spg_engine/select.rs
1//! SELECT execution — the window / meta-view / CTE variants and the
2//! subquery-resolution pre-pass. Lifted out of `lib.rs` (v7.32 engine
3//! modularisation). These `impl Engine` methods are dispatched from the
4//! bare-SELECT entry points and drive the non-trivial SELECT shapes.
5
6use alloc::borrow::Cow;
7use alloc::string::{String, ToString};
8use alloc::vec::Vec;
9
10use spg_sql::ast::{
11 ColumnName, Expr, FromClause, SelectItem, SelectStatement, Statement, TableRef, UnionKind,
12};
13use spg_storage::{
14 Catalog, ColumnSchema, DataType, Row, StorageError, TableSchema, Value, VecEncoding,
15};
16
17use crate::describe;
18use crate::eval::{EvalContext, EvalError};
19use crate::join::RowRef;
20use crate::system_catalog::collect_view_refs;
21use crate::{
22 ByteBudget, CancelToken, Engine, EngineError, OrderKey, QueryResult, aggregate,
23 apply_offset_and_limit, apply_offset_and_limit_tagged, approx_row_bytes, build_order_keys,
24 collect_meta_view_names, collect_qualified_refs, collect_scalar_subqueries,
25 collect_window_nodes, compute_window_partition, eval, expr_tree_has_subquery,
26 materialise_in_order, materialise_meta_view, memoize, order_by_value_cmp_in, partition_key_cmp,
27 rewrite_window_to_columns, select_has_window, select_references_meta_view, select_refers_to,
28 sort_by_keys, synth_info_key_column_usage, synth_info_referential_constraints,
29 synth_info_routines, synth_info_statistics, synth_information_schema_columns,
30 synth_information_schema_tables, synth_mysql_db, synth_mysql_user, synth_pg_attribute,
31 synth_pg_class, synth_pg_constraint, synth_pg_database, synth_pg_extension, synth_pg_index_raw,
32 synth_pg_indexes, synth_pg_namespace, synth_pg_operator, synth_pg_proc, synth_pg_roles,
33 synth_pg_sequence, synth_pg_settings, synth_pg_timezone_abbrevs, synth_pg_timezone_names,
34 synth_pg_trigger, synth_pg_type, synth_pg_views, topk_trim, try_gin_jsonb_seek, try_gin_seek,
35 try_index_seek, try_nsw_knn, try_pk_walk_top_n, try_trgm_seek, value_is_bigint,
36 value_is_integer, value_to_i64,
37};
38
39/// v7.39 (round 618) — a recursive term that can be run over the working set
40/// directly, instead of through a whole query execution per round.
41///
42/// PG plans the recursive term ONCE and re-scans a worktable each iteration.
43/// SPG emptied and refilled a real table and then called `exec_select_cancel`
44/// — FROM resolution, schema build, predicate compilation, projection build
45/// and result materialisation — for every round. Measured with the counting
46/// allocator on `WITH RECURSIVE r(n) AS (SELECT 1 UNION ALL SELECT n+1 FROM r
47/// WHERE n < N)`: about 40 allocations and 99 kB PER ROUND while the working
48/// set is one row, or 1.98 GB at N = 20000.
49///
50/// This is the shape that covers the ordinary recursive term: read the CTE,
51/// filter it, project it. Anything else — a join, an aggregate, a window, a
52/// subquery, DISTINCT, GROUP BY, ORDER BY, LIMIT, a locking clause, a
53/// non-table source — returns `None` and keeps the general path, so the
54/// answers it gives are the ones that path gave.
55struct RecursiveTermPlan<'t> {
56 items: Vec<&'t Expr>,
57 where_: Option<&'t Expr>,
58 alias: String,
59}
60
61fn plan_recursive_term<'t>(
62 t: &'t SelectStatement,
63 cte_name: &str,
64 ncols: usize,
65) -> Option<RecursiveTermPlan<'t>> {
66 if !t.unions.is_empty()
67 || !t.ctes.is_empty()
68 || t.distinct
69 || !t.distinct_on.is_empty()
70 || t.group_by.is_some()
71 || t.group_by_all
72 || t.having.is_some()
73 || !t.order_by.is_empty()
74 || t.limit.is_some()
75 || t.offset.is_some()
76 || t.limit_with_ties
77 || t.locking.is_some()
78 {
79 return None;
80 }
81 let from = t.from.as_ref()?;
82 if !from.joins.is_empty() {
83 return None;
84 }
85 let p = &from.primary;
86 if !p.name.eq_ignore_ascii_case(cte_name)
87 || p.as_of_segment.is_some()
88 || p.unnest_expr.is_some()
89 || !p.unnest_column_aliases.is_empty()
90 || p.with_ordinality
91 || p.generate_series_args.is_some()
92 || p.lateral_subquery.is_some()
93 || p.jsonb_each_text_arg.is_some()
94 || p.table_fn_call.is_some()
95 {
96 return None;
97 }
98 let unsupported = |e: &Expr| {
99 crate::aggregate::contains_aggregate(e)
100 || crate::subquery::expr_has_subquery(e)
101 || crate::window::expr_has_window_pub(e)
102 };
103 let mut items: Vec<&Expr> = Vec::with_capacity(t.items.len());
104 for it in &t.items {
105 match it {
106 SelectItem::Expr { expr, .. } => {
107 if unsupported(expr) {
108 return None;
109 }
110 items.push(expr);
111 }
112 // `*` would have to be expanded against the CTE's own schema;
113 // the general path already does that, so leave it there.
114 _ => return None,
115 }
116 }
117 if items.len() != ncols {
118 return None;
119 }
120 if let Some(w) = &t.where_
121 && unsupported(w)
122 {
123 return None;
124 }
125 Some(RecursiveTermPlan {
126 items,
127 where_: t.where_.as_ref(),
128 alias: p.alias.clone().unwrap_or_else(|| p.name.clone()),
129 })
130}
131
132impl Engine {
133 /// v4.12 window executor. Implements `ROW_NUMBER` / `RANK` /
134 /// `DENSE_RANK` and the partition-aware aggregates `SUM` /
135 /// `AVG` / `COUNT` / `MIN` / `MAX`. The plan is:
136 /// 1. Apply the WHERE filter.
137 /// 2. For each unique `WindowFunction` node in the projection,
138 /// partition + sort, compute the per-row value.
139 /// 3. Append the window values as synthetic columns (`__win_N`)
140 /// to the row schema.
141 /// 4. Rewrite the projection to read those columns.
142 /// 5. Hand off to the regular project / ORDER BY / LIMIT pipe.
143 #[allow(
144 clippy::too_many_lines,
145 clippy::type_complexity,
146 clippy::needless_range_loop
147 )] // window-eval is one cohesive pipe; splitting fragments
148 pub(crate) fn exec_select_with_window(
149 &self,
150 stmt: &SelectStatement,
151 cancel: CancelToken<'_>,
152 ) -> Result<QueryResult, EngineError> {
153 let from = stmt.from.as_ref().ok_or_else(|| {
154 EngineError::Unsupported("window functions require a FROM clause".into())
155 })?;
156 // v7.17.0 Phase 3.P0-43 — JOIN + window functions. Phase
157 // 3.6 rejected this combination outright ("queued for
158 // v5.x"); P0-43 materialises the join + WHERE through the
159 // existing nested-loop helper and runs the window pipeline
160 // on the joined row set with the combined `alias.col`
161 // schema. The window expressions resolve through the
162 // qualifier-aware column resolver same as the aggregate /
163 // projection paths on JOIN.
164 let (schema_cols_owned, alias_opt): (Vec<ColumnSchema>, Option<&str>);
165 // v7.39 (round 976) — rows this walk OWNS. A derived FROM item and
166 // a JOIN both produce rows that exist nowhere else, so they land
167 // here; a plain stored table does not, and borrows instead.
168 //
169 // It used to clone every row out of the table, on the reasoning
170 // that "the clone is cheap relative to the window computation that
171 // follows". Measured on 400k rows, `row_number() OVER ()` cost
172 // 31.881 ms against 46.520 with a 200-byte column added — so the
173 // clone tracks row width at about 36 ns per row per 200 bytes, and
174 // the window computation it was being compared against is a
175 // counter increment per row. Nothing downstream needs the rows
176 // owned: the very next statement used to be
177 // `filtered.iter().collect()` into the `&Row` slice the window
178 // pipeline actually reads.
179 let mut owned_rows: Vec<Row<'static>> = Vec::new();
180 // What the pipeline reads. Borrows `owned_rows` or the table.
181 let mut filtered: Vec<&Row<'static>> = Vec::new();
182 // Set by the branches that fill `owned_rows`, because "empty" is
183 // an answer a query can legitimately have and so cannot be the
184 // signal for which of the two holds the rows.
185 let mut rows_are_owned = false;
186 if from.joins.is_empty() {
187 let primary = &from.primary;
188 // v7.37 D.13 — window functions over a derived table (subquery /
189 // VALUES / unnest / generate_series). The catalog-by-name lookup
190 // below only finds real tables, so a derived primary threw
191 // TableNotFound. Materialise the derived rows + schema through the
192 // same helper the non-window FROM-primary path uses, then WHERE-
193 // filter and feed the identical window pipeline.
194 let is_derived = primary.lateral_subquery.is_some()
195 || primary.unnest_expr.is_some()
196 || primary.generate_series_args.is_some()
197 || primary.jsonb_each_text_arg.is_some()
198 || primary.table_fn_call.is_some();
199 if is_derived {
200 let (drows, dcols) = self.materialise_table_ref(primary)?;
201 schema_cols_owned = dcols;
202 alias_opt = primary.alias.as_deref();
203 let ctx = self.ev_ctx(&schema_cols_owned, alias_opt);
204 let mut owned: Vec<Row<'static>> = Vec::new();
205 for (i, row) in drows.into_iter().enumerate() {
206 if i.is_multiple_of(256) {
207 cancel.check()?;
208 }
209 if let Some(w) = &stmt.where_ {
210 let cond = eval::eval_expr(w, &row, &ctx)?;
211 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
212 continue;
213 }
214 }
215 owned.push(row);
216 }
217 owned_rows = owned;
218 rows_are_owned = true;
219 } else {
220 let table = self.active_catalog().get(&primary.name).ok_or_else(|| {
221 StorageError::TableNotFound {
222 name: primary.name.clone(),
223 }
224 })?;
225 let alias = primary.alias.as_deref().unwrap_or(primary.name.as_str());
226 schema_cols_owned = table.schema().columns.clone();
227 alias_opt = Some(alias);
228 let ctx = self.ev_ctx(&schema_cols_owned, alias_opt);
229 // The WHERE test, in ONE place, for all four ways a row can
230 // reach this walk. It deliberately does not touch the row
231 // collections: a closure that pushed into them would tie
232 // its argument to the closure body and no borrowed row
233 // could escape it, which is what forced the clone-shaped
234 // version of this loop in the first place.
235 let passes = |row: &Row<'static>| -> Result<bool, EngineError> {
236 if let Some(w) = &stmt.where_ {
237 let cond = eval::eval_expr(w, row, &ctx)?;
238 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
239 return Ok(false);
240 }
241 }
242 Ok(true)
243 };
244 // v7.37.15 Phase B — scan_visible filters rows by the
245 // engine's current snapshot. Phase B's `current_snapshot()`
246 // returns `Snapshot::unbounded()` so every row is visible,
247 // matching pre-v7.37.15 byte-for-byte. Phase C will wire
248 // real per-tx snapshots through this same callsite — no
249 // code change needed here when that lands.
250 let snap = self.current_snapshot();
251 if table.has_cold_rows_fast() {
252 // v7.36 (cold-tier coverage) — a cold segment's rows
253 // are produced on demand and live in a temporary this
254 // walk cannot borrow from, so a table carrying any owns
255 // its rows. Hot iter then cold iter, both through the
256 // same WHERE, as before.
257 let mut owned: Vec<Row<'static>> = Vec::new();
258 for (i, row) in table.scan_visible(&snap) {
259 if i.is_multiple_of(256) {
260 cancel.check()?;
261 }
262 if passes(row)? {
263 owned.push(row.clone());
264 }
265 }
266 let hot_len = table.row_count();
267 for (offset, row) in self.iter_cold_rows_of_table(table).iter().enumerate() {
268 let i = hot_len + offset;
269 if i.is_multiple_of(256) {
270 cancel.check()?;
271 }
272 if passes(row)? {
273 owned.push(row.clone());
274 }
275 }
276 owned_rows = owned;
277 rows_are_owned = true;
278 } else {
279 // v7.39 (round 975) — ask the indices first, the way
280 // the streaming walk has since round 970. This walk had
281 // the same hole and it is reached by any statement
282 // carrying a window function, so a WHERE that names an
283 // indexed column read the whole table: measured on 400k
284 // rows, `row_number() OVER () … WHERE id = 500` — a
285 // ONE-row answer on a primary key — took 13.762 ms
286 // against PG18.4's 0.151, while the same predicate
287 // without the window took 0.091. The cost was
288 // independent of how many rows survived (999 survivors
289 // cost 13.312 ms) and of row width (13.312 narrow vs
290 // 13.327 wide), which is what a full table walk looks
291 // like and what a result-shaped cost does not.
292 //
293 // The seek only NARROWS — `passes` still applies the
294 // whole WHERE — so no answer can change. Positions
295 // arrive visibility-filtered by the same predicate the
296 // scan applies and capped at a quarter of the table,
297 // and `None` walks the table exactly as before.
298 let seek_positions: Option<Vec<usize>> = stmt.where_.as_ref().and_then(|w| {
299 crate::index_access::try_index_seek_positions(
300 w,
301 &schema_cols_owned,
302 table,
303 alias,
304 &snap,
305 )
306 });
307 match seek_positions {
308 Some(mut positions) => {
309 // Table order, which is the order the scan
310 // would have produced.
311 positions.sort_unstable();
312 for (n, pos) in positions.into_iter().enumerate() {
313 if n.is_multiple_of(256) {
314 cancel.check()?;
315 }
316 let Some(row) = table.rows().get(pos) else {
317 continue;
318 };
319 if passes(row)? {
320 filtered.push(row);
321 }
322 }
323 }
324 None => {
325 for (i, row) in table.scan_visible(&snap) {
326 if i.is_multiple_of(256) {
327 cancel.check()?;
328 }
329 if passes(row)? {
330 filtered.push(row);
331 }
332 }
333 }
334 }
335 }
336 }
337 } else {
338 let deferred = self.build_joined_filtered_rows(
339 from,
340 stmt.where_.as_ref(),
341 cancel,
342 None,
343 &mut ByteBudget::new(self.max_query_bytes),
344 )?;
345 // A join's survivors are row-index tuples over its sources, so
346 // there is no single row to borrow — this branch owns them.
347 owned_rows = deferred.materialise();
348 rows_are_owned = true;
349 schema_cols_owned = deferred.combined_schema;
350 alias_opt = None;
351 }
352 if rows_are_owned {
353 filtered = owned_rows.iter().collect();
354 }
355 let schema_cols = &schema_cols_owned;
356 let ctx = self.ev_ctx(schema_cols, alias_opt);
357 let alias = alias_opt.unwrap_or("");
358 let n_rows = filtered.len();
359 // The window pipeline reads `&[&Row<'static>]`, and `filtered`
360 // already is one whichever branch produced it — the separate
361 // `filtered_refs` this used to build was the collect that made
362 // owning the rows look necessary.
363
364 // 2) Collect unique window function nodes from projection.
365 let mut window_nodes: Vec<Expr> = Vec::new();
366 for item in &stmt.items {
367 if let SelectItem::Expr { expr, .. } = item {
368 collect_window_nodes(expr, &mut window_nodes);
369 }
370 }
371 // v7.39 (round 592) — and from ORDER BY, which may name a window the
372 // select list never mentions. The order-key builder below rewrites
373 // window calls to `__win_N` columns, and a call that was never
374 // collected has no column to become.
375 for o in &stmt.order_by {
376 collect_window_nodes(&o.expr, &mut window_nodes);
377 }
378
379 // 3) For each window, compute per-row value.
380 // Index: same order as window_nodes; for row i, win_vals[w][i].
381 let mut win_vals: Vec<Vec<Value<'static>>> = Vec::with_capacity(window_nodes.len());
382 for wnode in &window_nodes {
383 let Expr::WindowFunction {
384 name,
385 args,
386 partition_by,
387 order_by,
388 frame,
389 null_treatment,
390 filter,
391 } = wnode
392 else {
393 unreachable!("collect_window_nodes pushes only WindowFunction");
394 };
395 // Compute (partition_key, order_key, original_index) for each row.
396 // v7.39 (round 593) — a key that is a plain column sits at the same
397 // position in every row, but was resolved BY NAME for each one. A
398 // per-library profile of `lag(id) OVER (ORDER BY id)` put
399 // `resolve_column` at 5.8% of the query on its own, with
400 // `rehydrate_cell` and the `eval_expr` dispatch behind it. Resolve
401 // once; anything that is not a plain column keeps the resolver.
402 let p_bound: Vec<Option<usize>> = partition_by
403 .iter()
404 .map(|e| crate::orderby::bound_column_position(e, schema_cols, alias_opt))
405 .collect();
406 let o_bound: Vec<Option<usize>> = order_by
407 .iter()
408 .map(|(e, _, _)| crate::orderby::bound_column_position(e, schema_cols, alias_opt))
409 .collect();
410 let arg_bound = args
411 .first()
412 .and_then(|a| crate::orderby::bound_column_position(a, schema_cols, alias_opt));
413 // v7.39 (round 690) — a window's ORDER BY over a column that
414 // declares a collation sorts by it, the same as a top-level
415 // ORDER BY. Resolved from the bound position, so only a bare
416 // column gets one; an expression produces a new value and the
417 // derivation that would give IT a collation is unbuilt.
418 let o_colls: Vec<Option<alloc::string::String>> = o_bound
419 .iter()
420 .map(|p| {
421 p.and_then(|pos| schema_cols.get(pos))
422 .and_then(|sc| sc.collation_name.clone())
423 .filter(|n| crate::collate::is_supported(n))
424 })
425 .collect();
426 let mut indexed: Vec<(Vec<Value<'static>>, Vec<(Value, bool, Option<bool>)>, usize)> =
427 Vec::with_capacity(n_rows);
428 // v7.39 (round 731) — single bound INT partition key, no window
429 // ORDER BY: group on the i64 directly. The generic build paid
430 // two heap Vecs per row (pkey + empty okey) plus a canonical
431 // string encode per row just to bucket 500k rows into 100
432 // groups; the whole per-row key apparatus disappears here.
433 // Neither key Vec is read downstream on this path: the hash
434 // grouping replaces partition_key_cmp, and okey is empty by
435 // construction.
436 let int_pkey_fast = order_by.is_empty()
437 && partition_by.len() == 1
438 && p_bound[0].is_some_and(|pos| {
439 matches!(
440 schema_cols.get(pos).map(|c| c.ty),
441 Some(
442 spg_storage::DataType::Int
443 | spg_storage::DataType::BigInt
444 | spg_storage::DataType::SmallInt
445 )
446 )
447 });
448 // v7.39 (round 979) — the same idea for a single bound INT
449 // window ORDER BY: sort on the i64 instead of on a heap vector
450 // per row.
451 //
452 // Measured at 400k rows (round 978, ablation, answer checked
453 // byte-for-byte against the general path on a key column that
454 // is a permutation): `row_number() OVER (ORDER BY k)` went
455 // 157.057-157.868 ms to 31.253-31.679, which is 79.8% and puts
456 // it on top of the `OVER ()` baseline — the sort essentially
457 // disappears. Round 977 had already shown the cost was
458 // key-shaped rather than row-shaped: the sort's share was
459 // 132.0 ms on a three-integer table and 132.5 with a 200-byte
460 // column added, and a per-row COPY does scale with width
461 // (round 976 measured that at +36 ns/row/200 bytes).
462 //
463 // Gated to ROW_NUMBER, which is the one function that reads
464 // neither key vector — it numbers the order it is handed.
465 // `rank` and `dense_rank` compare adjacent entries' order keys
466 // in `compute_window_partition`, so leaving those vectors
467 // empty would silently give every row rank 1. A wider version
468 // would carry the i64 in the entry and teach those two to use
469 // it; this one is the part that can be shown correct by
470 // construction.
471 let int_okey_fast = partition_by.is_empty()
472 && order_by.len() == 1
473 && frame.is_none()
474 && filter.is_none()
475 && matches!(null_treatment, spg_sql::ast::NullTreatment::Respect)
476 && name.eq_ignore_ascii_case("row_number")
477 && o_bound[0].is_some_and(|pos| {
478 matches!(
479 schema_cols.get(pos).map(|c| c.ty),
480 Some(
481 spg_storage::DataType::Int
482 | spg_storage::DataType::BigInt
483 | spg_storage::DataType::SmallInt
484 )
485 )
486 });
487 // Set when a cell in that column turns out not to be an
488 // integer after all. The declared type says it should be, but
489 // "should" is not a thing to sort 400k rows on, so the general
490 // path takes over and this build is discarded.
491 let mut int_okey_bailed = false;
492 if int_okey_fast {
493 let pos = o_bound[0].expect("gated bound");
494 let desc = order_by[0].1;
495 // PG orders NULLs last ascending and first descending
496 // unless the query says otherwise.
497 let nulls_first = order_by[0].2.unwrap_or(desc);
498 let mut keyed: Vec<(bool, i64, usize)> = Vec::with_capacity(n_rows);
499 for (i, row) in filtered.iter().enumerate() {
500 match row.values.get(pos) {
501 Some(Value::Int(n)) => keyed.push((false, i64::from(*n), i)),
502 Some(Value::BigInt(n)) => keyed.push((false, *n, i)),
503 Some(Value::SmallInt(n)) => keyed.push((false, i64::from(*n), i)),
504 Some(Value::Null) | None => keyed.push((true, 0, i)),
505 Some(_) => {
506 int_okey_bailed = true;
507 break;
508 }
509 }
510 }
511 if !int_okey_bailed {
512 // `null_rank` puts NULLs on the side the query asked
513 // for; the row's original index breaks every tie, so
514 // equal keys keep the order the scan produced — what
515 // the stable sort below would have given them.
516 let null_rank = |is_null: bool| -> u8 { u8::from(is_null != nulls_first) };
517 keyed.sort_unstable_by(|a, b| {
518 null_rank(a.0)
519 .cmp(&null_rank(b.0))
520 .then_with(|| {
521 if a.0 {
522 core::cmp::Ordering::Equal
523 } else if desc {
524 b.1.cmp(&a.1)
525 } else {
526 a.1.cmp(&b.1)
527 }
528 })
529 .then_with(|| a.2.cmp(&b.2))
530 });
531 for (_, _, i) in keyed {
532 indexed.push((Vec::new(), Vec::new(), i));
533 }
534 } else {
535 indexed.clear();
536 }
537 }
538 if int_okey_fast && !int_okey_bailed {
539 // Ordered above; nothing else to build.
540 } else if int_pkey_fast {
541 let pos = p_bound[0].expect("gated bound");
542 let mut slot: hashbrown::HashMap<Option<i64>, usize> = hashbrown::HashMap::new();
543 let mut groups: Vec<Vec<usize>> = Vec::new();
544 for (i, row) in filtered.iter().enumerate() {
545 let k: Option<i64> = match row.values.get(pos) {
546 Some(Value::BigInt(n)) => Some(*n),
547 Some(Value::Int(n)) => Some(i64::from(*n)),
548 Some(Value::SmallInt(n)) => Some(i64::from(*n)),
549 _ => None,
550 };
551 match slot.get(&k) {
552 Some(&gi) => groups[gi].push(i),
553 None => {
554 slot.insert(k, groups.len());
555 groups.push(alloc::vec![i]);
556 }
557 }
558 }
559 // The downstream partition-boundary scan compares pkeys
560 // of ADJACENT entries, so the key must ride along — one
561 // single-element Vec per row (half the generic build's
562 // allocations, no string encode).
563 for g in groups {
564 for i in g {
565 let k: Value<'static> = match filtered[i].values.get(pos) {
566 Some(v) => v.clone(),
567 None => Value::Null,
568 };
569 indexed.push((alloc::vec![k], Vec::new(), i));
570 }
571 }
572 } else {
573 for (i, row) in filtered.iter().enumerate() {
574 let pkey: Vec<Value<'static>> = partition_by
575 .iter()
576 .enumerate()
577 .map(
578 |(k, p)| match p_bound[k].and_then(|pos| row.values.get(pos)) {
579 Some(v) => Ok(v.clone()),
580 None => eval::eval_expr(p, row, &ctx),
581 },
582 )
583 .collect::<Result<_, _>>()?;
584 // v7.39 (read01 round 54) — a window's ORDER BY over an enum
585 // column must sort by MEMBER order (enumsortorder), not the
586 // label's text. Enum values are Text at runtime, so the raw
587 // value key sorted alphabetically — `row_number() OVER (ORDER
588 // BY mood)` numbered the rows happy,ok,sad. Substitute the
589 // member ordinal, the same key the top-level ORDER BY uses.
590 // (Closes the enum-order knife's recorded window residual.)
591 let okey: Vec<(Value, bool, Option<bool>)> = order_by
592 .iter()
593 .enumerate()
594 .map(|(k, (e, desc, nf))| -> Result<_, EngineError> {
595 let v = match o_bound[k].and_then(|pos| row.values.get(pos)) {
596 Some(v) => v.clone(),
597 None => eval::eval_expr(e, row, &ctx)?,
598 };
599 let v = match crate::orderby::enum_order_ordinal(e, &v, &ctx) {
600 Some(ord) => Value::Float(ord),
601 None => v,
602 };
603 Ok((v, *desc, *nf))
604 })
605 .collect::<Result<_, _>>()?;
606 indexed.push((pkey, okey, i));
607 }
608 }
609 // Sort by (partition_key, order_key). Partition key uses
610 // a stable encoded form; order key respects ASC/DESC.
611 // v7.39 (round 731) — with NO window ORDER BY the sort's only
612 // job was putting same-partition rows next to each other, and a
613 // 500k-row comparison sort is a spectacular way to hash-group:
614 // the panel's `sum(id) OVER (PARTITION BY g)` spent ~100 ms
615 // here. Group by encoded key instead, preserving row order
616 // inside each group — exactly what the stable sort preserved,
617 // so every function (row_number included) answers the same.
618 if int_okey_fast && !int_okey_bailed {
619 // Already ordered by the i64 key above.
620 } else if int_pkey_fast {
621 // Already grouped above; same-partition rows are adjacent
622 // in original row order.
623 } else if order_by.is_empty() && !partition_by.is_empty() {
624 let mut slot: hashbrown::HashMap<String, usize> = hashbrown::HashMap::new();
625 let mut groups: Vec<
626 Vec<(Vec<Value<'static>>, Vec<(Value, bool, Option<bool>)>, usize)>,
627 > = Vec::new();
628 let mut keybuf = String::new();
629 for entry in indexed.drain(..) {
630 keybuf.clear();
631 for v in &entry.0 {
632 crate::aggregate::push_canonical_key(&mut keybuf, v);
633 }
634 match slot.get(keybuf.as_str()) {
635 Some(&gi) => groups[gi].push(entry),
636 None => {
637 slot.insert(keybuf.clone(), groups.len());
638 groups.push(alloc::vec![entry]);
639 }
640 }
641 }
642 for g in groups {
643 indexed.extend(g);
644 }
645 } else {
646 indexed.sort_by(|a, b| {
647 let p_cmp = partition_key_cmp(&a.0, &b.0);
648 if p_cmp != core::cmp::Ordering::Equal {
649 return p_cmp;
650 }
651 crate::window::order_key_cmp_in(&a.1, &b.1, &o_colls)
652 });
653 }
654 // Per-partition compute.
655 let mut out_vals: Vec<Value<'static>> = alloc::vec![Value::Null; n_rows];
656 let mut p_start = 0;
657 while p_start < indexed.len() {
658 let mut p_end = p_start + 1;
659 while p_end < indexed.len()
660 && partition_key_cmp(&indexed[p_start].0, &indexed[p_end].0)
661 == core::cmp::Ordering::Equal
662 {
663 p_end += 1;
664 }
665 // Compute the function within this partition slice.
666 compute_window_partition(
667 name,
668 args,
669 arg_bound,
670 !order_by.is_empty(),
671 frame.as_ref(),
672 *null_treatment,
673 filter.as_deref(),
674 &indexed[p_start..p_end],
675 &filtered,
676 &ctx,
677 &mut out_vals,
678 )?;
679 p_start = p_end;
680 }
681 win_vals.push(out_vals);
682 }
683
684 // 4) Build extended schema: original columns + synthetic.
685 let mut ext_cols = schema_cols.clone();
686 for i in 0..window_nodes.len() {
687 ext_cols.push(ColumnSchema::new(
688 alloc::format!("__win_{i}"),
689 DataType::Text, // type doesn't matter for projection eval
690 true,
691 ));
692 }
693 // 6) Rewrite the projection: WindowFunction nodes → Column(__win_N).
694 let mut rewritten_items: Vec<SelectItem> = Vec::with_capacity(stmt.items.len());
695 for item in &stmt.items {
696 let new_item = match item {
697 SelectItem::Wildcard => SelectItem::Wildcard,
698 SelectItem::QualifiedWildcard(q) => SelectItem::QualifiedWildcard(q.clone()),
699 SelectItem::Expr { expr, alias } => {
700 let mut e = expr.clone();
701 rewrite_window_to_columns(&mut e, &window_nodes);
702 // The rewrite swaps the window call for a synthetic
703 // `__win_N` column, and the projection then reported
704 // THAT as the column name — `SELECT count(*) OVER ()`
705 // answered `__win_0`, an internal name, where PG18
706 // answers `count`. Pin the name while the call the
707 // column is named for is still in hand.
708 let alias = if alias.is_none() && e != *expr {
709 Some(default_output_name(expr, self.backslash_escapes))
710 } else {
711 alias.clone()
712 };
713 SelectItem::Expr { expr: e, alias }
714 }
715 };
716 rewritten_items.push(new_item);
717 }
718
719 // 7) Project into final rows. JOIN case uses None so the
720 // qualifier check in `resolve_column` falls through to the
721 // composite `alias.col` schema lookup; single-table case
722 // keeps the bare alias so `bare_col` resolution still
723 // works for the projection's per-row column references.
724 // v7.39 (read01 round 54) — build through `ev_ctx`, the canonical
725 // constructor: it threads the catalog (plus render style / tz / GUCs)
726 // that a bare `EvalContext::new` drops. Without the catalog the OUTER
727 // `ORDER BY <enum col>` of a windowed query sorted by TEXT — the
728 // window values were right, the row order silently was not.
729 let ext_ctx = self.ev_ctx(&ext_cols, alias_opt);
730 let projection = build_projection_hiding_tail(
731 &rewritten_items,
732 &ext_cols,
733 alias,
734 self.backslash_escapes,
735 window_nodes.len(),
736 )?;
737 let mut tagged: Vec<(Vec<OrderKey>, Row)> = Vec::with_capacity(n_rows);
738 // v7.39 (round 592) — the extended row (input columns plus the window
739 // values) used to be materialised for EVERY input row and kept until
740 // the projection had run: the input values cloned into a fresh Vec,
741 // then grown once to take the window columns. A counting allocator put
742 // the window path at 4 allocations a row where a plain derived table
743 // takes 1, and named all four — the input row, the clone, the growth,
744 // and the projected row. Only the last has to exist afterwards, so the
745 // extended row is one buffer refilled per row.
746 let mut ext_row: Row<'static> =
747 Row::new(Vec::with_capacity(schema_cols.len() + window_nodes.len()));
748 for i in 0..n_rows {
749 if i.is_multiple_of(256) {
750 cancel.check()?;
751 }
752 ext_row.values.clear();
753 ext_row.values.extend(filtered[i].values.iter().cloned());
754 for w in 0..window_nodes.len() {
755 ext_row.values.push(win_vals[w][i].clone());
756 }
757 let row = &ext_row;
758 let mut values = Vec::with_capacity(projection.len());
759 for p in &projection {
760 values.push(eval::eval_expr(&p.expr, row, &ext_ctx)?);
761 }
762 let order_keys = if stmt.order_by.is_empty() {
763 Vec::new()
764 } else {
765 let mut keys = Vec::with_capacity(stmt.order_by.len());
766 for o in &stmt.order_by {
767 let mut e = o.expr.clone();
768 rewrite_window_to_columns(&mut e, &window_nodes);
769 let key = eval::eval_expr(&e, row, &ext_ctx)?;
770 // v7.39 (read01 round 54) — this path builds its order keys
771 // itself instead of going through `build_order_keys`, so it
772 // skipped the enum-ordinal substitution: the OUTER
773 // `ORDER BY <enum col>` of a windowed query sorted by the
774 // label's TEXT, not by member order. The window values were
775 // right and only the row order was wrong — silently.
776 match crate::orderby::enum_order_ordinal(&e, &key, &ext_ctx) {
777 Some(ord) => keys.push(value_to_order_key(&Value::Float(ord))?),
778 None => keys.push(value_to_order_key(&key)?),
779 }
780 }
781 keys
782 };
783 tagged.push((order_keys, Row::new(values)));
784 }
785 // ORDER BY + LIMIT/OFFSET on the projected rows.
786 if !stmt.order_by.is_empty() {
787 let descs: Vec<bool> = stmt.order_by.iter().map(|o| o.desc).collect();
788 sort_by_keys(&mut tagged, &descs);
789 }
790 let mut out_rows: Vec<Row<'static>> = tagged.into_iter().map(|(_, r)| r).collect();
791 // v7.37 D.41 — `SELECT DISTINCT` over a window projection: the window
792 // pipeline builds one output row per input row, so DISTINCT must dedup the
793 // projected rows (PG evaluates window functions before DISTINCT). Applied
794 // after ORDER BY (duplicate rows share sort keys, so order is preserved)
795 // and before LIMIT.
796 if stmt.distinct {
797 out_rows = dedup_rows(out_rows, self.backslash_escapes);
798 }
799 apply_offset_and_limit(&mut out_rows, stmt.offset_literal(), stmt.limit_literal());
800 let final_cols: Vec<ColumnSchema> = projection
801 .into_iter()
802 .map(|p| {
803 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
804 c.user_enum_type = p.user_enum_type;
805 c.collation_name = p.collation_name;
806 c.mysql_fsp = p.mysql_fsp;
807 c
808 })
809 .collect();
810 Ok(QueryResult::Rows {
811 columns: final_cols,
812 rows: out_rows,
813 })
814 }
815
816 /// v4.11: materialise each CTE into a temp table inside a
817 /// cloned catalog, then run the body SELECT against a fresh
818 /// engine instance that owns the enriched catalog. The clone
819 /// is moderately expensive — only paid by CTE-bearing queries.
820 /// Subqueries inside CTE bodies / the main body resolve as
821 /// usual; `clock_fn` is propagated so `NOW()` lines up.
822 /// v7.16.2 — mailrs round-10 A.3. Materialise the
823 /// `information_schema.*` / `pg_catalog.*` virtual views
824 /// the SELECT references, then re-execute the SELECT
825 /// against an enriched catalog where those views are real
826 /// tables. Same pattern as `exec_with_ctes`. The temp
827 /// engine carries `meta_views_materialised = true` so its
828 /// own meta-dispatch short-circuits — without that we'd
829 /// infinite-recurse since the temp catalog's view name
830 /// still starts with `__spg_info_` and re-triggers the
831 /// check.
832 pub(crate) fn exec_select_with_meta_views(
833 &self,
834 stmt: &SelectStatement,
835 cancel: CancelToken<'_>,
836 ) -> Result<QueryResult, EngineError> {
837 let catalog = self.meta_view_catalog(stmt)?;
838 let mut temp = Engine::restore(catalog);
839 if let Some(c) = self.clock {
840 temp = temp.with_clock(c);
841 }
842 if let Some(f) = self.salt_fn {
843 temp = temp.with_salt_fn(f);
844 }
845 // v7.39 (round 522) — the temp engine holds the materialised
846 // catalog and, until now, nothing of the SESSION. So every
847 // session-scoped answer changed the moment a system view
848 // appeared in the FROM clause: `SELECT current_user` said
849 // `unmei` and `SELECT current_user FROM pg_class` said `admin`;
850 // `current_setting('work_mem')` fell back to the boot default
851 // after a SET; `application_name` read empty. A privilege check
852 // written against a catalog join was reading a different
853 // identity than the same check written without one.
854 //
855 // Carry what a session can be observed through — its parameters
856 // (which is also where the session user lives), the role store
857 // the privilege builtins read, the dialect, and the rendering
858 // settings a timestamp is spelled with.
859 temp.session_params.clone_from(&self.session_params);
860 temp.users.clone_from(&self.users);
861 temp.backslash_escapes = self.backslash_escapes;
862 temp.mysql_strict = self.mysql_strict;
863 temp.render_style = self.render_style;
864 temp.tz_offset_fn = self.tz_offset_fn;
865 temp.tz_localize_fn = self.tz_localize_fn;
866 temp.tz_abbrev_fn = self.tz_abbrev_fn;
867 temp.meta_views_materialised = true;
868 temp.exec_select_cancel(stmt, cancel)
869 }
870
871 /// v7.39 (round 462) — the catalog a meta-view SELECT resolves
872 /// against: this engine's catalog with every `__spg_*` view the
873 /// statement references materialised into it.
874 ///
875 /// Split out of `exec_select_with_meta_views` so Describe can reach
876 /// the same shapes execution reaches. Describe used to look the FROM
877 /// relation up in the plain catalog, where a system view does not
878 /// exist, and reported "no columns" for every one of them — so an
879 /// extended-protocol client reading `pg_stat_user_tables` got rows
880 /// with no column metadata. Sharing the materialisation means a
881 /// view added here is described correctly the day it is added.
882 pub(crate) fn meta_view_catalog(&self, stmt: &SelectStatement) -> Result<Catalog, EngineError> {
883 let mut needed: alloc::collections::BTreeSet<String> = alloc::collections::BTreeSet::new();
884 collect_meta_view_names(stmt, &mut needed);
885 let mut catalog = self.active_catalog().clone();
886 for view in &needed {
887 if catalog.get(view).is_some() {
888 continue;
889 }
890 match view.as_str() {
891 "__spg_info_columns" => {
892 let (schema, rows) = synth_information_schema_columns(
893 self.active_catalog(),
894 self.backslash_escapes,
895 );
896 materialise_meta_view(&mut catalog, view, schema, rows)?;
897 }
898 "__spg_info_tables" => {
899 let (schema, rows) = synth_information_schema_tables(self.active_catalog());
900 materialise_meta_view(&mut catalog, view, schema, rows)?;
901 }
902 "__spg_pg_class" => {
903 let (schema, rows) = synth_pg_class(
904 self.active_catalog(),
905 i64::try_from(self.vacuum_oldest_active()).unwrap_or(i64::MAX),
906 );
907 materialise_meta_view(&mut catalog, view, schema, rows)?;
908 }
909 "__spg_pg_attribute" => {
910 let (schema, rows) = synth_pg_attribute(self.active_catalog());
911 materialise_meta_view(&mut catalog, view, schema, rows)?;
912 }
913 // v7.17.0 Phase 3.P0-50 — pg_catalog.pg_type for
914 // sqlx / SQLAlchemy / Diesel / pgAdmin lookups.
915 "__spg_pg_type" => {
916 let (schema, rows) = synth_pg_type(self.active_catalog());
917 materialise_meta_view(&mut catalog, view, schema, rows)?;
918 }
919 // v7.39 (round 621) — pg_catalog.pg_operator, which did not
920 // exist at all.
921 "__spg_pg_operator" => {
922 let (schema, rows) = synth_pg_operator(self.active_catalog());
923 materialise_meta_view(&mut catalog, view, schema, rows)?;
924 }
925 // v7.17.0 Phase 3.P0-51 — pg_catalog.pg_proc for
926 // function-name introspection (ORM / pgAdmin).
927 "__spg_pg_proc" => {
928 let (schema, rows) = synth_pg_proc(self.active_catalog());
929 materialise_meta_view(&mut catalog, view, schema, rows)?;
930 }
931 // v7.24 (round-16 D) — pg_catalog.pg_trigger. The
932 // round-16 "why doesn't prod fire the trigger"
933 // question was unanswerable because triggers had NO
934 // introspection surface; tgname/tgenabled plus the
935 // pragmatic relname/timing/events/function columns
936 // make "is it registered and enabled" a one-liner.
937 "__spg_pg_trigger" => {
938 let (schema, rows) = synth_pg_trigger(self.active_catalog());
939 materialise_meta_view(&mut catalog, view, schema, rows)?;
940 }
941 // v7.17.0 Phase 3.P0-52 — pg_catalog.pg_namespace
942 // (schema list for admin tools' tree views).
943 "__spg_pg_namespace" => {
944 let (schema, rows) = synth_pg_namespace(self.active_catalog());
945 materialise_meta_view(&mut catalog, view, schema, rows)?;
946 }
947 // v7.39 — pg_tables convenience view (was a pgwire
948 // canned response that ignored projections).
949 "__spg_pg_tables" => {
950 let (schema, rows) =
951 crate::system_catalog::synth_pg_tables(self.active_catalog());
952 materialise_meta_view(&mut catalog, view, schema, rows)?;
953 }
954 // v7.37.24 (24.1) — pg_catalog.pg_enum (label list
955 // for ENUM types; sqlx / ORM enum codecs read this).
956 "__spg_pg_enum" => {
957 let (schema, rows) =
958 crate::system_catalog::synth_pg_enum(self.active_catalog());
959 materialise_meta_view(&mut catalog, view, schema, rows)?;
960 }
961 // v7.37.21 (21.13) — pg_catalog.pg_replication_slots
962 // (shape-stable empty until 21.12 persists slot state).
963 // v7.39 (round 277) — session-scoped prepared statements.
964 "__spg_pg_prepared_statements" => {
965 let (schema, rows) = crate::system_catalog::synth_pg_prepared_statements(
966 &self.prepared_statements,
967 );
968 materialise_meta_view(&mut catalog, view, schema, rows)?;
969 }
970 "__spg_pg_replication_slots" => {
971 let (schema, rows) =
972 crate::system_catalog::synth_pg_replication_slots(self.active_catalog());
973 materialise_meta_view(&mut catalog, view, schema, rows)?;
974 }
975 // v7.37.21 (21.13-b) — pg_catalog.pg_publication
976 // (one row per CREATE PUBLICATION).
977 "__spg_pg_publication" => {
978 let (schema, rows) = crate::system_catalog::synth_pg_publication(self);
979 materialise_meta_view(&mut catalog, view, schema, rows)?;
980 }
981 // v7.37.21 (21.13-c) — pg_catalog.pg_subscription
982 // (one row per CREATE SUBSCRIPTION; subconninfo
983 // redacted so dashboards can't leak credentials).
984 "__spg_pg_subscription" => {
985 let (schema, rows) = crate::system_catalog::synth_pg_subscription(self);
986 materialise_meta_view(&mut catalog, view, schema, rows)?;
987 }
988 // v7.37.22 (22.x-stat-db) — pg_catalog.pg_stat_database
989 // (one row for SPG's single database; counters are
990 // shape-stable 0 until wiring lands).
991 "__spg_pg_stat_database" => {
992 let (schema, rows) = crate::system_catalog::synth_pg_stat_database(
993 self,
994 self.stat_tup_inserted,
995 self.stat_tup_updated,
996 self.stat_tup_deleted,
997 );
998 materialise_meta_view(&mut catalog, view, schema, rows)?;
999 }
1000 // v7.37.22 (22.14) — pg_catalog.pg_stat_user_tables
1001 // (per-table churn counters; live_tup = row count).
1002 "__spg_pg_stat_user_tables" => {
1003 // r192 — DML counters come from the engine-side
1004 // non-transactional map, not the (tx-shadowed)
1005 // catalog tables.
1006 let (schema, rows) = crate::system_catalog::synth_pg_stat_user_tables(
1007 self.active_catalog(),
1008 &self.table_write_stats,
1009 );
1010 materialise_meta_view(&mut catalog, view, schema, rows)?;
1011 }
1012 // v7.37.22 (22.15) — pg_catalog.pg_stat_user_indexes
1013 // (per-index usage counters; flag unused indexes).
1014 "__spg_pg_stat_user_indexes" => {
1015 let (schema, rows) =
1016 crate::system_catalog::synth_pg_stat_user_indexes(self.active_catalog());
1017 materialise_meta_view(&mut catalog, view, schema, rows)?;
1018 }
1019 // v7.37.22 (22.16) — pg_catalog.pg_stat_bgwriter.
1020 "__spg_pg_stat_bgwriter" => {
1021 let (schema, rows) =
1022 crate::system_catalog::synth_pg_stat_bgwriter(self.active_catalog());
1023 materialise_meta_view(&mut catalog, view, schema, rows)?;
1024 }
1025 // v7.38 (read01 P3.14) — pg_catalog.pg_stat_checkpointer /
1026 // pg_stat_wal shell views (shape-stable, counters pending).
1027 "__spg_pg_stat_checkpointer" => {
1028 let (schema, rows) =
1029 crate::system_catalog::synth_pg_stat_checkpointer(self.active_catalog());
1030 materialise_meta_view(&mut catalog, view, schema, rows)?;
1031 }
1032 "__spg_pg_stat_wal" => {
1033 let (schema, rows) =
1034 crate::system_catalog::synth_pg_stat_wal(self.active_catalog());
1035 materialise_meta_view(&mut catalog, view, schema, rows)?;
1036 }
1037 // v7.38 (read01 P3.15) — pg_catalog.pg_stat_slru /
1038 // pg_stat_subscription_stats shell views.
1039 "__spg_pg_stat_slru" => {
1040 let (schema, rows) =
1041 crate::system_catalog::synth_pg_stat_slru(self.active_catalog());
1042 materialise_meta_view(&mut catalog, view, schema, rows)?;
1043 }
1044 "__spg_pg_stat_subscription_stats" => {
1045 let (schema, rows) = crate::system_catalog::synth_pg_stat_subscription_stats(
1046 self.active_catalog(),
1047 );
1048 materialise_meta_view(&mut catalog, view, schema, rows)?;
1049 }
1050 // v7.37.22 (22.17) — pg_catalog.pg_stat_archiver.
1051 "__spg_pg_stat_archiver" => {
1052 let (schema, rows) =
1053 crate::system_catalog::synth_pg_stat_archiver(self.active_catalog());
1054 materialise_meta_view(&mut catalog, view, schema, rows)?;
1055 }
1056 // v7.37.21 (21.13-d) — pg_catalog.pg_stat_replication.
1057 "__spg_pg_stat_replication" => {
1058 let (schema, rows) =
1059 crate::system_catalog::synth_pg_stat_replication(self.active_catalog());
1060 materialise_meta_view(&mut catalog, view, schema, rows)?;
1061 }
1062 // v7.37.24 (24.13) — pg_catalog.pg_am.
1063 "__spg_pg_am" => {
1064 let (schema, rows) = crate::system_catalog::synth_pg_am(self.active_catalog());
1065 materialise_meta_view(&mut catalog, view, schema, rows)?;
1066 }
1067 // v7.37.22 (22.18) — pg_catalog.pg_stat_io (PG 16+).
1068 "__spg_pg_stat_io" => {
1069 let (schema, rows) =
1070 crate::system_catalog::synth_pg_stat_io(self.active_catalog());
1071 materialise_meta_view(&mut catalog, view, schema, rows)?;
1072 }
1073 // v7.37.22 (22.19) — pg_catalog.pg_stat_user_functions.
1074 "__spg_pg_stat_user_functions" => {
1075 let (schema, rows) =
1076 crate::system_catalog::synth_pg_stat_user_functions(self.active_catalog());
1077 materialise_meta_view(&mut catalog, view, schema, rows)?;
1078 }
1079 // v7.39 (round 287) — pg_catalog.pg_largeobject{,_metadata}.
1080 "__spg_pg_largeobject" => {
1081 let (schema, rows) =
1082 crate::system_catalog::synth_pg_largeobject(self.active_catalog());
1083 materialise_meta_view(&mut catalog, view, schema, rows)?;
1084 }
1085 "__spg_pg_largeobject_metadata" => {
1086 let (schema, rows) =
1087 crate::system_catalog::synth_pg_largeobject_metadata(self.active_catalog());
1088 materialise_meta_view(&mut catalog, view, schema, rows)?;
1089 }
1090 // v7.37.23 (23.7-a) — pg_catalog.pg_statistic_ext.
1091 "__spg_pg_statistic_ext" => {
1092 let (schema, rows) =
1093 crate::system_catalog::synth_pg_statistic_ext(self.active_catalog());
1094 materialise_meta_view(&mut catalog, view, schema, rows)?;
1095 }
1096 // v7.37.24 (24.15) — pg_catalog.pg_statistic.
1097 "__spg_pg_statistic" => {
1098 let (schema, rows) =
1099 crate::system_catalog::synth_pg_statistic(self.active_catalog());
1100 materialise_meta_view(&mut catalog, view, schema, rows)?;
1101 }
1102 // v7.37.22 (22.20) — pg_catalog.pg_stat_progress_vacuum.
1103 "__spg_pg_stat_progress_vacuum" => {
1104 let (schema, rows) =
1105 crate::system_catalog::synth_pg_stat_progress_vacuum(self.active_catalog());
1106 materialise_meta_view(&mut catalog, view, schema, rows)?;
1107 }
1108 // v7.37.22 (22.21) — pg_catalog.pg_stat_progress_create_index.
1109 "__spg_pg_stat_progress_create_index" => {
1110 let (schema, rows) = crate::system_catalog::synth_pg_stat_progress_create_index(
1111 self.active_catalog(),
1112 );
1113 materialise_meta_view(&mut catalog, view, schema, rows)?;
1114 }
1115 // v7.37.22 (22.22) — pg_catalog.pg_stat_progress_analyze.
1116 "__spg_pg_stat_progress_analyze" => {
1117 let (schema, rows) = crate::system_catalog::synth_pg_stat_progress_analyze(
1118 self.active_catalog(),
1119 );
1120 materialise_meta_view(&mut catalog, view, schema, rows)?;
1121 }
1122 // v7.37.24 (24.16) — pg_catalog.pg_inherits
1123 // (partition parent → child OID mapping).
1124 "__spg_pg_inherits" => {
1125 let (schema, rows) =
1126 crate::system_catalog::synth_pg_inherits(self.active_catalog());
1127 materialise_meta_view(&mut catalog, view, schema, rows)?;
1128 }
1129 // v7.39 (round 650) — the text-search catalogs, filled
1130 // with what SPG actually has rather than PG's thirty.
1131 "__spg_pg_ts_config_map" => {
1132 let (schema, rows) =
1133 crate::system_catalog::synth_pg_ts_config_map(self.active_catalog());
1134 materialise_meta_view(&mut catalog, view, schema, rows)?;
1135 }
1136 "__spg_pg_ts_config" => {
1137 let (schema, rows) =
1138 crate::system_catalog::synth_pg_ts_config(self.active_catalog());
1139 materialise_meta_view(&mut catalog, view, schema, rows)?;
1140 }
1141 "__spg_pg_ts_dict" => {
1142 let (schema, rows) =
1143 crate::system_catalog::synth_pg_ts_dict(self.active_catalog());
1144 materialise_meta_view(&mut catalog, view, schema, rows)?;
1145 }
1146 "__spg_pg_ts_parser" => {
1147 let (schema, rows) =
1148 crate::system_catalog::synth_pg_ts_parser(self.active_catalog());
1149 materialise_meta_view(&mut catalog, view, schema, rows)?;
1150 }
1151 "__spg_pg_ts_template" => {
1152 let (schema, rows) =
1153 crate::system_catalog::synth_pg_ts_template(self.active_catalog());
1154 materialise_meta_view(&mut catalog, view, schema, rows)?;
1155 }
1156 // v7.37.24 (24.17) — pg_catalog.pg_depend
1157 // (dependency graph; shape-stable empty since
1158 // SPG's drop enforcement is per-kind, not per-object).
1159 "__spg_pg_depend" => {
1160 let (schema, rows) =
1161 crate::system_catalog::synth_pg_depend(self.active_catalog());
1162 materialise_meta_view(&mut catalog, view, schema, rows)?;
1163 }
1164 // 7.38.1 S5.1 — pg_catalog.pg_opclass (pg_dump wall #1).
1165 "__spg_pg_opclass" => {
1166 let (schema, rows) =
1167 crate::system_catalog::synth_pg_opclass(self.active_catalog());
1168 materialise_meta_view(&mut catalog, view, schema, rows)?;
1169 }
1170 "__spg_pg_opfamily" => {
1171 let (schema, rows) =
1172 crate::system_catalog::synth_pg_opfamily(self.active_catalog());
1173 materialise_meta_view(&mut catalog, view, schema, rows)?;
1174 }
1175 "__spg_pg_amop" => {
1176 let (schema, rows) =
1177 crate::system_catalog::synth_pg_amop(self.active_catalog());
1178 materialise_meta_view(&mut catalog, view, schema, rows)?;
1179 }
1180 "__spg_pg_amproc" => {
1181 let (schema, rows) =
1182 crate::system_catalog::synth_pg_amproc(self.active_catalog());
1183 materialise_meta_view(&mut catalog, view, schema, rows)?;
1184 }
1185 // v7.38 (read01) — pg_catalog.pg_attrdef (column defaults;
1186 // ORM reflection + pg_dump read the deparsed default text).
1187 "__spg_pg_attrdef" => {
1188 let (schema, rows) =
1189 crate::system_catalog::synth_pg_attrdef(self.active_catalog());
1190 materialise_meta_view(&mut catalog, view, schema, rows)?;
1191 }
1192 // v7.39 (RLS) — pg_catalog.pg_policy (raw) + pg_policies (view).
1193 "__spg_pg_policy" => {
1194 let (schema, rows) =
1195 crate::system_catalog::synth_pg_policy(self.active_catalog());
1196 materialise_meta_view(&mut catalog, view, schema, rows)?;
1197 }
1198 "__spg_pg_policies" => {
1199 let (schema, rows) =
1200 crate::system_catalog::synth_pg_policies(self.active_catalog());
1201 materialise_meta_view(&mut catalog, view, schema, rows)?;
1202 }
1203 // v7.37.24 (24.14) — pg_catalog.pg_collation.
1204 "__spg_pg_collation" => {
1205 let (schema, rows) =
1206 crate::system_catalog::synth_pg_collation(self.active_catalog());
1207 materialise_meta_view(&mut catalog, view, schema, rows)?;
1208 }
1209 // v7.37.23 (23.6-b) — pg_catalog.pg_tablespace.
1210 "__spg_pg_tablespace" => {
1211 let (schema, rows) =
1212 crate::system_catalog::synth_pg_tablespace(self.active_catalog());
1213 materialise_meta_view(&mut catalog, view, schema, rows)?;
1214 }
1215 // v7.17.0 Phase 3.P0-53 — pg_catalog.pg_indexes view
1216 // for pgAdmin / DataGrip "indexes per table" listings.
1217 "__spg_pg_indexes" => {
1218 let (schema, rows) = synth_pg_indexes(self.active_catalog());
1219 materialise_meta_view(&mut catalog, view, schema, rows)?;
1220 }
1221 // v7.39 (read01 round 50) — pg_catalog.pg_description, backing
1222 // psql's \d+ comment column and pg_dump's COMMENT ON emission.
1223 "__spg_pg_description" => {
1224 let (schema, rows) =
1225 crate::system_catalog::synth_pg_description(self.active_catalog());
1226 materialise_meta_view(&mut catalog, view, schema, rows)?;
1227 }
1228 // v7.17.0 Phase 3.P0-53 — pg_catalog.pg_index (raw)
1229 // for index introspection by ORM compilers.
1230 "__spg_pg_index" => {
1231 let (schema, rows) = synth_pg_index_raw(self.active_catalog());
1232 materialise_meta_view(&mut catalog, view, schema, rows)?;
1233 }
1234 // v7.17.0 Phase 3.P0-54 — pg_catalog.pg_constraint
1235 // for FK / UNIQUE / PK / CHECK introspection.
1236 "__spg_pg_constraint" => {
1237 let (schema, rows) = synth_pg_constraint(self.active_catalog());
1238 materialise_meta_view(&mut catalog, view, schema, rows)?;
1239 }
1240 // v7.37 U11 — pg_catalog.pg_sequence, one row per CREATE
1241 // SEQUENCE (psql \d <seq> + ORM sequence introspection).
1242 "__spg_pg_sequence" => {
1243 let (schema, rows) = synth_pg_sequence(self.active_catalog());
1244 materialise_meta_view(&mut catalog, view, schema, rows)?;
1245 }
1246 // v7.17.0 Phase 3.P0-55 — pg_catalog.pg_database /
1247 // pg_roles / pg_user. SPG is single-database so
1248 // pg_database surfaces just `postgres`; pg_roles
1249 // / pg_user walk the engine's UserStore.
1250 "__spg_pg_database" => {
1251 let (schema, rows) = synth_pg_database(self);
1252 materialise_meta_view(&mut catalog, view, schema, rows)?;
1253 }
1254 "__spg_pg_roles" => {
1255 let (schema, rows) = synth_pg_roles(self);
1256 materialise_meta_view(&mut catalog, view, schema, rows)?;
1257 }
1258 // v7.39 (round 542) — pg_user is a DIFFERENT view over the
1259 // same roles, with PG's own `use*` column names. It used to
1260 // publish pg_roles' columns under this name.
1261 "__spg_pg_user" => {
1262 let (schema, rows) = crate::system_catalog::synth_pg_user(self);
1263 materialise_meta_view(&mut catalog, view, schema, rows)?;
1264 }
1265 // v7.39 (read01 round 58) — role membership.
1266 "__spg_pg_auth_members" => {
1267 let (schema, rows) = crate::system_catalog::synth_pg_auth_members(self);
1268 materialise_meta_view(&mut catalog, view, schema, rows)?;
1269 }
1270 // v7.17.0 Phase 3.P0-56 — pg_catalog.pg_views. PG's
1271 // pg_views surfaces every CREATE VIEW result; SPG
1272 // ships one row per declared view from the catalog.
1273 "__spg_pg_views" => {
1274 let (schema, rows) = synth_pg_views(self.active_catalog());
1275 materialise_meta_view(&mut catalog, view, schema, rows)?;
1276 }
1277 // v7.39 (round 143) — pg_catalog.pg_rules: one row per
1278 // catalogued query-rewrite RULE.
1279 "__spg_pg_rules" => {
1280 let (schema, rows) =
1281 crate::system_catalog::synth_pg_rules(self.active_catalog());
1282 materialise_meta_view(&mut catalog, view, schema, rows)?;
1283 }
1284 // v7.39 (round 312) — pg_catalog.pg_rewrite: the rule
1285 // catalogue `pg_get_ruledef(oid)` resolves against.
1286 "__spg_pg_rewrite" => {
1287 let (schema, rows) =
1288 crate::system_catalog::synth_pg_rewrite(self.active_catalog());
1289 materialise_meta_view(&mut catalog, view, schema, rows)?;
1290 }
1291 // v7.39 (round 542) — pg_catalog.pg_matviews, with rows
1292 // and PG's own column names.
1293 "__spg_pg_matviews" => {
1294 let (schema, rows) =
1295 crate::system_catalog::synth_pg_matviews(self.active_catalog());
1296 materialise_meta_view(&mut catalog, view, schema, rows)?;
1297 }
1298 // pg_catalog.pg_extension — native capability list
1299 // (mailrs embed round-12).
1300 // v7.39 (round 546) — the catalogs SPG has real content
1301 // for, from the facts it already holds.
1302 "__spg_pg_db_role_setting" => {
1303 let (schema, rows) = crate::system_catalog::synth_pg_db_role_setting(self);
1304 materialise_meta_view(&mut catalog, view, schema, rows)?;
1305 }
1306 "__spg_pg_language" => {
1307 let (schema, rows) = crate::system_catalog::synth_pg_language();
1308 materialise_meta_view(&mut catalog, view, schema, rows)?;
1309 }
1310 "__spg_pg_sequences" => {
1311 let (schema, rows) =
1312 crate::system_catalog::synth_pg_sequences(self.active_catalog());
1313 materialise_meta_view(&mut catalog, view, schema, rows)?;
1314 }
1315 "__spg_pg_range" => {
1316 let (schema, rows) = crate::system_catalog::synth_pg_range();
1317 materialise_meta_view(&mut catalog, view, schema, rows)?;
1318 }
1319 "__spg_pg_partitioned_table" => {
1320 let (schema, rows) =
1321 crate::system_catalog::synth_pg_partitioned_table(self.active_catalog());
1322 materialise_meta_view(&mut catalog, view, schema, rows)?;
1323 }
1324 "__spg_pg_authid" => {
1325 let (schema, rows) = crate::system_catalog::synth_pg_authid(self);
1326 materialise_meta_view(&mut catalog, view, schema, rows)?;
1327 }
1328 "__spg_pg_group" => {
1329 let (schema, rows) = crate::system_catalog::synth_pg_group(self);
1330 materialise_meta_view(&mut catalog, view, schema, rows)?;
1331 }
1332 "__spg_pg_shadow" => {
1333 let (schema, rows) = crate::system_catalog::synth_pg_shadow(self);
1334 materialise_meta_view(&mut catalog, view, schema, rows)?;
1335 }
1336 // v7.39 (round 544) — pg_cast, probed from the real
1337 // cast implementation.
1338 "__spg_pg_cast" => {
1339 let (schema, rows) = crate::system_catalog::synth_pg_cast();
1340 materialise_meta_view(&mut catalog, view, schema, rows)?;
1341 }
1342 // v7.39 (round 541) — an empty catalog that exists.
1343 "__spg_pg_foreign_table" => {
1344 let (schema, rows) = crate::system_catalog::synth_pg_foreign_table();
1345 materialise_meta_view(&mut catalog, view, schema, rows)?;
1346 }
1347 "__spg_pg_extension" => {
1348 let (schema, rows) = synth_pg_extension();
1349 materialise_meta_view(&mut catalog, view, schema, rows)?;
1350 }
1351 // v7.39 (round 502) — the timezone catalogues.
1352 "__spg_pg_timezone_names" => {
1353 let (schema, rows) = synth_pg_timezone_names(self);
1354 materialise_meta_view(&mut catalog, view, schema, rows)?;
1355 }
1356 "__spg_pg_timezone_abbrevs" => {
1357 let (schema, rows) = synth_pg_timezone_abbrevs(self);
1358 materialise_meta_view(&mut catalog, view, schema, rows)?;
1359 }
1360 // v7.17.0 Phase 3.P0-57 — pg_catalog.pg_settings.
1361 "__spg_pg_settings" => {
1362 let (schema, rows) = synth_pg_settings(self);
1363 materialise_meta_view(&mut catalog, view, schema, rows)?;
1364 }
1365 // v7.17.0 Phase 3.P0-63 — information_schema.KEY_COLUMN_USAGE.
1366 // v7.39 (read01 round 51) — information_schema.role_table_grants
1367 // and .table_privileges. Both report the owner's seven implicit
1368 // table privileges; SPG's single role owns everything.
1369 // v7.39 (read01 round 59) — information_schema.column_privileges.
1370 "__spg_info_column_privileges" => {
1371 let (schema, rows) =
1372 crate::system_catalog::synth_info_column_privileges(self.active_catalog());
1373 materialise_meta_view(&mut catalog, view, schema, rows)?;
1374 }
1375 "__spg_info_role_table_grants" | "__spg_info_table_privileges" => {
1376 let grantee = self.current_role().to_string();
1377 let (schema, rows) = crate::system_catalog::synth_info_role_table_grants(
1378 self.active_catalog(),
1379 &grantee,
1380 );
1381 materialise_meta_view(&mut catalog, view, schema, rows)?;
1382 }
1383 "__spg_info_key_column_usage" => {
1384 let (schema, rows) = synth_info_key_column_usage(self.active_catalog());
1385 materialise_meta_view(&mut catalog, view, schema, rows)?;
1386 }
1387 // v7.17.0 Phase 3.P0-64 — information_schema.REFERENTIAL_CONSTRAINTS.
1388 "__spg_info_referential_constraints" => {
1389 let (schema, rows) = synth_info_referential_constraints(self.active_catalog());
1390 materialise_meta_view(&mut catalog, view, schema, rows)?;
1391 }
1392 // v7.17.0 Phase 3.P0-64 — information_schema.STATISTICS.
1393 "__spg_info_statistics" => {
1394 let (schema, rows) = synth_info_statistics(self.active_catalog());
1395 materialise_meta_view(&mut catalog, view, schema, rows)?;
1396 }
1397 // v7.17.0 Phase 3.P0-64 — information_schema.ROUTINES.
1398 "__spg_info_routines" => {
1399 let (schema, rows) = synth_info_routines();
1400 materialise_meta_view(&mut catalog, view, schema, rows)?;
1401 }
1402 // v7.37.24 (24.3) — information_schema.attributes.
1403 "__spg_info_attributes" => {
1404 let (schema, rows) = crate::system_catalog::synth_information_schema_attributes(
1405 self.active_catalog(),
1406 );
1407 materialise_meta_view(&mut catalog, view, schema, rows)?;
1408 }
1409 // v7.37.24 (24.2) — information_schema.domains.
1410 "__spg_info_domains" => {
1411 let (schema, rows) = crate::system_catalog::synth_information_schema_domains(
1412 self.active_catalog(),
1413 );
1414 materialise_meta_view(&mut catalog, view, schema, rows)?;
1415 }
1416 // v7.37.24 (24.9) — information_schema.schemata.
1417 "__spg_info_schemata" => {
1418 let (schema, rows) = crate::system_catalog::synth_information_schema_schemata(
1419 self.active_catalog(),
1420 );
1421 materialise_meta_view(&mut catalog, view, schema, rows)?;
1422 }
1423 // v7.37.24 (24.9) — information_schema.views.
1424 "__spg_info_views" => {
1425 let (schema, rows) = crate::system_catalog::synth_information_schema_views(
1426 self.active_catalog(),
1427 );
1428 materialise_meta_view(&mut catalog, view, schema, rows)?;
1429 }
1430 // v7.37.24 (24.9) — information_schema.table_constraints.
1431 "__spg_info_table_constraints" => {
1432 let (schema, rows) =
1433 crate::system_catalog::synth_information_schema_table_constraints(
1434 self.active_catalog(),
1435 );
1436 materialise_meta_view(&mut catalog, view, schema, rows)?;
1437 }
1438 // v7.37.17 — information_schema.constraint_column_usage.
1439 "__spg_info_constraint_column_usage" => {
1440 let (schema, rows) = crate::system_catalog::synth_info_constraint_column_usage(
1441 self.active_catalog(),
1442 );
1443 materialise_meta_view(&mut catalog, view, schema, rows)?;
1444 }
1445 // v7.37.17 — information_schema.triggers.
1446 "__spg_info_triggers" => {
1447 let (schema, rows) =
1448 crate::system_catalog::synth_info_triggers(self.active_catalog());
1449 materialise_meta_view(&mut catalog, view, schema, rows)?;
1450 }
1451 // v7.37.17 — information_schema.check_constraints.
1452 "__spg_info_check_constraints" => {
1453 let (schema, rows) =
1454 crate::system_catalog::synth_info_check_constraints(self.active_catalog());
1455 materialise_meta_view(&mut catalog, view, schema, rows)?;
1456 }
1457 // v7.37.17 — information_schema.sequences.
1458 "__spg_info_sequences" => {
1459 let (schema, rows) =
1460 crate::system_catalog::synth_info_sequences(self.active_catalog());
1461 materialise_meta_view(&mut catalog, view, schema, rows)?;
1462 }
1463 // v7.17.0 Phase 3.P0-65 — mysql.user / mysql.db.
1464 "__spg_mysql_user" => {
1465 let (schema, rows) = synth_mysql_user(self);
1466 materialise_meta_view(&mut catalog, view, schema, rows)?;
1467 }
1468 "__spg_mysql_db" => {
1469 let (schema, rows) = synth_mysql_db();
1470 materialise_meta_view(&mut catalog, view, schema, rows)?;
1471 }
1472 // v7.39 (round 541) — the catalogs PG has that SPG is
1473 // genuinely empty of. Table-driven; see EMPTY_PG_CATALOGS.
1474 other if crate::system_catalog::synth_empty_pg_catalog(other).is_some() => {
1475 let (schema, rows) =
1476 crate::system_catalog::synth_empty_pg_catalog(other).expect("just checked");
1477 materialise_meta_view(&mut catalog, view, schema, rows)?;
1478 }
1479 _ => {
1480 return Err(EngineError::Unsupported(alloc::format!(
1481 "meta view {view:?} is not yet materialisable; \
1482 v7.16.2 covers information_schema.columns / .tables \
1483 and pg_catalog.pg_class / pg_attribute; \
1484 v7.17.0 P0-50..P0-57 add pg_type / pg_proc / pg_namespace / \
1485 pg_indexes / pg_index / pg_constraint / pg_database / pg_roles / \
1486 pg_user / pg_views / pg_matviews / pg_settings"
1487 )));
1488 }
1489 }
1490 }
1491 Ok(catalog)
1492 }
1493
1494 pub(crate) fn exec_with_ctes(
1495 &self,
1496 stmt: &SelectStatement,
1497 cancel: CancelToken<'_>,
1498 ) -> Result<QueryResult, EngineError> {
1499 cancel.check()?;
1500 // v7.37.43-T4.4 — `&self` SELECT path: only read-only CTE
1501 // bodies are supported here. Writable CTEs on a SELECT
1502 // outer require `&mut self` and route through the
1503 // top-level `exec_select_cancel_mut` entry; sentori
1504 // 0065's WITH-INSERT-INSERT shape comes in as a top-level
1505 // INSERT, not a SELECT, so this restriction is harmless
1506 // in practice.
1507 if stmt.ctes.iter().any(|c| c.body.is_modifying()) {
1508 // v7.39 (read01 round 81) — PG's wording. A data-modifying CTE
1509 // (`WITH d AS (DELETE … RETURNING …) …`) is only legal at the top
1510 // of a statement, not nested inside a subquery; this path is
1511 // reached exactly when one is nested. The old text described SPG's
1512 // own executor plumbing ("the top-level mutable entry"), which
1513 // means nothing to a client.
1514 return Err(EngineError::Unsupported(
1515 "WITH clause containing a data-modifying statement must be at the top level".into(),
1516 ));
1517 }
1518 let catalog = self.materialise_ctes_readonly(&stmt.ctes, cancel)?;
1519 // Strip CTEs from the body before running on the temp engine
1520 // so we don't recurse forever.
1521 let mut body = stmt.clone();
1522 body.ctes = Vec::new();
1523 let mut temp = Engine::restore(catalog);
1524 if let Some(c) = self.clock {
1525 temp = temp.with_clock(c);
1526 }
1527 if let Some(f) = self.salt_fn {
1528 temp = temp.with_salt_fn(f);
1529 }
1530 temp.exec_select_cancel(&body, cancel)
1531 }
1532
1533 /// v7.37.43-T4.4 — read-only CTE materialiser used by the
1534 /// `&self` SELECT path. Caller guarantees no modifying CTE
1535 /// bodies are present.
1536 pub(crate) fn materialise_ctes_readonly(
1537 &self,
1538 ctes: &[spg_sql::ast::Cte],
1539 cancel: CancelToken<'_>,
1540 ) -> Result<crate::Catalog, EngineError> {
1541 cancel.check()?;
1542 let mut catalog = self.active_catalog().clone();
1543 for cte in ctes {
1544 let body_select = cte.body.as_select().ok_or_else(|| {
1545 EngineError::Unsupported(alloc::format!(
1546 "data-modifying CTE not supported on this SELECT entry"
1547 ))
1548 })?;
1549 // v7.39 (round 156) — a CTE may SHADOW a same-named real table
1550 // (PG scoping: the WITH name wins for the outer query and later
1551 // CTEs, while THIS body still sees the real table — a
1552 // non-recursive body's self-name is the table, probe P2). This
1553 // materialiser works on a CLONE, so the shadow is simply: run
1554 // the body against the untouched clone, then drop the real
1555 // table from the clone before installing the CTE's temp. A
1556 // RECURSIVE self-reference is the CTE itself (P6), so there the
1557 // drop happens before the iterating materialiser runs.
1558 let (columns, rows) = if cte.recursive && select_refers_to(body_select, &cte.name) {
1559 let synthetic = spg_sql::ast::Cte {
1560 name: cte.name.clone(),
1561 body: spg_sql::ast::CteBody::Select(body_select.clone()),
1562 recursive: true,
1563 column_overrides: cte.column_overrides.clone(),
1564 search: None,
1565 cycle: None,
1566 };
1567 if catalog.get(&cte.name).is_some() {
1568 let _ = catalog.drop_table(&cte.name);
1569 }
1570 self.materialise_recursive_cte(&synthetic, &catalog, cancel)?
1571 } else {
1572 let mut cte_engine = Engine::restore(catalog.clone());
1573 if let Some(c) = self.clock {
1574 cte_engine = cte_engine.with_clock(c);
1575 }
1576 if let Some(f) = self.salt_fn {
1577 cte_engine = cte_engine.with_salt_fn(f);
1578 }
1579 let body_result = cte_engine.exec_select_cancel(body_select, cancel)?;
1580 let QueryResult::Rows { columns, rows } = body_result else {
1581 return Err(EngineError::Unsupported(alloc::format!(
1582 "CTE {:?} body did not return rows",
1583 cte.name
1584 )));
1585 };
1586 (columns, rows)
1587 };
1588 let inferred = infer_column_types(&columns, &rows);
1589 let mut columns = inferred;
1590 if !cte.column_overrides.is_empty() {
1591 if cte.column_overrides.len() != columns.len() {
1592 return Err(EngineError::Unsupported(alloc::format!(
1593 "CTE {:?} column list has {} names but body returns {} columns",
1594 cte.name,
1595 cte.column_overrides.len(),
1596 columns.len()
1597 )));
1598 }
1599 for (col, name) in columns.iter_mut().zip(cte.column_overrides.iter()) {
1600 col.name.clone_from(name);
1601 }
1602 }
1603 let schema = TableSchema::new(cte.name.clone(), columns);
1604 // v7.39 (round 156) — the body ran against the untouched clone;
1605 // from here on the CTE name resolves to the temp (PG scoping).
1606 if catalog.get(&cte.name).is_some() {
1607 let _ = catalog.drop_table(&cte.name);
1608 }
1609 catalog.create_table(schema).map_err(EngineError::Storage)?;
1610 let table = catalog
1611 .get_mut(&cte.name)
1612 .expect("just-created CTE table must exist");
1613 for row in rows {
1614 table.insert(row).map_err(EngineError::Storage)?;
1615 }
1616 }
1617 Ok(catalog)
1618 }
1619
1620 /// v7.37.43-T4.4 — shared CTE materialiser (mutable variant).
1621 /// Retained for non-DML callers; the DML path (writable CTE on
1622 /// INSERT/UPDATE/DELETE outer) uses `run_with_cte_temps` in
1623 /// `dml.rs` which installs the CTE temps directly on the
1624 /// active catalog so the outer statement's writes hit real
1625 /// tables.
1626 #[allow(dead_code)]
1627 pub(crate) fn materialise_ctes(
1628 &mut self,
1629 ctes: &[spg_sql::ast::Cte],
1630 cancel: CancelToken<'_>,
1631 ) -> Result<crate::Catalog, EngineError> {
1632 cancel.check()?;
1633 // v7.37.43-T4.4 — modifying CTEs need to write through the
1634 // SAME catalog as the outer statement, not a clone (PG's
1635 // writable CTE puts all modifications in one transaction).
1636 // For the read-only case the original logic cloned, but
1637 // since the outer statement also goes through the cloned
1638 // engine and ALL writes must converge, we now drive the
1639 // accumulator off `self.active_catalog().clone()` and
1640 // commit the modifying writes directly to `self`'s active
1641 // catalog so the surface is consistent.
1642 let mut catalog = self.active_catalog().clone();
1643 // v7.39 (round 149) — a modifying CTE body's target must be a
1644 // real relation, never a sibling CTE (PG: relation does not
1645 // exist); checked before any alias lands in the accumulator.
1646 for cte in ctes {
1647 let body_target = match &cte.body {
1648 spg_sql::ast::CteBody::Select(_) => None,
1649 spg_sql::ast::CteBody::Insert(i) => Some(i.table.as_str()),
1650 spg_sql::ast::CteBody::Update(u) => Some(u.table.as_str()),
1651 spg_sql::ast::CteBody::Delete(d) => Some(d.table.as_str()),
1652 spg_sql::ast::CteBody::Merge(m) => Some(m.target.as_str()),
1653 };
1654 if let Some(t) = body_target
1655 && ctes.iter().any(|c| c.name.eq_ignore_ascii_case(t))
1656 && catalog.get(t).is_none()
1657 {
1658 return Err(EngineError::Storage(
1659 spg_storage::StorageError::TableNotFound { name: t.into() },
1660 ));
1661 }
1662 }
1663 for cte in ctes {
1664 if catalog.get(&cte.name).is_some() {
1665 return Err(EngineError::Unsupported(alloc::format!(
1666 "CTE name {:?} shadows an existing table; rename the CTE",
1667 cte.name
1668 )));
1669 }
1670 let (columns, rows) = match &cte.body {
1671 // v7.39 (round 145) — see the sibling site: only a body that
1672 // truly self-references takes the iterating materialiser.
1673 spg_sql::ast::CteBody::Select(body)
1674 if cte.recursive && select_refers_to(body, &cte.name) =>
1675 {
1676 // Recursive CTE — the existing helper takes a
1677 // SELECT body and the snapshot catalog.
1678 let synthetic = spg_sql::ast::Cte {
1679 name: cte.name.clone(),
1680 body: spg_sql::ast::CteBody::Select(body.clone()),
1681 recursive: true,
1682 column_overrides: cte.column_overrides.clone(),
1683 search: None,
1684 cycle: None,
1685 };
1686 self.materialise_recursive_cte(&synthetic, &catalog, cancel)?
1687 }
1688 spg_sql::ast::CteBody::Select(body) => {
1689 // v7.25 (round-17) — run against the accumulated
1690 // catalog so later CTEs can reference earlier
1691 // ones in the same WITH clause.
1692 let mut cte_engine = Engine::restore(catalog.clone());
1693 if let Some(c) = self.clock {
1694 cte_engine = cte_engine.with_clock(c);
1695 }
1696 if let Some(f) = self.salt_fn {
1697 cte_engine = cte_engine.with_salt_fn(f);
1698 }
1699 let body_result = cte_engine.exec_select_cancel(body, cancel)?;
1700 let QueryResult::Rows { columns, rows } = body_result else {
1701 return Err(EngineError::Unsupported(alloc::format!(
1702 "CTE {:?} body did not return rows",
1703 cte.name
1704 )));
1705 };
1706 (columns, rows)
1707 }
1708 spg_sql::ast::CteBody::Insert(body) => {
1709 self.exec_modifying_cte_insert(&cte.name, body, cancel)?
1710 }
1711 spg_sql::ast::CteBody::Update(body) => {
1712 self.exec_modifying_cte_update(&cte.name, body, cancel)?
1713 }
1714 spg_sql::ast::CteBody::Delete(body) => {
1715 self.exec_modifying_cte_delete(&cte.name, body, cancel)?
1716 }
1717 spg_sql::ast::CteBody::Merge(body) => {
1718 self.exec_modifying_cte_merge(&cte.name, body, cancel)?
1719 }
1720 };
1721 // v4.22: the projection builder labels any non-column
1722 // expression as Text — including literal SELECT 1.
1723 // Promote each column's type to whatever the rows
1724 // actually carry so the CTE storage table accepts them.
1725 let inferred = infer_column_types(&columns, &rows);
1726 let mut columns = inferred;
1727 if !cte.column_overrides.is_empty() {
1728 if cte.column_overrides.len() != columns.len() {
1729 return Err(EngineError::Unsupported(alloc::format!(
1730 "CTE {:?} column list has {} names but body returns {} columns",
1731 cte.name,
1732 cte.column_overrides.len(),
1733 columns.len()
1734 )));
1735 }
1736 for (col, name) in columns.iter_mut().zip(cte.column_overrides.iter()) {
1737 col.name.clone_from(name);
1738 }
1739 }
1740 let schema = TableSchema::new(cte.name.clone(), columns);
1741 catalog.create_table(schema).map_err(EngineError::Storage)?;
1742 let table = catalog
1743 .get_mut(&cte.name)
1744 .expect("just-created CTE table must exist");
1745 for row in rows {
1746 table.insert(row).map_err(EngineError::Storage)?;
1747 }
1748 }
1749 Ok(catalog)
1750 }
1751
1752 /// v7.37.43-T4.4 — execute an INSERT CTE body. Runs the INSERT
1753 /// against `self` (so the mutation lands in the active catalog
1754 /// inside the current transaction) and captures the RETURNING
1755 /// projection — column schema + rows — to materialise as the
1756 /// CTE alias's table. An INSERT without RETURNING produces a
1757 /// 0-row table with a synthetic single-column placeholder
1758 /// (matches PG: the CTE alias is still defined, but referencing
1759 /// it from the outer query without RETURNING raises a
1760 /// column-resolution error at scan time).
1761 fn exec_modifying_cte_insert(
1762 &mut self,
1763 cte_name: &str,
1764 body: &spg_sql::ast::InsertStatement,
1765 _cancel: CancelToken<'_>,
1766 ) -> Result<
1767 (
1768 Vec<spg_storage::ColumnSchema>,
1769 Vec<spg_storage::Row<'static>>,
1770 ),
1771 EngineError,
1772 > {
1773 // round 151 — a WITH-headed body keeps its own ctes; the body
1774 // statement routes through its writable-CTE entry (outer CTEs
1775 // are never copied into bodies, so no recursion risk).
1776 let body = body.clone();
1777 let result = self.exec_insert(body)?;
1778 match result {
1779 QueryResult::Rows { columns, rows } => Ok((columns, rows)),
1780 QueryResult::CommandOk { .. } => {
1781 // No RETURNING — emit a sentinel single-column
1782 // schema with zero rows so the alias is defined.
1783 let placeholder = spg_storage::ColumnSchema::new(
1784 alloc::format!("{cte_name}_returning_absent"),
1785 spg_storage::DataType::Text,
1786 true,
1787 );
1788 Ok((alloc::vec![placeholder], Vec::new()))
1789 }
1790 }
1791 }
1792
1793 /// v7.37.43-T4.4 — execute an UPDATE CTE body, same semantics
1794 /// as INSERT above.
1795 fn exec_modifying_cte_update(
1796 &mut self,
1797 cte_name: &str,
1798 body: &spg_sql::ast::UpdateStatement,
1799 cancel: CancelToken<'_>,
1800 ) -> Result<
1801 (
1802 Vec<spg_storage::ColumnSchema>,
1803 Vec<spg_storage::Row<'static>>,
1804 ),
1805 EngineError,
1806 > {
1807 let body = body.clone();
1808 let result = self.exec_update_cancel(&body, cancel)?;
1809 match result {
1810 QueryResult::Rows { columns, rows } => Ok((columns, rows)),
1811 QueryResult::CommandOk { .. } => {
1812 let placeholder = spg_storage::ColumnSchema::new(
1813 alloc::format!("{cte_name}_returning_absent"),
1814 spg_storage::DataType::Text,
1815 true,
1816 );
1817 Ok((alloc::vec![placeholder], Vec::new()))
1818 }
1819 }
1820 }
1821
1822 /// v7.37.43-T4.4 — execute a DELETE CTE body.
1823 fn exec_modifying_cte_delete(
1824 &mut self,
1825 cte_name: &str,
1826 body: &spg_sql::ast::DeleteStatement,
1827 cancel: CancelToken<'_>,
1828 ) -> Result<
1829 (
1830 Vec<spg_storage::ColumnSchema>,
1831 Vec<spg_storage::Row<'static>>,
1832 ),
1833 EngineError,
1834 > {
1835 let body = body.clone();
1836 let result = self.exec_delete_cancel(&body, cancel)?;
1837 match result {
1838 QueryResult::Rows { columns, rows } => Ok((columns, rows)),
1839 QueryResult::CommandOk { .. } => {
1840 let placeholder = spg_storage::ColumnSchema::new(
1841 alloc::format!("{cte_name}_returning_absent"),
1842 spg_storage::DataType::Text,
1843 true,
1844 );
1845 Ok((alloc::vec![placeholder], Vec::new()))
1846 }
1847 }
1848 }
1849
1850 /// v7.39 (round 149) — execute a MERGE CTE body (PG 17).
1851 fn exec_modifying_cte_merge(
1852 &mut self,
1853 cte_name: &str,
1854 body: &spg_sql::ast::MergeStatement,
1855 cancel: CancelToken<'_>,
1856 ) -> Result<
1857 (
1858 Vec<spg_storage::ColumnSchema>,
1859 Vec<spg_storage::Row<'static>>,
1860 ),
1861 EngineError,
1862 > {
1863 let body = body.clone();
1864 let result = self.exec_merge_cancel(&body, cancel)?;
1865 match result {
1866 QueryResult::Rows { columns, rows } => Ok((columns, rows)),
1867 QueryResult::CommandOk { .. } => {
1868 let placeholder = spg_storage::ColumnSchema::new(
1869 alloc::format!("{cte_name}_returning_absent"),
1870 spg_storage::DataType::Text,
1871 true,
1872 );
1873 Ok((alloc::vec![placeholder], Vec::new()))
1874 }
1875 }
1876 }
1877
1878 /// v4.22: materialise a WITH RECURSIVE CTE. The body must be a
1879 /// UNION (or UNION ALL) of an anchor that does not reference
1880 /// the CTE name, and one or more recursive terms that do. The
1881 /// anchor runs first; each subsequent iteration runs the
1882 /// recursive term against a temp catalog where the CTE name is
1883 /// bound to the *previous* iteration's output. Iteration stops
1884 /// when the recursive term yields no rows; UNION (DISTINCT)
1885 /// deduplicates against the accumulated result, UNION ALL does
1886 /// not. A hard cap on total rows prevents runaway queries.
1887 #[allow(clippy::too_many_lines)]
1888 pub(crate) fn materialise_recursive_cte(
1889 &self,
1890 cte: &spg_sql::ast::Cte,
1891 base_catalog: &Catalog,
1892 cancel: CancelToken<'_>,
1893 ) -> Result<(Vec<ColumnSchema>, Vec<Row<'static>>), EngineError> {
1894 const MAX_TOTAL_ROWS: usize = 1_000_000;
1895 const MAX_ITERATIONS: usize = 100_000;
1896 cancel.check()?;
1897 // v7.37.43-T4.4 — RECURSIVE only supports SELECT bodies;
1898 // a modifying recursive CTE is parser-rejectable but we
1899 // guard here defensively.
1900 let body_select = cte.body.as_select().ok_or_else(|| {
1901 EngineError::Unsupported(alloc::format!(
1902 "WITH RECURSIVE {:?} body must be a SELECT, not a data-modifying statement",
1903 cte.name
1904 ))
1905 })?;
1906 if body_select.unions.is_empty() {
1907 return Err(EngineError::Unsupported(alloc::format!(
1908 "WITH RECURSIVE {:?} body must be a UNION of an anchor and a recursive term",
1909 cte.name
1910 )));
1911 }
1912 // Anchor: the body's leading SELECT, with unions stripped.
1913 let mut anchor = body_select.clone();
1914 let all_union_terms = core::mem::take(&mut anchor.unions);
1915 anchor.ctes = Vec::new();
1916 // v7.37 D.42 — split the UNION members: those that do NOT reference the
1917 // CTE are additional ANCHOR terms, only the ones that do recurse. A
1918 // multi-row VALUES seed lowers to `SELECT r1 UNION ALL SELECT r2 UNION
1919 // ALL <recursive>`, so the leading SELECT alone is not the whole anchor —
1920 // treating the non-recursive `SELECT r2` as a recursive term made it
1921 // re-emit its constant row every iteration → runaway loop.
1922 let (anchor_terms, union_terms): (Vec<_>, Vec<_>) = all_union_terms
1923 .into_iter()
1924 .partition(|(_, t)| !select_refers_to(t, &cte.name));
1925 let anchor_result = self.exec_select_cancel(&anchor, cancel)?;
1926 let QueryResult::Rows {
1927 columns: anchor_cols,
1928 rows: mut anchor_rows,
1929 } = anchor_result
1930 else {
1931 return Err(EngineError::Unsupported(alloc::format!(
1932 "WITH RECURSIVE {:?}: anchor did not return rows",
1933 cte.name
1934 )));
1935 };
1936 // Append every non-recursive UNION member's rows to the anchor set.
1937 for (_, term) in &anchor_terms {
1938 let mut term = term.clone();
1939 term.ctes = Vec::new();
1940 if let QueryResult::Rows { rows, .. } = self.exec_select_cancel(&term, cancel)? {
1941 anchor_rows.extend(rows);
1942 }
1943 }
1944 // The projection builder labels non-column expressions Text;
1945 // refine column types from the anchor's actual values so the
1946 // intermediate iter-catalog tables accept them.
1947 let mut columns = infer_column_types(&anchor_cols, &anchor_rows);
1948 if !cte.column_overrides.is_empty() {
1949 if cte.column_overrides.len() != columns.len() {
1950 return Err(EngineError::Unsupported(alloc::format!(
1951 "CTE {:?} column list has {} names but anchor returns {} columns",
1952 cte.name,
1953 cte.column_overrides.len(),
1954 columns.len()
1955 )));
1956 }
1957 for (col, name) in columns.iter_mut().zip(cte.column_overrides.iter()) {
1958 col.name.clone_from(name);
1959 }
1960 }
1961 let mut all_rows: Vec<Row<'static>> = anchor_rows.clone();
1962 let mut working_set: Vec<Row<'static>> = anchor_rows;
1963 let mut seen: alloc::collections::BTreeSet<Vec<u8>> = alloc::collections::BTreeSet::new();
1964 // Track at least one "all UNION ALL" flag — if every union
1965 // kind is ALL we skip the dedup step (faster + matches PG).
1966 let all_union_all = union_terms.iter().all(|(k, _)| matches!(k, UnionKind::All));
1967 if !all_union_all {
1968 for r in &all_rows {
1969 seen.insert(encode_row_key(r));
1970 }
1971 }
1972 // v7.39 (round 598) — the engine and its catalog are built ONCE.
1973 // Each iteration used to clone the catalog, create the CTE table,
1974 // and construct a whole `Engine` — which initialises 82 fields — to
1975 // hold that round's working set. A counting allocator put the loop
1976 // at 63 allocations and 104 kB per iteration, or 1 GB for a
1977 // 10,000-row recursive CTE, and none of it varied with how much
1978 // else was in the catalog: the per-round rebuild WAS the cost. The
1979 // table is emptied and refilled instead.
1980 let mut iter_catalog = base_catalog.clone();
1981 let schema = TableSchema::new(cte.name.clone(), columns.clone());
1982 iter_catalog
1983 .create_table(schema)
1984 .map_err(EngineError::Storage)?;
1985 let mut iter_engine = Engine::restore(iter_catalog);
1986 if let Some(c) = self.clock {
1987 iter_engine = iter_engine.with_clock(c);
1988 }
1989 if let Some(f) = self.salt_fn {
1990 iter_engine = iter_engine.with_salt_fn(f);
1991 }
1992 // The recursive terms are cloned once too — the clone stripped the
1993 // CTE list off each of them, per term per iteration.
1994 let recursive_terms: Vec<SelectStatement> = union_terms
1995 .iter()
1996 .map(|(_, t)| {
1997 let mut t = t.clone();
1998 t.ctes = Vec::new();
1999 t
2000 })
2001 .collect();
2002 // v7.39 (round 618) — plan every recursive term once. Taken only if
2003 // ALL of them plan, so a query never runs half on each path.
2004 let term_plans: Option<Vec<RecursiveTermPlan<'_>>> = recursive_terms
2005 .iter()
2006 .map(|t| plan_recursive_term(t, &cte.name, columns.len()))
2007 .collect();
2008 let fast_ctx = term_plans.as_ref().map(|plans| {
2009 let alias = plans[0].alias.clone();
2010 (alias, ())
2011 });
2012 for iter in 0..MAX_ITERATIONS {
2013 cancel.check()?;
2014 if working_set.is_empty() {
2015 break;
2016 }
2017 if let (Some(plans), Some((_, ()))) = (term_plans.as_ref(), fast_ctx.as_ref()) {
2018 // The worktable IS the working set: no table to empty and
2019 // refill, and no query execution per round.
2020 let mut next_set: Vec<Row<'static>> = Vec::new();
2021 for plan in plans {
2022 let ctx = self.ev_ctx(&columns, Some(&plan.alias));
2023 for row in &working_set {
2024 cancel.check()?;
2025 if let Some(w) = plan.where_ {
2026 let v = eval::eval_expr(w, row, &ctx).map_err(EngineError::Eval)?;
2027 if !matches!(v, Value::Bool(true)) {
2028 continue;
2029 }
2030 }
2031 let mut vals: Vec<Value<'static>> = Vec::with_capacity(plan.items.len());
2032 for it in &plan.items {
2033 vals.push(eval::eval_expr(it, row, &ctx).map_err(EngineError::Eval)?);
2034 }
2035 let out = Row::new(vals);
2036 if !all_union_all {
2037 let key = encode_row_key(&out);
2038 if !seen.insert(key) {
2039 continue;
2040 }
2041 }
2042 next_set.push(out);
2043 }
2044 }
2045 if next_set.is_empty() {
2046 break;
2047 }
2048 all_rows.extend(next_set.iter().cloned());
2049 working_set = next_set;
2050 if all_rows.len() > MAX_TOTAL_ROWS {
2051 return Err(EngineError::Unsupported(alloc::format!(
2052 "WITH RECURSIVE {:?}: produced more than {MAX_TOTAL_ROWS} rows — likely runaway recursion",
2053 cte.name
2054 )));
2055 }
2056 if iter + 1 == MAX_ITERATIONS {
2057 return Err(EngineError::Unsupported(alloc::format!(
2058 "WITH RECURSIVE {:?}: exceeded {MAX_ITERATIONS} iterations",
2059 cte.name
2060 )));
2061 }
2062 continue;
2063 }
2064 {
2065 // Truncated rather than dropped and recreated: the table's
2066 // own structure is what dropping it throws away, and it is
2067 // identical every round.
2068 let cat = iter_engine.base_catalog_mut();
2069 let table = cat.get_mut(&cte.name).expect("created above");
2070 table.truncate();
2071 for row in &working_set {
2072 table.insert(row.clone()).map_err(EngineError::Storage)?;
2073 }
2074 }
2075 // Run each recursive term in sequence and collect new rows.
2076 let mut next_set: Vec<Row<'static>> = Vec::new();
2077 for term in &recursive_terms {
2078 let r = iter_engine.exec_select_cancel(term, cancel)?;
2079 let QueryResult::Rows {
2080 columns: rc,
2081 rows: rs,
2082 } = r
2083 else {
2084 return Err(EngineError::Unsupported(alloc::format!(
2085 "WITH RECURSIVE {:?}: recursive term did not return rows",
2086 cte.name
2087 )));
2088 };
2089 if rc.len() != columns.len() {
2090 return Err(EngineError::Unsupported(alloc::format!(
2091 "WITH RECURSIVE {:?}: column count of recursive term ({}) does not match anchor ({})",
2092 cte.name,
2093 rc.len(),
2094 columns.len()
2095 )));
2096 }
2097 for row in rs {
2098 if !all_union_all {
2099 let key = encode_row_key(&row);
2100 if !seen.insert(key) {
2101 continue;
2102 }
2103 }
2104 next_set.push(row);
2105 }
2106 }
2107 if next_set.is_empty() {
2108 break;
2109 }
2110 all_rows.extend(next_set.iter().cloned());
2111 working_set = next_set;
2112 if all_rows.len() > MAX_TOTAL_ROWS {
2113 return Err(EngineError::Unsupported(alloc::format!(
2114 "WITH RECURSIVE {:?}: produced more than {MAX_TOTAL_ROWS} rows — likely runaway recursion",
2115 cte.name
2116 )));
2117 }
2118 if iter + 1 == MAX_ITERATIONS {
2119 return Err(EngineError::Unsupported(alloc::format!(
2120 "WITH RECURSIVE {:?}: exceeded {MAX_ITERATIONS} iterations",
2121 cte.name
2122 )));
2123 }
2124 }
2125 Ok((columns, all_rows))
2126 }
2127
2128 pub(crate) fn resolve_select_subqueries(
2129 &self,
2130 stmt: &mut SelectStatement,
2131 cancel: CancelToken<'_>,
2132 ) -> Result<(), EngineError> {
2133 for item in &mut stmt.items {
2134 if let SelectItem::Expr { expr, alias } = item {
2135 // An UNCORRELATED subquery is replaced by its value right
2136 // here, and the shape the column was named for goes with
2137 // it: by projection time `SELECT EXISTS(SELECT 1)` is a
2138 // boolean literal, so SPG answered `?column?` where PG18
2139 // answers `exists`. Only a subquery at the TOP of the item
2140 // loses its name this way — one nested inside a call still
2141 // reports the call.
2142 if alias.is_none()
2143 && matches!(
2144 expr,
2145 Expr::ScalarSubquery(_)
2146 | Expr::Exists { .. }
2147 | Expr::InSubquery { .. }
2148 | Expr::RowInSubquery { .. }
2149 | Expr::RowCmpSubquery { .. }
2150 )
2151 {
2152 *alias = Some(default_output_name(expr, self.backslash_escapes));
2153 }
2154 self.resolve_expr_subqueries(expr, cancel)?;
2155 }
2156 }
2157 if let Some(w) = &mut stmt.where_ {
2158 self.resolve_expr_subqueries(w, cancel)?;
2159 }
2160 // v7.24.1 — JOIN ON conditions can carry subqueries too;
2161 // they were never walked, so even an UNCORRELATED subquery
2162 // in ON hit "subquery reached row eval".
2163 if let Some(from) = &mut stmt.from {
2164 for j in &mut from.joins {
2165 if let Some(on) = &mut j.on {
2166 self.resolve_expr_subqueries(on, cancel)?;
2167 }
2168 }
2169 }
2170 if let Some(gs) = &mut stmt.group_by {
2171 for g in gs {
2172 self.resolve_expr_subqueries(g, cancel)?;
2173 }
2174 }
2175 if let Some(h) = &mut stmt.having {
2176 self.resolve_expr_subqueries(h, cancel)?;
2177 }
2178 for o in &mut stmt.order_by {
2179 self.resolve_expr_subqueries(&mut o.expr, cancel)?;
2180 }
2181 for (_, peer) in &mut stmt.unions {
2182 self.resolve_select_subqueries(peer, cancel)?;
2183 }
2184 Ok(())
2185 }
2186
2187 #[allow(clippy::only_used_in_recursion)] // engine handle reads aren't really pure
2188 pub(crate) fn resolve_expr_subqueries(
2189 &self,
2190 e: &mut Expr,
2191 cancel: CancelToken<'_>,
2192 ) -> Result<(), EngineError> {
2193 // Replace-on-this-node cases first.
2194 if let Some(replacement) = self.subquery_replacement(e, cancel)? {
2195 *e = replacement;
2196 return Ok(());
2197 }
2198 match e {
2199 Expr::NamedArg { expr, .. } => self.resolve_expr_subqueries(expr, cancel)?,
2200 Expr::Variadic(expr) => self.resolve_expr_subqueries(expr, cancel)?,
2201 Expr::AggregateOrdered { call, order_by, .. } => {
2202 self.resolve_expr_subqueries(call, cancel)?;
2203 for o in order_by.iter_mut() {
2204 self.resolve_expr_subqueries(&mut o.expr, cancel)?;
2205 }
2206 }
2207 Expr::Binary { lhs, rhs, .. } => {
2208 self.resolve_expr_subqueries(lhs, cancel)?;
2209 self.resolve_expr_subqueries(rhs, cancel)?;
2210 }
2211 Expr::Unary { expr, .. }
2212 | Expr::Cast { expr, .. }
2213 | Expr::IsNull { expr, .. }
2214 | Expr::BoolTest { expr, .. }
2215 | Expr::FieldAccess { base: expr, .. } => {
2216 self.resolve_expr_subqueries(expr, cancel)?;
2217 }
2218 Expr::FunctionCall { args, .. } => {
2219 for a in args {
2220 self.resolve_expr_subqueries(a, cancel)?;
2221 }
2222 }
2223 Expr::Like { expr, pattern, .. } => {
2224 self.resolve_expr_subqueries(expr, cancel)?;
2225 self.resolve_expr_subqueries(pattern, cancel)?;
2226 }
2227 Expr::Extract { source, .. } => self.resolve_expr_subqueries(source, cancel)?,
2228 // v4.12 window functions — recurse into args + ORDER BY
2229 // + PARTITION BY in case they carry inner subqueries.
2230 Expr::WindowFunction {
2231 args,
2232 partition_by,
2233 order_by,
2234 ..
2235 } => {
2236 for a in args {
2237 self.resolve_expr_subqueries(a, cancel)?;
2238 }
2239 for p in partition_by {
2240 self.resolve_expr_subqueries(p, cancel)?;
2241 }
2242 for (e, _, _) in order_by {
2243 self.resolve_expr_subqueries(e, cancel)?;
2244 }
2245 }
2246 // Subquery nodes are handled in subquery_replacement
2247 // (which returned None — defensive no-op); Literal /
2248 // Column are leaves.
2249 Expr::ScalarSubquery(_)
2250 | Expr::Exists { .. }
2251 | Expr::InSubquery { .. }
2252 | Expr::RowInSubquery { .. }
2253 | Expr::RowCmpSubquery { .. }
2254 | Expr::Literal(_)
2255 | Expr::Placeholder(_)
2256 | Expr::Column(_) => {}
2257 // v7.30.2 — list elements can carry scalar subqueries
2258 // (`x IN (1, (SELECT …))`).
2259 Expr::InList { expr, list, .. } => {
2260 self.resolve_expr_subqueries(expr, cancel)?;
2261 for item in list {
2262 self.resolve_expr_subqueries(item, cancel)?;
2263 }
2264 }
2265 // v7.10.10 — recurse children.
2266 Expr::Array(items) => {
2267 for elem in items {
2268 self.resolve_expr_subqueries(elem, cancel)?;
2269 }
2270 }
2271 Expr::ArraySubscript { target, index } => {
2272 self.resolve_expr_subqueries(target, cancel)?;
2273 self.resolve_expr_subqueries(index, cancel)?;
2274 }
2275 Expr::ArraySlice { target, lo, hi } => {
2276 self.resolve_expr_subqueries(target, cancel)?;
2277 if let Some(l) = lo {
2278 self.resolve_expr_subqueries(l, cancel)?;
2279 }
2280 if let Some(h) = hi {
2281 self.resolve_expr_subqueries(h, cancel)?;
2282 }
2283 }
2284 Expr::AnyAll { expr, array, .. } => {
2285 self.resolve_expr_subqueries(expr, cancel)?;
2286 // Quantified subquery — an uncorrelated one
2287 // materialises up front; a correlated one stays for
2288 // the per-row resolver.
2289 if let Expr::ScalarSubquery(inner) = array.as_mut() {
2290 if !crate::subquery::select_is_correlated(inner) {
2291 let s = (**inner).clone();
2292 **array = self.materialize_quantified_rows(&s, cancel)?;
2293 }
2294 } else {
2295 self.resolve_expr_subqueries(array, cancel)?;
2296 }
2297 }
2298 Expr::Case {
2299 operand,
2300 branches,
2301 else_branch,
2302 } => {
2303 if let Some(o) = operand {
2304 self.resolve_expr_subqueries(o, cancel)?;
2305 }
2306 for (w, t) in branches {
2307 self.resolve_expr_subqueries(w, cancel)?;
2308 self.resolve_expr_subqueries(t, cancel)?;
2309 }
2310 if let Some(e) = else_branch {
2311 self.resolve_expr_subqueries(e, cancel)?;
2312 }
2313 }
2314 }
2315 Ok(())
2316 }
2317}
2318
2319impl Engine {
2320 /// v6.10.2 — projection for AS OF SEGMENT. Resolves
2321 /// `SelectItem::Wildcard` to all schema columns and
2322 /// `SelectItem::Expr` via the regular eval path.
2323 pub(crate) fn project_row_simple(
2324 &self,
2325 row: &Row<'static>,
2326 items: &[SelectItem],
2327 schema_cols: &[ColumnSchema],
2328 alias: &str,
2329 ) -> Result<Row<'static>, EngineError> {
2330 let ctx = self.ev_ctx(schema_cols, Some(alias));
2331 let cancel = CancelToken::none();
2332 let mut out_vals = Vec::new();
2333 for item in items {
2334 match item {
2335 // In a single-table projection (AS OF SEGMENT / RETURNING) a
2336 // qualified `t.*` covers exactly the same columns as a bare `*`.
2337 SelectItem::Wildcard | SelectItem::QualifiedWildcard(_) => {
2338 out_vals.extend(row.values.iter().cloned());
2339 }
2340 SelectItem::Expr { expr, .. } => {
2341 let v = self.eval_expr_with_correlated(expr, row, &ctx, cancel, None)?;
2342 out_vals.push(v);
2343 }
2344 }
2345 }
2346 Ok(Row::new(out_vals))
2347 }
2348
2349 /// v6.10.2 — derive the output `ColumnSchema` list for an
2350 /// AS OF SEGMENT projection. Wildcards take the full schema;
2351 /// expressions take the alias if present or a synthetic
2352 /// `?column?` (PG convention) otherwise.
2353 pub(crate) fn derive_output_columns(
2354 &self,
2355 items: &[SelectItem],
2356 schema_cols: &[ColumnSchema],
2357 table_alias: &str,
2358 ) -> Vec<ColumnSchema> {
2359 let mut out = Vec::new();
2360 for item in items {
2361 match item {
2362 // `t.*` / `OLD.*` / `NEW.*` all mirror the full table schema in
2363 // a single-table projection.
2364 SelectItem::Wildcard | SelectItem::QualifiedWildcard(_) => {
2365 out.extend(schema_cols.iter().cloned());
2366 }
2367 SelectItem::Expr { expr, alias } => {
2368 // Bare column references inherit the schema
2369 // column's name + type — PG names `RETURNING id`
2370 // "id" and types it BIGINT, and the sqlx embed
2371 // path type-checks RowDescription against the
2372 // Rust target (mailrs embed round-12).
2373 if let Expr::Column(col) = expr
2374 && let Some(sc) = schema_cols.iter().find(|c| c.name == col.name)
2375 {
2376 let name = alias.clone().unwrap_or_else(|| sc.name.clone());
2377 let mut c = ColumnSchema::new(name, sc.ty, sc.nullable);
2378 // v7.39 (read01 round 54) — carry the enum identity:
2379 // it lives outside the DataType lattice, so a derived
2380 // table built from this schema otherwise forgets it and
2381 // the OUTER `ORDER BY <enum col>` silently sorts by the
2382 // label's TEXT instead of member order.
2383 c.user_enum_type = sc.user_enum_type.clone();
2384 out.push(c);
2385 continue;
2386 }
2387 let name = alias.clone().unwrap_or_else(|| "?column?".to_string());
2388 // v7.30.4 (mailrs round-27, P0) — type the
2389 // expression with the same inference the SELECT
2390 // list uses (INT−INT=INT, BIGINT+INT=BIGINT…).
2391 // The old Text default broke every typed decode
2392 // of `RETURNING uidnext - 1 AS uid`: four days
2393 // of inbound mail indexed nowhere. Inference
2394 // failure keeps the old Text fallback rather
2395 // than inventing new error paths here.
2396 // v7.39 (round 258) — take the enum identity from the
2397 // same projection build, not just the type: a constant
2398 // SELECT (`SELECT 'ok'::mood AS x`, which is what a
2399 // VALUES row lowers to) is an EXPRESSION, so it landed
2400 // here and the derived table forgot the enum.
2401 let (ty, nullable) = build_projection(
2402 core::slice::from_ref(item),
2403 schema_cols,
2404 table_alias,
2405 self.backslash_escapes,
2406 )
2407 .ok()
2408 .and_then(|p| p.into_iter().next())
2409 .map_or((DataType::Text, true), |p| (p.ty, p.nullable));
2410 out.push(ColumnSchema::new(name, ty, nullable));
2411 }
2412 }
2413 }
2414 out
2415 }
2416
2417 /// v4.5: SELECT with cooperative cancellation. The token is
2418 /// honoured between UNION peers and inside the bare-SELECT row
2419 /// loop; HNSW kNN graph walks and the aggregate executor don't
2420 /// honour it yet (deferred — those paths bound their work
2421 /// internally by `LIMIT k` and `GROUP BY` cardinality).
2422 /// v7.38 (read01 P3.NEW3) — materialise a `spg_*` / `pg_*` meta-view by
2423 /// its (lowercased) name, or None if the name isn't a virtual view.
2424 /// Callers decide whether to return it directly (`SELECT *`) or stage
2425 /// it as a temp table for the full query pipeline.
2426 fn meta_view_result(&self, name: &str) -> Option<QueryResult> {
2427 Some(match name {
2428 "spg_statistic" => self.exec_spg_statistic(),
2429 "spg_stat_replication" => self.exec_spg_stat_replication(),
2430 "spg_stat_segment" => self.exec_spg_stat_segment(),
2431 "spg_memory_stats" => self.exec_spg_memory_stats(),
2432 "spg_stat_query" => self.exec_spg_stat_query(),
2433 "pg_stat_statements" => self.exec_pg_stat_statements(),
2434 "spg_stat_activity" => self.exec_spg_stat_activity(),
2435 "pg_stat_activity" => self.exec_pg_stat_activity(),
2436 "pg_locks" => self.exec_pg_locks(),
2437 "pg_statio_user_tables" => self.exec_pg_statio_user_tables(),
2438 "spg_stat_mvcc" => self.exec_spg_stat_mvcc(),
2439 "spg_partition_health" => self.exec_spg_partition_health(),
2440 "spg_audit_chain" => self.exec_spg_audit_chain(),
2441 "spg_audit_verify" => self.exec_spg_audit_verify(),
2442 "spg_table_ddl" => self.exec_spg_table_ddl(),
2443 "spg_role_ddl" => self.exec_spg_role_ddl(),
2444 "spg_database_ddl" => self.exec_spg_database_ddl(),
2445 _ => return None,
2446 })
2447 }
2448
2449 /// v7.39 (round 462) — the catalog an admin / stat view SELECT
2450 /// describes against: this engine's catalog with the view staged as a
2451 /// table, exactly as `exec_select_cancel_as` stages it for a
2452 /// non-bare query.
2453 ///
2454 /// These views never reach the catalog — each is a fixed row set built
2455 /// inside its own `exec_*` — so Describe reported no columns for all
2456 /// seventeen of them. Rows are deliberately not inserted: Describe
2457 /// only needs the shape, and `infer_column_types` reads the rows we
2458 /// already have in hand.
2459 pub(crate) fn admin_view_catalog(&self, stmt: &SelectStatement) -> Option<Catalog> {
2460 let from = stmt.from.as_ref()?;
2461 if !from.joins.is_empty() || self.active_catalog().get(&from.primary.name).is_some() {
2462 return None;
2463 }
2464 let lower = from.primary.name.to_ascii_lowercase();
2465 let QueryResult::Rows { columns, rows } = self.meta_view_result(&lower)? else {
2466 return None;
2467 };
2468 let mut catalog = self.active_catalog().clone();
2469 let cols = infer_column_types(&columns, &rows);
2470 catalog
2471 .create_table(TableSchema::new(from.primary.name.clone(), cols))
2472 .ok()?;
2473 Some(catalog)
2474 }
2475
2476 pub(crate) fn exec_select_cancel(
2477 &self,
2478 stmt: &SelectStatement,
2479 cancel: CancelToken<'_>,
2480 ) -> Result<QueryResult, EngineError> {
2481 self.exec_select_cancel_as(stmt, cancel, None)
2482 }
2483
2484 /// v7.39 (round 334, V55) — the same read core, authorised as
2485 /// `as_role`. A `SECURITY DEFINER` function's body runs as the
2486 /// function's OWNER: that is the entire point of the form, and without
2487 /// it every definer function failed with "permission denied" on the
2488 /// very table it exists to expose.
2489 /// v7.39 (round 559) — see the call site. `None` for anything but
2490 /// the bare shape, so every other query keeps its old path.
2491 fn try_bare_count_star(
2492 &self,
2493 stmt: &SelectStatement,
2494 as_role: Option<&str>,
2495 ) -> Result<Option<QueryResult>, EngineError> {
2496 use spg_sql::ast::SelectItem;
2497 if as_role.is_some()
2498 || !stmt.ctes.is_empty()
2499 || !stmt.unions.is_empty()
2500 || stmt.where_.is_some()
2501 || stmt.group_by.is_some()
2502 || stmt.having.is_some()
2503 || stmt.distinct
2504 || !stmt.order_by.is_empty()
2505 || stmt.limit.is_some()
2506 || stmt.offset.is_some()
2507 || stmt.items.len() != 1
2508 {
2509 return Ok(None);
2510 }
2511 let Some(from) = &stmt.from else {
2512 return Ok(None);
2513 };
2514 if !from.joins.is_empty()
2515 || stmt.locking.is_some()
2516 || from.primary.lateral_subquery.is_some()
2517 || from.primary.unnest_expr.is_some()
2518 || from.primary.generate_series_args.is_some()
2519 || from.primary.name.is_empty()
2520 || from.primary.name.starts_with("__spg_")
2521 {
2522 return Ok(None);
2523 }
2524 // A partition PARENT holds no rows of its own — they live in the
2525 // children — so its header count is 0 and the ordinary path has
2526 // to fan out. Caught by the partition conformance cases.
2527 //
2528 // v7.39 (round 645) — and an INHERITANCE parent holds only SOME
2529 // of them, which is worse: its header count is a real number,
2530 // just not the answer. `SELECT count(*) FROM par` returned 1
2531 // where PG returns 2, because this shortcut fired before the
2532 // fan-out could. The question is "does anything descend from
2533 // this", not "was it declared a partition parent".
2534 if crate::partition::has_children(self.active_catalog(), &from.primary.name) {
2535 return Ok(None);
2536 }
2537 let SelectItem::Expr { expr, alias } = &stmt.items[0] else {
2538 return Ok(None);
2539 };
2540 let spg_sql::ast::Expr::FunctionCall { name, args } = expr else {
2541 return Ok(None);
2542 };
2543 if !name.eq_ignore_ascii_case("count_star") || !args.is_empty() {
2544 return Ok(None);
2545 }
2546 // A row-security policy filters rows, so the header count is not
2547 // the answer; the ordinary path applies the policy.
2548 let Some(table) = self.active_catalog().get(&from.primary.name) else {
2549 return Ok(None);
2550 };
2551 if table.schema().row_security {
2552 return Ok(None);
2553 }
2554 // Rows frozen to the cold tier are not in `headers`, so the
2555 // header count would miss them. Caught by the cold-tier e2e.
2556 if table.has_cold_rows_fast() {
2557 return Ok(None);
2558 }
2559 let n = table.count_visible(&self.current_snapshot());
2560 let col = alias.clone().unwrap_or_else(|| String::from("count"));
2561 Ok(Some(QueryResult::Rows {
2562 columns: alloc::vec![ColumnSchema::new(col, DataType::BigInt, false)],
2563 rows: alloc::vec![Row::new(alloc::vec![Value::BigInt(
2564 i64::try_from(n).unwrap_or(i64::MAX)
2565 )])],
2566 }))
2567 }
2568
2569 /// v7.39 (round 560) — `SELECT <indexed col> FROM t WHERE <range on
2570 /// that col>` served from the index, never reading a row.
2571 ///
2572 /// Measured over pgwire on a 500k table, a 100k-row range: PG18's
2573 /// Index Only Scan 3.6 ms against SPG's 30 ms, widening with the row
2574 /// count (2x at 1k). PG needs its visibility map for this — a heap
2575 /// tuple carries its own visibility, so an index entry alone cannot
2576 /// say whether the row is live, and PG reads the heap for any page
2577 /// the map does not mark all-visible. SPG keeps a header array
2578 /// beside the rows, so the locator answers it directly and there is
2579 /// no map to be stale.
2580 /// v7.39 (round 564) — the shape test, once, for both the
2581 /// materialising scan and the streaming one.
2582 ///
2583 /// Two callers asking the same question in two places is how a fact
2584 /// starts drifting; the answer here is the single copy. Returns the
2585 /// table, the alias the predicate is written against, the projected
2586 /// column's position, and the name the single output column takes.
2587 pub(crate) fn index_only_shape<'s>(
2588 &'s self,
2589 stmt: &'s SelectStatement,
2590 ) -> Option<(&'s spg_storage::Table, &'s str, usize, String)> {
2591 use spg_sql::ast::SelectItem;
2592 if !stmt.ctes.is_empty()
2593 || !stmt.unions.is_empty()
2594 || stmt.group_by.is_some()
2595 || stmt.having.is_some()
2596 || stmt.distinct
2597 || stmt.locking.is_some()
2598 || !stmt.order_by.is_empty()
2599 || stmt.limit.is_some()
2600 || stmt.offset.is_some()
2601 || stmt.items.len() != 1
2602 {
2603 return None;
2604 }
2605 let (Some(from), Some(_)) = (&stmt.from, &stmt.where_) else {
2606 return None;
2607 };
2608 if !from.joins.is_empty()
2609 || from.primary.lateral_subquery.is_some()
2610 || from.primary.unnest_expr.is_some()
2611 || from.primary.generate_series_args.is_some()
2612 || from.primary.name.is_empty()
2613 || from.primary.name.starts_with("__spg_")
2614 {
2615 return None;
2616 }
2617 // v7.39 (round 645) — see the note on the sibling shortcut above:
2618 // an inheritance parent's own header count is not the answer.
2619 if crate::partition::has_children(self.active_catalog(), &from.primary.name) {
2620 return None;
2621 }
2622 let SelectItem::Expr { expr, alias } = &stmt.items[0] else {
2623 return None;
2624 };
2625 let spg_sql::ast::Expr::Column(c) = expr else {
2626 return None;
2627 };
2628 let alias_name = from.primary.alias.as_deref().unwrap_or(&from.primary.name);
2629 if let Some(q) = c.qualifier.as_deref()
2630 && !q.eq_ignore_ascii_case(alias_name)
2631 {
2632 return None;
2633 }
2634 let table = self.active_catalog().get(&from.primary.name)?;
2635 if table.schema().row_security {
2636 return None;
2637 }
2638 let cols = &table.schema().columns;
2639 let pos = cols
2640 .iter()
2641 .position(|s| s.name.eq_ignore_ascii_case(&c.name))?;
2642 let out = alias.clone().unwrap_or_else(|| cols[pos].name.clone());
2643 Some((table, alias_name, pos, out))
2644 }
2645
2646 /// v7.39 (round 565) — would this statement be answered out of the
2647 /// index alone?
2648 ///
2649 /// EXPLAIN has to name the node the executor will actually run, and
2650 /// the only honest way to know is to ask the same two questions the
2651 /// executor asks: the statement's shape, and everything decidable
2652 /// about the scan before it walks. Neither is re-stated here.
2653 pub(crate) fn stmt_takes_index_only_scan(&self, stmt: &SelectStatement) -> bool {
2654 let Some((table, alias_name, pos, _)) = self.index_only_shape(stmt) else {
2655 return false;
2656 };
2657 let Some(where_) = stmt.where_.as_ref() else {
2658 return false;
2659 };
2660 crate::index_access::index_only_precheck(
2661 where_,
2662 &table.schema().columns,
2663 table,
2664 alias_name,
2665 pos,
2666 )
2667 .is_some()
2668 }
2669
2670 fn try_index_only_scan(
2671 &self,
2672 stmt: &SelectStatement,
2673 ) -> Result<Option<QueryResult>, EngineError> {
2674 let Some((table, alias_name, pos, out_name)) = self.index_only_shape(stmt) else {
2675 return Ok(None);
2676 };
2677 // r1058 — same declines as `try_exec_joined_streaming`: CTEs
2678 // are not materialised here, and a partition parent's own
2679 // heap/indexes are empty (its rows live in the children).
2680 if !stmt.ctes.is_empty() {
2681 return Ok(None);
2682 }
2683 if let Some(from) = &stmt.from
2684 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
2685 {
2686 return Ok(None);
2687 }
2688 let where_ = stmt.where_.as_ref().expect("shape checked it");
2689 let cols = &table.schema().columns;
2690 let Some(values) = crate::index_access::try_index_only_range(
2691 where_,
2692 cols,
2693 table,
2694 alias_name,
2695 &self.current_snapshot(),
2696 pos,
2697 ) else {
2698 return Ok(None);
2699 };
2700 let schema = alloc::vec![ColumnSchema::new(
2701 out_name,
2702 cols[pos].ty,
2703 cols[pos].nullable
2704 )];
2705 Ok(Some(QueryResult::Rows {
2706 columns: schema,
2707 rows: values
2708 .into_iter()
2709 .map(|v| Row::new(alloc::vec![v]))
2710 .collect(),
2711 }))
2712 }
2713
2714 /// v7.39 (round 564) — the same scan, emitting each value instead of
2715 /// building a `Vec<Row>` for the encoder to walk once and drop.
2716 ///
2717 /// A profile of the server serving a 50k-row range put 10.2% of the
2718 /// connection thread's CPU on BUILDING that vector and another 9.7%
2719 /// on dropping it — a fifth of the query, spent allocating and
2720 /// freeing one single-element `Vec` per output row so that the wire
2721 /// encoder could borrow each value for a few nanoseconds. The
2722 /// streaming interface it then hands them to takes `&[Value]`
2723 /// already.
2724 ///
2725 /// Returns `None` when the shape does not apply, so the caller falls
2726 /// back before anything has been emitted.
2727 pub(crate) fn try_index_only_stream<F>(
2728 &self,
2729 stmt: &SelectStatement,
2730 emit: &mut F,
2731 ) -> Result<Option<usize>, EngineError>
2732 where
2733 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
2734 {
2735 let Some((table, alias_name, pos, out_name)) = self.index_only_shape(stmt) else {
2736 return Ok(None);
2737 };
2738 // r1058 — same declines as `try_exec_joined_streaming`: CTEs
2739 // are not materialised here, and a partition parent's own
2740 // heap/indexes are empty (its rows live in the children).
2741 if !stmt.ctes.is_empty() {
2742 return Ok(None);
2743 }
2744 if let Some(from) = &stmt.from
2745 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
2746 {
2747 return Ok(None);
2748 }
2749 let where_ = stmt.where_.as_ref().expect("shape checked it");
2750 let cols = &table.schema().columns;
2751 let schema = alloc::vec![ColumnSchema::new(
2752 out_name,
2753 cols[pos].ty,
2754 cols[pos].nullable
2755 )];
2756 let snapshot = self.current_snapshot();
2757 // The header goes out only once the walk has agreed to run — a
2758 // shape rejection after it would leave the client with a
2759 // RowDescription for a result that never comes.
2760 let mut wrote_header = false;
2761 let counted = crate::index_access::index_only_range_each(
2762 where_,
2763 cols,
2764 table,
2765 alias_name,
2766 &snapshot,
2767 pos,
2768 &mut |v: spg_storage::Value<'_>| {
2769 if !wrote_header {
2770 emit(crate::StreamItem::Header(&schema))?;
2771 wrote_header = true;
2772 }
2773 emit(crate::StreamItem::Row(crate::RowCells::Refs(&[&v])))
2774 },
2775 );
2776 match counted {
2777 None => Ok(None),
2778 Some(Err(e)) => Err(e),
2779 Some(Ok(n)) => {
2780 if !wrote_header {
2781 emit(crate::StreamItem::Header(&schema))?;
2782 }
2783 Ok(Some(n))
2784 }
2785 }
2786 }
2787
2788 /// `DISTINCT ON`'s de-duplication, which runs after the inner
2789 /// SELECT has produced its rows.
2790 ///
2791 /// `#[inline(never)]` and out of `exec_select_cancel_as` for the
2792 /// reason round 848 established: a debug build gives every branch's
2793 /// locals a slot in the frame whichever branch runs, and this one is
2794 /// eighty lines of hashing, key slicing and survivor sorting that a
2795 /// statement without `DISTINCT ON` never touches. Round 867
2796 /// measured `exec_select_cancel_as` holding ~46 KB on a path that
2797 /// reaches none of it — the segment that had been blamed on
2798 /// `exec_bare_select_cancel`, which turned out to hold 2 KB.
2799 #[inline(never)]
2800 fn apply_distinct_on(
2801 &self,
2802 result: QueryResult,
2803 don_hidden: usize,
2804 don_limit: &(
2805 Option<spg_sql::ast::LimitExpr>,
2806 Option<spg_sql::ast::LimitExpr>,
2807 ),
2808 don_top1: usize,
2809 orig_order_by: &[spg_sql::ast::OrderBy],
2810 ) -> Result<QueryResult, EngineError> {
2811 let QueryResult::Rows { columns, rows } = result else {
2812 return Ok(result);
2813 };
2814 // The keys are the hidden trailing columns appended above.
2815 // v7.39 (round 729) — top-1 mode: the trailing columns are the
2816 // DON keys plus the ORDER tail; keep each group's best in one
2817 // hash pass, then sort the SURVIVORS with the original spec.
2818 let mut kept: alloc::vec::Vec<Row<'static>>;
2819 let key_start;
2820 if don_top1 > 0 {
2821 let tail = don_top1 - 1;
2822 key_start = columns.len().saturating_sub(don_hidden + tail);
2823 let ord_start = key_start + don_hidden;
2824 let tail_dirs: alloc::vec::Vec<(bool, Option<bool>)> = orig_order_by[don_hidden..]
2825 .iter()
2826 .map(|o| (o.desc, o.nulls_first))
2827 .collect();
2828 let mysql = self.backslash_escapes;
2829 let better = |a: &Row<'static>, b: &Row<'static>| -> bool {
2830 for (k, (desc, nf)) in tail_dirs.iter().enumerate() {
2831 let av = a.values.get(ord_start + k).unwrap_or(&Value::Null);
2832 let bv = b.values.get(ord_start + k).unwrap_or(&Value::Null);
2833 match crate::order_by_value_cmp_in(*desc, *nf, av, bv, mysql) {
2834 core::cmp::Ordering::Less => return true,
2835 core::cmp::Ordering::Greater => return false,
2836 core::cmp::Ordering::Equal => {}
2837 }
2838 }
2839 false
2840 };
2841 let mut slot: hashbrown::HashMap<String, usize> = hashbrown::HashMap::new();
2842 let mut best: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
2843 let mut keybuf = String::new();
2844 for row in rows {
2845 keybuf.clear();
2846 for v in row.values.get(key_start..ord_start).unwrap_or(&[]) {
2847 aggregate::push_canonical_key(&mut keybuf, v);
2848 }
2849 match slot.get(keybuf.as_str()) {
2850 Some(&i) => {
2851 if better(&row, &best[i]) {
2852 best[i] = row;
2853 }
2854 }
2855 None => {
2856 slot.insert(keybuf.clone(), best.len());
2857 best.push(row);
2858 }
2859 }
2860 }
2861 // Survivors sort with the FULL original spec (keys are still
2862 // aboard as hidden columns).
2863 let full_dirs: alloc::vec::Vec<(bool, Option<bool>)> = orig_order_by
2864 .iter()
2865 .map(|o| (o.desc, o.nulls_first))
2866 .collect();
2867 best.sort_by(|a, b| {
2868 for (k, (desc, nf)) in full_dirs.iter().enumerate() {
2869 let av = a.values.get(key_start + k).unwrap_or(&Value::Null);
2870 let bv = b.values.get(key_start + k).unwrap_or(&Value::Null);
2871 match crate::order_by_value_cmp_in(*desc, *nf, av, bv, mysql) {
2872 core::cmp::Ordering::Equal => {}
2873 o => return o,
2874 }
2875 }
2876 core::cmp::Ordering::Equal
2877 });
2878 for r in &mut best {
2879 r.values.truncate(key_start);
2880 }
2881 kept = best;
2882 } else {
2883 key_start = columns.len().saturating_sub(don_hidden);
2884 let mut seen: alloc::vec::Vec<alloc::vec::Vec<Value<'static>>> = alloc::vec::Vec::new();
2885 kept = alloc::vec::Vec::new();
2886 for mut row in rows {
2887 let key: alloc::vec::Vec<Value<'static>> =
2888 row.values.get(key_start..).unwrap_or(&[]).to_vec();
2889 if seen.iter().any(|k| k == &key) {
2890 continue;
2891 }
2892 seen.push(key);
2893 row.values.truncate(key_start);
2894 kept.push(row);
2895 }
2896 }
2897 let mut columns = columns;
2898 columns.truncate(key_start);
2899 // PG limits what DISTINCT ON left, not what fed it.
2900 let kept = apply_deferred_limit(kept, don_limit);
2901 Ok(QueryResult::Rows {
2902 columns,
2903 rows: kept,
2904 })
2905 }
2906
2907 pub(crate) fn exec_select_cancel_as(
2908 &self,
2909 stmt: &SelectStatement,
2910 cancel: CancelToken<'_>,
2911 as_role: Option<&str>,
2912 ) -> Result<QueryResult, EngineError> {
2913 // v7.39 (round 763, F31-C1) — `SELECT *, count(*) … GROUP BY
2914 // <all columns>` is legal PG (the wildcard expands to grouped
2915 // columns); SPG refused the whole shape. Expand the wildcard
2916 // into explicit column refs up front — the aggregate layer's
2917 // existing "must appear in the GROUP BY clause" validation
2918 // then answers PG's sentence for any non-grouped column.
2919 if let Some(expanded) = self.expand_aggregate_wildcard(stmt) {
2920 return self.exec_select_cancel_as(&expanded, cancel, as_role);
2921 }
2922 // v7.39 (round 559) — `SELECT count(*) FROM t` without touching
2923 // a row.
2924 //
2925 // The aggregate layer already short-circuits this to
2926 // `rows.len()`, so the O(1) part was never the problem — the
2927 // cost is UPSTREAM, materialising every visible row so that
2928 // layer can take its length. Measured over pgwire on 500k rows:
2929 // PG18 8.2 ms with two parallel workers, 10.3 ms with
2930 // parallelism off, SPG 16.5 ms — 1.6x slower than a
2931 // single-threaded PG on the commonest aggregate there is, and no
2932 // ledger entry recorded it.
2933 //
2934 // Counting visible HEADERS needs no row at all. PG cannot do
2935 // this: its visibility lives in the heap tuples themselves, so
2936 // it has to read them (that is why its own count(*) is a full
2937 // scan, parallel or not).
2938 // v7.39 (read01 round 57) — the table-privilege gate on the common
2939 // read core. A superuser session returns from it immediately.
2940 // v7.39 (round 529) — resolve an ORDER BY that names an output
2941 // ALIAS. The statement-level pass never reached a SELECT nested in
2942 // a FROM clause, a CTE or a scalar subquery, so the same query
2943 // worked on its own and failed the moment anything wrapped it —
2944 // which is what generated SQL does constantly.
2945 let aliased;
2946 let stmt = if crate::orderby::order_by_names_an_alias(stmt) {
2947 let mut s = stmt.clone();
2948 crate::orderby::resolve_order_by_position(&mut s);
2949 aliased = s;
2950 &aliased
2951 } else {
2952 stmt
2953 };
2954 // v7.39 (round 529) — DISTINCT ON needs two things it did not have.
2955 //
2956 // Its keys were evaluated against the PROJECTED row, so a key that
2957 // is not in the select list — `SELECT DISTINCT ON (g) v FROM t
2958 // ORDER BY g, v DESC`, the canonical "latest row per group" — could
2959 // not be read at all and the query failed. PG evaluates them on the
2960 // input. They are projected as hidden columns here and stripped
2961 // again below, the same way the grouping-set ordering columns
2962 // already travel.
2963 //
2964 // And the dedup ran AFTER the inner statement's LIMIT, so
2965 // `… DISTINCT ON (g) … LIMIT 2` on four rows answered ONE row where
2966 // PG answers two: the limit had already taken two rows of the same
2967 // group before anything deduplicated them. A paginated DISTINCT ON
2968 // returned short pages, with no error. The limit is deferred to
2969 // after the dedup, which is PG's order.
2970 let don_stmt;
2971 // v7.39 (round 729) — the top-1 consumer needs the ORIGINAL
2972 // order spec (the rewritten stmt's is emptied).
2973 let orig_order_by = stmt.order_by.clone();
2974 let (stmt, don_hidden, don_limit, don_top1) = if stmt.distinct_on.is_empty() {
2975 (stmt, 0, (None, None), 0usize)
2976 } else {
2977 let mut s = stmt.clone();
2978 let hidden = s.distinct_on.len();
2979 for (i, e) in stmt.distinct_on.iter().enumerate() {
2980 s.items.push(SelectItem::Expr {
2981 expr: e.clone(),
2982 alias: Some(alloc::format!("__distinct_on_{i}")),
2983 });
2984 }
2985 // v7.39 (round 729) — group-top-1 short circuit. When the
2986 // DISTINCT ON keys are exactly the ORDER BY's leading keys,
2987 // the answer is "per group, the row that wins the remaining
2988 // order" — a single O(n) hash pass. The old path sorted the
2989 // ENTIRE input first (500k rows, ~180 ms on the panel cell)
2990 // to keep 100. The inner query runs UNSORTED with every
2991 // order key appended as a hidden column; the dedup below
2992 // keeps each group's best, then sorts the SURVIVORS.
2993 // Declared-collation order keys stay on the sorting path
2994 // (the value comparator here is collation-blind).
2995 let prefix_matches = s.order_by.len() >= hidden
2996 && stmt
2997 .distinct_on
2998 .iter()
2999 .zip(s.order_by.iter())
3000 .all(|(d, o)| *d == o.expr && !o.desc && o.nulls_first.is_none());
3001 let colls_plain =
3002 crate::orderby::order_by_collations(&s.order_by, &self.ev_ctx(&[], None))
3003 .map(|cs| cs.iter().all(Option::is_none))
3004 .unwrap_or(false);
3005 let top1_tail = if prefix_matches && colls_plain && s.group_by.is_none() {
3006 let tail = s.order_by.len() - hidden;
3007 for (j, o) in s.order_by[hidden..].iter().enumerate() {
3008 s.items.push(SelectItem::Expr {
3009 expr: o.expr.clone(),
3010 alias: Some(alloc::format!("__don_ord_{j}")),
3011 });
3012 }
3013 // Carry the tail's direction flags through the aliases'
3014 // ORDER; the survivors re-sort below with the full spec.
3015 s.order_by = Vec::new();
3016 tail + 1 // sentinel: 1 + number of tail keys (0 tail is still active)
3017 } else {
3018 0
3019 };
3020 // Only a folded literal is deferred; a placeholder or an
3021 // expression keeps the path it has today rather than being
3022 // resolved a second way here.
3023 let deferrable = matches!(
3024 (&s.limit, &s.offset),
3025 (
3026 None | Some(spg_sql::ast::LimitExpr::Literal(_)),
3027 None | Some(spg_sql::ast::LimitExpr::Literal(_))
3028 )
3029 );
3030 let deferred = if deferrable {
3031 (s.limit.take(), s.offset.take())
3032 } else {
3033 (None, None)
3034 };
3035 don_stmt = s;
3036 (&don_stmt, hidden, deferred, top1_tail)
3037 };
3038 self.acl_check_select_as(stmt, as_role)?;
3039 validate_aggregate_placement(stmt)?;
3040 // v7.39 (round 559) — the bare `count(*)` fast path, AFTER the
3041 // privilege gate above. Placed before it at first, and the
3042 // security-definer e2e caught it immediately: a SECURITY INVOKER
3043 // function whose body is `SELECT count(*) FROM t` answered
3044 // instead of being refused, because the fast path never reached
3045 // the check.
3046 if let Some(r) = self.try_bare_count_star(stmt, as_role)? {
3047 return Ok(r);
3048 }
3049 // v7.39 (round 560) — an index-only range scan. Same placement
3050 // reasoning as the count above: after the privilege gate.
3051 if let Some(r) = self.try_index_only_scan(stmt)? {
3052 return Ok(r);
3053 }
3054 validate_locking_clause(stmt)?;
3055 let result = self.exec_select_cancel_inner(stmt, cancel)?;
3056 // v7.39 (round 135) — drop the synthetic `__grp_ord_*` ordering columns
3057 // the parser injects for GROUPING() in ORDER BY on a grouping-set query.
3058 // They carry the per-branch mask through the UNION-ALL sort and must not
3059 // appear in the output. Stripped per SELECT level (grouping-set queries
3060 // are often wrapped in a derived subquery), before DISTINCT ON.
3061 let result = strip_synthetic_order_cols(result);
3062 // v7.37.17 (17.6 siblings) — `SELECT DISTINCT ON (exprs)`:
3063 // rows arrive here already ORDER BY'd; keep the FIRST row of
3064 // each group the expressions define (PG semantics). The
3065 // expressions evaluate against the projected schema — an
3066 // expression that isn't in the select list errors honestly.
3067 if stmt.distinct_on.is_empty() {
3068 return Ok(result);
3069 }
3070 self.apply_distinct_on(result, don_hidden, &don_limit, don_top1, &orig_order_by)
3071 }
3072
3073 /// The UNION chain: execute the head as a bare block, then fold each
3074 /// peer in with left-associative dedup.
3075 ///
3076 /// `#[inline(never)]` and out of `exec_select_cancel_inner` for the
3077 /// reason round 848 established. A statement with no unions returns
3078 /// one line above the call — and every nested subquery on a deep
3079 /// path is such a statement, so each level of the recursion carried
3080 /// 170 lines of locals it could not reach. Round 867 measured that
3081 /// frame at 34,800 bytes, the largest single one on the descent,
3082 /// after two earlier attributions had blamed its caller and then its
3083 /// callee: the gap between two marks is the frame of everything
3084 /// BETWEEN them, and this function had no mark of its own.
3085 #[inline(never)]
3086 fn exec_union_chain(
3087 &self,
3088 stmt_ref: &SelectStatement,
3089 stmt: &SelectStatement,
3090 cancel: CancelToken<'_>,
3091 ) -> Result<QueryResult, EngineError> {
3092 // UNION path: clone-strip the head into a bare block (its own
3093 // DISTINCT and any inner ORDER BY are dropped by parser rule —
3094 // the wrapper SelectStatement carries them), execute, then chain
3095 // peers with left-associative dedup semantics.
3096 // v7.39 (round 232) — the wrapper's ORDER BY addresses the head's
3097 // output columns; a position past their count is PG's 42P10.
3098 crate::orderby::check_order_by_positions(stmt_ref)?;
3099 let mut head_unknown = branch_unknown_mask(stmt_ref);
3100 let head_regcast = branch_regcast_mask(stmt_ref);
3101 let mut head = stmt_ref.clone();
3102 head.unions = Vec::new();
3103 head.order_by = Vec::new();
3104 head.limit = None;
3105 let QueryResult::Rows {
3106 mut columns,
3107 mut rows,
3108 } = self.exec_bare_select_cancel(&head, cancel)?
3109 else {
3110 unreachable!("bare SELECT cannot return CommandOk")
3111 };
3112 for (kind, peer) in &stmt_ref.unions {
3113 // v7.37.17 (17.6 siblings) — a peer carrying its own
3114 // unions is a nested INTERSECT group (the parser's
3115 // precedence regrouping); recurse through the
3116 // union-aware wrapper for it.
3117 let peer_result = if peer.unions.is_empty() {
3118 self.exec_bare_select_cancel(peer, cancel)?
3119 } else {
3120 self.exec_select_cancel(peer, cancel)?
3121 };
3122 let QueryResult::Rows {
3123 columns: peer_cols,
3124 rows: mut peer_rows,
3125 } = peer_result
3126 else {
3127 unreachable!("bare SELECT cannot return CommandOk")
3128 };
3129 if peer_cols.len() != columns.len() {
3130 // v7.39 (round 232) — PG's wording, which clients match on.
3131 return Err(EngineError::Unsupported(alloc::format!(
3132 "each {} query must have the same number of columns",
3133 set_op_name(*kind)
3134 )));
3135 }
3136 // v7.39 (round 232+233) — PG resolves each result column to one
3137 // type before it merges anything, and refuses the query when the
3138 // two branches have no common type. SPG's unifier
3139 // (`unify_union_columns`) is value-driven and deliberately
3140 // conservative — "a column where any cell fails to coerce is left
3141 // exactly as it was" — so a mismatch produced a column holding
3142 // BOTH types (`SELECT a, b FROM t UNION SELECT b, a FROM t` came
3143 // back with integers and text interleaved) instead of an error.
3144 //
3145 // The check has to read the branch ASTs, not just their schemas:
3146 // SPG has no `Unknown` DataType, so a bare `'a'` literal describes
3147 // as TEXT and is indistinguishable from a real text column by
3148 // schema alone — yet PG treats the two completely differently
3149 // (`SELECT 1 UNION SELECT 'a'` is an input-syntax error on the
3150 // literal, `SELECT 1 UNION SELECT 'a'::text` is a type mismatch).
3151 let peer_unknown = branch_unknown_mask(peer);
3152 let peer_regcast = branch_regcast_mask(peer);
3153 for i in 0..columns.len() {
3154 let hu = head_unknown.get(i).copied().unwrap_or(false);
3155 let pu = peer_unknown.get(i).copied().unwrap_or(false);
3156 let (ht, pt) = (columns[i].ty, peer_cols[i].ty);
3157 let reg_dual = peer_regcast.get(i).copied().unwrap_or(false)
3158 || head_regcast.get(i).copied().unwrap_or(false);
3159 match (hu, pu) {
3160 // Both sides carry a real type: they must share a category.
3161 (false, false) => {
3162 if !reg_dual && !crate::conversions::types_unify(ht, pt) {
3163 return Err(EngineError::Unsupported(alloc::format!(
3164 "{} types {} and {} cannot be matched",
3165 set_op_name(*kind),
3166 crate::conversions::pg_type_name_for_error(ht),
3167 crate::conversions::pg_type_name_for_error(pt),
3168 )));
3169 }
3170 }
3171 // One side is an untyped literal: it takes the other's
3172 // type, and failing to convert is the error PG reports.
3173 (true, false) => {
3174 coerce_branch_column(&mut rows, i, pt, &columns[i].name)?;
3175 columns[i].ty = pt;
3176 head_unknown[i] = false;
3177 }
3178 (false, true) => {
3179 coerce_branch_column(&mut peer_rows, i, ht, &columns[i].name)?;
3180 }
3181 // Both untyped — nothing to resolve against yet.
3182 (true, true) => {}
3183 }
3184 }
3185 // v7.37 D.26 — a UNION result column is nullable when ANY branch is
3186 // nullable (PG semantics). Previously the result kept only the head's
3187 // nullability, so `VALUES (1),(NULL)` (a UNION-ALL chain seeded by the
3188 // non-null `1`) wrongly reported the column NOT NULL, which let
3189 // `count(col)`'s NOT-NULL fast-path count the NULL row.
3190 for (i, pc) in peer_cols.iter().enumerate() {
3191 if pc.nullable {
3192 columns[i].nullable = true;
3193 }
3194 }
3195 // v7.39 (round 410) — under MySQL, set-op dedup / matching folds
3196 // text by the session collation (CI + accent + PAD SPACE), like
3197 // GROUP BY. PG stays byte-exact.
3198 let mysql = self.backslash_escapes;
3199 match kind {
3200 UnionKind::All => rows.extend(peer_rows),
3201 UnionKind::Distinct => {
3202 rows.extend(peer_rows);
3203 rows = dedup_rows(rows, mysql);
3204 }
3205 // v7.37.17 (17.6 siblings) — PG set semantics.
3206 // v7.39 (round 591) — all four ask the same question of the
3207 // right side, and all four used to answer it by scanning it
3208 // once per left row. `PeerIndex` buckets it by the hash
3209 // DISTINCT already uses, so the answer is a lookup.
3210 // INTERSECT: distinct rows present on both sides.
3211 UnionKind::Intersect => {
3212 let idx = PeerIndex::build(&peer_rows, mysql);
3213 rows = dedup_rows(rows, mysql)
3214 .into_iter()
3215 .filter(|r| idx.contains(r))
3216 .collect();
3217 }
3218 // INTERSECT ALL: multiset intersection — each row
3219 // keeps min(left count, right count) occurrences.
3220 UnionKind::IntersectAll => {
3221 let mut idx = PeerIndex::build(&peer_rows, mysql);
3222 let mut kept: Vec<Row<'static>> = Vec::new();
3223 for r in rows {
3224 if idx.take_one(&r) {
3225 kept.push(r);
3226 }
3227 }
3228 rows = kept;
3229 }
3230 // EXCEPT: distinct left rows absent from the right.
3231 UnionKind::Except => {
3232 let idx = PeerIndex::build(&peer_rows, mysql);
3233 rows = dedup_rows(rows, mysql)
3234 .into_iter()
3235 .filter(|r| !idx.contains(r))
3236 .collect();
3237 }
3238 // EXCEPT ALL: multiset subtraction — each right
3239 // occurrence cancels one left occurrence.
3240 UnionKind::ExceptAll => {
3241 let mut idx = PeerIndex::build(&peer_rows, mysql);
3242 let mut kept: Vec<Row<'static>> = Vec::new();
3243 for r in rows {
3244 if !idx.take_one(&r) {
3245 kept.push(r);
3246 }
3247 }
3248 rows = kept;
3249 }
3250 }
3251 }
3252 // PG resolves a UNION / VALUES result column to one common type
3253 // and casts every branch to it (`SELECT '2020-01-01'::date UNION
3254 // ALL SELECT '2020-01-02'` → both DATE, not DATE + TEXT). SPG
3255 // built each branch independently, leaving mixed-type columns
3256 // that broke ORDER BY, comparisons, and value-based window
3257 // frames. Unify + coerce before the combined ORDER BY sees them.
3258 unify_union_columns(&mut columns, &mut rows);
3259 // ORDER BY at the top of a UNION applies to the combined result.
3260 // Eval against the projected schema (NOT the source table).
3261 if !stmt.order_by.is_empty() {
3262 // v7.39 (read01 round 54) — the combined-result ctx must carry the
3263 // catalog, and the projected columns must keep their enum identity
3264 // (`user_enum_type`), or `ORDER BY <enum col>` over a UNION sorts
3265 // by TEXT instead of member order — silently wrong rows, not an
3266 // error. (Same shape as the enum-order knife's GROUP BY fix.)
3267 let synth_ctx = EvalContext::new(&columns, None).with_catalog(self.active_catalog());
3268 // v7.37.17 (17.6 siblings) — positional keys (ORDER BY 1)
3269 // survive to here when the head projects a Wildcard (the
3270 // group-tail wrapper shape): map them onto the Nth
3271 // projected column so the combined sort works.
3272 let resolved_order: Vec<spg_sql::ast::OrderBy> = stmt
3273 .order_by
3274 .iter()
3275 .map(|o| {
3276 let mut o = o.clone();
3277 if let Expr::Literal(spg_sql::ast::Literal::Integer(n)) = &o.expr
3278 && *n >= 1
3279 && let Ok(idx) = usize::try_from(*n - 1)
3280 && idx < columns.len()
3281 {
3282 o.expr = Expr::Column(spg_sql::ast::ColumnName {
3283 qualifier: None,
3284 name: columns[idx].name.clone(),
3285 });
3286 }
3287 o
3288 })
3289 .collect();
3290 let descs: Vec<bool> = resolved_order.iter().map(|o| o.desc).collect();
3291 let mut tagged: Vec<(Vec<OrderKey>, Row)> = Vec::with_capacity(rows.len());
3292 for r in rows {
3293 let keys = build_order_keys(&resolved_order, &r, &synth_ctx)?;
3294 tagged.push((keys, r));
3295 }
3296 sort_by_keys(&mut tagged, &descs);
3297 rows = tagged.into_iter().map(|(_, r)| r).collect();
3298 }
3299 apply_offset_and_limit(&mut rows, stmt.offset_literal(), stmt.limit_literal());
3300 Ok(QueryResult::Rows { columns, rows })
3301 }
3302
3303 fn exec_select_cancel_inner(
3304 &self,
3305 stmt: &SelectStatement,
3306 cancel: CancelToken<'_>,
3307 ) -> Result<QueryResult, EngineError> {
3308 cancel.check()?;
3309 // v7.38 P0 元机制 A — first observable point inside the
3310 // planner / executor. Tests use this to inject a delay or
3311 // a cancellation race before any row is produced. Release
3312 // build expands to `let _ = (...);` — zero cost.
3313 crate::injection_point!("planner_first_row_fetch", &stmt.from);
3314 // v7.39 (round 705) — WINDOW-clause definitions nothing referenced.
3315 // PG analyses every definition, referenced or not, so `SELECT i FROM
3316 // t WINDOW w AS (ORDER BY nosuch)` fails there and silently
3317 // succeeded here (the parser used to drop the unreferenced defs
3318 // whole). The check is the CREATE VIEW check's shape (round 700): a
3319 // LIMIT-0 run of the same FROM with the definitions' key
3320 // expressions as the projection — it cannot disagree with what a
3321 // referencing window would have done, because it resolves the same
3322 // names the same way. Zero cost for the ordinary statement: the
3323 // list is empty unless a WINDOW clause left unreferenced defs.
3324 if !stmt.window_check_exprs.is_empty() {
3325 let mut probe = stmt.clone();
3326 probe.items = stmt
3327 .window_check_exprs
3328 .iter()
3329 .map(|e| spg_sql::ast::SelectItem::Expr {
3330 expr: e.clone(),
3331 alias: None,
3332 })
3333 .collect();
3334 probe.window_check_exprs = Vec::new();
3335 probe.distinct = false;
3336 probe.distinct_on = Vec::new();
3337 probe.group_by = None;
3338 probe.group_by_all = false;
3339 probe.having = None;
3340 probe.unions = Vec::new();
3341 probe.order_by = Vec::new();
3342 probe.locking = None;
3343 probe.limit = Some(spg_sql::ast::LimitExpr::Literal(0));
3344 probe.offset = None;
3345 probe.limit_with_ties = false;
3346 self.exec_select_cancel_inner(&probe, cancel)?;
3347 }
3348 // v7.39 (read01 round 74) — lower `(f(args)).*`. Naming a record's fields
3349 // takes the catalog, so the parser leaves a marker and the rewrite lands
3350 // here: the call moves into a LATERAL FROM item and the item becomes one
3351 // reference per declared column. `SELECT 'p', (rows_of(2)).*` is
3352 // `SELECT 'p', __rec.id, __rec.v FROM rows_of(2) AS __rec` — reusing the
3353 // set-returning FROM machinery of rounds 65 and 69 rather than growing a
3354 // second one.
3355 if let Some(lowered) = self.lower_record_expansion(stmt)? {
3356 return self.exec_select_cancel_inner(&lowered, cancel);
3357 }
3358 // v7.17.0 Phase 1.2 — user-defined VIEW expansion. If the
3359 // FROM / JOIN graph references any catalogued view name,
3360 // re-parse the view body and prepend it as a synthetic
3361 // CTE. Recurses on views-in-views via the regular CTE
3362 // dispatch below. Fast-path: skip the walker entirely when
3363 // the catalog has no views (the typical OLTP load).
3364 if !self.active_catalog().views_all().is_empty() {
3365 if let Some(rewritten) = self.expand_views_in_select(stmt)? {
3366 return self.exec_select_cancel(&rewritten, cancel);
3367 }
3368 }
3369 // v7.37.6-B(sentori Epic 2 P0)— `SELECT … FROM <partition-parent>`
3370 // gets rewritten to a UNION-ALL over the children that overlap
3371 // the WHERE-derived key range. Uses the same CTE-injection
3372 // trick as VIEW expansion above so downstream resolution
3373 // doesn't need a partition-aware code path.
3374 if let Some(rewritten) = self.expand_partition_parents_in_select(stmt)? {
3375 return self.exec_select_cancel(&rewritten, cancel);
3376 }
3377 // v7.16.2 — information_schema / pg_catalog virtual
3378 // views (mailrs round-10 A.3). If the SELECT touches a
3379 // synthetic meta-table name (`__spg_info_*` /
3380 // `__spg_pg_*` — produced by the parser for
3381 // `information_schema.X` / `pg_catalog.X`), clone the
3382 // catalog, materialise the requested view as a real
3383 // temporary table, and re-execute against an enriched
3384 // engine. Same pattern as `exec_with_ctes` for CTEs.
3385 if !self.meta_views_materialised && select_references_meta_view(stmt) {
3386 return self.exec_select_with_meta_views(stmt, cancel);
3387 }
3388 // v6.10.2 — cold-tier time-travel short-circuit. When the
3389 // primary TableRef carries `AS OF SEGMENT '<id>'`, run a
3390 // dedicated cold-segment scan instead of the regular
3391 // hot+index path. The scope is intentionally narrow for
3392 // v6.10.2 — bare `SELECT * FROM <t> AS OF SEGMENT 'id'`,
3393 // optionally with a single-column-equality WHERE. JOINs /
3394 // aggregates / ORDER BY / subqueries on top of a time-
3395 // travelled scan are STABILITY § "Out of v6.10".
3396 if let Some(from) = &stmt.from
3397 && let Some(seg_id) = from.primary.as_of_segment
3398 {
3399 return self.exec_select_as_of_segment(stmt, from, seg_id);
3400 }
3401 // v6.2.0 / v6.5.0 — virtual-table short-circuits. Detected
3402 // pre-CTE because they don't read from the catalog and
3403 // shouldn't participate in regular FROM resolution.
3404 // v6.2.0 / v6.5.0 / v7.38 (read01 P3.NEW3) — virtual-table
3405 // short-circuits. A meta-view FROM materialises to a fixed row
3406 // set. For a bare `SELECT *` we return it directly; otherwise we
3407 // stage it as a temp table and run the normal pipeline, so
3408 // projection / WHERE / ORDER BY / aggregates work over these views
3409 // (they were `SELECT *`-only before). A real table shadowing the
3410 // name wins (checked first), which also stops the staged re-run
3411 // from recursing back into meta-view detection.
3412 if let Some(from) = &stmt.from
3413 && from.joins.is_empty()
3414 && self.active_catalog().get(&from.primary.name).is_none()
3415 {
3416 let lower = from.primary.name.to_ascii_lowercase();
3417 if let Some(result) = self.meta_view_result(&lower) {
3418 let bare = stmt.where_.is_none()
3419 && stmt.group_by.is_none()
3420 && stmt.having.is_none()
3421 && stmt.unions.is_empty()
3422 && stmt.order_by.is_empty()
3423 && stmt.limit.is_none()
3424 && stmt.offset.is_none()
3425 && !stmt.distinct
3426 && stmt.items.iter().all(|i| matches!(i, SelectItem::Wildcard));
3427 if bare {
3428 return Ok(result);
3429 }
3430 if let QueryResult::Rows { columns, rows } = result {
3431 let mut catalog = self.active_catalog().clone();
3432 let cols = infer_column_types(&columns, &rows);
3433 let schema = TableSchema::new(from.primary.name.clone(), cols);
3434 catalog.create_table(schema).map_err(EngineError::Storage)?;
3435 let t = catalog
3436 .get_mut(&from.primary.name)
3437 .expect("just-created meta-view table must exist");
3438 for row in rows {
3439 t.insert(row).map_err(EngineError::Storage)?;
3440 }
3441 let mut eng = Engine::restore(catalog);
3442 if let Some(c) = self.clock {
3443 eng = eng.with_clock(c);
3444 }
3445 if let Some(f) = self.salt_fn {
3446 eng = eng.with_salt_fn(f);
3447 }
3448 // v7.39 (read01 pgstatfuncs.c) — carry the calling-
3449 // connection identity so `WHERE pid = pg_backend_pid()`
3450 // matches inside the staged meta-view run.
3451 if let Some(f) = self.backend_pid_fn {
3452 eng.set_backend_pid_fn(f);
3453 }
3454 return eng.exec_select_cancel(stmt, cancel);
3455 }
3456 return Ok(result);
3457 }
3458 }
3459 // v4.11: CTEs materialise into a temporary enriched catalog
3460 // *before* anything else — the body SELECT can then refer
3461 // to CTE names via the regular FROM-clause resolution.
3462 // Uncorrelated only: each CTE body runs once against the
3463 // current catalog, not against later CTEs' results (left-
3464 // to-right materialisation would relax this, but we keep
3465 // it simple for v4.11 MVP).
3466 if !stmt.ctes.is_empty() {
3467 return self.exec_with_ctes(stmt, cancel);
3468 }
3469 // v4.10: subqueries (uncorrelated) are resolved here, before
3470 // the executor sees the row loop. We clone the statement so
3471 // we can mutate without disturbing the caller's AST — most
3472 // queries pass through with no subquery nodes and the clone
3473 // is cheap; with subqueries the materialisation cost
3474 // dominates anyway.
3475 let mut stmt_owned;
3476 let stmt_ref: &SelectStatement = if expr_tree_has_subquery(stmt) {
3477 stmt_owned = stmt.clone();
3478 // v7.33 (mailrs 7.32.1) — sublink pull-up first: an
3479 // aggregate-wrapped correlated scalar subquery whose
3480 // correlation key is UNIQUE/PK becomes a LEFT JOIN, so the
3481 // executor streams one join instead of splicing a per-row
3482 // subplan. Runs before the per-row/batch resolver, which then
3483 // only sees the subqueries the pull-up left behind.
3484 self.pull_up_unique_correlated_agg_subqueries(&mut stmt_owned);
3485 // v7.37.4 (A — correlated LIMIT 1 ORDER BY DESC pull-up) —
3486 // the "per-key latest" scalar subquery shape (inbox / feed
3487 // / timeline applications) becomes a CTE + LEFT JOIN
3488 // against a GROUP BY pre-aggregation that reuses the v7.33
3489 // first_ordered argmax executor. Runs AFTER unique-key
3490 // pull-up (so the unique-key fast path still wins for
3491 // single-PK lookups) and BEFORE the EXISTS sublink rewrite.
3492 // Phase 1 (this commit) is skeleton only — no-op pass.
3493 self.pull_up_correlated_limit_one_subqueries(&mut stmt_owned);
3494 // v7.34.2 (mailrs prod NOT EXISTS) — plan-time `[NOT] EXISTS`
3495 // sublink pull-up to semi/anti-join, before the resolver gets
3496 // a chance to walk per-row.
3497 self.pull_up_exists_sublinks(&mut stmt_owned);
3498 // v7.37.4 — if the LIMIT 1 pullup added CTEs, route through
3499 // exec_with_ctes so they materialise once before the body
3500 // SELECT runs. exec_with_ctes strips ctes from the body
3501 // clone, then re-enters select.
3502 if !stmt_owned.ctes.is_empty() {
3503 return self.exec_with_ctes(&stmt_owned, cancel);
3504 }
3505 // v7.37.x (docker-fair INSUBQ attack) — short-circuit
3506 // SELECT COUNT(*) FROM A WHERE A.pk IN (<uncorrelated subquery>)
3507 // BEFORE `resolve_select_subqueries` materialises the inner
3508 // result as `Vec<Expr::Literal>` (~150 µs for the 6 k-row
3509 // INSUBQ benchmark). Run the inner once, collect the result
3510 // values into a `HashSet<i64>` directly, then probe A.pk per
3511 // value and tally. Returns `Some` when the shape matches.
3512 if let Some(out) = self.try_count_star_pk_in_subquery_fast(&stmt_owned, cancel)? {
3513 return Ok(out);
3514 }
3515 self.resolve_select_subqueries(&mut stmt_owned, cancel)?;
3516 &stmt_owned
3517 } else {
3518 stmt
3519 };
3520 if stmt_ref.unions.is_empty() {
3521 return self.exec_bare_select_cancel(stmt_ref, cancel);
3522 }
3523 self.exec_union_chain(stmt_ref, stmt, cancel)
3524 }
3525
3526 #[allow(clippy::too_many_lines)]
3527 #[allow(clippy::too_many_lines)] // huge match — splitting fragments the planner
3528 /// v7.11.7 — execute `SELECT … FROM unnest(expr) [AS] alias …`.
3529 /// Synthesises a single-column virtual table whose column type
3530 /// is TEXT and whose rows are the array elements. Routes
3531 /// through the regular projection / WHERE / ORDER BY / LIMIT
3532 /// machinery so set-returning UNNEST composes naturally with
3533 /// the rest of the SELECT surface.
3534 fn exec_select_unnest(
3535 &self,
3536 stmt: &SelectStatement,
3537 primary: &TableRef,
3538 cancel: CancelToken<'_>,
3539 ) -> Result<QueryResult, EngineError> {
3540 let expr = primary
3541 .unnest_expr
3542 .as_deref()
3543 .expect("caller guards unnest_expr.is_some()");
3544 // Multi-arg unnest(a, b, …) — parallel zip, NULL-padded.
3545 // N value columns instead of one; the shared builder does
3546 // the work and the tail below (WHERE / agg / projection)
3547 // runs against the wider schema.
3548 let multi: Option<(alloc::vec::Vec<DataType>, alloc::vec::Vec<Row<'static>>)> =
3549 match unnest_zip_args(expr) {
3550 Some(args) => Some(unnest_zip_rows(args)?),
3551 None => None,
3552 };
3553 // Evaluate the array expression once. Empty schema / empty
3554 // row — uncorrelated UNNEST cannot reference outer columns.
3555 // v7.39 (read01 round 49) — the ctx must carry the catalog: the enum
3556 // introspection family (enum_range / enum_first / enum_last) resolves
3557 // its labels from the argument's STATIC enum type against the
3558 // catalog's enum registry. Without it `unnest(enum_range(NULL::mood))`
3559 // fell through to the generic arm, got NULL, and expanded to zero rows
3560 // — while the bare `SELECT enum_range(NULL::mood)` (whose ctx does
3561 // carry the catalog) worked.
3562 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
3563 let ctx = EvalContext::new(&empty_schema, None).with_catalog(self.active_catalog());
3564 let dummy_row = Row::new(alloc::vec::Vec::new());
3565 // v7.11.13 — unnest dispatches per array element type so
3566 // INT[] / BIGINT[] surface their PG types in projection.
3567 // v7.39 (round 758, F31-B8a) — the composite SRF names its own
3568 // columns (PG: lexeme | positions | weights); everything else
3569 // keeps the alias / "unnest" defaults below.
3570 let mut composite_names: Option<&[&str]> = None;
3571 let (dtypes, rows): (alloc::vec::Vec<DataType>, alloc::vec::Vec<Row<'static>>) =
3572 if let Some(m) = multi {
3573 m
3574 } else {
3575 // v7.39 (round 236) — flatten a multidimensional array into
3576 // its row-major elements (PG) before the 1-D-only match.
3577 let unnest_src = {
3578 let v = eval::eval_expr(expr, &dummy_row, &ctx).map_err(EngineError::Eval)?;
3579 crate::eval::values::flatten_2d(&v).unwrap_or(v)
3580 };
3581 let mut return_multi: Option<(
3582 alloc::vec::Vec<DataType>,
3583 alloc::vec::Vec<Row<'static>>,
3584 )> = None;
3585 let (elem_dtype, rows): (DataType, alloc::vec::Vec<Row<'static>>) = match unnest_src
3586 {
3587 Value::Null => (DataType::Text, alloc::vec::Vec::new()),
3588 Value::TextArray(items) => {
3589 let rows = items
3590 .into_iter()
3591 .map(|item| {
3592 Row::new(alloc::vec![match item {
3593 Some(s) => Value::text(s),
3594 None => Value::Null,
3595 }])
3596 })
3597 .collect();
3598 (DataType::Text, rows)
3599 }
3600 Value::IntArray(items) => {
3601 let rows = items
3602 .into_iter()
3603 .map(|item| {
3604 Row::new(alloc::vec![match item {
3605 Some(n) => Value::Int(n),
3606 None => Value::Null,
3607 }])
3608 })
3609 .collect();
3610 (DataType::Int, rows)
3611 }
3612 Value::BigIntArray(items) => {
3613 let rows = items
3614 .into_iter()
3615 .map(|item| {
3616 Row::new(alloc::vec![match item {
3617 Some(n) => Value::BigInt(n),
3618 None => Value::Null,
3619 }])
3620 })
3621 .collect();
3622 (DataType::BigInt, rows)
3623 }
3624 Value::Multirange { kind, ranges } => {
3625 let rows = ranges
3626 .iter()
3627 .map(|sp| {
3628 Row::new(alloc::vec![Value::Range {
3629 kind,
3630 lower: sp.lower.clone(),
3631 upper: sp.upper.clone(),
3632 lower_inc: sp.lower_inc,
3633 upper_inc: sp.upper_inc,
3634 empty: false,
3635 }])
3636 })
3637 .collect();
3638 (DataType::Range(kind), rows)
3639 }
3640 // v7.39 (round 758, F31-B8a) — unnest(tsvector):
3641 // one row per lexeme, PG18-measured columns
3642 // lexeme | positions | weights (`a | {1,3} |
3643 // {D,D}`); a position-less lexeme (a stripped
3644 // vector) reads NULL in both array columns.
3645 Value::TsVector(lexemes) => {
3646 composite_names = Some(&["lexeme", "positions", "weights"]);
3647 let rows = lexemes
3648 .iter()
3649 .map(|l| {
3650 let (pos, wts) = if l.positions.is_empty() {
3651 (Value::Null, Value::Null)
3652 } else {
3653 let letter = match l.weight {
3654 3 => "A",
3655 2 => "B",
3656 1 => "C",
3657 _ => "D",
3658 };
3659 (
3660 Value::SmallIntArray(
3661 l.positions
3662 .iter()
3663 .map(|p| {
3664 Some(i16::try_from(*p).unwrap_or(i16::MAX))
3665 })
3666 .collect(),
3667 ),
3668 Value::TextArray(
3669 l.positions
3670 .iter()
3671 .map(|_| Some(letter.into()))
3672 .collect(),
3673 ),
3674 )
3675 };
3676 Row::new(alloc::vec![Value::text(l.word.clone()), pos, wts])
3677 })
3678 .collect();
3679 return_multi = Some((
3680 alloc::vec![
3681 DataType::Text,
3682 DataType::SmallIntArray,
3683 DataType::TextArray
3684 ],
3685 rows,
3686 ));
3687 (DataType::Text, alloc::vec::Vec::new())
3688 }
3689 other => {
3690 // v7.39 (round 622, S05a) — see table_access.rs:
3691 // the same sentence, and it is a type mismatch.
3692 return Err(EngineError::Eval(EvalError::TypeMismatch {
3693 detail: alloc::format!(
3694 "unnest() expects an array argument, got {}",
3695 crate::conversions::pg_type_name_for_error_opt(other.data_type())
3696 ),
3697 }));
3698 }
3699 };
3700 if let Some(m) = return_multi {
3701 m
3702 } else {
3703 (alloc::vec![elem_dtype], rows)
3704 }
3705 };
3706 let alias = primary
3707 .alias
3708 .clone()
3709 .unwrap_or_else(|| "unnest".to_string());
3710 // v7.13.2 — mailrs round-6 S5. Honour PG-standard
3711 // `UNNEST(arr) AS p(col_name)` column-list aliasing:
3712 // entries map positionally over the value columns. Without
3713 // the column list, a single column falls back to the table
3714 // alias (pre-v7.13.2 behaviour); multi-arg columns default
3715 // to PG's `unnest`.
3716 let n_vals = dtypes.len();
3717 let mut schema_cols: alloc::vec::Vec<ColumnSchema> = dtypes
3718 .iter()
3719 .enumerate()
3720 .map(|(i, dt)| {
3721 let name = primary
3722 .unnest_column_aliases
3723 .get(i)
3724 .cloned()
3725 .unwrap_or_else(|| {
3726 if let Some(names) = composite_names {
3727 names
3728 .get(i)
3729 .map_or_else(|| "unnest".to_string(), |n| (*n).to_string())
3730 } else if n_vals == 1 {
3731 alias.clone()
3732 } else {
3733 "unnest".to_string()
3734 }
3735 });
3736 ColumnSchema::new(name, *dt, true)
3737 })
3738 .collect();
3739 // v7.39 (read01 round 78) — the item's row type IS this scalar when the
3740 // parser desugared a base-type-returning function here (see
3741 // TableRef::scalar_fn_item); the marker rides the column so it survives
3742 // every EvalContext an inner stage rebuilds.
3743 if primary.scalar_fn_item && schema_cols.len() == 1 {
3744 schema_cols[0].scalar_row_source = true;
3745 }
3746 // WITH ORDINALITY — trailing BIGINT counting rows from 1
3747 // in element order. The alias entry after the value
3748 // columns renames it (PG default: `ordinality`).
3749 let rows = if primary.with_ordinality {
3750 let ord_name = primary
3751 .unnest_column_aliases
3752 .get(n_vals)
3753 .cloned()
3754 .unwrap_or_else(|| "ordinality".to_string());
3755 schema_cols.push(ColumnSchema::new(ord_name, DataType::BigInt, false));
3756 rows.into_iter()
3757 .enumerate()
3758 .map(|(i, row)| {
3759 let mut vals = row.values.clone();
3760 vals.push(Value::BigInt(i as i64 + 1));
3761 Row::new(vals)
3762 })
3763 .collect()
3764 } else {
3765 rows
3766 };
3767 // v7.39 (read01 round 54) — `ev_ctx` threads the catalog; a bare
3768 // `EvalContext::new` drops it and every catalog-dependent cast
3769 // (regclass / enum / composite / domain) silently degrades.
3770 let scan_ctx = self.ev_ctx(&schema_cols, Some(&alias));
3771 // Apply WHERE.
3772 let filtered: alloc::vec::Vec<Row<'static>> = if let Some(w) = &stmt.where_ {
3773 let mut out = alloc::vec::Vec::with_capacity(rows.len());
3774 for row in rows {
3775 cancel.check()?;
3776 let v = eval::eval_expr(w, &row, &scan_ctx).map_err(EngineError::Eval)?;
3777 if matches!(v, Value::Bool(true)) {
3778 out.push(row);
3779 }
3780 }
3781 out
3782 } else {
3783 rows
3784 };
3785 // v7.17.0 Phase 3.P0-48 — aggregate dispatch over the
3786 // unnest source. Same routing the relational scan path
3787 // already takes — without it `SELECT COUNT(*) FROM
3788 // unnest(ARRAY[…])` either errored at projection time or
3789 // returned the wrong shape.
3790 if aggregate::uses_aggregate(stmt) {
3791 // v7.29 — a per-query memo so correlated scalar
3792 // subqueries batch-evaluate once (group map) instead of
3793 // executing per group.
3794 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
3795 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
3796 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
3797 .map_err(|err| match err {
3798 EngineError::Eval(ev) => ev,
3799 other => eval::EvalError::TypeMismatch {
3800 detail: alloc::format!("{other}"),
3801 },
3802 })
3803 };
3804 // v7.39 (round 656) — hand the rows over as they are rather than
3805 // collecting a second vector of `RowRef` wrappers. Note this is
3806 // a set-returning-function path, NOT the relational scan: the
3807 // measured O(rows) cost lived in `run_single_table_aggregate`,
3808 // and converting these four first was a miss that cost a full
3809 // round — every test stayed green and the number did not move.
3810 let agg = aggregate::run(
3811 stmt,
3812 crate::join::AggRows::Owned(&filtered),
3813 &schema_cols,
3814 Some(&alias),
3815 Some(&agg_correlated),
3816 self.parallel_runner.0.as_deref(),
3817 Some(self.active_catalog()),
3818 Some(self),
3819 )?;
3820 return self.finish_agg_result(agg, stmt, cancel);
3821 }
3822 // Projection.
3823 let projection =
3824 build_projection(&stmt.items, &schema_cols, &alias, self.backslash_escapes)?;
3825 let mut projected_rows: alloc::vec::Vec<Row<'static>> =
3826 alloc::vec::Vec::with_capacity(filtered.len());
3827 // v7.19 P5 — Set-Returning-Function in projection
3828 // position (PG `SELECT unnest(arr) FROM t` shape). When a
3829 // SELECT item evaluates to a top-level unnest(arr) call,
3830 // expand it: for each input row, evaluate the array, emit
3831 // one output row per element, broadcasting non-SRF
3832 // projections from the same input row. Multi-SRF + LCM
3833 // padding stays a documented carve-out; mailrs uses
3834 // single-SRF for redirect_uris.
3835 // v7.39 (read01 round 67) — EVERY set-returning item expands, in lockstep
3836 // (see `expand_srf_row`); a user `RETURNS SETOF` function counts too.
3837 let srf_idxs = self.srf_target_idxs(&projection);
3838 // v7.39 (round 621) — which input row each output row came from. An
3839 // SRF turns one input row into many, and the ORDER BY below used to
3840 // index the EXPANDED rows by the INPUT row's position: the result was
3841 // silently truncated to the input row count and left unsorted, so
3842 // `SELECT unnest(ARRAY[1,2]), y FROM unnest(ARRAY[5,6,7]) y ORDER BY 1`
3843 // answered three of its six rows, in no order. Without the ORDER BY
3844 // the same query was already right.
3845 let mut src_of_row: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
3846 if !srf_idxs.is_empty() {
3847 let (rows, src) =
3848 expand_projection_srfs(self, &projection, &srf_idxs, &filtered, &scan_ctx)?;
3849 projected_rows = rows;
3850 src_of_row = src;
3851 } else {
3852 // v7.24 (round-16 B) — select-list subqueries resolve
3853 // per row (correlated-aware; plain exprs take the fast
3854 // path inside).
3855 let mut proj_memo = memoize::MemoizeCache::default();
3856 for row in &filtered {
3857 let mut vals = alloc::vec::Vec::with_capacity(projection.len());
3858 for p in &projection {
3859 vals.push(self.eval_expr_with_correlated(
3860 &p.expr,
3861 row,
3862 &scan_ctx,
3863 cancel,
3864 Some(&mut proj_memo),
3865 )?);
3866 }
3867 projected_rows.push(Row::new(vals));
3868 }
3869 }
3870 // ORDER BY / LIMIT — apply on the projected rows (cheap;
3871 // unnest result sets are small by design).
3872 let columns: alloc::vec::Vec<ColumnSchema> = projection
3873 .iter()
3874 // v7.39 (read01 round 54) — keep the column's enum identity through
3875 // the projection (it lives outside the DataType lattice), or a
3876 // derived table / UNION / windowed result forgets it and any outer
3877 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
3878 .map(|p| {
3879 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
3880 c.user_enum_type = p.user_enum_type.clone();
3881 c.mysql_fsp = p.mysql_fsp;
3882 c
3883 })
3884 .collect();
3885 // Re-evaluate ORDER BY against the source schema (pre-projection
3886 // so col refs by name still resolve through `scan_ctx`).
3887 // v7.39 (read01 round 80) — a positional key means the Nth OUTPUT
3888 // column. Evaluated as an expression it is just the constant N: the same
3889 // key for every row, so the sort ran and changed nothing.
3890 let order_by = resolve_positional_order_by(&stmt.order_by, &projection);
3891 if !order_by.is_empty() {
3892 // v7.39 (round 621) — one entry per OUTPUT row, not per input row.
3893 // A key that names a select-list item reads it out of the expanded
3894 // row (PG sorts AFTER the expansion); one that names a source
3895 // column the query does not project is evaluated on the input row
3896 // it came from, which is what `srf_order_output_cols` decides.
3897 let out_cols = if srf_idxs.is_empty() {
3898 alloc::vec![None; order_by.len()]
3899 } else {
3900 srf_order_output_cols(&order_by, &projection)
3901 };
3902 let mut indexed: alloc::vec::Vec<(usize, Vec<Value<'static>>)> = projected_rows
3903 .iter()
3904 .enumerate()
3905 .map(|(k, out)| -> Result<_, EngineError> {
3906 let src = src_of_row.get(k).copied().unwrap_or(k);
3907 let keys: Result<Vec<Value<'static>>, EngineError> = order_by
3908 .iter()
3909 .zip(out_cols.iter())
3910 .map(|(ob, oc)| srf_order_key(ob, *oc, out, &filtered[src], &scan_ctx))
3911 .collect();
3912 Ok((k, keys?))
3913 })
3914 .collect::<Result<_, _>>()?;
3915 indexed.sort_by(|a, b| {
3916 for (idx, (ka, kb)) in a.1.iter().zip(b.1.iter()).enumerate() {
3917 let o = &order_by[idx];
3918 let cmp = order_by_value_cmp_in(
3919 o.desc,
3920 o.nulls_first,
3921 ka,
3922 kb,
3923 scan_ctx.mysql_dialect && !crate::eval::is_binary_coerced(&o.expr),
3924 );
3925 if cmp != core::cmp::Ordering::Equal {
3926 return cmp;
3927 }
3928 }
3929 core::cmp::Ordering::Equal
3930 });
3931 projected_rows = indexed
3932 .into_iter()
3933 .map(|(i, _)| projected_rows[i].clone())
3934 .collect();
3935 }
3936 // v7.38 (read01) — DISTINCT over a synthetic source was dropped here.
3937 if stmt.distinct {
3938 projected_rows = dedup_rows(projected_rows, scan_ctx.mysql_dialect);
3939 }
3940 // LIMIT / OFFSET — apply at the tail.
3941 if let Some(offset) = stmt.offset_literal() {
3942 let off = (offset as usize).min(projected_rows.len());
3943 projected_rows.drain(..off);
3944 }
3945 if let Some(limit) = stmt.limit_literal() {
3946 projected_rows.truncate(limit as usize);
3947 }
3948 Ok(QueryResult::Rows {
3949 columns,
3950 rows: projected_rows,
3951 })
3952 }
3953
3954 /// v7.17.0 Phase 3.10 — `FROM generate_series(start, stop [,
3955 /// step])` set-returning source. Mirrors `exec_select_unnest`'s
3956 /// shape: evaluate the arg list once against an empty row,
3957 /// materialise the row stream by stepping start → stop, then
3958 /// route through the standard WHERE / projection / ORDER BY /
3959 /// LIMIT pipeline. Two arg-type combos in v7.17:
3960 /// * integer / integer [/ integer] — SmallInt, Int, BigInt
3961 /// (widened to BigInt internally; step defaults to 1)
3962 /// * timestamp / timestamp / interval — date-range
3963 /// iteration (mailrs's daily-report pattern)
3964 fn exec_select_generate_series(
3965 &self,
3966 stmt: &SelectStatement,
3967 primary: &TableRef,
3968 cancel: CancelToken<'_>,
3969 ) -> Result<QueryResult, EngineError> {
3970 let args = primary
3971 .generate_series_args
3972 .as_ref()
3973 .expect("caller guards generate_series_args.is_some()");
3974 let (elem_dtype, rows) = generate_series_rows(args, &cancel)?;
3975 let alias = primary
3976 .alias
3977 .clone()
3978 .unwrap_or_else(|| "generate_series".to_string());
3979 // `AS t(n)` — the first column-alias entry renames the
3980 // series column (PG semantics); bare alias keeps the
3981 // pre-existing behaviour of naming the column after it.
3982 let col_name = primary
3983 .unnest_column_aliases
3984 .first()
3985 .cloned()
3986 .unwrap_or_else(|| alias.clone());
3987 let col_schema = ColumnSchema::new(col_name, elem_dtype, true);
3988 let mut schema_cols = alloc::vec![col_schema.clone()];
3989 // WITH ORDINALITY — trailing BIGINT counting rows from 1;
3990 // the second column-alias entry renames it.
3991 let rows = if primary.with_ordinality {
3992 let ord_name = primary
3993 .unnest_column_aliases
3994 .get(1)
3995 .cloned()
3996 .unwrap_or_else(|| "ordinality".to_string());
3997 schema_cols.push(ColumnSchema::new(ord_name, DataType::BigInt, false));
3998 rows.into_iter()
3999 .enumerate()
4000 .map(|(i, row)| {
4001 let mut vals = row.values.clone();
4002 vals.push(Value::BigInt(i as i64 + 1));
4003 Row::new(vals)
4004 })
4005 .collect()
4006 } else {
4007 rows
4008 };
4009 // v7.39 (read01 round 54) — `ev_ctx` threads the catalog; a bare
4010 // `EvalContext::new` drops it and every catalog-dependent cast
4011 // (regclass / enum / composite / domain) silently degrades.
4012 let scan_ctx = self.ev_ctx(&schema_cols, Some(&alias));
4013 // WHERE.
4014 let filtered: alloc::vec::Vec<Row<'static>> = if let Some(w) = &stmt.where_ {
4015 let mut out = alloc::vec::Vec::with_capacity(rows.len());
4016 for row in rows {
4017 cancel.check()?;
4018 let v = eval::eval_expr(w, &row, &scan_ctx).map_err(EngineError::Eval)?;
4019 if matches!(v, Value::Bool(true)) {
4020 out.push(row);
4021 }
4022 }
4023 out
4024 } else {
4025 rows
4026 };
4027 // v7.17.0 Phase 3.P0-48 — aggregate dispatch for set-
4028 // returning sources. When the SELECT projection contains
4029 // aggregate functions (COUNT/SUM/MIN/MAX/AVG/string_agg/
4030 // …) we route the filtered row stream through the same
4031 // aggregate executor the relational scan path uses, so
4032 // `SELECT COUNT(*) FROM generate_series(1, 100)` returns
4033 // a single 100 row instead of erroring at projection
4034 // time. GROUP BY / HAVING / ORDER BY over the aggregate
4035 // output all ride through `aggregate::run`.
4036 if aggregate::uses_aggregate(stmt) {
4037 // v7.29 — a per-query memo so correlated scalar
4038 // subqueries batch-evaluate once (group map) instead of
4039 // executing per group.
4040 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
4041 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
4042 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
4043 .map_err(|err| match err {
4044 EngineError::Eval(ev) => ev,
4045 other => eval::EvalError::TypeMismatch {
4046 detail: alloc::format!("{other}"),
4047 },
4048 })
4049 };
4050 // v7.39 (round 656) — hand the rows over as they are rather than
4051 // collecting a second vector of `RowRef` wrappers. Note this is
4052 // a set-returning-function path, NOT the relational scan: the
4053 // measured O(rows) cost lived in `run_single_table_aggregate`,
4054 // and converting these four first was a miss that cost a full
4055 // round — every test stayed green and the number did not move.
4056 let agg = aggregate::run(
4057 stmt,
4058 crate::join::AggRows::Owned(&filtered),
4059 &schema_cols,
4060 Some(&alias),
4061 Some(&agg_correlated),
4062 self.parallel_runner.0.as_deref(),
4063 Some(self.active_catalog()),
4064 Some(self),
4065 )?;
4066 return self.finish_agg_result(agg, stmt, cancel);
4067 }
4068 // Projection.
4069 let projection =
4070 build_projection(&stmt.items, &schema_cols, &alias, self.backslash_escapes)?;
4071 // v7.39 (round 621) — and here, for the same reason.
4072 let srf_idxs = self.srf_target_idxs(&projection);
4073 let mut src_of_row: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
4074 let mut projected_rows: alloc::vec::Vec<Row<'static>> =
4075 alloc::vec::Vec::with_capacity(filtered.len());
4076 let mut proj_memo = memoize::MemoizeCache::default();
4077 if !srf_idxs.is_empty() {
4078 let (rows, src) =
4079 expand_projection_srfs(self, &projection, &srf_idxs, &filtered, &scan_ctx)?;
4080 projected_rows = rows;
4081 src_of_row = src;
4082 } else {
4083 for row in &filtered {
4084 let mut vals = alloc::vec::Vec::with_capacity(projection.len());
4085 for p in &projection {
4086 // v7.24 (round-16 B) — correlated-aware.
4087 vals.push(self.eval_expr_with_correlated(
4088 &p.expr,
4089 row,
4090 &scan_ctx,
4091 cancel,
4092 Some(&mut proj_memo),
4093 )?);
4094 }
4095 projected_rows.push(Row::new(vals));
4096 }
4097 }
4098 let columns: alloc::vec::Vec<ColumnSchema> = projection
4099 .iter()
4100 // v7.39 (read01 round 54) — keep the column's enum identity through
4101 // the projection (it lives outside the DataType lattice), or a
4102 // derived table / UNION / windowed result forgets it and any outer
4103 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
4104 .map(|p| {
4105 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
4106 c.user_enum_type = p.user_enum_type.clone();
4107 c.mysql_fsp = p.mysql_fsp;
4108 c
4109 })
4110 .collect();
4111 // ORDER BY against the source schema.
4112 // v7.39 (round 621) — one entry per OUTPUT row (a target-list SRF makes
4113 // more of them than there were inputs), and a positional key means the
4114 // Nth OUTPUT column, which is what `resolve_positional_order_by` does
4115 // and what the other two synthetic-source tails already did.
4116 let order_by = resolve_positional_order_by(&stmt.order_by, &projection);
4117 if !order_by.is_empty() {
4118 let out_cols = if srf_idxs.is_empty() {
4119 alloc::vec![None; order_by.len()]
4120 } else {
4121 srf_order_output_cols(&order_by, &projection)
4122 };
4123 let mut indexed: alloc::vec::Vec<(usize, Vec<Value<'static>>)> = projected_rows
4124 .iter()
4125 .enumerate()
4126 .map(|(k, out)| -> Result<_, EngineError> {
4127 let r = &filtered[src_of_row.get(k).copied().unwrap_or(k)];
4128 let keys: Result<Vec<Value<'static>>, EngineError> = order_by
4129 .iter()
4130 .zip(out_cols.iter())
4131 .map(|(ob, oc)| srf_order_key(ob, *oc, out, r, &scan_ctx))
4132 .collect();
4133 Ok((k, keys?))
4134 })
4135 .collect::<Result<_, _>>()?;
4136 indexed.sort_by(|a, b| {
4137 for (idx, (ka, kb)) in a.1.iter().zip(b.1.iter()).enumerate() {
4138 let o = &stmt.order_by[idx];
4139 let cmp = order_by_value_cmp_in(
4140 o.desc,
4141 o.nulls_first,
4142 ka,
4143 kb,
4144 scan_ctx.mysql_dialect && !crate::eval::is_binary_coerced(&o.expr),
4145 );
4146 if cmp != core::cmp::Ordering::Equal {
4147 return cmp;
4148 }
4149 }
4150 core::cmp::Ordering::Equal
4151 });
4152 projected_rows = indexed
4153 .into_iter()
4154 .map(|(i, _)| projected_rows[i].clone())
4155 .collect();
4156 }
4157 // v7.38 (read01) — DISTINCT over a synthetic source was dropped here.
4158 if stmt.distinct {
4159 projected_rows = dedup_rows(projected_rows, scan_ctx.mysql_dialect);
4160 }
4161 if let Some(offset) = stmt.offset_literal() {
4162 let off = (offset as usize).min(projected_rows.len());
4163 projected_rows.drain(..off);
4164 }
4165 if let Some(limit) = stmt.limit_literal() {
4166 projected_rows.truncate(limit as usize);
4167 }
4168 Ok(QueryResult::Rows {
4169 columns,
4170 rows: projected_rows,
4171 })
4172 }
4173
4174 /// The FROM shapes that are not an ordinary table scan — joins, the
4175 /// set-returning sources, JSON_TABLE, a derived table, and the rest.
4176 ///
4177 /// `#[inline(never)]` and out of `exec_bare_select_cancel` for the
4178 /// reason round 848 established in the parser: a debug build gives
4179 /// EVERY branch's locals a slot in the frame, whichever branch runs.
4180 /// `exec_bare_select_cancel` measured 64,784 bytes and a nested query
4181 /// stacks several of them; a plain scan reaches none of these
4182 /// branches. Moving them out took the frame to 52,336.
4183 ///
4184 /// `Ok(None)` means "not one of these shapes, carry on".
4185 #[inline(never)]
4186 fn try_from_shape_paths(
4187 &self,
4188 stmt: &SelectStatement,
4189 from: &spg_sql::ast::FromClause,
4190 cancel: CancelToken<'_>,
4191 ) -> Result<Option<QueryResult>, EngineError> {
4192 if !from.joins.is_empty() {
4193 // v7.37.x (docker-fair LEFTJOIN 71 % attack) — LEFT JOIN
4194 // elimination: when a LEFT JOIN's right side is referenced
4195 // ONLY in the ON equality and the right-side join key is
4196 // UNIQUE/PK, the join preserves outer cardinality exactly
4197 // and contributes no values used downstream. Drop the
4198 // entire join. PG does this on the
4199 // `SELECT COUNT(*) FROM A LEFT JOIN B ON B.pk = A.fk` shape
4200 // — A's row count is what survives, B never has to be
4201 // touched.
4202 if let Some(eliminated) = self.try_eliminate_redundant_left_joins(stmt) {
4203 return self.exec_bare_select_cancel(&eliminated, cancel).map(Some);
4204 }
4205 // v7.38 P0 元机制 D — `SPG_TEST_DISABLE_JOINFOLD=1` skips
4206 // the v7.32 joinfold rewrite that turns inner JOINs into a
4207 // single-table scan when the catalogue can prove key-only
4208 // dependency. Tests use this to assert "without joinfold,
4209 // the join still executes correctly" (joinfold is a
4210 // semantically-equivalent rewrite, not a correctness fix).
4211 if !self.env_cfg().disable_joinfold {
4212 if let Some(folded) = self.try_fold_inner_joins(stmt, cancel)? {
4213 return self.exec_bare_select_cancel(&folded, cancel).map(Some);
4214 }
4215 }
4216 return self.exec_joined_select(stmt, from, cancel).map(Some);
4217 }
4218 // v7.11.7 — `FROM unnest(<expr>) [AS] <alias>`. Synthesise a
4219 // single-column table at SELECT entry by evaluating the
4220 // expression once against the empty row (UNNEST is
4221 // uncorrelated in v7.11; correlated / LATERAL unnest is a
4222 // v7.12 carve-out). Build a virtual `Table` in a heap-only
4223 // catalog, then route to the regular scan path.
4224 if from.primary.unnest_expr.is_some() {
4225 return self
4226 .exec_select_unnest(stmt, &from.primary, cancel)
4227 .map(Some);
4228 }
4229 // v7.37.43-T4.5 — `FROM jsonb_each_text(<expr>)` set-
4230 // returning function. Same dispatch shape as unnest but
4231 // emits a two-column (key TEXT, value TEXT) row stream.
4232 if from.primary.jsonb_each_text_arg.is_some() {
4233 return self
4234 .exec_select_jsonb_each_text(stmt, &from.primary, cancel)
4235 .map(Some);
4236 }
4237 // v7.39 (read01 partitionfuncs.c) — FROM-position table functions
4238 // (pg_partition_tree / pg_partition_ancestors) dispatched by name.
4239 // v7.39 (read01 round 74) — `ROWS FROM (f(a), g(b))` whose entries have no
4240 // array form. Each function runs; the results zip in LOCKSTEP with the
4241 // shorter padded to NULL — the SAME rule the target-list SRFs follow
4242 // (round 67), which is why `srf_values` is what evaluates each entry.
4243 if from.primary.rows_from.is_some() {
4244 let (rows, mut schema_cols) = self.rows_from_rows(&from.primary)?;
4245 for (i, new_name) in from.primary.unnest_column_aliases.iter().enumerate() {
4246 if let Some(col) = schema_cols.get_mut(i) {
4247 col.name = new_name.clone();
4248 }
4249 }
4250 let alias = from
4251 .primary
4252 .alias
4253 .clone()
4254 .unwrap_or_else(|| from.primary.name.clone());
4255 return self
4256 .exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
4257 .map(Some);
4258 }
4259 // v7.39 (round 205, JSON_TABLE) — `FROM JSON_TABLE(doc, '$p'
4260 // COLUMNS (...))`. Materialise the row stream + schema by
4261 // walking the row path, then run the regular pipeline over it.
4262 if let Some(jt) = &from.primary.json_table {
4263 let (rows, schema_cols) = self.json_table_rows(jt, None)?;
4264 let alias = from
4265 .primary
4266 .alias
4267 .clone()
4268 .unwrap_or_else(|| from.primary.name.clone());
4269 return self
4270 .exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
4271 .map(Some);
4272 }
4273 if from.primary.table_fn_call.is_some() {
4274 let (rows, mut schema_cols) = self.table_fn_rows(&from.primary)?;
4275 // v7.39 (read01 round 68) — WITH ORDINALITY appends a BIGINT counter
4276 // (from 1, in output order) AFTER the function's own columns. The
4277 // alias list names it like any other, which is why it is appended
4278 // BEFORE the renaming pass below.
4279 let rows = if from.primary.with_ordinality {
4280 schema_cols.push(ColumnSchema::new(
4281 "ordinality".to_string(),
4282 DataType::BigInt,
4283 false,
4284 ));
4285 rows.into_iter()
4286 .enumerate()
4287 .map(|(i, r)| {
4288 let mut vals = r.values;
4289 vals.push(Value::BigInt(i as i64 + 1));
4290 Row::new(vals)
4291 })
4292 .collect()
4293 } else {
4294 rows
4295 };
4296 for (i, new_name) in from.primary.unnest_column_aliases.iter().enumerate() {
4297 if let Some(col) = schema_cols.get_mut(i) {
4298 col.name = new_name.clone();
4299 }
4300 }
4301 let alias = from
4302 .primary
4303 .alias
4304 .clone()
4305 .unwrap_or_else(|| from.primary.name.clone());
4306 return self
4307 .exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
4308 .map(Some);
4309 }
4310 // v7.37.17 (17.6 siblings) — plain derived table in primary
4311 // position: `FROM ( SELECT … ) alias` (no joins). The inner
4312 // SELECT materialises once (it is uncorrelated by
4313 // construction), then the outer projection / WHERE /
4314 // aggregate / ORDER BY pipeline runs over the synthetic
4315 // table. Joined derived tables keep riding the LATERAL
4316 // machinery in join.rs.
4317 if from.joins.is_empty() && from.primary.lateral_subquery.is_some() {
4318 // v7.39 (round 727) — flatten first. A simple derived table
4319 // (bare-column projection over one stored table, nothing that
4320 // changes cardinality or order) used to force the inner
4321 // SELECT through the SERIAL row-at-a-time projection pipeline
4322 // just to materialise a synthetic table the outer query then
4323 // re-scans: `count(*) FROM (SELECT id v FROM d WHERE …) q`
4324 // measured 18.6 ms against PG's 5 — and bare count over the
4325 // same filter WITHOUT the wrapper is 2 ms here, because it
4326 // rides the fused parallel lane. Rewriting to the unwrapped
4327 // form is PG's subquery pull-up; the whole tree gets the
4328 // fast lanes back.
4329 if let Some(flat) = try_flatten_derived(stmt, &from.primary) {
4330 return self.exec_select_cancel(&flat, cancel).map(Some);
4331 }
4332 // v7.39 (round 742) — `SELECT count(*) FROM (SELECT … ORDER
4333 // BY … OFFSET k) q` is `greatest(count_of_inner - k, 0)`:
4334 // ORDER BY never changes the row count, and OFFSET drops
4335 // exactly k. The materialising path sorted 500k rows to
4336 // count 10k (57 ms); PG runs its parallel sort anyway
4337 // (28 ms). The rewrite skips the sort entirely on both
4338 // counts — a plan PG itself does not have.
4339 if let Some(rewritten) = try_count_over_offset(stmt, &from.primary) {
4340 return self.exec_select_cancel(&rewritten, cancel).map(Some);
4341 }
4342 // v7.39 (round 743) — `count(*) OVER a derived whose only
4343 // item is unnest(ARRAY[k elements])` is `k * count(WHERE)`:
4344 // a constant-length array unnests to exactly k rows per
4345 // input row, NULL elements included. PG expands the set to
4346 // count it (6.6 ms on the panel cell); the identity doesn't.
4347 if let Some(rewritten) = try_count_over_const_unnest(stmt, &from.primary) {
4348 return self.exec_select_cancel(&rewritten, cancel).map(Some);
4349 }
4350 return self
4351 .exec_select_derived(stmt, &from.primary, cancel)
4352 .map(Some);
4353 }
4354 // v7.17.0 Phase 3.10 — `FROM generate_series(start, stop
4355 // [, step])` set-returning source. Dispatch mirrors UNNEST:
4356 // materialise the row stream from a single eval pass, then
4357 // run the regular projection / WHERE / ORDER BY / LIMIT
4358 // pipeline over the synthetic single-column table.
4359 if from.primary.generate_series_args.is_some() {
4360 return self
4361 .exec_select_generate_series(stmt, &from.primary, cancel)
4362 .map(Some);
4363 }
4364 Ok(None)
4365 }
4366
4367 /// Pick an index seek for this WHERE, if any of the four apply:
4368 /// BTree equality, GIN `@@`, trigram LIKE, or JSONB `@>`.
4369 ///
4370 /// `#[inline(never)]` and out of `exec_bare_select_cancel` for the
4371 /// frame reason on `try_from_shape_paths`: in a debug build a
4372 /// closure's locals belong to the enclosing frame, and this one is
4373 /// four seek attempts wide on a function that nests.
4374 #[inline(never)]
4375 fn pick_indexed_rows<'r>(
4376 &'r self,
4377 stmt: &SelectStatement,
4378 table: &'r spg_storage::Table,
4379 schema_cols: &[spg_storage::ColumnSchema],
4380 alias: &str,
4381 ctx: &crate::eval::EvalContext<'_>,
4382 seek_snapshot: &crate::Snapshot,
4383 ) -> Option<Vec<Cow<'r, Row<'static>>>> {
4384 stmt.where_.as_ref().and_then(|w| {
4385 // BTree / col=literal seek first — covers the v7.11.3 multi-
4386 // column AND case and the leading-column equality lookup.
4387 try_index_seek(
4388 w,
4389 schema_cols,
4390 self.active_catalog(),
4391 table,
4392 alias,
4393 seek_snapshot,
4394 )
4395 .or_else(|| {
4396 // v7.12.3 — GIN-accelerated `WHERE col @@
4397 // tsquery` when the column has a `USING gin`
4398 // index. Returns an over-approximate candidate
4399 // set; the WHERE re-eval loop below verifies
4400 // the full `@@` predicate per row.
4401 try_gin_seek(
4402 w,
4403 schema_cols,
4404 self.active_catalog(),
4405 table,
4406 alias,
4407 ctx,
4408 seek_snapshot,
4409 )
4410 })
4411 .or_else(|| {
4412 // v7.15.0 — trigram-GIN-accelerated
4413 // `WHERE col LIKE / ILIKE '<pat>'` when the
4414 // column has a `gin_trgm_ops` GIN index.
4415 // Over-approximate candidate set; the WHERE
4416 // re-eval verifies the LIKE per row.
4417 try_trgm_seek(w, schema_cols, table, alias, seek_snapshot)
4418 })
4419 .or_else(|| {
4420 // v7.37.8(sentori Epic 5 P2)— real JSONB-GIN
4421 // accelerated `WHERE col @> <jsonb_literal>`
4422 // when the column has a `USING gin` index. The
4423 // posting-list intersection returns an over-
4424 // approximate candidate set; the WHERE re-eval
4425 // verifies the full `@>` predicate per row.
4426 try_gin_jsonb_seek(w, schema_cols, table, alias, seek_snapshot)
4427 })
4428 })
4429 }
4430
4431 /// Index-seek fast paths: NSW kNN, the primary-key top-N walk, and
4432 /// the two `count(*)` short-circuits. Out-of-line for the frame
4433 /// reason on `try_from_shape_paths` — an ordinary scan reaches none
4434 /// of them, and in a debug build their locals sit in the frame
4435 /// regardless.
4436 #[inline(never)]
4437 fn try_seek_fast_paths(
4438 &self,
4439 stmt: &SelectStatement,
4440 table: &spg_storage::Table,
4441 schema_cols: &[spg_storage::ColumnSchema],
4442 alias: &str,
4443 seek_snapshot: &crate::Snapshot,
4444 cancel: CancelToken<'_>,
4445 ) -> Result<Option<QueryResult>, EngineError> {
4446 if let Some(nsw_rows) = try_nsw_knn(stmt, table, schema_cols, alias, seek_snapshot) {
4447 // NSW kNN dispatches against the hot-tier vector index only
4448 // (vector cells aren't promoted to cold segments), so wrap
4449 // the returned row indices as `Cow::Borrowed` for the
4450 // unified `materialise_in_order` shape.
4451 let ordered: Vec<Cow<'_, Row<'static>>> = nsw_rows
4452 .into_iter()
4453 .filter_map(|i| table.rows().get(i).map(Cow::Borrowed))
4454 .collect();
4455 return materialise_in_order(
4456 stmt,
4457 schema_cols,
4458 alias,
4459 &ordered,
4460 self.backslash_escapes,
4461 )
4462 .map(Some);
4463 }
4464
4465 // v7.34.5 — ORDER BY <indexed col> [DESC|ASC] LIMIT N drives
4466 // the scan via the BTree iterator in the requested direction
4467 // and stops after `OFFSET + LIMIT` candidates pass WHERE. The
4468 // 80 ms `mailrs_prod_plain_limit` baseline at 250 k rows is
4469 // the load-bearing consumer; this skips the materialise-every-
4470 // row + partial-sort tail entirely. Walker output is already
4471 // in ORDER BY order so `materialise_in_order` (no extra sort)
4472 // is the natural sink.
4473 if let Some(walked) = try_pk_walk_top_n(
4474 stmt,
4475 self.active_catalog(),
4476 table,
4477 schema_cols,
4478 alias,
4479 self,
4480 cancel,
4481 ) {
4482 return materialise_in_order(stmt, schema_cols, alias, &walked, self.backslash_escapes)
4483 .map(Some);
4484 }
4485
4486 // Index seek: if WHERE is `col = literal` (or commuted) and the
4487 // referenced column has an index, dispatch each locator through
4488 // the catalog (hot tier → borrow, cold tier → page-read +
4489 // decode) and iterate just those rows. Otherwise fall back to a
4490 // v7.37.x (docker-fair INSUBQ attack) — short-circuit COUNT(*)
4491 // FROM A WHERE A.pk IN (large literal list). The post-subquery-
4492 // replacement shape of INSUBQ. Runs BEFORE `indexed_rows` so
4493 // we don't pay the row materialisation cost twice. Returns
4494 // a bare `Rows{count}` if the shape matches.
4495 if aggregate::uses_aggregate(stmt)
4496 && let Some(out) = self.try_count_star_pk_in_list_fast(stmt, table, schema_cols, alias)
4497 {
4498 return Ok(Some(out));
4499 }
4500 // v7.38 (perf) — `count(*) WHERE <indexed BETWEEN>`: count the in-range
4501 // locators directly, skipping row materialisation + WHERE re-eval.
4502 if aggregate::uses_aggregate(stmt)
4503 && let Some(out) = self.try_count_star_indexed_range_fast(
4504 stmt,
4505 table,
4506 schema_cols,
4507 alias,
4508 seek_snapshot,
4509 )
4510 {
4511 return Ok(Some(out));
4512 }
4513 Ok(None)
4514 }
4515
4516 /// The two rewrites that must happen before the FROM clause is even
4517 /// looked at: a meta-view reference needs the catalog views
4518 /// materialised, and a windowed projection belongs to the window
4519 /// executor. Out-of-line for the frame reason on
4520 /// `try_from_shape_paths`.
4521 #[inline(never)]
4522 fn try_pre_from_paths(
4523 &self,
4524 stmt: &SelectStatement,
4525 cancel: CancelToken<'_>,
4526 ) -> Result<Option<QueryResult>, EngineError> {
4527 if !self.meta_views_materialised && select_references_meta_view(stmt) {
4528 return self.exec_select_with_meta_views(stmt, cancel).map(Some);
4529 }
4530 // v4.12: window-function path. When the projection contains
4531 // any `name(args) OVER (...)` we route to the dedicated
4532 // executor — partition + sort + per-row window value before
4533 // the regular projection.
4534 if select_has_window(stmt) {
4535 // v7.37 D.23 — window functions run AFTER GROUP BY aggregation.
4536 // `SELECT g, sum(v), rank() OVER (ORDER BY sum(v)) FROM t GROUP BY g`
4537 // needs the aggregation done first, then windows over the grouped
4538 // rows. Rewrite to an aggregate derived subquery + outer window query
4539 // (which the window-over-derived path, D.13, executes). Only fires on
4540 // the currently-erroring agg+window+GROUP BY shape, so it can't
4541 // regress working window-only or aggregate-only queries.
4542 if let Some(rewritten) = rewrite_agg_before_window(stmt) {
4543 return self.exec_select_cancel(&rewritten, cancel).map(Some);
4544 }
4545 return self.exec_select_with_window(stmt, cancel).map(Some);
4546 }
4547 Ok(None)
4548 }
4549
4550 /// A projection naming `ctid` or another system column: the schema
4551 /// has to be widened with them before the scan. Out-of-line for the
4552 /// frame reason on `try_from_shape_paths`.
4553 #[inline(never)]
4554 fn try_ctid_projection(
4555 &self,
4556 stmt: &SelectStatement,
4557 primary: &spg_sql::ast::TableRef,
4558 table: &spg_storage::Table,
4559 schema_cols: &[spg_storage::ColumnSchema],
4560 alias: &str,
4561 cancel: CancelToken<'_>,
4562 ) -> Result<Option<QueryResult>, EngineError> {
4563 if references_ctid(stmt) {
4564 let snapshot = self.current_snapshot();
4565 let mut ext_cols = schema_cols.to_vec();
4566 for name in SYSTEM_COLUMNS {
4567 ext_cols.push(ColumnSchema::new(name.to_string(), DataType::Text, false));
4568 }
4569 let table_oid =
4570 crate::system_catalog::relation_oid(self.active_catalog(), &primary.name)
4571 .unwrap_or(0);
4572 let headers = table.headers();
4573 let rows: Vec<Row<'static>> = table
4574 .scan_visible(&snapshot)
4575 .map(|(i, r)| {
4576 let mut vals = r.values.clone();
4577 // One block, offsets from 1, as PG numbers them.
4578 vals.push(Value::Tid(0, i as u32 + 1));
4579 let h = headers.get(i);
4580 vals.push(Value::Xid(h.map_or(0, |h| h.xmin as u32)));
4581 vals.push(Value::Xid(h.map_or(0, |h| h.xmax as u32)));
4582 // SPG keeps no per-statement command ids; PG shows 0 for
4583 // every row a reader can see, which is every row here.
4584 vals.push(Value::Cid(0));
4585 vals.push(Value::Cid(0));
4586 vals.push(Value::BigInt(table_oid));
4587 Row::new(vals)
4588 })
4589 .collect();
4590 return self
4591 .exec_select_over_rows(stmt, rows, ext_cols, alias, cancel)
4592 .map(Some);
4593 }
4594 Ok(None)
4595 }
4596
4597 /// A sequence read as a one-row relation (`SELECT last_value FROM
4598 /// seq`), which PG allows and psql's \\d relies on. Out-of-line for
4599 /// the frame reason on `try_from_shape_paths`.
4600 #[inline(never)]
4601 fn try_sequence_relation(
4602 &self,
4603 stmt: &SelectStatement,
4604 primary: &spg_sql::ast::TableRef,
4605 cancel: CancelToken<'_>,
4606 ) -> Result<Option<QueryResult>, EngineError> {
4607 if self.active_catalog().get(&primary.name).is_none()
4608 && let Some(seq) = self.active_catalog().sequence(&primary.name)
4609 {
4610 let rows = alloc::vec![Row::new(alloc::vec![
4611 Value::BigInt(seq.last_value),
4612 Value::BigInt(0),
4613 Value::Bool(seq.is_called),
4614 ])];
4615 let schema_cols = alloc::vec![
4616 ColumnSchema::new("last_value", DataType::BigInt, false),
4617 ColumnSchema::new("log_cnt", DataType::BigInt, false),
4618 ColumnSchema::new("is_called", DataType::Bool, false),
4619 ];
4620 let alias = primary
4621 .alias
4622 .clone()
4623 .unwrap_or_else(|| primary.name.clone());
4624 return self
4625 .exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
4626 .map(Some);
4627 }
4628 Ok(None)
4629 }
4630
4631 pub(crate) fn exec_bare_select_cancel(
4632 &self,
4633 stmt: &SelectStatement,
4634 cancel: CancelToken<'_>,
4635 ) -> Result<QueryResult, EngineError> {
4636 // v7.17.0 Phase 3.P0-49 — `FETCH FIRST N ROWS WITH TIES`
4637 // is meaningless without an ORDER BY; PG raises a hard
4638 // error and SPG mirrors the surface so the same DDL/app
4639 // path behaves identically on cutover.
4640 check_with_ties_requires_order_by(stmt)?;
4641 // v7.39 (round 229) — WHERE / HAVING run before the window pass, so
4642 // PG rejects window calls there outright. Checked here rather than
4643 // on the window path: `HAVING row_number() OVER () = 1` has no
4644 // window in its projection at all.
4645 crate::window::reject_window_in_row_clauses(stmt)?;
4646 // v7.39 (round 232) — the ORDER BY legality rules (positional
4647 // bounds, DISTINCT, DISTINCT ON). Same placement as the window
4648 // check: before anything scans.
4649 crate::orderby::check_order_by_legality(stmt)?;
4650 // v7.37.16 — resolve `USING` column-merge + `NATURAL JOIN` into an
4651 // equivalent statement the regular executor handles (merged join
4652 // columns collapse to a single unqualified output column; NATURAL
4653 // gets its common-column ON synthesised). The rewrite clears the
4654 // flags, so this re-entrant call is a no-op on the second pass.
4655 if let Some(rewritten) = self.desugar_using_natural(stmt)? {
4656 return self.exec_bare_select_cancel(&rewritten, cancel);
4657 }
4658 // v7.39 (RLS) Phase 3 — cross-table joins: wrap each RLS-enabled join
4659 // operand in a security-barrier subquery, then re-enter (the wrapped
4660 // operands are no longer bare RLS tables, so this is a no-op on the
4661 // second pass).
4662 if let Some(rewritten) = self.rls_rewrite_joins(stmt) {
4663 return self.exec_bare_select_cancel(&rewritten, cancel);
4664 }
4665 // v7.39 (RLS) Phase 1 — for a policy-subject (non-superuser) session,
4666 // AND the RLS USING predicate into a single-table SELECT's WHERE.
4667 // Superuser sessions and non-RLS tables get `None` (no clone, no
4668 // change). Applied inline (shadowing `stmt`) rather than via re-entry
4669 // so it can't re-inject on a recursive pass.
4670 let rls_stmt;
4671 let stmt = match self.rls_select_predicate(stmt)? {
4672 Some(pred) => {
4673 let mut s = stmt.clone();
4674 s.where_ = Some(match s.where_.take() {
4675 Some(existing) => spg_sql::ast::Expr::Binary {
4676 lhs: alloc::boxed::Box::new(existing),
4677 op: spg_sql::ast::BinOp::And,
4678 rhs: alloc::boxed::Box::new(pred),
4679 },
4680 None => pred,
4681 });
4682 rls_stmt = s;
4683 &rls_stmt
4684 }
4685 None => stmt,
4686 };
4687 // v7.16.2 — same meta-view dispatch as
4688 // `exec_select_cancel`, applied here too because
4689 // `subquery_replacement` enters this function directly
4690 // for Exists / ScalarSubquery / InSubquery resolution
4691 // (bypassing the top-level entry to avoid double
4692 // subquery walking). Without this dispatch the subquery
4693 // hits `__spg_info_columns` and reports TableNotFound.
4694 if let Some(done) = self.try_pre_from_paths(stmt, cancel)? {
4695 return Ok(done);
4696 }
4697 // Constant SELECT (no FROM) — evaluate each item once against an
4698 // empty dummy row. Useful for `SELECT 1`, `SELECT coalesce(...)`,
4699 // `SELECT '7'::INT`. Column references will surface as
4700 // ColumnNotFound on eval since the schema is empty.
4701 let Some(from) = &stmt.from else {
4702 return self.exec_constant_select(stmt);
4703 };
4704 // Multi-table FROM (one or more joined peers) goes through the
4705 // nested-loop join executor. Single-table FROM stays on the
4706 // existing scan + index-seek path.
4707 if let Some(done) = self.try_from_shape_paths(stmt, from, cancel)? {
4708 return Ok(done);
4709 }
4710 // NOT hooked up. `try_spill_sorted_scan` is written, correct and
4711 // tested — eight ORDER BY shapes byte-identical spilled against
4712 // in-memory, with 103 runs opened to prove the spill ran — and it
4713 // loses on wall clock, which is a hard stop whatever the memory
4714 // buys. Measured round 865, same psql client both sides, same
4715 // machine, row counts verified, and both sides confirmed to be
4716 // doing an external merge rather than an indexed walk:
4717 //
4718 // PG18 178.7 - 187.0 ms Sort Method: external merge, 85 MB
4719 // SPG spilled 269.7 - 299.6 ms 33 spill files at peak
4720 //
4721 // Non-overlapping, about 1.55x. Re-enable by restoring the call
4722 // below once that closes; nothing else has to change, which is
4723 // the point of it being a separate path.
4724 //
4725 // if let Some(done) = self.try_spill_sorted_scan(stmt, from, cancel)? {
4726 // return Ok(done);
4727 // }
4728 //
4729 // v7.37 (round 882) — this walk stays unhooked, but its streaming
4730 // twin `try_spill_sorted_stream` IS hooked, above the ORDER BY
4731 // bail in `try_exec_joined_streaming`. Collecting the answer was
4732 // most of what this one cost: handing rows over as the merge
4733 // produces them holds peak to the budget plus one row, and the
4734 // wall clock lands inside PG18's range rather than 1.55x outside
4735 // it. Numbers in `extsort.rs`'s header.
4736 let primary = &from.primary;
4737 // v7.39 (round 244) — a sequence is selectable as a one-row relation
4738 // in PG (`SELECT last_value FROM seq` — psql's \d and several ORMs
4739 // read it). Synthesize PG's three columns.
4740 if let Some(done) = self.try_sequence_relation(stmt, primary, cancel)? {
4741 return Ok(done);
4742 }
4743 let table = self.active_catalog().get(&primary.name).ok_or_else(|| {
4744 StorageError::TableNotFound {
4745 name: primary.name.clone(),
4746 }
4747 })?;
4748 let schema_cols = &table.schema().columns;
4749 // The qualifier accepted on column refs is the alias (if any) else the
4750 // bare table name.
4751 let alias = primary.alias.as_deref().unwrap_or(primary.name.as_str());
4752 // v7.39 (round 511) — `ctid`, PG's physical row identity. SPG had no
4753 // system columns at all: `SELECT ctid FROM t` answered "column
4754 // \"ctid\" does not exist", which takes out the dedup idiom every
4755 // PG user knows — `DELETE … WHERE ctid NOT IN (SELECT min(ctid) …
4756 // GROUP BY key)`.
4757 //
4758 // The value comes from the row's position, which the scan already
4759 // yields; the column is appended to the schema and the rows only
4760 // when the statement asks for it, so nothing else pays for it. That
4761 // also routes the query down the general path, past the index fast
4762 // paths below — they hand back rows without positions, and a ctid
4763 // that was sometimes right would be worse than none.
4764 if let Some(done) =
4765 self.try_ctid_projection(stmt, primary, table, schema_cols, alias, cancel)?
4766 {
4767 return Ok(done);
4768 }
4769 let ctx = self.ev_ctx(schema_cols, Some(alias));
4770
4771 // NSW kNN planner: `ORDER BY col <-> literal LIMIT k` with no
4772 // WHERE and an NSW index on `col` skips the full scan. The
4773 // walk returns rows already in ascending-distance order, so
4774 // ORDER BY / LIMIT are honoured implicitly.
4775 // Phase C.3 step 2c — compute the reader's MVCC snapshot once
4776 // and thread it into every index-seek fast path below. No-op
4777 // today (every hot header is committed-alive).
4778 let seek_snapshot = self.current_snapshot();
4779 if let Some(done) =
4780 self.try_seek_fast_paths(stmt, table, schema_cols, alias, &seek_snapshot, cancel)?
4781 {
4782 return Ok(done);
4783 }
4784 // full scan over the hot tier (cold-tier rows are only reached
4785 // via index seek in v5.1 — full table scans against cold-tier
4786 // data ship in v5.2 with the freezer's per-segment scan API).
4787 let indexed_rows =
4788 self.pick_indexed_rows(stmt, table, schema_cols, alias, &ctx, &seek_snapshot);
4789
4790 // Aggregate path: filter rows first, then hand off to the
4791 // aggregate executor which does its own projection + ORDER BY.
4792 if aggregate::uses_aggregate(stmt) {
4793 return self.run_single_table_aggregate(
4794 stmt,
4795 table,
4796 schema_cols,
4797 alias,
4798 indexed_rows,
4799 cancel,
4800 );
4801 }
4802 self.run_single_table_scan(stmt, table, schema_cols, alias, indexed_rows, cancel)
4803 }
4804
4805 /// v7.37.43-T4.5 — execute `SELECT … FROM jsonb_each_text(<expr>)`.
4806 /// Sentori migration 0067 uses this with `CROSS JOIN LATERAL`; the
4807 /// uncorrelated FROM-primary case is the simpler shape, used by
4808 /// e2e pins. Materialises the (key, value) pair stream into a
4809 /// synthetic two-column TEXT table, then routes through the
4810 /// regular projection / WHERE / ORDER BY pipeline.
4811 /// v7.39 (read01 partitionfuncs.c) — materialise a FROM-position
4812 /// v7.39 (round 205, JSON_TABLE) — materialise a JSON_TABLE FROM
4813 /// item into (rows, schema). `outer_doc` is `Some` only when this
4814 /// is a NESTED level being expanded against a parent row item's
4815 /// already-parsed sub-document; the top-level call parses the doc
4816 /// expr itself. Row/column paths reuse the existing jsonpath
4817 /// evaluator (`json::json_table_path`); coercion reuses
4818 /// `coerce_value` on the JSON scalar text, so a json string
4819 /// coerces to DATE by its content, matching PG.
4820 #[allow(clippy::type_complexity)]
4821 pub(crate) fn json_table_rows(
4822 &self,
4823 jt: &spg_sql::ast::JsonTable,
4824 outer_doc: Option<&crate::json::JsonValue>,
4825 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
4826 // Column schema is static (independent of data): flatten the
4827 // COLUMNS tree in declaration order (NESTED contributes its
4828 // children inline, the PG output shape).
4829 let schema = json_table_schema(&jt.columns);
4830
4831 // PASSING variables → a single JsonValue object the jsonpath
4832 // engine reads `$name` from.
4833 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
4834 let ctx = EvalContext::new(&empty_schema, None);
4835 let dummy = Row::new(alloc::vec::Vec::new());
4836 let vars: Option<crate::json::JsonValue> = if jt.passing.is_empty() {
4837 None
4838 } else {
4839 let mut entries = alloc::vec::Vec::new();
4840 for (name, e) in &jt.passing {
4841 let v = eval::eval_expr(e, &dummy, &ctx).map_err(EngineError::Eval)?;
4842 entries.push((name.clone(), value_to_json_value(&v)));
4843 }
4844 Some(crate::json::JsonValue::Object(entries))
4845 };
4846
4847 // The document root: a NESTED level gets it from the parent;
4848 // the top level parses its doc expr.
4849 let root_owned;
4850 let root: &crate::json::JsonValue = match outer_doc {
4851 Some(d) => d,
4852 None => {
4853 let doc_val = eval::eval_expr(&jt.doc, &dummy, &ctx).map_err(EngineError::Eval)?;
4854 let src = match &doc_val {
4855 Value::Null => return Ok((alloc::vec::Vec::new(), schema)),
4856 Value::Json(s) | Value::Text(s) => s.as_ref().to_string(),
4857 other => {
4858 return Err(EngineError::Unsupported(alloc::format!(
4859 "JSON_TABLE document must be json/text, got {}",
4860 crate::conversions::pg_type_name_for_error_opt(other.data_type())
4861 )));
4862 }
4863 };
4864 root_owned = crate::json::parse_doc(&src).map_err(EngineError::Eval)?;
4865 &root_owned
4866 }
4867 };
4868
4869 let items = crate::json::json_table_path(root, &jt.row_path, vars.as_ref())
4870 .map_err(EngineError::Eval)?;
4871 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
4872 for (idx, item) in items.iter().enumerate() {
4873 self.json_table_emit_item(jt, item, idx, vars.as_ref(), &mut rows)?;
4874 }
4875 Ok((rows, schema))
4876 }
4877
4878 /// v7.39 (round 205) — emit the row(s) for one row-pattern item.
4879 /// Regular columns produce one value each; a NESTED column expands
4880 /// as an outer join (each nested match → one row sharing the
4881 /// parent cells; no nested match → one row with the nested cells
4882 /// NULL). Sibling NESTED at one level cross by concatenation of
4883 /// their independent expansions (PG's UNION-of-outer shape).
4884 fn json_table_emit_item(
4885 &self,
4886 jt: &spg_sql::ast::JsonTable,
4887 item: &crate::json::JsonValue,
4888 ordinality: usize,
4889 vars: Option<&crate::json::JsonValue>,
4890 out: &mut alloc::vec::Vec<Row<'static>>,
4891 ) -> Result<(), EngineError> {
4892 use spg_sql::ast::JsonTableColumn as C;
4893 // Parent cells (regular + ordinality), left-to-right; NESTED
4894 // columns contribute a run of child cells appended after.
4895 let mut parent_cells: alloc::vec::Vec<Value<'static>> = alloc::vec::Vec::new();
4896 let mut nested_runs: alloc::vec::Vec<alloc::vec::Vec<Row<'static>>> =
4897 alloc::vec::Vec::new();
4898 let mut nested_widths: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
4899 for col in &jt.columns {
4900 match col {
4901 C::Ordinality { .. } => {
4902 parent_cells.push(Value::BigInt(ordinality as i64 + 1));
4903 }
4904 C::Regular { .. } => {
4905 parent_cells.push(self.json_table_column_value(col, item, vars)?);
4906 }
4907 C::Nested { path, columns } => {
4908 // Recurse: a nested JSON_TABLE over `item` filtered
4909 // by `path`, with the same PASSING vars.
4910 let sub = spg_sql::ast::JsonTable {
4911 doc: jt.doc.clone(), // unused (outer_doc provided)
4912 row_path: path.clone(),
4913 columns: columns.clone(),
4914 passing: alloc::vec::Vec::new(),
4915 };
4916 let (nrows, nschema) = self.json_table_rows(&sub, Some(item))?;
4917 nested_widths.push(nschema.len());
4918 nested_runs.push(nrows);
4919 }
4920 }
4921 }
4922 if nested_runs.is_empty() {
4923 out.push(Row::new(parent_cells));
4924 return Ok(());
4925 }
4926 // PG sibling-NESTED semantics: each sibling expands
4927 // INDEPENDENTLY and the results CONCATENATE — a row from
4928 // sibling s fills only s's cells, every other sibling's cells
4929 // NULL. An empty sibling contributes ZERO rows (not a NULL
4930 // row). Only when EVERY sibling is empty does the parent still
4931 // emit one all-NULL row (the outer-join guarantee that a parent
4932 // item is never dropped). Verified vs PG18 (r207): a=1,b=2 → 3
4933 // rows; a=1,b=[] → 1 row; all-empty → 1 NULL row.
4934 let before = out.len();
4935 for (s_idx, run) in nested_runs.iter().enumerate() {
4936 for nrow in run {
4937 let mut cells = parent_cells.clone();
4938 for (o_idx, w) in nested_widths.iter().enumerate() {
4939 if o_idx == s_idx {
4940 cells.extend(nrow.values.iter().cloned());
4941 } else {
4942 for _ in 0..*w {
4943 cells.push(Value::Null);
4944 }
4945 }
4946 }
4947 out.push(Row::new(cells));
4948 }
4949 }
4950 if out.len() == before {
4951 // Every sibling empty → one all-NULL nested row.
4952 let mut cells = parent_cells.clone();
4953 for w in &nested_widths {
4954 for _ in 0..*w {
4955 cells.push(Value::Null);
4956 }
4957 }
4958 out.push(Row::new(cells));
4959 }
4960 Ok(())
4961 }
4962
4963 /// v7.39 (round 205) — evaluate one Regular column against a row
4964 /// item: EXISTS → bool; else path → at most one value, coerced to
4965 /// the declared type with ON EMPTY / ON ERROR / DEFAULT behaviour.
4966 fn json_table_column_value(
4967 &self,
4968 col: &spg_sql::ast::JsonTableColumn,
4969 item: &crate::json::JsonValue,
4970 vars: Option<&crate::json::JsonValue>,
4971 ) -> Result<Value<'static>, EngineError> {
4972 use spg_sql::ast::{JsonTableColumn as C, JsonTableOnBehavior as B};
4973 let C::Regular {
4974 name,
4975 ty,
4976 path,
4977 exists,
4978 format_json,
4979 wrapper,
4980 on_empty,
4981 on_error,
4982 } = col
4983 else {
4984 unreachable!("caller guards Regular");
4985 };
4986 let matches = crate::json::json_table_path(item, path, vars).map_err(EngineError::Eval)?;
4987 if *exists {
4988 return Ok(Value::Bool(!matches.is_empty()));
4989 }
4990 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
4991 let ctx = EvalContext::new(&empty_schema, None);
4992 let dummy = Row::new(alloc::vec::Vec::new());
4993 let default_of = |b: &B| -> Result<Option<Value<'static>>, EngineError> {
4994 match b {
4995 B::Null => Ok(Some(Value::Null)),
4996 B::Error => Ok(None),
4997 B::Default(e) => Ok(Some(
4998 eval::eval_expr(e, &dummy, &ctx).map_err(EngineError::Eval)?,
4999 )),
5000 }
5001 };
5002 // Empty match set → ON EMPTY.
5003 if matches.is_empty() {
5004 return match default_of(on_empty)? {
5005 Some(v) => coerce_json_table_default(v, *ty, name),
5006 None => Err(EngineError::Unsupported(alloc::format!(
5007 "no SQL/JSON item found for JSON_TABLE column {name:?}"
5008 ))),
5009 };
5010 }
5011 let first = &matches[0];
5012 // FORMAT JSON: return the PG-canonical json representation.
5013 // WITH WRAPPER wraps the whole match SET in an array (even a
5014 // single scalar → `[5]`); without it, the single match's json.
5015 if *format_json {
5016 let text = if *wrapper {
5017 crate::json::JsonValue::Array(matches.clone()).canonical_json_text()
5018 } else {
5019 first.canonical_json_text()
5020 };
5021 return Ok(Value::Json(alloc::borrow::Cow::Owned(text)));
5022 }
5023 if first.is_json_null() {
5024 return Ok(Value::Null);
5025 }
5026 // Coerce the scalar text to the declared type; on failure → ON
5027 // ERROR (default NULL, DEFAULT expr, or raise).
5028 let dt = crate::conversions::column_type_to_data_type(*ty);
5029 let scalar = Value::Text(alloc::borrow::Cow::Owned(first.scalar_text()));
5030 match crate::conversions::coerce_value(scalar, dt, name, 0) {
5031 Ok(v) => Ok(v),
5032 Err(e) => match default_of(on_error)? {
5033 Some(v) => coerce_json_table_default(v, *ty, name),
5034 None => Err(e),
5035 },
5036 }
5037 }
5038
5039 /// table function into (rows, default schema). Dispatch by name.
5040 pub(crate) fn table_fn_rows(
5041 &self,
5042 primary: &TableRef,
5043 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
5044 let (fn_name, args) = primary
5045 .table_fn_call
5046 .as_deref()
5047 .expect("caller guards table_fn_call.is_some()");
5048 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
5049 let ctx = EvalContext::new(&empty_schema, None);
5050 let dummy_row = Row::new(alloc::vec::Vec::new());
5051 let arg0: Option<Value<'static>> = match args.first() {
5052 Some(e) => Some(eval::eval_expr(e, &dummy_row, &ctx).map_err(EngineError::Eval)?),
5053 None => None,
5054 };
5055 match fn_name.as_str() {
5056 // v7.39 (read01 round 76) — `jsonb_populate_record(NULL::t, j)` /
5057 // `…_recordset` (+ json_ variants). The row shape is the BASE
5058 // argument's declared type — a table's or a composite type's
5059 // column list — which only the catalog knows, so the parser hands
5060 // the raw arguments here rather than desugaring blind.
5061 "jsonb_populate_record"
5062 | "json_populate_record"
5063 | "jsonb_populate_recordset"
5064 | "json_populate_recordset" => {
5065 let type_name = match args.first() {
5066 Some(Expr::Cast {
5067 target: spg_sql::ast::CastTarget::Named(n),
5068 ..
5069 }) => n.clone(),
5070 _ => {
5071 return Err(EngineError::Unsupported(alloc::format!(
5072 "{fn_name}(): first argument must name a row type, \
5073 e.g. NULL::mytable"
5074 )));
5075 }
5076 };
5077 let cat = self.active_catalog();
5078 let cols: alloc::vec::Vec<ColumnSchema> = if let Some(t) = cat.get(&type_name) {
5079 t.schema().columns.clone()
5080 } else if let Some(c) = cat.composite_types().get(&type_name) {
5081 c.fields
5082 .iter()
5083 .map(|(n, ty)| ColumnSchema::new(n.clone(), *ty, true))
5084 .collect()
5085 } else {
5086 return Err(EngineError::Unsupported(alloc::format!(
5087 "type \"{type_name}\" does not exist"
5088 )));
5089 };
5090 let json_arg = match args.get(1) {
5091 Some(e) => eval::eval_expr(e, &dummy_row, &ctx).map_err(EngineError::Eval)?,
5092 None => Value::Null,
5093 };
5094 // The set form iterates the JSON array; the scalar form is
5095 // the one-element case of the same walk.
5096 let docs: alloc::vec::Vec<Value<'static>> = if fn_name.ends_with("recordset") {
5097 crate::json::array_element_rows(&json_arg, false, fn_name)
5098 .map_err(EngineError::Eval)?
5099 .into_iter()
5100 .map(|s| s.map_or(Value::Null, Value::json))
5101 .collect()
5102 } else if matches!(json_arg, Value::Null) {
5103 alloc::vec::Vec::new()
5104 } else {
5105 alloc::vec![json_arg]
5106 };
5107 let mut rows = alloc::vec::Vec::with_capacity(docs.len());
5108 for doc in &docs {
5109 let mut vals = alloc::vec::Vec::with_capacity(cols.len());
5110 for c in &cols {
5111 // `->>` semantics: a missing key is NULL, present keys
5112 // arrive as text and cast to the declared column type.
5113 let raw = crate::json::path_get(doc, &Value::text(c.name.clone()), true)
5114 .map_err(EngineError::Eval)?;
5115 let v = if matches!(raw, Value::Null) {
5116 Value::Null
5117 } else {
5118 crate::conversions::coerce_value(raw, c.ty, "", 0)
5119 .map_err(|e| EngineError::Unsupported(alloc::format!("{e:?}")))?
5120 };
5121 vals.push(v);
5122 }
5123 rows.push(Row::new(vals));
5124 }
5125 Ok((rows, cols))
5126 }
5127 // 7.38.1 S5.1 (pg_dump wall #3) — pg_options_to_table:
5128 // a text[] of 'name=value' reloptions/fdw options → one
5129 // (option_name, option_value) row per element. NULL or an
5130 // empty array yields zero rows (PG); an element without
5131 // '=' carries a NULL option_value, matching PG's split.
5132 "pg_options_to_table" => {
5133 let schema = alloc::vec![
5134 ColumnSchema::new("option_name", DataType::Text, true),
5135 ColumnSchema::new("option_value", DataType::Text, true),
5136 ];
5137 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
5138 if let Some(Value::TextArray(items)) = arg0 {
5139 for item in items.into_iter().flatten() {
5140 let (name, value) = match item.split_once('=') {
5141 Some((n, v)) => (Value::text(n), Value::text(v)),
5142 None => (Value::text(item.as_str()), Value::Null),
5143 };
5144 rows.push(Row::new(alloc::vec![name, value]));
5145 }
5146 }
5147 Ok((rows, schema))
5148 }
5149 // 7.38.1 S5.1 (pg_dump wall) — pg_get_sequence_data(oid):
5150 // PG18's per-sequence state SRF, (last_value, is_called).
5151 // pg_dump reads it joined to pg_sequence for every dumped
5152 // sequence's setval line. The oid resolves through the
5153 // same relation_oid mapping seqrelid publishes.
5154 "pg_get_sequence_data" => {
5155 let schema = alloc::vec![
5156 ColumnSchema::new("last_value", DataType::BigInt, false),
5157 ColumnSchema::new("is_called", DataType::Bool, false),
5158 ];
5159 let want = match arg0 {
5160 Some(Value::Int(n)) => i64::from(n),
5161 Some(Value::BigInt(n)) => n,
5162 _ => {
5163 return Err(EngineError::Unsupported(
5164 "pg_get_sequence_data(): argument must be a sequence oid".into(),
5165 ));
5166 }
5167 };
5168 let cat = self.active_catalog();
5169 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
5170 for (name, def) in cat.sequences_all() {
5171 if crate::system_catalog::relation_oid(cat, name) == Some(want) {
5172 rows.push(Row::new(alloc::vec![
5173 Value::BigInt(def.last_value),
5174 Value::Bool(def.is_called),
5175 ]));
5176 break;
5177 }
5178 }
5179 Ok((rows, schema))
5180 }
5181 "pg_partition_tree" => {
5182 let cols = alloc::vec![
5183 ColumnSchema::new("relid".to_string(), DataType::Text, true),
5184 ColumnSchema::new("parentrelid".to_string(), DataType::Text, true),
5185 ColumnSchema::new("isleaf".to_string(), DataType::Bool, true),
5186 ColumnSchema::new("level".to_string(), DataType::Int, true),
5187 ];
5188 let Some(Value::Text(name)) = &arg0 else {
5189 // NULL (or missing) argument → zero rows (PG).
5190 return Ok((alloc::vec::Vec::new(), cols));
5191 };
5192 let entries = crate::partition_walks::tree_of(self.active_catalog(), name.as_ref());
5193 if entries.is_empty() && self.active_catalog().get(name.as_ref()).is_none() {
5194 return Err(EngineError::Unsupported(alloc::format!(
5195 "relation \"{name}\" does not exist"
5196 )));
5197 }
5198 let rows = entries
5199 .into_iter()
5200 .map(|(relid, parent, isleaf, level)| {
5201 Row::new(alloc::vec![
5202 Value::text(relid),
5203 parent.map_or(Value::Null, Value::text),
5204 Value::Bool(isleaf),
5205 #[allow(clippy::cast_possible_truncation)]
5206 Value::Int(level as i32),
5207 ])
5208 })
5209 .collect();
5210 Ok((rows, cols))
5211 }
5212 "pg_partition_ancestors" => {
5213 let cols =
5214 alloc::vec![ColumnSchema::new("relid".to_string(), DataType::Text, true)];
5215 let Some(Value::Text(name)) = &arg0 else {
5216 return Ok((alloc::vec::Vec::new(), cols));
5217 };
5218 let cat = self.active_catalog();
5219 if cat.get(name.as_ref()).is_none() {
5220 return Err(EngineError::Unsupported(alloc::format!(
5221 "relation \"{name}\" does not exist"
5222 )));
5223 }
5224 // A relation outside any partition tree yields no rows (PG).
5225 let in_tree = cat
5226 .get(name.as_ref())
5227 .is_some_and(|t| t.schema().partition_role.is_some());
5228 let rows = if in_tree {
5229 crate::partition_walks::ancestors_of(cat, name.as_ref())
5230 .into_iter()
5231 .map(|n| Row::new(alloc::vec![Value::text(n)]))
5232 .collect()
5233 } else {
5234 alloc::vec::Vec::new()
5235 };
5236 Ok((rows, cols))
5237 }
5238 // v7.39 (round 651) — `ts_debug(config, text)`: what the parser
5239 // saw, what each token was called, which dictionary took it
5240 // and what came out. It is a projection of the same tokenizer
5241 // and the same map the indexer uses, so it cannot describe a
5242 // pipeline other than the one that runs.
5243 "ts_debug" => {
5244 use crate::fts::{TokenType, TsDict};
5245 let cols = alloc::vec![
5246 ColumnSchema::new("alias".to_string(), DataType::Text, false),
5247 ColumnSchema::new("description".to_string(), DataType::Text, false),
5248 ColumnSchema::new("token".to_string(), DataType::Text, false),
5249 ColumnSchema::new("dictionaries".to_string(), DataType::TextArray, false),
5250 ColumnSchema::new("dictionary".to_string(), DataType::Text, true),
5251 ColumnSchema::new("lexemes".to_string(), DataType::TextArray, true),
5252 ];
5253 // PG's one-arg form uses the session configuration; the
5254 // two-arg form names one.
5255 let (cfg_name, text) = match (&arg0, args.get(1)) {
5256 (Some(Value::Text(c)), Some(t)) => {
5257 let v = eval::eval_expr(t, &dummy_row, &ctx).map_err(EngineError::Eval)?;
5258 (c.to_string(), crate::eval::value_to_text(&v))
5259 }
5260 (Some(v), None) => (
5261 alloc::string::String::from("english"),
5262 crate::eval::value_to_text(v),
5263 ),
5264 _ => return Ok((alloc::vec::Vec::new(), cols)),
5265 };
5266 let english = match cfg_name
5267 .trim()
5268 .trim_start_matches("pg_catalog.")
5269 .to_ascii_lowercase()
5270 .as_str()
5271 {
5272 "english" => true,
5273 "simple" => false,
5274 other => {
5275 return Err(EngineError::Unsupported(alloc::format!(
5276 "text search configuration \"{other}\" does not exist"
5277 )));
5278 }
5279 };
5280 let rows = crate::fts::tokenize_typed(&text)
5281 .into_iter()
5282 .map(|tok| {
5283 let dict = tok.ty.dictionary(english);
5284 let dname = dict.map(|d| match d {
5285 TsDict::Simple => "simple",
5286 TsDict::EnglishStem => "english_stem",
5287 });
5288 let folded = tok.text.to_lowercase();
5289 let lexemes = dict.map(|d| match d {
5290 TsDict::Simple => alloc::vec![Some(folded.clone())],
5291 TsDict::EnglishStem => {
5292 if crate::fts::is_english_stopword(&folded) {
5293 alloc::vec::Vec::new()
5294 } else {
5295 alloc::vec![Some(crate::fts::porter_stem(&folded))]
5296 }
5297 }
5298 });
5299 Row::new(alloc::vec![
5300 Value::text(tok.ty.alias()),
5301 Value::text(tok.ty.description()),
5302 Value::text(tok.text),
5303 Value::TextArray(
5304 dname
5305 .map(|n| alloc::vec![Some(alloc::string::String::from(n))])
5306 .unwrap_or_default(),
5307 ),
5308 dname.map_or(Value::Null, Value::text),
5309 lexemes.map_or(Value::Null, Value::TextArray),
5310 ])
5311 })
5312 .collect();
5313 let _ = TokenType::AsciiWord;
5314 Ok((rows, cols))
5315 }
5316 // v7.39 (round 651) — `ts_token_type('default')`, the list the
5317 // parser actually produces. It is a projection of the
5318 // `TokenType` enum the tokenizer and `pg_ts_config_map` both
5319 // read, so the three cannot disagree about what a token is.
5320 "ts_token_type" => {
5321 use crate::fts::TokenType as T;
5322 let cols = alloc::vec![
5323 ColumnSchema::new("tokid".to_string(), DataType::Int, false),
5324 ColumnSchema::new("alias".to_string(), DataType::Text, false),
5325 ColumnSchema::new("description".to_string(), DataType::Text, false),
5326 ];
5327 // PG takes the parser by name or oid; SPG has the one.
5328 if let Some(Value::Text(p)) = &arg0
5329 && !p.eq_ignore_ascii_case("default")
5330 && !p.eq_ignore_ascii_case("pg_catalog.default")
5331 {
5332 return Err(EngineError::Unsupported(alloc::format!(
5333 "text search parser \"{p}\" does not exist"
5334 )));
5335 }
5336 const TYPES: &[T] = &[
5337 T::AsciiWord,
5338 T::Word,
5339 T::NumWord,
5340 T::Email,
5341 T::Url,
5342 T::Host,
5343 T::SFloat,
5344 T::Version,
5345 T::HwordNumPart,
5346 T::HwordPart,
5347 T::HwordAsciiPart,
5348 T::Blank,
5349 T::Tag,
5350 T::Protocol,
5351 T::NumHword,
5352 T::AsciiHword,
5353 T::Hword,
5354 T::UrlPath,
5355 T::File,
5356 T::Float,
5357 T::Int,
5358 T::Uint,
5359 T::Entity,
5360 ];
5361 let rows = TYPES
5362 .iter()
5363 .map(|t| {
5364 Row::new(alloc::vec![
5365 Value::Int(*t as i32),
5366 Value::text(t.alias()),
5367 Value::text(t.description()),
5368 ])
5369 })
5370 .collect();
5371 Ok((rows, cols))
5372 }
5373 // v7.39 (read01 round 65) — a set-returning USER function in FROM
5374 // (`FROM rows_of(2)`). Its body runs through the real executor, like
5375 // every other function body since round 63.
5376 other => {
5377 if !self.active_catalog().functions_named(other).is_empty() {
5378 return self.exec_setof_user_function(other, args, primary.alias.as_deref());
5379 }
5380 Err(EngineError::Unsupported(alloc::format!(
5381 "table function {other}() is not supported in FROM"
5382 )))
5383 }
5384 }
5385 }
5386
5387 /// v7.39 (read01 round 65) — run a `RETURNS SETOF <type>` / `RETURNS
5388 /// TABLE(…)` function in FROM position. The body is a SELECT; the arguments
5389 /// are bound into it as literals and it goes through the read path, so the
5390 /// rows it yields are exactly the rows a hand-written query would see.
5391 ///
5392 /// The column NAMES come from the declared shape: `RETURNS TABLE(id int, v
5393 /// text)` names them, and a `SETOF <scalar>` yields a single column named
5394 /// after the function — PG's rule, and what a bare `SELECT * FROM f()`
5395 /// shows.
5396 fn exec_setof_user_function(
5397 &self,
5398 name: &str,
5399 args: &[spg_sql::ast::Expr],
5400 // v7.39 (read01 round 65) — `FROM evens() AS x` names the single column
5401 // `x`: for a scalar SETOF, the table alias IS the column name (PG).
5402 alias: Option<&str>,
5403 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
5404 // The call's arguments belong to the ENCLOSING query, so they are
5405 // evaluated here and the body sees values.
5406 let empty: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
5407 let arg_ctx = self.ev_ctx(&empty, None);
5408 let dummy = Row::new(alloc::vec::Vec::new());
5409 let mut vals: alloc::vec::Vec<Value<'static>> = alloc::vec::Vec::new();
5410 for a in args {
5411 vals.push(eval::eval_expr(a, &dummy, &arg_ctx).map_err(EngineError::Eval)?);
5412 }
5413 self.setof_rows_of(name, &vals, alias)
5414 }
5415
5416 /// v7.39 (read01 round 67) — the set-returning core, on already-evaluated
5417 /// arguments. Shared by the FROM position and the target-list expansion, so
5418 /// a function cannot behave differently depending on where it is called.
5419 pub(crate) fn setof_rows_of(
5420 &self,
5421 name: &str,
5422 arg_values: &[Value<'static>],
5423 alias: Option<&str>,
5424 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
5425 let cat = self.active_catalog();
5426 let overloads = cat.functions_named(name);
5427 let def = overloads
5428 .iter()
5429 .find(|f| spg_storage::function_arg_types(&f.args_repr).len() == arg_values.len())
5430 .ok_or_else(|| {
5431 EngineError::Unsupported(alloc::format!(
5432 "function {name} does not exist with {} argument(s)",
5433 arg_values.len()
5434 ))
5435 })?;
5436 let declared = def.returns.trim().to_string();
5437 let upper = declared.to_ascii_uppercase();
5438 if !upper.starts_with("SETOF") && !upper.starts_with("TABLE(") {
5439 return Err(EngineError::Unsupported(alloc::format!(
5440 "function {name}() does not return a set — it cannot be used in FROM"
5441 )));
5442 }
5443
5444 let arg_names_pl = spg_storage::function_arg_names(&def.args_repr);
5445 // v7.39 (read01 round 66) — a plpgsql SETOF body builds its rows with
5446 // RETURN NEXT / RETURN QUERY; the interpreter collects them.
5447 if def.language.eq_ignore_ascii_case("plpgsql") {
5448 let out_rows = self
5449 .call_plpgsql_setof_fn(def, &arg_names_pl, arg_values)
5450 .map_err(EngineError::Eval)?;
5451 let cols = setof_column_shape(&declared, name, alias, out_rows.first());
5452 let rows = out_rows.into_iter().map(Row::new).collect();
5453 return Ok((rows, cols));
5454 }
5455 let body = def.body.trim().trim_end_matches(';');
5456 let stmt = spg_sql::parser::parse_statement(body).map_err(|e| {
5457 EngineError::Unsupported(alloc::format!("function {name} body does not parse: {e}"))
5458 })?;
5459 let spg_sql::ast::Statement::Select(body_select) = stmt else {
5460 return Err(EngineError::Unsupported(alloc::format!(
5461 "function {name}(): a set-returning body must be a SELECT"
5462 )));
5463 };
5464 let arg_names = spg_storage::function_arg_names(&def.args_repr);
5465 let bound = crate::eval::bind_user_fn_args(
5466 self.active_catalog(),
5467 &body_select,
5468 &arg_names,
5469 arg_values,
5470 )
5471 .map_err(EngineError::Eval)?;
5472 let out = self.exec_select_cancel(&bound, crate::CancelToken::none())?;
5473 let QueryResult::Rows { columns, rows } = out else {
5474 return Ok((alloc::vec::Vec::new(), alloc::vec::Vec::new()));
5475 };
5476 // Name the columns from the DECLARED shape — the same rule the plpgsql
5477 // path above uses, so a body's language cannot change the row shape.
5478 let cols = setof_column_shape_from(&declared, name, alias, &columns);
5479 Ok((rows, cols))
5480 }
5481
5482 fn exec_select_jsonb_each_text(
5483 &self,
5484 stmt: &SelectStatement,
5485 primary: &TableRef,
5486 cancel: CancelToken<'_>,
5487 ) -> Result<QueryResult, EngineError> {
5488 let (each_fn, arg_expr) = primary
5489 .jsonb_each_text_arg
5490 .as_ref()
5491 .map(|(name, expr)| (name.as_str(), expr.as_ref()))
5492 .expect("caller guards jsonb_each_text_arg.is_some()");
5493 // v7.37.17 (17.6 siblings) — the plain jsonb_each / json_each
5494 // forms keep JSON rendering in the value column (JSON null
5495 // stays jsonb 'null', strings keep their quotes).
5496 let as_text = each_fn.ends_with("_text");
5497 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
5498 let ctx = EvalContext::new(&empty_schema, None);
5499 let dummy_row = Row::new(alloc::vec::Vec::new());
5500 let arg_value = eval::eval_expr(arg_expr, &dummy_row, &ctx).map_err(EngineError::Eval)?;
5501 let pairs =
5502 crate::json::each_rows(&arg_value, as_text, each_fn).map_err(EngineError::Eval)?;
5503 let rows: alloc::vec::Vec<Row<'static>> = pairs
5504 .into_iter()
5505 .map(|(k, v)| {
5506 let key_val = Value::text(k);
5507 let value_val = match v {
5508 Some(s) if as_text => Value::text(s),
5509 Some(s) => Value::Json(alloc::borrow::Cow::Owned(s)),
5510 None => Value::Null,
5511 };
5512 Row::new(alloc::vec![key_val, value_val])
5513 })
5514 .collect();
5515 let alias = primary.alias.clone().unwrap_or_else(|| each_fn.to_string());
5516 let value_dtype = if as_text {
5517 spg_storage::DataType::Text
5518 } else {
5519 spg_storage::DataType::Json
5520 };
5521 let key_col = ColumnSchema::new("key".to_string(), spg_storage::DataType::Text, false);
5522 let value_col = ColumnSchema::new("value".to_string(), value_dtype, as_text);
5523 let mut schema_cols = alloc::vec![key_col, value_col];
5524 // `AS t(k, v)` renames key/value positionally (PG behaviour); the
5525 // LATERAL-position form of the same call already honours it.
5526 for (i, new_name) in primary.unnest_column_aliases.iter().enumerate() {
5527 if let Some(col) = schema_cols.get_mut(i) {
5528 col.name = new_name.clone();
5529 }
5530 }
5531 // v7.39 (read01 round 54) — `ev_ctx` threads the catalog; a bare
5532 // `EvalContext::new` drops it and every catalog-dependent cast
5533 // (regclass / enum / composite / domain) silently degrades.
5534 let scan_ctx = self.ev_ctx(&schema_cols, Some(&alias));
5535 // WHERE.
5536 let filtered: alloc::vec::Vec<Row<'static>> = if let Some(w) = &stmt.where_ {
5537 let mut out = alloc::vec::Vec::with_capacity(rows.len());
5538 for row in rows {
5539 cancel.check()?;
5540 let v = eval::eval_expr(w, &row, &scan_ctx).map_err(EngineError::Eval)?;
5541 if matches!(v, Value::Bool(true)) {
5542 out.push(row);
5543 }
5544 }
5545 out
5546 } else {
5547 rows
5548 };
5549 // Aggregate dispatch (e.g. SELECT COUNT(*) FROM jsonb_each_text…).
5550 if aggregate::uses_aggregate(stmt) {
5551 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
5552 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
5553 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
5554 .map_err(|err| match err {
5555 EngineError::Eval(ev) => ev,
5556 other => eval::EvalError::TypeMismatch {
5557 detail: alloc::format!("{other}"),
5558 },
5559 })
5560 };
5561 // v7.39 (round 656) — hand the rows over as they are rather than
5562 // collecting a second vector of `RowRef` wrappers. Note this is
5563 // a set-returning-function path, NOT the relational scan: the
5564 // measured O(rows) cost lived in `run_single_table_aggregate`,
5565 // and converting these four first was a miss that cost a full
5566 // round — every test stayed green and the number did not move.
5567 let agg = aggregate::run(
5568 stmt,
5569 crate::join::AggRows::Owned(&filtered),
5570 &schema_cols,
5571 Some(&alias),
5572 Some(&agg_correlated),
5573 self.parallel_runner.0.as_deref(),
5574 Some(self.active_catalog()),
5575 Some(self),
5576 )?;
5577 return self.finish_agg_result(agg, stmt, cancel);
5578 }
5579 // Projection.
5580 let projection =
5581 build_projection(&stmt.items, &schema_cols, &alias, self.backslash_escapes)?;
5582 let mut projected_rows: alloc::vec::Vec<Row<'static>> =
5583 alloc::vec::Vec::with_capacity(filtered.len());
5584 for row in &filtered {
5585 let mut vals = alloc::vec::Vec::with_capacity(projection.len());
5586 for p in &projection {
5587 let v = eval::eval_expr(&p.expr, row, &scan_ctx).map_err(EngineError::Eval)?;
5588 vals.push(v);
5589 }
5590 projected_rows.push(Row::new(vals));
5591 }
5592 let columns: alloc::vec::Vec<ColumnSchema> = projection
5593 .iter()
5594 // v7.39 (read01 round 54) — keep the column's enum identity through
5595 // the projection (it lives outside the DataType lattice), or a
5596 // derived table / UNION / windowed result forgets it and any outer
5597 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
5598 .map(|p| {
5599 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
5600 c.user_enum_type = p.user_enum_type.clone();
5601 c.mysql_fsp = p.mysql_fsp;
5602 c
5603 })
5604 .collect();
5605 // ORDER BY.
5606 if !stmt.order_by.is_empty() {
5607 let mut indexed: alloc::vec::Vec<(usize, Vec<Value<'static>>)> = filtered
5608 .iter()
5609 .enumerate()
5610 .map(|(i, r)| -> Result<_, EngineError> {
5611 let keys: Result<Vec<Value<'static>>, EngineError> = stmt
5612 .order_by
5613 .iter()
5614 .map(|ob| {
5615 eval::eval_expr(&ob.expr, r, &scan_ctx).map_err(EngineError::Eval)
5616 })
5617 .collect();
5618 Ok((i, keys?))
5619 })
5620 .collect::<Result<_, _>>()?;
5621 indexed.sort_by(|a, b| {
5622 for (idx, (ka, kb)) in a.1.iter().zip(b.1.iter()).enumerate() {
5623 let o = &stmt.order_by[idx];
5624 let cmp = order_by_value_cmp_in(
5625 o.desc,
5626 o.nulls_first,
5627 ka,
5628 kb,
5629 scan_ctx.mysql_dialect && !crate::eval::is_binary_coerced(&o.expr),
5630 );
5631 if cmp != core::cmp::Ordering::Equal {
5632 return cmp;
5633 }
5634 }
5635 core::cmp::Ordering::Equal
5636 });
5637 projected_rows = indexed
5638 .into_iter()
5639 .map(|(i, _)| projected_rows[i].clone())
5640 .collect();
5641 }
5642 // v7.38 (read01) — DISTINCT over a synthetic source was dropped here.
5643 if stmt.distinct {
5644 projected_rows = dedup_rows(projected_rows, scan_ctx.mysql_dialect);
5645 }
5646 if let Some(offset) = stmt.offset_literal() {
5647 let off = (offset as usize).min(projected_rows.len());
5648 projected_rows.drain(..off);
5649 }
5650 if let Some(limit) = stmt.limit_literal() {
5651 projected_rows.truncate(limit as usize);
5652 }
5653 Ok(QueryResult::Rows {
5654 columns,
5655 rows: projected_rows,
5656 })
5657 }
5658
5659 /// v7.37.17 (17.6 siblings) — execute `SELECT … FROM
5660 /// ( SELECT … ) alias` in primary position. The inner SELECT
5661 /// materialises once through the regular bare-select executor
5662 /// (UNION tails included), then the outer WHERE / aggregate /
5663 /// projection / ORDER BY / LIMIT pipeline runs over the
5664 /// synthetic table — the same post-materialisation shape as
5665 /// exec_select_jsonb_each_text, generalised to N columns.
5666 fn exec_select_derived(
5667 &self,
5668 stmt: &SelectStatement,
5669 primary: &TableRef,
5670 cancel: CancelToken<'_>,
5671 ) -> Result<QueryResult, EngineError> {
5672 let inner = primary
5673 .lateral_subquery
5674 .as_deref()
5675 .expect("caller guards lateral_subquery.is_some()");
5676 // exec_select_cancel is the union-aware wrapper — the inner
5677 // SELECT may carry UNION tails on stmt.unions.
5678 let QueryResult::Rows {
5679 columns: inner_cols,
5680 rows,
5681 } = self.exec_select_cancel(inner, cancel)?
5682 else {
5683 return Err(EngineError::Unsupported(
5684 "derived table subquery must return rows".into(),
5685 ));
5686 };
5687 let alias = primary
5688 .alias
5689 .clone()
5690 .unwrap_or_else(|| primary.name.clone());
5691 // `AS t(a, b)` renames the materialised columns positionally
5692 // (extra inner columns keep their own names, PG behaviour).
5693 let mut schema_cols: alloc::vec::Vec<ColumnSchema> = inner_cols;
5694 // v7.39 (read01 round 78) — a column-alias list longer than the item is
5695 // the error PG reports; SPG used to let the extra names through and then
5696 // fail two layers downstream with "column not found: <the extra name>".
5697 let n_out = schema_cols.len() + usize::from(primary.with_ordinality);
5698 if primary.unnest_column_aliases.len() > n_out {
5699 return Err(EngineError::Unsupported(alloc::format!(
5700 "table \"{alias}\" has {n_out} columns available but {} columns specified",
5701 primary.unnest_column_aliases.len()
5702 )));
5703 }
5704 if primary.scalar_fn_item && schema_cols.len() == 1 {
5705 schema_cols[0].scalar_row_source = true;
5706 }
5707 // v7.39 (read01 round 78) — WITH ORDINALITY on a table function that
5708 // rides this channel (regexp_matches): a trailing bigint counter, 1-based.
5709 // The column-alias list, if given, names it like any other column.
5710 let mut rows = rows;
5711 if primary.with_ordinality {
5712 schema_cols.push(ColumnSchema::new(
5713 "ordinality".to_string(),
5714 DataType::BigInt,
5715 false,
5716 ));
5717 rows = rows
5718 .into_iter()
5719 .enumerate()
5720 .map(|(i, r)| {
5721 let mut v = r.values;
5722 #[allow(clippy::cast_possible_wrap)]
5723 v.push(Value::BigInt(i as i64 + 1));
5724 Row::new(v)
5725 })
5726 .collect();
5727 }
5728 for (i, new_name) in primary.unnest_column_aliases.iter().enumerate() {
5729 if let Some(col) = schema_cols.get_mut(i) {
5730 col.name = new_name.clone();
5731 }
5732 }
5733 self.exec_select_over_rows(stmt, rows, schema_cols, &alias, cancel)
5734 }
5735
5736 /// v7.39 (read01 partitionfuncs.c) — shared synthetic-source SELECT
5737 /// pipeline (WHERE / aggregate / projection / ORDER BY / DISTINCT /
5738 /// OFFSET / LIMIT) over a pre-materialised row set. Drives the
5739 /// derived-table executor and the FROM-position table functions.
5740 fn exec_select_over_rows(
5741 &self,
5742 stmt: &SelectStatement,
5743 rows: alloc::vec::Vec<Row<'static>>,
5744 schema_cols: alloc::vec::Vec<ColumnSchema>,
5745 alias: &str,
5746 cancel: CancelToken<'_>,
5747 ) -> Result<QueryResult, EngineError> {
5748 let scan_ctx = self.ev_ctx(&schema_cols, Some(alias));
5749 // v7.37 D.21 — correlated subqueries in the WHERE / projection may
5750 // reference this derived table's columns (`… WHERE u.gg = t.g` where t
5751 // is `(VALUES …) t`). Resolve them per-row via eval_expr_with_correlated
5752 // (the same path the aggregate branch uses); the old plain eval_expr let
5753 // a ScalarSubquery reach row-eval unresolved ("engine resolver bug").
5754 let corr_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
5755 // WHERE.
5756 let filtered: alloc::vec::Vec<Row<'static>> = if let Some(w) = &stmt.where_ {
5757 let mut out = alloc::vec::Vec::with_capacity(rows.len());
5758 for row in rows {
5759 cancel.check()?;
5760 let v = self.eval_expr_with_correlated(
5761 w,
5762 &row,
5763 &scan_ctx,
5764 cancel,
5765 Some(&mut corr_memo.borrow_mut()),
5766 )?;
5767 if matches!(v, Value::Bool(true)) {
5768 out.push(row);
5769 }
5770 }
5771 out
5772 } else {
5773 rows
5774 };
5775 // Aggregate dispatch.
5776 if aggregate::uses_aggregate(stmt) {
5777 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
5778 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
5779 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
5780 .map_err(|err| match err {
5781 EngineError::Eval(ev) => ev,
5782 other => eval::EvalError::TypeMismatch {
5783 detail: alloc::format!("{other}"),
5784 },
5785 })
5786 };
5787 // v7.39 (round 656) — hand the rows over as they are rather than
5788 // collecting a second vector of `RowRef` wrappers. Note this is
5789 // a set-returning-function path, NOT the relational scan: the
5790 // measured O(rows) cost lived in `run_single_table_aggregate`,
5791 // and converting these four first was a miss that cost a full
5792 // round — every test stayed green and the number did not move.
5793 let agg = aggregate::run(
5794 stmt,
5795 crate::join::AggRows::Owned(&filtered),
5796 &schema_cols,
5797 Some(alias),
5798 Some(&agg_correlated),
5799 self.parallel_runner.0.as_deref(),
5800 Some(self.active_catalog()),
5801 Some(self),
5802 )?;
5803 return self.finish_agg_result(agg, stmt, cancel);
5804 }
5805 // Projection.
5806 let projection =
5807 build_projection(&stmt.items, &schema_cols, alias, self.backslash_escapes)?;
5808 // v7.39 (round 621) — a target-list SRF expands here too. This tail
5809 // serves VALUES, a derived table and `ROWS FROM (…)`, and knew nothing
5810 // about them: `SELECT unnest(ARRAY[1,2]), x FROM (VALUES (3),(4)) v(x)`
5811 // answered `function unnest(integer[]) does not exist` for a query PG
5812 // answers.
5813 let srf_idxs = self.srf_target_idxs(&projection);
5814 let mut src_of_row: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
5815 let mut projected_rows: alloc::vec::Vec<Row<'static>> =
5816 alloc::vec::Vec::with_capacity(filtered.len());
5817 if !srf_idxs.is_empty() {
5818 let (rows, src) =
5819 expand_projection_srfs(self, &projection, &srf_idxs, &filtered, &scan_ctx)?;
5820 projected_rows = rows;
5821 src_of_row = src;
5822 } else {
5823 for row in &filtered {
5824 let mut vals = alloc::vec::Vec::with_capacity(projection.len());
5825 for p in &projection {
5826 let v = self.eval_expr_with_correlated(
5827 &p.expr,
5828 row,
5829 &scan_ctx,
5830 cancel,
5831 Some(&mut corr_memo.borrow_mut()),
5832 )?;
5833 vals.push(v);
5834 }
5835 projected_rows.push(Row::new(vals));
5836 }
5837 }
5838 let columns: alloc::vec::Vec<ColumnSchema> = projection
5839 .iter()
5840 // v7.39 (read01 round 54) — keep the column's enum identity through
5841 // the projection (it lives outside the DataType lattice), or a
5842 // derived table / UNION / windowed result forgets it and any outer
5843 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
5844 .map(|p| {
5845 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
5846 c.user_enum_type = p.user_enum_type.clone();
5847 c.mysql_fsp = p.mysql_fsp;
5848 c
5849 })
5850 .collect();
5851 // ORDER BY over the source rows (same shape as the other
5852 // synthetic-table executors).
5853 // v7.39 (read01 round 80) — a positional key (`ORDER BY 1`) means the Nth
5854 // OUTPUT column. Evaluated as an expression, as it was here, the literal
5855 // `1` is just the constant 1: the same sort key for every row, so the
5856 // sort ran and changed nothing. `SELECT unnest(ARRAY['B','a','A','b'])
5857 // ORDER BY 1` (which the parser turns into `SELECT * FROM unnest(…)`,
5858 // landing on this executor) came back in input order.
5859 let order_by = resolve_positional_order_by(&stmt.order_by, &projection);
5860 if !order_by.is_empty() {
5861 // v7.39 (round 621) — one entry per OUTPUT row, since a target-list
5862 // SRF makes more of them than there were inputs.
5863 let out_cols = if srf_idxs.is_empty() {
5864 alloc::vec![None; order_by.len()]
5865 } else {
5866 srf_order_output_cols(&order_by, &projection)
5867 };
5868 let mut indexed: alloc::vec::Vec<(usize, Vec<Value<'static>>)> = projected_rows
5869 .iter()
5870 .enumerate()
5871 .map(|(k, out)| -> Result<_, EngineError> {
5872 let r = &filtered[src_of_row.get(k).copied().unwrap_or(k)];
5873 let keys: Result<Vec<Value<'static>>, EngineError> = order_by
5874 .iter()
5875 .zip(out_cols.iter())
5876 .map(|(ob, oc)| {
5877 // v7.39 (read01 round 54) — this path builds its
5878 // sort keys itself instead of going through
5879 // `build_order_keys`, so it skipped the enum-ordinal
5880 // substitution: an OUTER `ORDER BY <enum col>` over
5881 // a DERIVED TABLE sorted by the label TEXT, not by
5882 // member order. Silently wrong rows, not an error.
5883 let v = srf_order_key(ob, *oc, out, r, &scan_ctx)?;
5884 Ok(
5885 match crate::orderby::enum_order_ordinal(&ob.expr, &v, &scan_ctx) {
5886 Some(ord) => Value::Float(ord),
5887 None => v,
5888 },
5889 )
5890 })
5891 .collect();
5892 Ok((k, keys?))
5893 })
5894 .collect::<Result<_, _>>()?;
5895 indexed.sort_by(|a, b| {
5896 for (idx, (ka, kb)) in a.1.iter().zip(b.1.iter()).enumerate() {
5897 let o = &stmt.order_by[idx];
5898 let cmp = order_by_value_cmp_in(
5899 o.desc,
5900 o.nulls_first,
5901 ka,
5902 kb,
5903 scan_ctx.mysql_dialect && !crate::eval::is_binary_coerced(&o.expr),
5904 );
5905 if cmp != core::cmp::Ordering::Equal {
5906 return cmp;
5907 }
5908 }
5909 core::cmp::Ordering::Equal
5910 });
5911 projected_rows = indexed
5912 .into_iter()
5913 .map(|(i, _)| projected_rows[i].clone())
5914 .collect();
5915 }
5916 // v7.38 (read01) — DISTINCT over a synthetic source was dropped here.
5917 if stmt.distinct {
5918 projected_rows = dedup_rows(projected_rows, scan_ctx.mysql_dialect);
5919 }
5920 if let Some(offset) = stmt.offset_literal() {
5921 let off = (offset as usize).min(projected_rows.len());
5922 projected_rows.drain(..off);
5923 }
5924 if let Some(limit) = stmt.limit_literal() {
5925 projected_rows.truncate(limit as usize);
5926 }
5927 Ok(QueryResult::Rows {
5928 columns,
5929 rows: projected_rows,
5930 })
5931 }
5932
5933 /// Constant `SELECT` with no FROM: evaluate each projection item
5934 /// once against an empty dummy row (`SELECT 1`, `SELECT '7'::INT`).
5935 fn exec_constant_select(&self, stmt: &SelectStatement) -> Result<QueryResult, EngineError> {
5936 let empty_schema: Vec<ColumnSchema> = Vec::new();
5937 let ctx = self.ev_ctx(&empty_schema, None);
5938 // v7.39 (read01 round 106) — an aggregate with no FROM runs over the
5939 // single implicit row (`SELECT count(*)` → 1, `SELECT sum(5)` → 5,
5940 // `SELECT string_agg('x',',')` → x). Before this it fell through to the
5941 // scalar projection, where the aggregate name looked like an unknown
5942 // function. The WHERE filters that one row, so `… WHERE false` leaves
5943 // the aggregate zero input rows (`count(*)` → 0).
5944 if aggregate::uses_aggregate(stmt) {
5945 let dummy = Row::new(Vec::new());
5946 let passes = match &stmt.where_ {
5947 Some(w) => matches!(eval::eval_expr(w, &dummy, &ctx)?, Value::Bool(true)),
5948 None => true,
5949 };
5950 let rows: Vec<RowRef<'_>> = if passes {
5951 alloc::vec![RowRef::Owned(&dummy)]
5952 } else {
5953 Vec::new()
5954 };
5955 let agg = aggregate::run(
5956 stmt,
5957 crate::join::AggRows::Refs(&rows),
5958 &empty_schema,
5959 None,
5960 None,
5961 self.parallel_runner.0.as_deref(),
5962 Some(self.active_catalog()),
5963 Some(self),
5964 )?;
5965 return self.finish_agg_result(agg, stmt, CancelToken::none());
5966 }
5967 let projection = build_projection(&stmt.items, &empty_schema, "", self.backslash_escapes)?;
5968 // `SELECT … WHERE cond` with no FROM — the one conceptual
5969 // row survives only when the condition is true (previously
5970 // the WHERE was silently ignored: `SELECT 1 WHERE false`
5971 // returned a row).
5972 let dummy_row = Row::new(Vec::new());
5973 if let Some(w) = &stmt.where_ {
5974 let cond = eval::eval_expr(w, &dummy_row, &ctx)?;
5975 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
5976 let columns: Vec<ColumnSchema> = projection
5977 .into_iter()
5978 .map(|p| {
5979 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
5980 c.user_enum_type = p.user_enum_type;
5981 c.collation_name = p.collation_name;
5982 c.mysql_fsp = p.mysql_fsp;
5983 c
5984 })
5985 .collect();
5986 return Ok(QueryResult::Rows {
5987 columns,
5988 rows: Vec::new(),
5989 });
5990 }
5991 }
5992 // v7.38 (read01, T15) — a top-level SRF that the parser did NOT rewrite
5993 // into a FROM item (regexp_matches, whose rows are arrays and so cannot
5994 // desugar to unnest) expands here: one output row per SRF row, sibling
5995 // scalar columns repeated. unnest / array_elements / path_query reach a
5996 // real FROM via the parser rewrite and never land here.
5997 // v7.39 (read01 round 67) — every SRF in the list, in lockstep.
5998 let srf_idxs = self.srf_target_idxs(&projection);
5999 if !srf_idxs.is_empty() {
6000 let mut rows = expand_srf_row(self, &projection, &srf_idxs, &dummy_row, &ctx)?;
6001 let columns: Vec<ColumnSchema> = projection
6002 .into_iter()
6003 .map(|p| {
6004 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
6005 c.user_enum_type = p.user_enum_type;
6006 c.collation_name = p.collation_name;
6007 c.mysql_fsp = p.mysql_fsp;
6008 c
6009 })
6010 .collect();
6011 // v7.39 (read01 round 80) — a FROM-less SELECT still has an ORDER BY,
6012 // an OFFSET and a LIMIT, and they apply to the rows the SRF expanded
6013 // to. This returned straight out of the expansion, so
6014 // `SELECT unnest(ARRAY['B','a','A','b']) ORDER BY 1` came back in
6015 // input order — the sort was not wrong, it never ran. (There is
6016 // exactly one conceptual input row here, which is why the ordinary
6017 // scan pipeline is not on this path at all.)
6018 if !stmt.order_by.is_empty() {
6019 let synth_ctx =
6020 EvalContext::new(&columns, None).with_catalog(self.active_catalog());
6021 let resolved: Vec<spg_sql::ast::OrderBy> = stmt
6022 .order_by
6023 .iter()
6024 .map(|o| {
6025 let mut o = o.clone();
6026 if let Expr::Literal(spg_sql::ast::Literal::Integer(n)) = &o.expr
6027 && *n >= 1
6028 && let Ok(idx) = usize::try_from(*n - 1)
6029 && idx < columns.len()
6030 {
6031 o.expr = Expr::Column(spg_sql::ast::ColumnName {
6032 qualifier: None,
6033 name: columns[idx].name.clone(),
6034 });
6035 }
6036 o
6037 })
6038 .collect();
6039 let descs: Vec<bool> = resolved.iter().map(|o| o.desc).collect();
6040 let mut tagged: Vec<(Vec<OrderKey>, Row)> = Vec::with_capacity(rows.len());
6041 for r in rows {
6042 let keys = build_order_keys(&resolved, &r, &synth_ctx)?;
6043 tagged.push((keys, r));
6044 }
6045 sort_by_keys(&mut tagged, &descs);
6046 rows = tagged.into_iter().map(|(_, r)| r).collect();
6047 }
6048 apply_offset_and_limit(&mut rows, stmt.offset_literal(), stmt.limit_literal());
6049 return Ok(QueryResult::Rows { columns, rows });
6050 }
6051 let mut values = Vec::with_capacity(projection.len());
6052 for p in &projection {
6053 values.push(eval::eval_expr(&p.expr, &dummy_row, &ctx)?);
6054 }
6055 let columns: Vec<ColumnSchema> = projection
6056 .into_iter()
6057 .map(|p| {
6058 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
6059 c.user_enum_type = p.user_enum_type;
6060 c.collation_name = p.collation_name;
6061 c.mysql_fsp = p.mysql_fsp;
6062 c
6063 })
6064 .collect();
6065 // v7.39 (round 239) — the FROM-less scalar path ignored LIMIT and
6066 // OFFSET entirely, so `SELECT 1 LIMIT 0` returned its row where PG
6067 // returns none. (The SRF and aggregate arms above already applied
6068 // them; this tail was the one that didn't.)
6069 let mut rows = alloc::vec![Row::new(values)];
6070 apply_offset_and_limit(&mut rows, stmt.offset_literal(), stmt.limit_literal());
6071 Ok(QueryResult::Rows { columns, rows })
6072 }
6073
6074 /// v7.37.x (docker-fair INSUBQ attack) — pre-replacement short-
6075 /// circuit. Catches
6076 /// SELECT COUNT(*) FROM A WHERE A.pk IN (<uncorrelated subquery>)
6077 /// BEFORE `resolve_select_subqueries` materialises the inner result
6078 /// as `Vec<Expr::Literal>`. Runs the inner once, collects the
6079 /// values into a `HashSet<i64>` directly, then probes A.pk per
6080 /// HashSet entry and tallies. Saves the Expr-literal roundtrip
6081 /// (~150 µs / query at INSUBQ benchmark scale).
6082 pub(crate) fn try_count_star_pk_in_subquery_fast(
6083 &self,
6084 stmt: &SelectStatement,
6085 cancel: CancelToken<'_>,
6086 ) -> Result<Option<QueryResult>, EngineError> {
6087 use spg_sql::ast::SelectItem;
6088 if stmt.distinct
6089 || stmt.limit_with_ties
6090 || stmt.group_by.is_some()
6091 || stmt.having.is_some()
6092 || !stmt.unions.is_empty()
6093 || !stmt.order_by.is_empty()
6094 || stmt.limit.is_some()
6095 || stmt.offset.is_some()
6096 || stmt.items.len() != 1
6097 {
6098 return Ok(None);
6099 }
6100 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
6101 return Ok(None);
6102 };
6103 let is_count_star = matches!(expr, Expr::FunctionCall { name, args }
6104 if name.eq_ignore_ascii_case("count_star") && args.is_empty());
6105 if !is_count_star {
6106 return Ok(None);
6107 }
6108 let Some(from) = stmt.from.as_ref() else {
6109 return Ok(None);
6110 };
6111 if !from.joins.is_empty()
6112 || from.primary.lateral_subquery.is_some()
6113 || from.primary.unnest_expr.is_some()
6114 || from.primary.generate_series_args.is_some()
6115 || from.primary.table_fn_call.is_some()
6116 || from.primary.as_of_segment.is_some()
6117 {
6118 return Ok(None);
6119 }
6120 let Some(where_expr) = stmt.where_.as_ref() else {
6121 return Ok(None);
6122 };
6123 // The WHERE conjunct must be a bare `<col> IN (subquery)` with
6124 // negated=false; no other predicates.
6125 let Expr::InSubquery {
6126 expr: col_expr,
6127 subquery,
6128 negated: false,
6129 } = where_expr
6130 else {
6131 return Ok(None);
6132 };
6133 let Expr::Column(c) = col_expr.as_ref() else {
6134 return Ok(None);
6135 };
6136 let outer_alias = from
6137 .primary
6138 .alias
6139 .as_deref()
6140 .unwrap_or(from.primary.name.as_str());
6141 if let Some(q) = c.qualifier.as_deref()
6142 && !q.eq_ignore_ascii_case(outer_alias)
6143 {
6144 return Ok(None);
6145 }
6146 // Outer column must be a single-column PK on integer family.
6147 let catalog = self.active_catalog();
6148 let Some(outer_table) = catalog.get(from.primary.name.as_str()) else {
6149 return Ok(None);
6150 };
6151 let outer_schema = outer_table.schema();
6152 let Some(outer_pos) = outer_schema
6153 .columns
6154 .iter()
6155 .position(|s| s.name.eq_ignore_ascii_case(&c.name))
6156 else {
6157 return Ok(None);
6158 };
6159 if !matches!(
6160 outer_schema.columns[outer_pos].ty,
6161 spg_storage::DataType::BigInt
6162 | spg_storage::DataType::Int
6163 | spg_storage::DataType::SmallInt
6164 ) {
6165 return Ok(None);
6166 }
6167 if !outer_schema
6168 .uniqueness_constraints
6169 .iter()
6170 .any(|u| u.is_primary_key && u.columns.as_slice() == [outer_pos])
6171 {
6172 return Ok(None);
6173 }
6174 let Some(idx) = outer_table.index_on(outer_pos) else {
6175 return Ok(None);
6176 };
6177 // Inner must be uncorrelated. The cheap-correlation pre-check
6178 // exists upstream; here we just attempt the bare exec.
6179 if crate::subquery::select_is_correlated(subquery) {
6180 return Ok(None);
6181 }
6182 let mut inner = (**subquery).clone();
6183 self.resolve_select_subqueries(&mut inner, cancel)?;
6184 let r = match self.exec_bare_select_cancel(&inner, cancel) {
6185 Ok(r) => r,
6186 Err(_) => return Ok(None),
6187 };
6188 let QueryResult::Rows { columns, rows, .. } = r else {
6189 return Ok(None);
6190 };
6191 if columns.len() != 1 {
6192 return Ok(None);
6193 }
6194 // v7.37.43 (INSUBQ B-1) — inner-uniqueness check. If the inner
6195 // subquery projects a column known to be UNIQUE/PK on its table
6196 // (statically: `SELECT <col> FROM <tbl> WHERE …` where <col> is
6197 // in `tbl.uniqueness_constraints`), survivor values are
6198 // guaranteed distinct and the per-survivor `HashSet::insert`
6199 // dedup check is redundant. ~25 ns × N_inner-survivors saved.
6200 //
6201 // Inlined check — gated on: no DISTINCT/GROUP/UNION/JOIN, single
6202 // projection that is a bare Column ref, table-column lookup in
6203 // catalog confirms the column appears as a unique constraint's
6204 // sole member. UNIQUE NOT NULL is required — a nullable unique
6205 // column may have multiple NULLs, but NULLs are already skipped
6206 // above (`Value::Null => continue`), so a UNIQUE-only column is
6207 // still safe to dedup-skip.
6208 let inner_unique = (|| -> bool {
6209 if inner.distinct
6210 || inner.group_by.is_some()
6211 || !inner.unions.is_empty()
6212 || inner.having.is_some()
6213 || inner.items.len() != 1
6214 {
6215 return false;
6216 }
6217 let Some(inner_from) = inner.from.as_ref() else {
6218 return false;
6219 };
6220 if !inner_from.joins.is_empty()
6221 || inner_from.primary.lateral_subquery.is_some()
6222 || inner_from.primary.unnest_expr.is_some()
6223 || inner_from.primary.generate_series_args.is_some()
6224 || inner_from.primary.table_fn_call.is_some()
6225 {
6226 return false;
6227 }
6228 let SelectItem::Expr { expr: proj, .. } = &inner.items[0] else {
6229 return false;
6230 };
6231 let Expr::Column(pc) = proj else {
6232 return false;
6233 };
6234 let inner_alias = inner_from
6235 .primary
6236 .alias
6237 .as_deref()
6238 .unwrap_or(inner_from.primary.name.as_str());
6239 if let Some(q) = pc.qualifier.as_deref()
6240 && !q.eq_ignore_ascii_case(inner_alias)
6241 {
6242 return false;
6243 }
6244 let Some(inner_table) = catalog.get(inner_from.primary.name.as_str()) else {
6245 return false;
6246 };
6247 let isch = inner_table.schema();
6248 let Some(ipos) = isch
6249 .columns
6250 .iter()
6251 .position(|s| s.name.eq_ignore_ascii_case(&pc.name))
6252 else {
6253 return false;
6254 };
6255 isch.uniqueness_constraints
6256 .iter()
6257 .any(|u| u.columns.as_slice() == [ipos])
6258 })();
6259 // Collect inner i64 values directly into a HashSet, then probe.
6260 let mut count: i64 = 0;
6261 let mut probed = if inner_unique {
6262 hashbrown::HashSet::<i64>::new()
6263 } else {
6264 hashbrown::HashSet::<i64>::with_capacity(rows.len())
6265 };
6266 for row in &rows {
6267 let v = row.values.first().cloned().unwrap_or(Value::Null);
6268 let n = match v {
6269 Value::BigInt(n) => n,
6270 Value::Int(n) => i64::from(n),
6271 Value::SmallInt(n) => i64::from(n),
6272 Value::Null => continue,
6273 _ => return Ok(None),
6274 };
6275 // De-duplicate inner key set so a duplicate inner value
6276 // doesn't double-count the same outer row. Skipped when
6277 // the inner projection is statically unique.
6278 if !inner_unique && !probed.insert(n) {
6279 continue;
6280 }
6281 // v7.37.43 (INSUBQ B-2 + B-4) — direct i64 PK probe, skipping
6282 // the `IndexKey::from_value` enum-dispatch and the per-call
6283 // `IndexKey` wrapper construction. The outer column is
6284 // already gated to integer-family above, so an i64 key
6285 // always corresponds to a valid PK lookup.
6286 if !idx.lookup_eq_i64(n).is_empty() {
6287 count += 1;
6288 }
6289 }
6290 let columns_out = alloc::vec![ColumnSchema::new(
6291 "count".to_string(),
6292 spg_storage::DataType::BigInt,
6293 false,
6294 )];
6295 let rows_out = alloc::vec![Row::new(alloc::vec![Value::BigInt(count)])];
6296 Ok(Some(QueryResult::Rows {
6297 columns: columns_out,
6298 rows: rows_out,
6299 }))
6300 }
6301
6302 /// v7.37.x (docker-fair INSUBQ attack) — short-circuit
6303 /// SELECT COUNT(*) FROM A WHERE A.pk IN (literal list)
6304 /// (the post-subquery-replacement shape of the INSUBQ probe
6305 /// `SELECT COUNT(*) FROM A WHERE A.pk IN (SELECT k FROM B WHERE …)`).
6306 /// The general aggregate path materialises every seeked row into
6307 /// a `Vec<Cow<Row>>`, then runs the aggregate executor over it.
6308 /// For COUNT(*) we only care how many keys hit; iterate the list
6309 /// and tally `idx.lookup_eq(key)` non-empty results, skipping the
6310 /// row materialisation, the aggregate state machine, and the per-
6311 /// row WHERE re-eval (the seek already filtered by the same list).
6312 /// Returns `None` when the shape doesn't match.
6313 fn try_count_star_pk_in_list_fast(
6314 &self,
6315 stmt: &SelectStatement,
6316 table: &spg_storage::Table,
6317 schema_cols: &[ColumnSchema],
6318 alias: &str,
6319 ) -> Option<QueryResult> {
6320 use spg_sql::ast::{ColumnName, SelectItem};
6321 // Gates on the SELECT shape.
6322 if stmt.distinct
6323 || stmt.limit_with_ties
6324 || stmt.group_by.is_some()
6325 || stmt.having.is_some()
6326 || !stmt.unions.is_empty()
6327 || !stmt.order_by.is_empty()
6328 || stmt.limit.is_some()
6329 || stmt.offset.is_some()
6330 || stmt.items.len() != 1
6331 {
6332 return None;
6333 }
6334 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
6335 return None;
6336 };
6337 let is_count_star = matches!(expr, Expr::FunctionCall { name, args }
6338 if name.eq_ignore_ascii_case("count_star") && args.is_empty());
6339 if !is_count_star {
6340 return None;
6341 }
6342 // WHERE must be `<col> IN (literal list)` with no other
6343 // conjuncts (the seek result is a true subset of the row
6344 // population for this predicate).
6345 let where_expr = stmt.where_.as_ref()?;
6346 let Expr::InList {
6347 expr: col_expr,
6348 list,
6349 negated: false,
6350 } = where_expr
6351 else {
6352 return None;
6353 };
6354 let Expr::Column(c) = col_expr.as_ref() else {
6355 return None;
6356 };
6357 if let Some(q) = c.qualifier.as_deref()
6358 && !q.eq_ignore_ascii_case(alias)
6359 {
6360 return None;
6361 }
6362 let col_pos = schema_cols
6363 .iter()
6364 .position(|s| s.name.eq_ignore_ascii_case(&c.name))?;
6365 // The column must be a single-column PK on an integer family
6366 // — the same gate the SCALARSQ + LEFT-ANTI-JOIN fast paths use,
6367 // so the antiset stays collision-free under `HashSet<i64>`.
6368 let schema = table.schema();
6369 if !matches!(
6370 schema.columns[col_pos].ty,
6371 spg_storage::DataType::BigInt
6372 | spg_storage::DataType::Int
6373 | spg_storage::DataType::SmallInt
6374 ) {
6375 return None;
6376 }
6377 if !schema
6378 .uniqueness_constraints
6379 .iter()
6380 .any(|u| u.is_primary_key && u.columns.as_slice() == [col_pos])
6381 {
6382 return None;
6383 }
6384 let idx = table.index_on(col_pos)?;
6385 // Tally non-empty seek results across all literal values.
6386 let mut count: i64 = 0;
6387 for lit in list {
6388 let Expr::Literal(l) = lit else {
6389 return None;
6390 };
6391 // r1039 — through the shared resolver, so a literal spelled
6392 // in another type ('5' against an integer PK) is read as the
6393 // column's before it becomes a key. This tally answers from
6394 // the index alone, so a key in the wrong space would return a
6395 // COUNT of zero rather than fall back to a scan.
6396 let col = schema.columns.get(col_pos)?;
6397 let v = crate::index_access::literal_as_column_value(l, col, col_pos)?;
6398 let key = spg_storage::IndexKey::from_value_for_column(&v, col.ty)?;
6399 if !idx.lookup_eq(&key).is_empty() {
6400 count += 1;
6401 }
6402 }
6403 let columns = alloc::vec![ColumnSchema::new(
6404 "count".to_string(),
6405 spg_storage::DataType::BigInt,
6406 false,
6407 )];
6408 let rows = alloc::vec![Row::new(alloc::vec![Value::BigInt(count)])];
6409 let _ = ColumnName {
6410 qualifier: None,
6411 name: String::new(),
6412 };
6413 Some(QueryResult::Rows { columns, rows })
6414 }
6415
6416 /// v7.38 (perf, exact-range count) — `SELECT count(*) FROM t WHERE <col>
6417 /// BETWEEN a AND b` on an indexed column. The index range walk yields
6418 /// exactly the matching (visible) rows, so we count locators directly —
6419 /// skipping the row materialisation, the aggregate state machine, and the
6420 /// per-row WHERE re-eval the general path pays. Turns the `range_count`
6421 /// endpoint from tied-with-PG (superset re-eval) into a clear win. None
6422 /// when the shape doesn't match.
6423 fn try_count_star_indexed_range_fast(
6424 &self,
6425 stmt: &SelectStatement,
6426 table: &spg_storage::Table,
6427 schema_cols: &[ColumnSchema],
6428 alias: &str,
6429 snapshot: &spg_storage::snapshot::Snapshot,
6430 ) -> Option<QueryResult> {
6431 use spg_sql::ast::SelectItem;
6432 if stmt.distinct
6433 || stmt.limit_with_ties
6434 || stmt.group_by.is_some()
6435 || stmt.having.is_some()
6436 || !stmt.unions.is_empty()
6437 || !stmt.order_by.is_empty()
6438 || stmt.limit.is_some()
6439 || stmt.offset.is_some()
6440 || stmt.items.len() != 1
6441 {
6442 return None;
6443 }
6444 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
6445 return None;
6446 };
6447 let is_count_star = matches!(expr, Expr::FunctionCall { name, args }
6448 if name.eq_ignore_ascii_case("count_star") && args.is_empty());
6449 if !is_count_star {
6450 return None;
6451 }
6452 let where_expr = stmt.where_.as_ref()?;
6453 let count =
6454 crate::index_access::try_range_count(where_expr, schema_cols, table, alias, snapshot)?;
6455 let columns = alloc::vec![ColumnSchema::new(
6456 "count".to_string(),
6457 spg_storage::DataType::BigInt,
6458 false,
6459 )];
6460 let rows = alloc::vec![Row::new(alloc::vec![Value::BigInt(count)])];
6461 Some(QueryResult::Rows { columns, rows })
6462 }
6463
6464 /// Single-table aggregate path: filter the (optionally index-seeked)
6465 /// rows, then hand off to the aggregate executor which does its own
6466 /// projection + ORDER BY before `finish_agg_result` applies LIMIT.
6467 fn run_single_table_aggregate<'a>(
6468 &self,
6469 stmt: &SelectStatement,
6470 table: &'a spg_storage::Table,
6471 schema_cols: &'a [ColumnSchema],
6472 alias: &str,
6473 indexed_rows: Option<Vec<Cow<'a, Row<'static>>>>,
6474 cancel: CancelToken<'_>,
6475 ) -> Result<QueryResult, EngineError> {
6476 // v7.38 (read01 U15) — per-scan sampler cell for TABLESAMPLE
6477 // REPEATABLE (see run_single_table_scan). Aggregates
6478 // (`count(*) FROM t TABLESAMPLE …`) filter through this ctx too.
6479 let sample_cell: core::cell::Cell<Option<u64>> = core::cell::Cell::new(None);
6480 let ctx = self
6481 .ev_ctx(schema_cols, Some(alias))
6482 .with_sample_rng(&sample_cell);
6483 // v7.39 (round 657) — pre-sized. Pushing 500k pointers into a
6484 // `Vec::new()` walks the doubling chain 8, 16, … 262144, 524288,
6485 // and every abandoned buffer on the way stays resident: RSS is a
6486 // high-water mark, so the intermediates are paid for even though
6487 // they are freed. Round 656 measured the scan at 17 bytes/row
6488 // where the survivor list itself only needs 8.
6489 let mut filtered: Vec<&Row<'static>> = if stmt.where_.is_none() {
6490 Vec::with_capacity(table.rows().len())
6491 } else {
6492 // With a WHERE, the row count is an UPPER bound and reserving it
6493 // is the worse trade: `… WHERE id = 5` over 50M rows would take
6494 // 400 MB of pointers to hold one survivor. Let it grow.
6495 Vec::new()
6496 };
6497 // v6.2.6 — Memoize: per-query LRU cache for correlated
6498 // scalar subqueries. Fresh per row-loop entry so each
6499 // SELECT execution gets an isolated cache.
6500 let mut memo = memoize::MemoizeCache::new();
6501 // v7.37 (perf) — single-table aggregate's WHERE filter
6502 // pre-7.37 ran the slow tree-walker (`eval_expr_with_
6503 // correlated`) per row, even for subquery-free WHEREs that
6504 // the single-table SCAN path has compiled since v7.32
6505 // (perf knife D). The asymmetry meant a fold-to-filter
6506 // rewrite (joinfold) that swapped a JOIN for a single-table
6507 // aggregate over a compiled WHERE saw the tree-walker
6508 // instead — 25 k rows × `m.mailbox_id IN (25 lits)` cost
6509 // ~9 ms via the walker, vs ~1 ms via the compiled InSet
6510 // step. Compile once if eligible; fall back to the walker
6511 // for subquery-bearing or non-compilable WHEREs.
6512 let compiled_where: Option<eval::CompiledExpr> = stmt
6513 .where_
6514 .as_ref()
6515 .filter(|w| eval::fully_compilable(w))
6516 .map(|w| {
6517 // v7.38.8 — the scan filter runs the cheap half of its
6518 // conjunction first. Called from HERE and not from
6519 // `eval::compiled`, deliberately: the row loop lives in
6520 // that file, and adding a function to it cost this
6521 // query 11 % through layout alone while doing no work
6522 // for it. See `crate::qualorder`.
6523 match crate::qualorder::reordered(w) {
6524 Some(r) => eval::compile_expr(&r, &ctx),
6525 None => eval::compile_expr(w, &ctx),
6526 }
6527 });
6528 let mut eval_stack: Vec<Value<'static>> = Vec::new();
6529 let mut row_passes_where = |row: &Row<'static>,
6530 eval_stack: &mut Vec<Value<'static>>,
6531 memo: &mut memoize::MemoizeCache|
6532 -> Result<bool, EngineError> {
6533 match (&compiled_where, &stmt.where_) {
6534 (Some(cw), _) => {
6535 // v7.39 (round 479) — the predicate wants a bool, not a
6536 // Value. The owned entry ended in `Value::into_owned`
6537 // and the caller then dropped it, once per row; round
6538 // 478's profile put that pair above the comparison
6539 // itself.
6540 Ok(eval::compiled::eval_compiled_pred(
6541 cw,
6542 row,
6543 &ctx,
6544 eval_stack,
6545 ctx.mysql_dialect,
6546 )
6547 .map_err(EngineError::Eval)?)
6548 }
6549 (None, Some(w)) => {
6550 let cond = self.eval_expr_with_correlated(w, row, &ctx, cancel, Some(memo))?;
6551 Ok(crate::eval::predicate_is_true(
6552 &cond,
6553 "WHERE",
6554 ctx.mysql_dialect,
6555 )?)
6556 }
6557 (None, None) => Ok(true),
6558 }
6559 };
6560 if let Some(rows) = &indexed_rows {
6561 for cow in rows {
6562 let row = cow.as_ref();
6563 if !row_passes_where(row, &mut eval_stack, &mut memo)? {
6564 continue;
6565 }
6566 filtered.push(row);
6567 }
6568 }
6569 // v7.36 (cold-tier coverage) — single-table aggregate's
6570 // non-indexed full scan was hot-only and silently lost cold
6571 // rows on COUNT/SUM/etc. Materialise cold rows once into
6572 // `cold_rows_storage` (Vec<Row<'static>>) so the `filtered: Vec<&Row<'static>>`
6573 // shape stays unchanged; the cold rows live until the end of
6574 // the aggregate run.
6575 let cold_rows_storage = if indexed_rows.is_none() {
6576 self.iter_cold_rows_of_table(table)
6577 } else {
6578 Vec::new()
6579 };
6580 if indexed_rows.is_none() {
6581 // v7.37.15 (Phase C.3, step 2) — MVCC visibility gate for the
6582 // single-table aggregate full-scan path. Mirrors the gate on
6583 // `run_single_table_scan`: this is a user-query result path,
6584 // so under gate-on (`SPG_MVCC_INPLACE`) it must skip rows the
6585 // reader's snapshot cannot see (e.g. tombstoned versions),
6586 // otherwise COUNT/SUM/etc. would tally dead rows. A no-op
6587 // under the default gate-off: every hot row is frozen or
6588 // committed-and-alive, so `is_row_visible` returns true.
6589 // Cold-tier rows are frozen (visible) by definition — left
6590 // ungated, matching the plain-scan path.
6591 let scan_snapshot = self.current_snapshot();
6592 // v7.39 (pg_stat knife B) — this full-scan branch walks
6593 // headers directly (serial and sharded alike); count the
6594 // sequential scan here.
6595 table.note_seq_scan();
6596 // v7.39 (parallel-agg P2) — the visibility probe + WHERE
6597 // filter dominate the pre-aggregate wall time on big
6598 // scans (P1's ground truth: accumulation is only ~17%).
6599 // Shard THAT work when the host injected an executor and
6600 // the WHERE is compiled (the compiled evaluator is pure
6601 // over &row; the tree-walker fallback can hit correlated
6602 // subqueries and stays serial). Shards return surviving
6603 // ROW INDICES — &Row can't cross the Box<dyn Any>'s
6604 // 'static bound — and the main thread only dereferences.
6605 let n = table.row_count();
6606 let par = self.parallel_runner.0.as_deref().filter(|_| {
6607 n >= crate::PARALLEL_MIN_ROWS && (stmt.where_.is_none() || compiled_where.is_some())
6608 });
6609 if let Some(r) = par {
6610 let n_shards = (n / crate::PARALLEL_MIN_ROWS).clamp(2, 8);
6611 let chunk = n.div_ceil(n_shards);
6612 type ShardOut = Result<alloc::vec::Vec<usize>, EngineError>;
6613 let cw = &compiled_where;
6614 let snap_ref = &scan_snapshot;
6615 let results = r.run_shards(n_shards, &|s| {
6616 let lo = s * chunk;
6617 let hi = ((s + 1) * chunk).min(n);
6618 let mut keep: alloc::vec::Vec<usize> = alloc::vec::Vec::with_capacity(hi - lo);
6619 // EvalContext carries Cells (sampler / row counters)
6620 // and is !Sync — each shard builds its own from the
6621 // same Sync inputs. The compiled WHERE is gated to
6622 // the pure-scalar whitelist, which reads none of the
6623 // session state the engine-built ctx would add
6624 // (TABLESAMPLE's __tsm_fract is not whitelisted, so
6625 // sampled scans never take this branch).
6626 let shard_ctx = EvalContext::new(schema_cols, Some(alias));
6627 let mut stack: Vec<Value<'static>> = Vec::new();
6628 let out: ShardOut = (|| {
6629 for i in lo..hi {
6630 if !table.is_row_visible(i, snap_ref) {
6631 continue;
6632 }
6633 let row = &table.rows()[i];
6634 // v7.39 (round 480) — the parallel full-scan
6635 // shard is the path the aggregate benchmark
6636 // actually takes, and it was still on the OWNED
6637 // entry: round 480's profile attributed 68.7 %
6638 // of `drop_glue<Value>` to this closure, which
6639 // is why round 479's fix to the indexed path
6640 // barely moved the total.
6641 //
6642 // The `matches!(…, Value::Bool(true))` form was
6643 // also a narrower reading than the rest of the
6644 // engine uses — `predicate_is_true` is what
6645 // handles NULL and MySQL truthiness — so the
6646 // bool entry fixes the shape as well as the cost.
6647 let pass = match cw {
6648 Some(c) => eval::compiled::eval_compiled_pred(
6649 c,
6650 row,
6651 &shard_ctx,
6652 &mut stack,
6653 shard_ctx.mysql_dialect,
6654 )
6655 .map_err(EngineError::Eval)?,
6656 None => true,
6657 };
6658 if pass {
6659 keep.push(i);
6660 }
6661 }
6662 Ok(keep)
6663 })();
6664 alloc::boxed::Box::new(out)
6665 });
6666 // v7.39 (round 567) — `rows()` is a 32-way trie, so
6667 // indexing it is four dependent loads and a scan that
6668 // reads every row paid them every row. A profile of
6669 // `SELECT sum(id)` over 500k rows put 37.8% of the
6670 // connection thread's CPU on THIS ONE LINE. The cursor
6671 // holds the leaf, making that one descent per 32.
6672 let mut rows_cur = table.rows().run_cursor();
6673 for boxed in results {
6674 let shard = boxed
6675 .downcast::<ShardOut>()
6676 .expect("runner echoes the closure's box");
6677 for i in (*shard)? {
6678 if let Some(row) = rows_cur.get(i) {
6679 filtered.push(row);
6680 }
6681 }
6682 }
6683 } else {
6684 let mut rows_cur = table.rows().run_cursor();
6685 for i in 0..n {
6686 if !table.is_row_visible(i, &scan_snapshot) {
6687 continue;
6688 }
6689 let Some(row) = rows_cur.get(i) else { continue };
6690 if !row_passes_where(row, &mut eval_stack, &mut memo)? {
6691 continue;
6692 }
6693 filtered.push(row);
6694 }
6695 }
6696 for row in &cold_rows_storage {
6697 if !row_passes_where(row, &mut eval_stack, &mut memo)? {
6698 continue;
6699 }
6700 filtered.push(row);
6701 }
6702 }
6703 // v7.29 — a per-query memo so correlated scalar
6704 // subqueries batch-evaluate once (group map) instead of
6705 // executing per group.
6706 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
6707 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
6708 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
6709 .map_err(|err| match err {
6710 EngineError::Eval(ev) => ev,
6711 other => eval::EvalError::TypeMismatch {
6712 detail: alloc::format!("{other}"),
6713 },
6714 })
6715 };
6716 // v7.39 (round 656) — the plain relational scan. This collect() was
6717 // the measured defect: one 64-byte `RowRef` per surviving row to
6718 // wrap an 8-byte pointer `filtered` already holds. Scalar
6719 // aggregates measured ~81 bytes/row of working memory because of
6720 // it — 40 MB at 500k rows, 3.2 GB at 50M, for a query that returns
6721 // one number. `AggRows::Ptrs` reads the pointers directly.
6722 let agg = aggregate::run(
6723 stmt,
6724 crate::join::AggRows::Ptrs(&filtered),
6725 schema_cols,
6726 Some(alias),
6727 Some(&agg_correlated),
6728 self.parallel_runner.0.as_deref(),
6729 Some(self.active_catalog()),
6730 Some(self),
6731 )?;
6732 self.finish_agg_result(agg, stmt, cancel)
6733 }
6734
6735 /// Single-table scan + projection path: WHERE filter (compiled when
6736 /// subquery-free), ORDER BY keying, SRF expansion / projection, then
6737 /// sort + WITH TIES / DISTINCT / OFFSET-LIMIT.
6738 fn run_single_table_scan<'a>(
6739 &self,
6740 stmt: &SelectStatement,
6741 table: &'a spg_storage::Table,
6742 schema_cols: &'a [ColumnSchema],
6743 alias: &str,
6744 indexed_rows: Option<Vec<Cow<'a, Row<'static>>>>,
6745 cancel: CancelToken<'_>,
6746 ) -> Result<QueryResult, EngineError> {
6747 // v7.38 (read01 U15) — a fresh per-scan sampler cell for
6748 // `TABLESAMPLE … REPEATABLE(seed)`. Created before the ctx so the
6749 // deterministic `__tsm_fract(seed)` draws share one scan-local
6750 // state (isolated from the global random() PRNG); a fresh cell per
6751 // scan makes a repeat / rescan reproduce the same sample. Unused
6752 // and cheap when the query carries no sample.
6753 let sample_cell: core::cell::Cell<Option<u64>> = core::cell::Cell::new(None);
6754 let ctx = self
6755 .ev_ctx(schema_cols, Some(alias))
6756 .with_sample_rng(&sample_cell);
6757 let projection = build_projection(&stmt.items, schema_cols, alias, self.backslash_escapes)?;
6758 // v7.19 P5 — single-table SELECT path for SRF
6759 // `SELECT unnest(arr) FROM t` shape. Detect a top-level
6760 // unnest in the projection list. When present, the
6761 // per-row processor emits one output row per array
6762 // element (broadcasting non-SRF projections from the
6763 // same input row). Empty / NULL arrays emit zero rows
6764 // for that input — PG semantics.
6765 // v7.39 (read01 round 67) — every SRF in the target list, in lockstep.
6766 let srf_idxs = self.srf_target_idxs(&projection);
6767 let srf_position = srf_idxs.first().copied();
6768 // v7.39 (round 599) — the SRF analysis is per QUERY, not per row.
6769 let mut srf_plan = if srf_position.is_some() {
6770 Some(build_srf_plan(self, &projection, &srf_idxs, &ctx)?)
6771 } else {
6772 None
6773 };
6774
6775 // Materialise the filter pass into `(order_key, projected_row)`
6776 // tuples. The order key is `None` when there's no ORDER BY clause.
6777 let mut tagged: Vec<(Vec<OrderKey>, Row<'static>)> = Vec::new();
6778 // v7.33 (C1, ceiling-first/never-die) — charge each accumulated
6779 // output row to the per-query byte budget as it is built, so a
6780 // fat single-table scan / sort REJECTS with QueryBytesExceeded
6781 // at ~the ceiling instead of materialising the whole table and
6782 // only noticing at the final enforce_row_limit check. Without
6783 // this, N concurrent fat scans peak at N×table and OOM the host.
6784 // `max_query_bytes = None` (the embedded default) = no ceiling,
6785 // so existing unbudgeted behaviour is byte-identical.
6786 let mut budget = ByteBudget::new(self.max_query_bytes);
6787 // v6.2.6 — Memoize per-row WHERE eval shares one cache.
6788 let mut memo = memoize::MemoizeCache::new();
6789 // v7.32 (perf knife D) — subquery-free WHERE compiles once;
6790 // the row loop then runs a flat step program instead of a
6791 // tree interpretation per row.
6792 let compiled_where: Option<eval::CompiledExpr> = stmt
6793 .where_
6794 .as_ref()
6795 .filter(|w| eval::fully_compilable(w))
6796 .map(|w| {
6797 // v7.38.8 — the scan filter runs the cheap half of its
6798 // conjunction first. Called from HERE and not from
6799 // `eval::compiled`, deliberately: the row loop lives in
6800 // that file, and adding a function to it cost this
6801 // query 11 % through layout alone while doing no work
6802 // for it. See `crate::qualorder`.
6803 match crate::qualorder::reordered(w) {
6804 Some(r) => eval::compile_expr(&r, &ctx),
6805 None => eval::compile_expr(w, &ctx),
6806 }
6807 });
6808 let mut eval_stack: Vec<Value<'static>> = Vec::new();
6809 // v7.37.x (docker-fair SCALARSQ attack) — pre-analyse every
6810 // SELECT-item scalar subquery for the PK-probe fast path. The
6811 // analysis (gate checks + catalog lookups) takes ~500 ns; doing
6812 // it once per query instead of once per row × 100 rows saves
6813 // ~50 µs and lets the per-row evaluation reduce to a single
6814 // index probe + outer-column read.
6815 let scalarsq_fast: Vec<Option<crate::ScalarPkProbeFastPath>> = projection
6816 .iter()
6817 .map(|p| {
6818 if let Expr::ScalarSubquery(inner) = &p.expr {
6819 self.analyse_scalar_count_pk_eq_probe(inner, schema_cols, alias)
6820 } else {
6821 None
6822 }
6823 })
6824 .collect();
6825 let any_scalarsq_fast = scalarsq_fast.iter().any(Option::is_some);
6826 // v7.39 (round 487) — a projection item that is a bare column
6827 // reference binds its position ONCE per query.
6828 //
6829 // Per row it used to walk `eval_expr_with_correlated` (a memo
6830 // lookup for "does this have a subquery", then an un-memoised
6831 // `expr_may_use_in_set` tree walk), then `eval_expr`'s dispatch,
6832 // then `resolve_column`, which finds the column by scanning the
6833 // schema and comparing NAMES. On `SELECT g FROM h` that chain was
6834 // 19 % of self time for what is ultimately one cell read.
6835 //
6836 // `compile_column_pos` is the Step VM's resolver, already
6837 // `pub(crate)` and already reused by the aggregate's bind-once
6838 // path: it mirrors `resolve_column`'s happy layers and returns
6839 // None for anything that would reach an error, an ambiguity, or a
6840 // miss, so those still go the interpreter's way and keep its
6841 // exact message. A composite column is excluded for the same
6842 // reason `compile_into` excludes it — it must be rehydrated from
6843 // stored JSON, which is not a cell read.
6844 let proj_direct = bind_direct_columns(&projection, &ctx);
6845 let any_proj_direct = proj_direct.iter().any(Option::is_some);
6846 // v7.39 (round 605) — a projection item that cannot depend on the row
6847 // is evaluated once. `SELECT ('{"a":1}')::JSONB FROM j` cost TEN
6848 // allocations a row against one for a plain column, `'abc' || 'def'`
6849 // six and `upper('abc')` five, all of them producing the same value
6850 // 50,000 times. An item that fails to evaluate is left alone, so its
6851 // error still comes from the row loop in the interpreter's wording.
6852 let proj_const: Vec<Option<Value<'static>>> = projection
6853 .iter()
6854 .map(|p| crate::eval::compiled::constant_projection_value(&p.expr, &ctx))
6855 .collect();
6856 let any_proj_const = proj_const.iter().any(Option::is_some);
6857 crate::bump_counter!(crate::select::SCAN_PATH_ENTERED);
6858 // v7.39 (read01 round 80) — positional ORDER BY over a WILDCARD
6859 // projection. Statement prep (`resolve_order_by_position`) can only map
6860 // `ORDER BY 1` onto the first SELECT item when that item is an
6861 // expression; a `*` is not one, so the literal survived to here and was
6862 // evaluated as the CONSTANT 1 — the same key for every row, i.e. no sort
6863 // at all. The parser rewrites `SELECT unnest(a) x` into
6864 // `SELECT * FROM unnest(a) x`, so that innocuous-looking shape landed
6865 // exactly here: `SELECT unnest(ARRAY['B','a','A','b']) ORDER BY 1` came
6866 // back in input order. The projection is built by now, so the Nth output
6867 // column is known — resolve against it.
6868 let order_by = resolve_positional_order_by(&stmt.order_by, &projection);
6869 // v7.39 (round 600) — the ORDER BY of an SRF query is decided on the
6870 // EXPANDED rows, so a key naming a select-list item reads that item.
6871 let srf_order_cols: Vec<Option<usize>> = if srf_position.is_some() {
6872 srf_order_output_cols(&order_by, &projection)
6873 } else {
6874 Vec::new()
6875 };
6876 let srf_key_bound: Vec<Option<usize>> = (0..order_by.len()).map(Some).collect();
6877 // v7.37.x (docker-fair SCALARSQ attack) — early-limit gate for
6878 // the no-ORDER-BY-no-DISTINCT-no-TIES-no-SRF-no-WHERE shape.
6879 // Hoisted above the closure so the projection-eval path can
6880 // gate `memo` passing on it: the SELECT-item correlated-scalar
6881 // batch path scans the FULL inner table once (~5 ms for 12.5 k
6882 // rows) and is only a win when N outer rows is large; for small
6883 // LIMITed shapes a per-row PK seek (~5 µs × 100 = 500 µs) wins.
6884 let early_cap: Option<usize> = if order_by.is_empty()
6885 && !stmt.distinct
6886 && !stmt.limit_with_ties
6887 && srf_position.is_none()
6888 && stmt.where_.is_none()
6889 {
6890 stmt.limit_literal()
6891 .map(|n| n.saturating_add(stmt.offset_literal().unwrap_or(0)) as usize)
6892 } else {
6893 None
6894 };
6895 // v7.38 (read01 B8) — streaming top-N budget. For `ORDER BY …
6896 // LIMIT k` (no DISTINCT / WITH TIES / SRF, and not forced to
6897 // full-sort by the test gate) keep only the running top-`keep`
6898 // rows in memory instead of materialising every projected row,
6899 // so a `… ORDER BY col LIMIT 10` over a huge table is O(keep)
6900 // space, not O(rows). `None` = accumulate everything (the prior
6901 // behaviour). The final `partial_sort_tagged(keep)` below still
6902 // runs and produces the identical rows.
6903 // v7.39 (round 683) — the declared collation for each ORDER BY
6904 // position, resolved once and carried beside `descs` for the same
6905 // reason `descs` is carried: it is per key position, not per row.
6906 let order_colls = crate::orderby::order_by_collations(&order_by, &ctx)?;
6907 let topk_stream: Option<(usize, Vec<bool>)> = if !order_by.is_empty()
6908 && !stmt.distinct
6909 && !stmt.limit_with_ties
6910 && srf_position.is_none()
6911 && !self.env_cfg().disable_topk
6912 {
6913 stmt.limit_literal().and_then(|l| {
6914 let keep = (l as usize).saturating_add(stmt.offset_literal().unwrap_or(0) as usize);
6915 (keep >= 1).then(|| (keep, order_by.iter().map(|o| o.desc).collect()))
6916 })
6917 } else {
6918 None
6919 };
6920 // v7.37.16 — streaming DISTINCT seen-set: norm-hash → indices of
6921 // kept rows in `tagged`. Probing on the PROJECTED row as soon as
6922 // it is built means a duplicate costs neither a build_order_keys
6923 // eval (the dominant per-row cost of `DISTINCT … ORDER BY`) nor
6924 // a tagged slot, and the sort below runs over u survivors, not
6925 // n input rows — PG's hash-distinct-then-sort plan shape.
6926 let mut seen_distinct: hashbrown::HashMap<u64, crate::distinct::DistinctBucket> =
6927 hashbrown::HashMap::new();
6928 let distinct_hb = hashbrown::DefaultHashBuilder::default();
6929 // v7.39 (round 485) — one projection buffer for the whole scan
6930 // rather than a fresh `Vec` per input row. A row that survives
6931 // the DISTINCT probe takes the buffer with it (`mem::take`) and
6932 // the next row allocates a new one; a row that duplicates an
6933 // earlier one leaves the buffer — and its capacity — in place.
6934 // The round-485 counter says 49 900 of `distinct_proj`'s 50 000
6935 // projected rows are duplicates, so that is 49 900 allocate /
6936 // free pairs the scan no longer performs. Shapes where every row
6937 // survives (plain projection, `DISTINCT` over a unique column)
6938 // allocate exactly as often as before.
6939 let mut proj_buf: Vec<Value<'static>> = Vec::new();
6940 // v7.39 (round 571) — buffers handed back by the top-N trim.
6941 // Round 485 made the scan share ONE projection buffer, but a
6942 // surviving row takes it (`mem::take`) and without DISTINCT
6943 // almost every row survives, so the next one starts from zero
6944 // capacity and allocates. The trim drops `keep` rows at a time
6945 // and their buffers come back here instead of being freed.
6946 let mut proj_pool: Vec<Vec<Value<'static>>> = Vec::new();
6947 let mut key_pool: Vec<Vec<crate::orderby::OrderKey>> = Vec::new();
6948 // v7.39 (round 581) — the worst row the accumulator is currently
6949 // keeping. Anything that loses to it cannot reach the answer, so
6950 // it is dropped before its projection is ever built.
6951 let mut topk_boundary: Option<Vec<crate::orderby::OrderKey>> = None;
6952 // v7.39 (round 582) — resolve each ORDER BY column once, not
6953 // once per row. See `order_by_bound_positions`.
6954 let order_bound =
6955 crate::orderby::order_by_bound_positions(&order_by, schema_cols, Some(alias));
6956 // v7.39 (round 581) — and it stops asking when the answer is
6957 // always "keep".
6958 //
6959 // The check earns its place only on rows it rejects. Over
6960 // ascending ids, `ORDER BY id DESC` never rejects one — every
6961 // row beats the current worst — so the comparison is pure
6962 // overhead there, measured at +5.5% in three batches out of
6963 // three. After a window of rows it looks at what it has
6964 // actually rejected and switches itself off if the shape is not
6965 // paying. The answers do not depend on it either way.
6966 const BOUNDARY_WINDOW: u32 = 8192;
6967 let mut boundary_checks: u32 = 0;
6968 let mut boundary_rejects: u32 = 0;
6969 let mut boundary_check_on = true;
6970 // Inline the per-row work in a closure so the indexed and full-
6971 // scan branches share the body.
6972 let mut process_row = |row: &Row<'static>, loop_idx: usize| -> Result<(), EngineError> {
6973 if loop_idx.is_multiple_of(256) {
6974 cancel.check()?;
6975 }
6976 if let Some(cw) = &compiled_where {
6977 let cond = eval::eval_compiled(cw, row, &ctx, &mut eval_stack)
6978 .map_err(EngineError::Eval)?;
6979 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
6980 return Ok(());
6981 }
6982 } else if let Some(where_expr) = &stmt.where_ {
6983 let cond =
6984 self.eval_expr_with_correlated(where_expr, row, &ctx, cancel, Some(&mut memo))?;
6985 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
6986 return Ok(());
6987 }
6988 }
6989 // Under DISTINCT the keys are built AFTER the dup probe
6990 // (survivors only); the non-distinct order is unchanged.
6991 // v7.39 (round 600) — an SRF query's keys are built per EXPANDED
6992 // row further down, and building them here would evaluate the
6993 // ORDER BY against the INPUT row: a key naming the SRF's own
6994 // output became a scalar call to it, which is where
6995 // "function unnest(integer[]) does not exist" came from.
6996 let order_keys = if order_by.is_empty() || stmt.distinct || srf_position.is_some() {
6997 Vec::new()
6998 } else {
6999 let mut buf = key_pool.pop().unwrap_or_default();
7000 crate::orderby::build_order_keys_bound(
7001 &order_by,
7002 &order_bound,
7003 row,
7004 &ctx,
7005 &mut buf,
7006 )?;
7007 // v7.39 (round 581) — reject before projecting.
7008 //
7009 // `ORDER BY g DESC, id DESC LIMIT 10` over 500k rows with
7010 // 50 distinct `g` decides nearly every row on the FIRST
7011 // key, and PG answers it FASTER than the single-key form
7012 // (7.4 ms against 10.4) because a rejected row costs it
7013 // one comparison. SPG built both keys AND the projected
7014 // row for all 500k before throwing them away. The keys
7015 // are needed to compare; the projection is not.
7016 if boundary_check_on
7017 && let Some((_, descs)) = &topk_stream
7018 && let Some(b) = &topk_boundary
7019 {
7020 boundary_checks += 1;
7021 let loses = crate::orderby::cmp_multi_key_in(&buf, b, descs, &order_colls)
7022 == core::cmp::Ordering::Greater;
7023 if loses {
7024 boundary_rejects += 1;
7025 }
7026 if boundary_checks == BOUNDARY_WINDOW {
7027 // Keep asking only if it has been rejecting at
7028 // least a quarter of what it saw.
7029 boundary_check_on = boundary_rejects.saturating_mul(4) >= boundary_checks;
7030 }
7031 if loses {
7032 buf.clear();
7033 key_pool.push(buf);
7034 return Ok(());
7035 }
7036 }
7037 buf
7038 };
7039 if srf_position.is_some() {
7040 let plan = srf_plan.as_mut().expect("srf_position implies a plan");
7041 for out in expand_srf_row_with(self, plan, &projection, row, &ctx)? {
7042 if stmt.distinct {
7043 let bucket = seen_distinct
7044 .entry(norm_hash_row(&out, &distinct_hb, ctx.mysql_dialect))
7045 .or_default();
7046 if bucket
7047 .iter()
7048 .any(|i| row_eq_norm(&tagged[i].1, &out, ctx.mysql_dialect))
7049 {
7050 continue;
7051 }
7052 bucket.push(tagged.len());
7053 }
7054 budget.charge(approx_row_bytes(&out))?;
7055 // The keys come from THIS expanded row: a key naming a
7056 // select-list item reads its value, anything else is
7057 // still evaluated against the input row.
7058 let keys = if order_by.is_empty() {
7059 Vec::new()
7060 } else {
7061 let mut kv: Vec<Value<'static>> = Vec::with_capacity(order_by.len());
7062 for (k, ob) in order_by.iter().enumerate() {
7063 kv.push(match srf_order_cols.get(k).copied().flatten() {
7064 Some(p) => out.values.get(p).cloned().unwrap_or(Value::Null),
7065 None => eval::eval_expr(&ob.expr, row, &ctx)
7066 .map_err(EngineError::Eval)?,
7067 });
7068 }
7069 // Packed by the same code every other ORDER BY uses,
7070 // so DESC / NULLS FIRST / the MySQL rule are not
7071 // restated here.
7072 let key_row = Row::new(kv);
7073 let mut buf = Vec::new();
7074 crate::orderby::build_order_keys_bound(
7075 &order_by,
7076 &srf_key_bound,
7077 &key_row,
7078 &ctx,
7079 &mut buf,
7080 )?;
7081 buf
7082 };
7083 tagged.push((keys, out));
7084 }
7085 } else {
7086 let values = &mut proj_buf;
7087 values.clear();
7088 values.reserve(projection.len());
7089 for (i, p) in projection.iter().enumerate() {
7090 // v7.37.x (docker-fair SCALARSQ attack) — pre-
7091 // analysed PK-probe fast path. The per-row work is
7092 // a read of outer.col from the row plus an index
7093 // probe — no Expr clone, no walker, no
7094 // `eval_expr_with_correlated` framework.
7095 if any_scalarsq_fast && let Some(fp) = &scalarsq_fast[i] {
7096 values.push(self.probe_with_pk_fast_path(fp, row));
7097 continue;
7098 }
7099 // v7.39 (round 605) — the same value every row.
7100 if any_proj_const && let Some(v) = &proj_const[i] {
7101 values.push(v.clone());
7102 continue;
7103 }
7104 // v7.39 (round 487) — bound column: read the cell.
7105 // This is `rehydrate_cell`'s body for a non-composite
7106 // column, which is what the whole chain below reduces
7107 // to once the name has been resolved.
7108 if any_proj_direct && let Some(pos) = proj_direct[i] {
7109 crate::bump_counter!(crate::select::PROJ_DIRECT_FIRE);
7110 values.push(row.values[pos].clone().into_owned());
7111 continue;
7112 }
7113 // v7.24 (round-16 B) — correlated-aware.
7114 // v7.37.x (docker-fair SCALARSQ attack) — share the
7115 // per-row memo with projection. Required for the
7116 // batch-evaluated correlated-scalar path to fire on
7117 // SELECT-item scalar subqueries; otherwise each row
7118 // re-executes the inner.
7119 //
7120 // Skip the memo when the outer row count is small
7121 // (early-limited): the batch path scans the FULL
7122 // inner table to build a GroupMap (~5 ms for a
7123 // 12.5 k-row inner), while per-row execution with a
7124 // PK index seek is ~5 µs per call — much cheaper for
7125 // N ≤ ~1000 outer rows.
7126 let pass_memo = early_cap.is_none_or(|cap| cap > 1000);
7127 let memo_arg = if pass_memo { Some(&mut memo) } else { None };
7128 values.push(
7129 self.eval_expr_with_correlated(&p.expr, row, &ctx, cancel, memo_arg)?,
7130 );
7131 }
7132 crate::bump_counter!(crate::select::PROJ_ROW_BUILT);
7133 if stmt.distinct {
7134 let bucket = seen_distinct
7135 .entry(norm_hash_values(&proj_buf, &distinct_hb, ctx.mysql_dialect))
7136 .or_default();
7137 if bucket
7138 .iter()
7139 .any(|i| values_eq_norm(&tagged[i].1.values, &proj_buf, ctx.mysql_dialect))
7140 {
7141 crate::bump_counter!(crate::select::DISTINCT_DUP_DROPPED);
7142 return Ok(());
7143 }
7144 bucket.push(tagged.len());
7145 }
7146 let out = Row::new(core::mem::replace(
7147 &mut proj_buf,
7148 proj_pool.pop().unwrap_or_default(),
7149 ));
7150 let order_keys = if stmt.distinct && !order_by.is_empty() {
7151 build_order_keys(&order_by, row, &ctx)?
7152 } else {
7153 order_keys
7154 };
7155 budget.charge(approx_row_bytes(&out))?;
7156 tagged.push((order_keys, out));
7157 }
7158 // Streaming top-N: bound the accumulator to O(keep) rows.
7159 if let Some((k, descs)) = &topk_stream {
7160 crate::orderby::topk_trim_recycling(
7161 &mut tagged,
7162 *k,
7163 descs,
7164 &mut proj_pool,
7165 &mut key_pool,
7166 &mut topk_boundary,
7167 );
7168 }
7169 Ok(())
7170 };
7171 // v7.37.15 (Phase C.3, step 2) — MVCC visibility gate for the
7172 // load-bearing full-scan path. This is the primary single-table
7173 // executor; pre-C.3 it read every hot-tier row raw. Once C.3's
7174 // in-place writers retain dead/old versions, an ungated scan
7175 // here would return them, so the gate must land BEFORE the
7176 // writers flip (see the plan's activation-order rule). A no-op
7177 // today: every hot row is frozen or committed-and-alive under
7178 // the reader's snapshot, so `is_row_visible` returns true for
7179 // all of them (verified by the full e2e suite staying green).
7180 let scan_snapshot = self.current_snapshot();
7181 let mut emitted: usize = 0;
7182 if let Some(rows) = &indexed_rows {
7183 for (loop_idx, cow) in rows.iter().enumerate() {
7184 if let Some(cap) = early_cap
7185 && emitted >= cap
7186 {
7187 break;
7188 }
7189 process_row(cow.as_ref(), loop_idx)?;
7190 emitted = emitted.saturating_add(1);
7191 }
7192 } else {
7193 // v7.39 (round 570) — the row store is a 32-way trie, so
7194 // indexing it is four dependent loads. Round 567 measured
7195 // -18% on the aggregate scan from holding the leaf between
7196 // rows; this is the same loop for the projecting scan.
7197 let mut rows_cur = table.rows().run_cursor();
7198 for i in 0..table.row_count() {
7199 if let Some(cap) = early_cap
7200 && emitted >= cap
7201 {
7202 break;
7203 }
7204 // Skip rows this snapshot cannot see (invisible rows do
7205 // not count toward the LIMIT).
7206 if !table.is_row_visible(i, &scan_snapshot) {
7207 continue;
7208 }
7209 let Some(row) = rows_cur.get(i) else { continue };
7210 process_row(row, i)?;
7211 emitted = emitted.saturating_add(1);
7212 }
7213 // v7.35.1 (mailrs prod #6 follow-up) — fold cold-tier
7214 // rows into the same loop. The full-scan path here is the
7215 // load-bearing single-table SELECT executor, and pre-
7216 // 7.35.1 it only walked `table.rows()` (hot), so any
7217 // `SELECT … FROM t` against a table with cold segments
7218 // silently returned a subset.
7219 let cold_rows = self.iter_cold_rows_of_table(table);
7220 for (offset, row) in cold_rows.iter().enumerate() {
7221 if let Some(cap) = early_cap
7222 && emitted >= cap
7223 {
7224 break;
7225 }
7226 process_row(row, table.row_count() + offset)?;
7227 emitted = emitted.saturating_add(1);
7228 }
7229 }
7230
7231 // (DISTINCT already de-duped STREAMING inside process_row, so the
7232 // sort below only sees the u survivors and the partial-sort
7233 // budget applies to DISTINCT too.)
7234 if !order_by.is_empty() {
7235 // Partial-sort fast path: when LIMIT is small relative to
7236 // the row count, select_nth_unstable + sort just the
7237 // prefix is O(n + k log k) instead of O(n log n).
7238 // WITH TIES needs the full sort so the tie extension can
7239 // scan past `limit` to find rows that share the last-kept
7240 // row's key.
7241 let keep = if stmt.limit_with_ties
7242 // v7.38 元机制 D acceptor — `SPG_TEST_DISABLE_TOPK=1`
7243 // forces the full-sort fallback by suppressing the
7244 // partial-sort `keep` budget. See
7245 // `xtests/sigil/test-mode-gucs.md`.
7246 || self.env_cfg().disable_topk
7247 {
7248 None
7249 } else {
7250 stmt.limit_literal()
7251 .map(|l| l as usize + stmt.offset_literal().map_or(0, |o| o as usize))
7252 };
7253 let descs: Vec<bool> = order_by.iter().map(|o| o.desc).collect();
7254 crate::orderby::partial_sort_tagged_in(&mut tagged, keep, &descs, &order_colls);
7255 }
7256
7257 // v7.17.0 Phase 3.P0-49 — `FETCH FIRST … WITH TIES` extends
7258 // past the truncated tail through every row that shares the
7259 // last-kept row's ORDER BY key. The tie check uses the
7260 // already-computed `(order_keys, row)` pairs so it matches
7261 // the sort comparator exactly. DISTINCT + WITH TIES falls
7262 // through to the no-ties path (PG also disallows their
7263 // combination; SPG silently drops the tie extension here so
7264 // the customer doesn't see a hard error mid-query — the
7265 // user-visible result is still correct, just narrower).
7266 let output_rows: Vec<Row<'static>> = if stmt.limit_with_ties && !stmt.distinct {
7267 apply_offset_and_limit_tagged(
7268 &mut tagged,
7269 stmt.offset_literal(),
7270 stmt.limit_literal(),
7271 true,
7272 );
7273 tagged.into_iter().map(|(_, r)| r).collect()
7274 } else {
7275 // DISTINCT already de-duped pre-sort above.
7276 let mut output_rows: Vec<Row<'static>> = tagged.into_iter().map(|(_, r)| r).collect();
7277 apply_offset_and_limit(
7278 &mut output_rows,
7279 stmt.offset_literal(),
7280 stmt.limit_literal(),
7281 );
7282 output_rows
7283 };
7284
7285 let columns: Vec<ColumnSchema> = projection
7286 .into_iter()
7287 .map(|p| {
7288 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
7289 c.user_enum_type = p.user_enum_type;
7290 c.collation_name = p.collation_name;
7291 c.mysql_fsp = p.mysql_fsp;
7292 c
7293 })
7294 .collect();
7295
7296 Ok(QueryResult::Rows {
7297 columns,
7298 rows: output_rows,
7299 })
7300 }
7301
7302 /// v7.31 (perf — PG lesson #1): shared aggregate finisher. Apply
7303 /// OFFSET/LIMIT first, then evaluate the deferred subquery-bearing
7304 /// select items for the surviving rows only — PG's Result-above-
7305 /// Limit shape, where SubPlan loops equal the OUTPUT row count
7306 /// (50) instead of the group count (24k).
7307 fn finish_agg_result(
7308 &self,
7309 mut agg: aggregate::AggResult,
7310 stmt: &SelectStatement,
7311 cancel: CancelToken<'_>,
7312 ) -> Result<QueryResult, EngineError> {
7313 apply_offset_and_limit(&mut agg.rows, stmt.offset_literal(), stmt.limit_literal());
7314 if !agg.deferred.is_empty() {
7315 apply_offset_and_limit(
7316 &mut agg.synth_rows,
7317 stmt.offset_literal(),
7318 stmt.limit_literal(),
7319 );
7320 let ctx = EvalContext::new(&agg.synth_schema, None);
7321 let mut memo = memoize::MemoizeCache::default();
7322 // v7.32 (architecture v2 P3) — keyed index-probe seeding.
7323 // Deferred subqueries are referenced only by surviving
7324 // select-list rows (≤ LIMIT), so their correlation keys are
7325 // exactly the ≤LIMIT group keys in `synth_rows`. Pre-build
7326 // each batchable subquery's group map over just those keys
7327 // via per-key index seek; the per-row splice loop below then
7328 // reuses the seeded map. A join-shaped or un-indexed inner
7329 // falls through to the all-keys batch inside the call (built
7330 // eagerly here instead of lazily on row 0 — same cost), so
7331 // it still pays the full scan, never the 715 ms per-row
7332 // direct eval; its index-nested-loop probe is the next
7333 // knife. Genuinely non-batchable shapes return None and are
7334 // left unseeded for the loop's per-row resolver, as before.
7335 for (_, expr) in &agg.deferred {
7336 let mut subs: Vec<&SelectStatement> = Vec::new();
7337 collect_scalar_subqueries(expr, &mut subs);
7338 for sub in subs {
7339 let repr = alloc::format!("{sub}");
7340 if memo.group_maps.contains_key(&repr) {
7341 continue;
7342 }
7343 if let Some(gm) = self.try_batch_correlated_scalar(
7344 sub,
7345 Some((&agg.synth_rows, &ctx)),
7346 cancel,
7347 )? {
7348 memo.group_maps.insert(repr, Some(alloc::rc::Rc::new(gm)));
7349 }
7350 }
7351 }
7352 for (ri, srow) in agg.synth_rows.iter().enumerate() {
7353 cancel.check()?;
7354 for (col, expr) in &agg.deferred {
7355 let v =
7356 self.eval_expr_with_correlated(expr, srow, &ctx, cancel, Some(&mut memo))?;
7357 if let Some(cell) = agg.rows[ri].values.get_mut(*col) {
7358 *cell = v;
7359 }
7360 }
7361 }
7362 }
7363 Ok(QueryResult::Rows {
7364 columns: agg.columns,
7365 rows: agg.rows,
7366 })
7367 }
7368
7369 /// v7.37 — streaming projection for the joined-non-aggregate
7370 /// shape (multi-table FROM, all projection items bound, no
7371 /// ORDER BY / DISTINCT / GROUP BY / HAVING / LIMIT / OFFSET /
7372 /// UNION). Walks the deferred join survivors and emits
7373 /// `&[&Value]` borrowed straight out of the source tables — no
7374 /// `.cloned()`, no `Vec<Row<'static>>`. Skips the 25 k × 3-TEXT clone tax
7375 /// on the mailrs `PROJ` shape (about 4 ms saved).
7376 ///
7377 /// Returns `Ok(None)` when the shape doesn't qualify; the caller
7378 /// then falls back to the materialising path.
7379 /// v7.37 (round 831) — stream a joinless SELECT straight off the
7380 /// stored table, one row at a time, without ever building a row set.
7381 ///
7382 /// Returns `Ok(None)` for anything this cannot serve, and the caller
7383 /// falls through to the deferred-join path exactly as before: a
7384 /// missing table, or a cold tier whose hydration the fallback handles.
7385 /// Sort a single-table scan through the external sorter, so the
7386 /// answer's size is bounded by `work_mem` and not by the input.
7387 ///
7388 /// Sorting held every row twice — the scan's `Vec<Row>` and the
7389 /// sort's `Vec<(keys, Row)>` beside it — with nothing bounding
7390 /// either: 807 MB at 400k rows, whatever `work_mem` said. A large
7391 /// enough ORDER BY took the server down, which is a liveness
7392 /// problem before it is a performance one.
7393 ///
7394 /// A SEPARATE walk rather than a change to `run_single_table_scan`,
7395 /// following what round 831 did for the joinless shape. That
7396 /// function is 552 lines whose projection loop is entangled with
7397 /// DISTINCT (which indexes back into the tagged vector) and with
7398 /// streaming top-N (whose boundary moves as the scan runs); both
7399 /// assume the projection has already happened when a row is
7400 /// pushed, which is exactly what spilling has to defer. Two earlier
7401 /// attempts tried to rework that loop and were reverted. Here the
7402 /// existing path is untouched and this one only claims shapes it
7403 /// can serve, so a decline costs nothing.
7404 ///
7405 /// Records are SOURCE rows, not projected ones: `finish` re-derives
7406 /// keys from what it decodes, and an ORDER BY key need not be in
7407 /// the projection — `SELECT pad FROM big ORDER BY id` (round 835).
7408 fn try_spill_sorted_scan(
7409 &self,
7410 stmt: &SelectStatement,
7411 from: &FromClause,
7412 cancel: CancelToken<'_>,
7413 ) -> Result<Option<QueryResult>, EngineError> {
7414 // Shapes this walk does not serve. Each one either needs the
7415 // whole tagged vector addressable (DISTINCT probes back into
7416 // it, WITH TIES re-reads its tail) or is already bounded
7417 // without spilling (a LIMIT makes the partial sort O(keep)).
7418 if !self.can_spill()
7419 || stmt.order_by.is_empty()
7420 || stmt.distinct
7421 || stmt.limit_with_ties
7422 || stmt.limit_literal().is_some()
7423 || !from.joins.is_empty()
7424 || from.primary.lateral_subquery.is_some()
7425 || from.primary.unnest_expr.is_some()
7426 || from.primary.generate_series_args.is_some()
7427 || select_has_window(stmt)
7428 {
7429 return Ok(None);
7430 }
7431 // A parent's rows are its children's. These walks scan the named
7432 // relation alone, so a partitioned or inherited parent comes back
7433 // short — and silently: the corpus caught `SELECT id FROM pr
7434 // ORDER BY id` and `SELECT k FROM pl ORDER BY k` returning the
7435 // parent's own rows instead of the partitions'. `ONLY` is exactly
7436 // the case that does not fan out, so it stays, which is the test
7437 // the FROM-clause fan-out itself makes.
7438 if !from.primary.only
7439 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
7440 {
7441 return Ok(None);
7442 }
7443 let Some(table) = self.active_catalog().get(&from.primary.name) else {
7444 return Ok(None);
7445 };
7446 // Cold-tier rows live outside `rows()`; this walk would drop
7447 // them silently, the same reason round 831's walk declines.
7448 if table.has_cold_rows_fast() {
7449 return Ok(None);
7450 }
7451
7452 let alias = from
7453 .primary
7454 .alias
7455 .as_deref()
7456 .unwrap_or(from.primary.name.as_str());
7457 let cols = table.schema().columns.clone();
7458 let sess = self.dml_session();
7459 let ctx = EvalContext::new(&cols, Some(alias))
7460 .with_catalog(self.active_catalog())
7461 .with_session(&sess);
7462 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
7463 let order_by = stmt.order_by.clone();
7464 // The same one-shot resolution the general path does (round
7465 // 582): each ORDER BY column is bound once, not once per row.
7466 let order_bound = crate::orderby::order_by_bound_positions(&order_by, &cols, Some(alias));
7467 let descs: Vec<bool> = order_by.iter().map(|o| o.desc).collect();
7468 // Resolved BEFORE the scan, because it now decides what the sort
7469 // STORES and not just what it decodes (round 995).
7470 let needed = Self::sort_record_columns_needed(&stmt.items, &order_bound, cols.len(), &ctx);
7471
7472 let mut sorter = crate::extsort::ExternalSorter::new(
7473 self.temp_run_factory,
7474 self.session_work_mem_bytes(),
7475 cols.clone(),
7476 &descs,
7477 )
7478 .with_stats(&self.spill_stats)
7479 .with_pruned(&needed);
7480 let snapshot = self.current_snapshot();
7481 // One key buffer for the whole scan: `push` drains it and leaves
7482 // the capacity behind.
7483 let mut keys: Vec<OrderKey> = Vec::new();
7484 // r1024 — compile the predicate once for the scan.
7485 //
7486 // These two sorted-spill scans are the paths a single-table SELECT
7487 // with an ORDER BY takes, and they were the last row-returning ones
7488 // still walking the expression tree per row. r1023 did the
7489 // no-ORDER-BY sibling; the sweep's two remaining losing cells are
7490 // exactly this shape.
7491 //
7492 // Found from the profile's CALL TREE rather than its leaves. The
7493 // leaves say what is expensive — `eval_expr` 320, `apply_binary`
7494 // 261, `mod_op` 178 — and two attempts at reasoning out which
7495 // function asked for it were both wrong. The tree names the caller
7496 // chain, and it named this one.
7497 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
7498 .where_
7499 .as_ref()
7500 .filter(|w| crate::eval::fully_compilable(w))
7501 .map(|w| crate::eval::compile_expr(w, &ctx));
7502 let mut eval_stack: Vec<Value<'static>> = Vec::new();
7503 for (i, row) in table.scan_visible_from(0, &snapshot) {
7504 if i.is_multiple_of(256) {
7505 cancel.check()?;
7506 }
7507 if let Some(c) = &compiled_where {
7508 if !crate::eval::compiled::eval_compiled_pred(
7509 c,
7510 row,
7511 &ctx,
7512 &mut eval_stack,
7513 ctx.mysql_dialect,
7514 )? {
7515 continue;
7516 }
7517 } else if let Some(w) = &stmt.where_ {
7518 let cond = crate::eval::eval_expr(w, row, &ctx).map_err(EngineError::Eval)?;
7519 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
7520 continue;
7521 }
7522 }
7523 keys.clear();
7524 crate::orderby::build_order_keys_bound(&order_by, &order_bound, row, &ctx, &mut keys)?;
7525 sorter.push(&mut keys, row)?;
7526 }
7527
7528 let key_ctx = &ctx;
7529 let rows = sorter.finish(
7530 |src, buf| {
7531 crate::orderby::build_order_keys_bound(&order_by, &order_bound, src, key_ctx, buf)
7532 },
7533 |src| {
7534 let mut values = Vec::with_capacity(projection.len());
7535 for p in &projection {
7536 values.push(
7537 crate::eval::eval_expr(&p.expr, src, key_ctx).map_err(EngineError::Eval)?,
7538 );
7539 }
7540 Ok(Row::new(values))
7541 },
7542 )?;
7543
7544 let columns: Vec<ColumnSchema> = projection
7545 .iter()
7546 .map(|p| {
7547 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
7548 c.user_enum_type = p.user_enum_type.clone();
7549 c.mysql_fsp = p.mysql_fsp;
7550 c
7551 })
7552 .collect();
7553 Ok(Some(QueryResult::Rows { columns, rows }))
7554 }
7555
7556 /// v7.37 (round 882) — the bounded sort of `try_spill_sorted_scan`,
7557 /// handing each row to the consumer instead of collecting the answer.
7558 ///
7559 /// That walk bounds the SORT and then returns `QueryResult::Rows`,
7560 /// which holds every output row. Measured at `work_mem = 4 MB` over
7561 /// 200-byte rows, RSS above the server's own baseline while the
7562 /// query runs grew +30 MB at 100k rows, +68 MB at 200k and +137 MB
7563 /// at 400k — linear — while the spill underneath worked correctly
7564 /// (9 / 17 / 33 runs, witnessed DURING the query; `FileRun::drop`
7565 /// removes each file, so a count taken afterwards reads 0 whatever
7566 /// happened, and an earlier reading of "no spill at all" was that
7567 /// blind witness). The growth is the collected result, not the sort.
7568 ///
7569 /// Emitting makes peak the budget, one buffer per run and a single
7570 /// row — the state a merge already holds at every step. It also
7571 /// frees each projected row as the next is built rather than
7572 /// accumulating them, which is where the time is: a profile of the
7573 /// collecting walk put the allocator at 586 samples, more than every
7574 /// sort comparison combined (420), against 19 for `push` itself.
7575 /// v7.37 (round 923) — which of a sort record's columns the output half
7576 /// reads. The record is the SOURCE row (round 836), so a narrow projection
7577 /// decoded every column: skipping one 200-byte text halves a decode
7578 /// (2.17 -> 1.14 ms per pass at 10k rows, priced additively).
7579 ///
7580 /// Timid on purpose — a wrong mask is a SILENT wrong answer, a pruned
7581 /// column reads NULL. Answers only when every projection item is a bare
7582 /// column reference AND every ORDER BY key is a bound column; anything
7583 /// else returns empty, decoding everything as before.
7584 /// `explain.rs`'s `collect_column_refs` is NOT used: its `_ => {}` arm
7585 /// drops references from expression kinds it does not enumerate.
7586 ///
7587 /// ORDER BY columns are included — the merge re-derives keys from the
7588 /// decoded row on the spilled path, so pruning one would sort NULLs.
7589 pub(crate) fn sort_record_columns_needed(
7590 items: &[SelectItem],
7591 order_bound: &[Option<usize>],
7592 arity: usize,
7593 ctx: &EvalContext,
7594 ) -> Vec<bool> {
7595 let all_bare = items.iter().all(|i| {
7596 matches!(
7597 i,
7598 SelectItem::Expr {
7599 expr: Expr::Column(_),
7600 ..
7601 }
7602 )
7603 });
7604 if !all_bare || order_bound.iter().any(Option::is_none) {
7605 return Vec::new();
7606 }
7607 let mut mask = alloc::vec![false; arity];
7608 for item in items {
7609 if let SelectItem::Expr {
7610 expr: Expr::Column(c),
7611 ..
7612 } = item
7613 {
7614 match crate::eval::find_column_pos(c, ctx) {
7615 Some(p) if p < arity => mask[p] = true,
7616 _ => return Vec::new(),
7617 }
7618 }
7619 }
7620 for p in order_bound.iter().flatten() {
7621 if *p < arity {
7622 mask[*p] = true;
7623 } else {
7624 return Vec::new();
7625 }
7626 }
7627 mask
7628 }
7629
7630 /// r1025 — `ORDER BY <indexed NOT NULL column>` walks the index instead
7631 /// of sorting.
7632 ///
7633 /// PG serves such an ordering from the index and never sorts. We sorted:
7634 /// measured at 400,000 rows, `SELECT pad FROM t ORDER BY id` costs
7635 /// 138-144 ms against PG18's 64-75, and the call tree puts the cost in
7636 /// the sorter's own round trip — `ExternalSorter::finish_each` →
7637 /// `next_row` → `decode_row_body_dense_pruned` → `read_value_body`.
7638 /// Every row is encoded into the sorter's arena and decoded back out,
7639 /// for an order the index already holds.
7640 ///
7641 /// The walk exists — `try_pk_walk_top_n` — and requires a `LIMIT`,
7642 /// because it was built for top-N. This is the unbounded sibling.
7643 ///
7644 /// NOT NULL is a hard gate, not a simplification: a NULL key is absent
7645 /// from a btree, so walking one would silently drop those rows. That is
7646 /// exactly the defect r1020 fixed on the top-N path, where it had
7647 /// shipped.
7648 /// r1044 — the index this statement's ORDER BY can be WALKED on,
7649 /// instead of sorted, or `None`.
7650 ///
7651 /// Extracted so `EXPLAIN` can ask the same question the executor
7652 /// answers. It could not, and said so: `SELECT pad FROM t ORDER BY
7653 /// id` on a 400,000-row table planned as `Sort` over `Seq Scan`
7654 /// while the executor walked the primary key — 34.9 ms against
7655 /// 147.0 for the same query ordered by an unindexed column, so the
7656 /// walk was plainly running. Round 551 fixed a different case of
7657 /// this and wrote the reason down: EXPLAIN is the first thing any
7658 /// performance question opens, and an instrument that misnames the
7659 /// access path is worse than one that says nothing.
7660 ///
7661 /// The gate is here once. Two copies of it is how the plan and the
7662 /// executor come to disagree again.
7663 pub(crate) fn index_order_walk_target(
7664 &self,
7665 stmt: &SelectStatement,
7666 from: &FromClause,
7667 ) -> Option<(String, usize)> {
7668 if stmt.order_by.len() != 1
7669 || !stmt.distinct_on.is_empty()
7670 || stmt.limit_with_ties
7671 || stmt.limit.is_some()
7672 || stmt.offset.is_some()
7673 || stmt.having.is_some()
7674 || stmt.group_by.is_some()
7675 || !stmt.unions.is_empty()
7676 || !from.joins.is_empty()
7677 || from.primary.lateral_subquery.is_some()
7678 || from.primary.unnest_expr.is_some()
7679 || from.primary.as_of_segment.is_some()
7680 || from.primary.generate_series_args.is_some()
7681 || select_has_window(stmt)
7682 || aggregate::uses_aggregate(stmt)
7683 {
7684 return None;
7685 }
7686 if stmt
7687 .items
7688 .iter()
7689 .any(|i| matches!(i, SelectItem::Expr { expr, .. } if is_top_level_unnest(expr)))
7690 {
7691 return None;
7692 }
7693 let table = self.active_catalog().get(&from.primary.name)?;
7694 if table.has_cold_rows_fast() {
7695 return None;
7696 }
7697 if !from.primary.only
7698 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
7699 {
7700 return None;
7701 }
7702 let alias = from
7703 .primary
7704 .alias
7705 .as_deref()
7706 .unwrap_or(from.primary.name.as_str());
7707 let cols = &table.schema().columns;
7708 let order = &stmt.order_by[0];
7709 let Expr::Column(oc) = &order.expr else {
7710 return None;
7711 };
7712 if let Some(q) = &oc.qualifier
7713 && !q.eq_ignore_ascii_case(alias)
7714 {
7715 return None;
7716 }
7717 let order_pos = cols
7718 .iter()
7719 .position(|c| c.name.eq_ignore_ascii_case(&oc.name))?;
7720 // r1047 — DISTINCT joins the walk when the projection IS the
7721 // order column, and only then. The index's keys are canonical
7722 // (r1039: representation equality is value equality — the
7723 // property every seek already depends on), so one key is one
7724 // distinct value and the walk can emit the first passing row of
7725 // each key group instead of hashing every row. On the release
7726 // sweep's `SELECT DISTINCT n FROM t ORDER BY n` — 400,000 rows,
7727 // 1,000 distinct values — the hash path priced at 21.3-22.7 ms
7728 // with an ablation floor of 14.8, because the hash must
7729 // normalize and probe ALL the rows; the walk visits each key
7730 // once. A wider projection makes DISTINCT about the whole tuple,
7731 // not the key, so anything else still declines.
7732 if stmt.distinct {
7733 let only_the_order_column = stmt.items.len() == 1
7734 && match &stmt.items[0] {
7735 SelectItem::Expr {
7736 expr: Expr::Column(c),
7737 ..
7738 } => {
7739 c.name.eq_ignore_ascii_case(&oc.name)
7740 && match &c.qualifier {
7741 Some(q) => q.eq_ignore_ascii_case(alias),
7742 None => true,
7743 }
7744 }
7745 _ => false,
7746 };
7747 if !only_the_order_column {
7748 return None;
7749 }
7750 }
7751 // r1046 — a nullable key no longer refuses the walk; it changes
7752 // what the walk has to do. A NULL key is not in the btree, so
7753 // walking alone would silently drop those rows — the r1020
7754 // defect, which shipped once. The walk emits them separately, at
7755 // the end SQL puts them.
7756 //
7757 // Refusing was costing every nullable indexed column a 3.4x:
7758 // `SELECT id FROM t ORDER BY b` over 400,000 rows measured
7759 // 72.0 ms with the column nullable and 20.2 with the same data
7760 // under NOT NULL. `NOT NULL` is not the default, so that was the
7761 // common case paying for the uncommon one.
7762 let index = table.index_on(order_pos)?;
7763 if !matches!(index.kind, spg_storage::IndexKind::BTree(_))
7764 || index.expression.is_some()
7765 || index.partial_predicate.is_some()
7766 {
7767 return None;
7768 }
7769 Some((index.name.clone(), order_pos))
7770 }
7771
7772 fn try_index_order_stream<F>(
7773 &self,
7774 stmt: &SelectStatement,
7775 from: &FromClause,
7776 cancel: CancelToken<'_>,
7777 emit: &mut F,
7778 ) -> Result<Option<usize>, EngineError>
7779 where
7780 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
7781 {
7782 // r1044 — the shape gate lives in `index_order_walk_target`, so
7783 // `EXPLAIN` answers the same question. What stays here is the
7784 // part that RAISES (an illegal ORDER BY has to keep erroring
7785 // from where it did) and the bindings the walk needs.
7786 crate::orderby::check_order_by_legality(stmt)?;
7787 crate::orderby::check_order_by_positions(stmt)?;
7788 crate::window::reject_window_in_row_clauses(stmt)?;
7789 let Some((_, order_pos)) = self.index_order_walk_target(stmt, from) else {
7790 return Ok(None);
7791 };
7792 let Some(table) = self.active_catalog().get(&from.primary.name) else {
7793 return Ok(None);
7794 };
7795 let alias = from
7796 .primary
7797 .alias
7798 .as_deref()
7799 .unwrap_or(from.primary.name.as_str());
7800 let cols = table.schema().columns.clone();
7801 let order = &stmt.order_by[0];
7802 let Some(index) = table.index_on(order_pos) else {
7803 return Ok(None);
7804 };
7805
7806 let sess = self.dml_session();
7807 let ctx = EvalContext::new(&cols, Some(alias))
7808 .with_catalog(self.active_catalog())
7809 .with_session(&sess);
7810 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
7811 let columns: Vec<ColumnSchema> = projection
7812 .iter()
7813 .map(|p| {
7814 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
7815 c.user_enum_type = p.user_enum_type.clone();
7816 c.mysql_fsp = p.mysql_fsp;
7817 c
7818 })
7819 .collect();
7820 emit(crate::StreamItem::Header(&columns))?;
7821 let bound_pos: Vec<Option<usize>> = projection
7822 .iter()
7823 .map(|p| match &p.expr {
7824 Expr::Column(c) => match crate::eval::locate_column(c, &ctx) {
7825 Ok(Some(pos)) => Some(pos),
7826 _ => None,
7827 },
7828 _ => None,
7829 })
7830 .collect();
7831
7832 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
7833 .where_
7834 .as_ref()
7835 .filter(|w| crate::eval::fully_compilable(w))
7836 .map(|w| crate::eval::compile_expr(w, &ctx));
7837 let mut eval_stack: Vec<Value<'static>> = Vec::new();
7838 let mut values: Vec<Value<'static>> = Vec::with_capacity(projection.len());
7839 let snapshot = self.current_snapshot();
7840
7841 // A btree holds one locator per row VERSION, so a row whose key was
7842 // updated can sit under two keys and a dead one can sit beside its
7843 // replacement. The visibility gate drops the dead; `seen` drops a
7844 // live row that the walk reaches twice, which would otherwise be a
7845 // duplicated output row rather than a slow one.
7846 let mut emitted_rows = alloc::vec![false; table.rows().len()];
7847
7848 // r1046 — the rows the index cannot hold.
7849 //
7850 // A NULL key is not in the btree, so the walk below never reaches
7851 // those rows; they are emitted here, at the end SQL puts them.
7852 // PG's default is NULLS LAST ascending and NULLS FIRST
7853 // descending, and an explicit `NULLS FIRST` / `NULLS LAST` wins —
7854 // the same rule `order_by_value_cmp_raw` applies to the sort this
7855 // replaces, so the two orders agree.
7856 //
7857 // Finding them costs one pass over the column. That pass is why
7858 // this is still worth doing: the sort it replaces encodes and
7859 // decodes every row, and the walk plus the pass measured 72.0 ms
7860 // down to about 22 on 400,000 rows.
7861 let nulls_first = order.nulls_first.unwrap_or(order.desc);
7862 // r1047 — under DISTINCT the walk emits the FIRST passing row of
7863 // each key group and skips the rest; the gate admits DISTINCT
7864 // only when the projection is the order column itself, so one
7865 // canonical key is one output row. NULL is one distinct value,
7866 // so the NULL pass stops at its first emit too.
7867 let distinct = stmt.distinct;
7868 let mut count = 0usize;
7869 let mut visited = 0usize;
7870 let mut emit_null_rows = |emitted_rows: &mut alloc::vec::Vec<bool>,
7871 eval_stack: &mut Vec<Value<'static>>,
7872 values: &mut Vec<Value<'static>>,
7873 visited: &mut usize,
7874 emit: &mut F|
7875 -> Result<usize, EngineError> {
7876 if !cols[order_pos].nullable {
7877 return Ok(0);
7878 }
7879 let mut n = 0usize;
7880 for (ri, row) in table.rows().iter().enumerate() {
7881 if !matches!(row.values.get(order_pos), Some(Value::Null)) {
7882 continue;
7883 }
7884 if emitted_rows.get(ri).copied().unwrap_or(true) {
7885 continue;
7886 }
7887 if !table.is_row_visible(ri, &snapshot) {
7888 continue;
7889 }
7890 *visited += 1;
7891 if visited.is_multiple_of(256) {
7892 cancel.check()?;
7893 }
7894 emitted_rows[ri] = true;
7895 if Self::stream_project_row(
7896 row,
7897 stmt.where_.as_ref(),
7898 compiled_where.as_ref(),
7899 eval_stack,
7900 &projection,
7901 &bound_pos,
7902 &ctx,
7903 values,
7904 emit,
7905 )? {
7906 n += 1;
7907 if distinct {
7908 break;
7909 }
7910 }
7911 }
7912 Ok(n)
7913 };
7914
7915 if nulls_first {
7916 count += emit_null_rows(
7917 &mut emitted_rows,
7918 &mut eval_stack,
7919 &mut values,
7920 &mut visited,
7921 emit,
7922 )?;
7923 }
7924
7925 let walker: alloc::boxed::Box<
7926 dyn Iterator<Item = (&spg_storage::IndexKey, &spg_storage::PostingList)>,
7927 > = if order.desc {
7928 alloc::boxed::Box::new(index.iter_desc())
7929 } else {
7930 alloc::boxed::Box::new(index.iter_asc())
7931 };
7932 for (_key, locators) in walker {
7933 for loc in locators {
7934 let spg_storage::RowLocator::Hot(ri) = *loc else {
7935 continue;
7936 };
7937 if emitted_rows.get(ri).copied().unwrap_or(true) {
7938 continue;
7939 }
7940 if !table.is_row_visible(ri, &snapshot) {
7941 continue;
7942 }
7943 let Some(row) = table.rows().get(ri) else {
7944 continue;
7945 };
7946 visited += 1;
7947 if visited.is_multiple_of(256) {
7948 cancel.check()?;
7949 }
7950 emitted_rows[ri] = true;
7951 if Self::stream_project_row(
7952 row,
7953 stmt.where_.as_ref(),
7954 compiled_where.as_ref(),
7955 &mut eval_stack,
7956 &projection,
7957 &bound_pos,
7958 &ctx,
7959 &mut values,
7960 emit,
7961 )? {
7962 count += 1;
7963 // One row per key group: the rest are the same value.
7964 if distinct {
7965 break;
7966 }
7967 }
7968 }
7969 }
7970
7971 if !nulls_first {
7972 count += emit_null_rows(
7973 &mut emitted_rows,
7974 &mut eval_stack,
7975 &mut values,
7976 &mut visited,
7977 emit,
7978 )?;
7979 }
7980 Ok(Some(count))
7981 }
7982
7983 /// r1031 — `ORDER BY` over NOT NULL integer columns, sorted without
7984 /// building an `OrderKey` vector per row.
7985 ///
7986 /// The row-returning sorted scan allocates twice per row: one
7987 /// `Vec<OrderKey>` for the sort keys and one `Vec<Value>` for the
7988 /// projection. Counted over 400 k rows (r1030,
7989 /// `docs/PERF_SORTED_SCAN_ALLOCATIONS_2026-08-15.md`), that is 800,067
7990 /// allocations and 208 MB of traffic for an answer of four hundred
7991 /// thousand integers.
7992 ///
7993 /// The key half is pure ceremony on this shape.
7994 /// `sort_tagged_by_inline_int_key` already sorts indices rather than
7995 /// rows, so the per-row vector is built, has one integer taken out of
7996 /// it, and is then dragged through the permutation — it exists to carry
7997 /// a number the row's column already held. This lane carries the number
7998 /// instead, in a fixed-size array that lives inside the buffer element
7999 /// and allocates nothing. Same idea as the predicate VM's integer lane.
8000 ///
8001 /// Declines to `None` for anything it does not cover, and every caller
8002 /// falls through to the general path, so the gate list is the
8003 /// specification.
8004 ///
8005 /// Ties: equal keys keep scan order, as the stable sort on the general
8006 /// path does. Rows that tie on every ORDER BY term are entitled to any
8007 /// order among themselves either way — see `STABILITY.md`.
8008 fn try_int_key_sorted_stream<F>(
8009 &self,
8010 stmt: &SelectStatement,
8011 from: &FromClause,
8012 cancel: CancelToken<'_>,
8013 emit: &mut F,
8014 ) -> Result<Option<usize>, EngineError>
8015 where
8016 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
8017 {
8018 /// Sort terms this lane carries inline. Four covers every ORDER BY
8019 /// in the endpoint sweep and in the dogfood corpus; wider ones fall
8020 /// through rather than growing the buffer element for everybody.
8021 const MAX_KEYS: usize = 4;
8022
8023 if stmt.order_by.is_empty()
8024 || stmt.order_by.len() > MAX_KEYS
8025 || stmt.distinct
8026 || stmt.limit_with_ties
8027 || stmt.limit.is_some()
8028 || stmt.offset.is_some()
8029 || stmt.having.is_some()
8030 || stmt.group_by.is_some()
8031 || !stmt.unions.is_empty()
8032 || !from.joins.is_empty()
8033 || from.primary.lateral_subquery.is_some()
8034 || from.primary.unnest_expr.is_some()
8035 || from.primary.as_of_segment.is_some()
8036 || from.primary.generate_series_args.is_some()
8037 || select_has_window(stmt)
8038 || aggregate::uses_aggregate(stmt)
8039 {
8040 return Ok(None);
8041 }
8042 if stmt
8043 .items
8044 .iter()
8045 .any(|i| matches!(i, SelectItem::Expr { expr, .. } if is_top_level_unnest(expr)))
8046 {
8047 return Ok(None);
8048 }
8049 crate::orderby::check_order_by_legality(stmt)?;
8050 crate::orderby::check_order_by_positions(stmt)?;
8051 crate::window::reject_window_in_row_clauses(stmt)?;
8052 let Some(table) = self.active_catalog().get(&from.primary.name) else {
8053 return Ok(None);
8054 };
8055 if table.has_cold_rows_fast() {
8056 return Ok(None);
8057 }
8058 if !from.primary.only
8059 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
8060 {
8061 return Ok(None);
8062 }
8063 let alias = from
8064 .primary
8065 .alias
8066 .as_deref()
8067 .unwrap_or(from.primary.name.as_str());
8068 let cols = table.schema().columns.clone();
8069
8070 // Every ORDER BY term must be a NOT NULL integer column of this
8071 // table. NOT NULL is what lets the key be a bare integer: with
8072 // NULLs the lane would have to carry their ordering too, and
8073 // getting that subtly wrong is the r1020 defect.
8074 let mut key_pos = [0usize; MAX_KEYS];
8075 let mut descs = [false; MAX_KEYS];
8076 // PG's default is NULLS LAST for ASC and NULLS FIRST for DESC,
8077 // which the AST records as `None`; `unwrap_or(desc)` is how the
8078 // rest of the engine resolves it.
8079 let mut nulls_first = [false; MAX_KEYS];
8080 let n_keys = stmt.order_by.len();
8081 for (slot, order) in stmt.order_by.iter().enumerate() {
8082 let Expr::Column(oc) = &order.expr else {
8083 return Ok(None);
8084 };
8085 if let Some(q) = &oc.qualifier
8086 && !q.eq_ignore_ascii_case(alias)
8087 {
8088 return Ok(None);
8089 }
8090 let Some(pos) = cols
8091 .iter()
8092 .position(|c| c.name.eq_ignore_ascii_case(&oc.name))
8093 else {
8094 return Ok(None);
8095 };
8096 if !matches!(
8097 cols[pos].ty,
8098 spg_storage::DataType::SmallInt
8099 | spg_storage::DataType::Int
8100 | spg_storage::DataType::BigInt
8101 ) {
8102 return Ok(None);
8103 }
8104 key_pos[slot] = pos;
8105 descs[slot] = order.desc;
8106 nulls_first[slot] = order.nulls_first.unwrap_or(order.desc);
8107 }
8108
8109 let sess = self.dml_session();
8110 let ctx = EvalContext::new(&cols, Some(alias))
8111 .with_catalog(self.active_catalog())
8112 .with_session(&sess);
8113 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
8114 let columns: Vec<ColumnSchema> = projection
8115 .iter()
8116 .map(|p| {
8117 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
8118 c.user_enum_type = p.user_enum_type.clone();
8119 c.mysql_fsp = p.mysql_fsp;
8120 c
8121 })
8122 .collect();
8123 let bound_pos: Vec<Option<usize>> = projection
8124 .iter()
8125 .map(|p| match &p.expr {
8126 Expr::Column(c) => match crate::eval::locate_column(c, &ctx) {
8127 Ok(Some(pos)) => Some(pos),
8128 _ => None,
8129 },
8130 _ => None,
8131 })
8132 .collect();
8133 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
8134 .where_
8135 .as_ref()
8136 .filter(|w| crate::eval::fully_compilable(w))
8137 .map(|w| crate::eval::compile_expr(w, &ctx));
8138
8139 // The same first-observable point the materialising planner fires,
8140 // placed after the gates so it fires exactly once: this lane runs
8141 // BEFORE that planner and would otherwise be a hole in the
8142 // panic-isolation and cancellation-race coverage rather than a
8143 // faster path through it.
8144 crate::injection_point!("planner_first_row_fetch", &stmt.from);
8145
8146 let mut eval_stack: Vec<Value<'static>> = Vec::new();
8147 let mut values: Vec<Value<'static>> = Vec::with_capacity(projection.len());
8148 let mut budget = ByteBudget::new(self.max_query_bytes);
8149 let snapshot = self.current_snapshot();
8150 // Keys, a NULL bit per key slot, and the row. The bitmask keeps
8151 // the element small: a nullable key still costs one bit rather
8152 // than a second array.
8153 let mut sorted: Vec<([i64; MAX_KEYS], u8, Vec<Value<'static>>)> = Vec::new();
8154
8155 for (ri, row) in table.rows().iter().enumerate() {
8156 if ri.is_multiple_of(256) {
8157 cancel.check()?;
8158 }
8159 if !table.is_row_visible(ri, &snapshot) {
8160 continue;
8161 }
8162 // The key comes from the STORED row, before projection: an
8163 // ORDER BY column need not appear in the select list.
8164 let mut keys = [0i64; MAX_KEYS];
8165 let mut nulls = 0u8;
8166 let mut keyed = true;
8167 for slot in 0..n_keys {
8168 match row.values.get(key_pos[slot]) {
8169 Some(Value::SmallInt(v)) => keys[slot] = i64::from(*v),
8170 Some(Value::Int(v)) => keys[slot] = i64::from(*v),
8171 Some(Value::BigInt(v)) => keys[slot] = *v,
8172 Some(Value::Null) | None => nulls |= 1 << slot,
8173 // An integer column holding something else is a row
8174 // this lane cannot order; hand the whole query back
8175 // rather than guess at it.
8176 _ => {
8177 keyed = false;
8178 break;
8179 }
8180 }
8181 }
8182 if !keyed {
8183 return Ok(None);
8184 }
8185 if !Self::stream_filter_project(
8186 row,
8187 stmt.where_.as_ref(),
8188 compiled_where.as_ref(),
8189 &mut eval_stack,
8190 &projection,
8191 &bound_pos,
8192 &ctx,
8193 &mut values,
8194 )? {
8195 continue;
8196 }
8197 budget.charge(crate::bytebudget::approx_values_bytes(&values))?;
8198 sorted.push((keys, nulls, core::mem::take(&mut values)));
8199 values.reserve(projection.len());
8200 }
8201
8202 sorted.sort_by(|a, b| {
8203 use core::cmp::Ordering;
8204 for slot in 0..n_keys {
8205 let bit = 1u8 << slot;
8206 let ord = match (a.1 & bit != 0, b.1 & bit != 0) {
8207 (true, true) => Ordering::Equal,
8208 // Where the NULLs go is already decided — `nulls_first`
8209 // resolved DESC's default when it was read. Reversing
8210 // this for DESC as well would apply the direction
8211 // twice and put them at the wrong end.
8212 (true, false) => {
8213 if nulls_first[slot] {
8214 Ordering::Less
8215 } else {
8216 Ordering::Greater
8217 }
8218 }
8219 (false, true) => {
8220 if nulls_first[slot] {
8221 Ordering::Greater
8222 } else {
8223 Ordering::Less
8224 }
8225 }
8226 (false, false) => {
8227 let o = a.0[slot].cmp(&b.0[slot]);
8228 if descs[slot] { o.reverse() } else { o }
8229 }
8230 };
8231 if ord != Ordering::Equal {
8232 return ord;
8233 }
8234 }
8235 Ordering::Equal
8236 });
8237
8238 emit(crate::StreamItem::Header(&columns))?;
8239 let count = sorted.len();
8240 for (_, _, vals) in &sorted {
8241 emit(crate::StreamItem::Row(crate::RowCells::Values(vals)))?;
8242 }
8243 Ok(Some(count))
8244 }
8245
8246 fn try_spill_sorted_stream<F>(
8247 &self,
8248 stmt: &SelectStatement,
8249 from: &FromClause,
8250 cancel: CancelToken<'_>,
8251 emit: &mut F,
8252 ) -> Result<Option<usize>, EngineError>
8253 where
8254 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
8255 {
8256 // The shapes `try_spill_sorted_scan` declines, plus the ones the
8257 // streaming executor does not carry (a LIMIT is already bounded
8258 // by a partial sort; the rest need the answer addressable).
8259 if !self.can_spill()
8260 || stmt.order_by.is_empty()
8261 || stmt.distinct
8262 || stmt.limit_with_ties
8263 || stmt.limit.is_some()
8264 || stmt.offset.is_some()
8265 || stmt.having.is_some()
8266 || stmt.group_by.is_some()
8267 || !stmt.unions.is_empty()
8268 || !from.joins.is_empty()
8269 || from.primary.lateral_subquery.is_some()
8270 || from.primary.unnest_expr.is_some()
8271 || from.primary.as_of_segment.is_some()
8272 || from.primary.generate_series_args.is_some()
8273 || select_has_window(stmt)
8274 || aggregate::uses_aggregate(stmt)
8275 {
8276 return Ok(None);
8277 }
8278 if stmt
8279 .items
8280 .iter()
8281 .any(|i| matches!(i, SelectItem::Expr { expr, .. } if is_top_level_unnest(expr)))
8282 {
8283 return Ok(None);
8284 }
8285 // Everything `exec_bare_select_cancel` does before it scans runs
8286 // BELOW this path, so a statement claimed here skips it. Three of
8287 // those were missed on the way in and each was caught by a
8288 // different gate — the ORDER BY rules by an e2e (`SELECT a FROM t
8289 // ORDER BY 2` sorted happily instead of raising 42P10), the
8290 // cancellation check by another, the partition fan-out by the
8291 // differential corpus. What is reconciled, item by item: with-ties
8292 // needs ORDER BY (gated above), USING/NATURAL and RLS join
8293 // rewrites (joins gated above), the single-table RLS predicate
8294 // (the dispatcher declines a policy-subject table before this is
8295 // reached), the meta-view dispatch (those names are not in the
8296 // catalog, so the lookup below declines). These three are calls,
8297 // so the message and SQLSTATE are the ones the fall-back gives —
8298 // `select_has_window` above reads the select list and ORDER BY but
8299 // not WHERE, which is the case the third one covers.
8300 crate::orderby::check_order_by_legality(stmt)?;
8301 crate::orderby::check_order_by_positions(stmt)?;
8302 crate::window::reject_window_in_row_clauses(stmt)?;
8303 // A parent's rows are its children's. These walks scan the named
8304 // relation alone, so a partitioned or inherited parent comes back
8305 // short — and silently: the corpus caught `SELECT id FROM pr
8306 // ORDER BY id` and `SELECT k FROM pl ORDER BY k` returning the
8307 // parent's own rows instead of the partitions'. `ONLY` is exactly
8308 // the case that does not fan out, so it stays, which is the test
8309 // the FROM-clause fan-out itself makes.
8310 if !from.primary.only
8311 && crate::partition::has_children(self.active_catalog(), &from.primary.name)
8312 {
8313 return Ok(None);
8314 }
8315 let Some(table) = self.active_catalog().get(&from.primary.name) else {
8316 return Ok(None);
8317 };
8318 // Cold-tier rows live outside `rows()`; this walk would drop
8319 // them silently, the same reason round 831's walk declines.
8320 if table.has_cold_rows_fast() {
8321 return Ok(None);
8322 }
8323
8324 let alias = from
8325 .primary
8326 .alias
8327 .as_deref()
8328 .unwrap_or(from.primary.name.as_str());
8329 let cols = table.schema().columns.clone();
8330 let sess = self.dml_session();
8331 let ctx = EvalContext::new(&cols, Some(alias))
8332 .with_catalog(self.active_catalog())
8333 .with_session(&sess);
8334 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
8335 let order_by = stmt.order_by.clone();
8336 // The same one-shot resolution the general path does (round
8337 // 582): each ORDER BY column is bound once, not once per row.
8338 let order_bound = crate::orderby::order_by_bound_positions(&order_by, &cols, Some(alias));
8339 let descs: Vec<bool> = order_by.iter().map(|o| o.desc).collect();
8340 // Resolved BEFORE the scan, because it now decides what the sort
8341 // STORES and not just what it decodes (round 995).
8342 let needed = Self::sort_record_columns_needed(&stmt.items, &order_bound, cols.len(), &ctx);
8343
8344 let mut sorter = crate::extsort::ExternalSorter::new(
8345 self.temp_run_factory,
8346 self.session_work_mem_bytes(),
8347 cols.clone(),
8348 &descs,
8349 )
8350 .with_stats(&self.spill_stats)
8351 .with_pruned(&needed);
8352 let snapshot = self.current_snapshot();
8353 // One key buffer for the whole scan: `push` drains it and leaves
8354 // the capacity behind.
8355 let mut keys: Vec<OrderKey> = Vec::new();
8356 // r1024 — compile the predicate once for the scan.
8357 //
8358 // These two sorted-spill scans are the paths a single-table SELECT
8359 // with an ORDER BY takes, and they were the last row-returning ones
8360 // still walking the expression tree per row. r1023 did the
8361 // no-ORDER-BY sibling; the sweep's two remaining losing cells are
8362 // exactly this shape.
8363 //
8364 // Found from the profile's CALL TREE rather than its leaves. The
8365 // leaves say what is expensive — `eval_expr` 320, `apply_binary`
8366 // 261, `mod_op` 178 — and two attempts at reasoning out which
8367 // function asked for it were both wrong. The tree names the caller
8368 // chain, and it named this one.
8369 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
8370 .where_
8371 .as_ref()
8372 .filter(|w| crate::eval::fully_compilable(w))
8373 .map(|w| crate::eval::compile_expr(w, &ctx));
8374 let mut eval_stack: Vec<Value<'static>> = Vec::new();
8375 for (i, row) in table.scan_visible_from(0, &snapshot) {
8376 if i.is_multiple_of(256) {
8377 cancel.check()?;
8378 }
8379 if let Some(c) = &compiled_where {
8380 if !crate::eval::compiled::eval_compiled_pred(
8381 c,
8382 row,
8383 &ctx,
8384 &mut eval_stack,
8385 ctx.mysql_dialect,
8386 )? {
8387 continue;
8388 }
8389 } else if let Some(w) = &stmt.where_ {
8390 let cond = crate::eval::eval_expr(w, row, &ctx).map_err(EngineError::Eval)?;
8391 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
8392 continue;
8393 }
8394 }
8395 keys.clear();
8396 crate::orderby::build_order_keys_bound(&order_by, &order_bound, row, &ctx, &mut keys)?;
8397 sorter.push(&mut keys, row)?;
8398 }
8399
8400 let columns: Vec<ColumnSchema> = projection
8401 .iter()
8402 .map(|p| {
8403 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
8404 c.user_enum_type = p.user_enum_type.clone();
8405 c.mysql_fsp = p.mysql_fsp;
8406 c
8407 })
8408 .collect();
8409 emit(crate::StreamItem::Header(&columns))?;
8410
8411 let key_ctx = &ctx;
8412 let mut emitted_since_check = 0usize;
8413 let n = sorter.finish_each(
8414 |src, buf| {
8415 crate::orderby::build_order_keys_bound(&order_by, &order_bound, src, key_ctx, buf)
8416 },
8417 |src, values| {
8418 for p in &projection {
8419 values.push(
8420 crate::eval::eval_expr(&p.expr, src, key_ctx).map_err(EngineError::Eval)?,
8421 );
8422 }
8423 Ok(())
8424 },
8425 |cells| {
8426 // The merge is the long half of a big sort, and the scan's
8427 // check above stops running once it ends: a cancelled
8428 // `SELECT pad FROM big ORDER BY id` delivered all 120k rows
8429 // anyway. Same stride as the scan.
8430 emitted_since_check += 1;
8431 if emitted_since_check >= 256 {
8432 emitted_since_check = 0;
8433 cancel.check()?;
8434 }
8435 emit(crate::StreamItem::Row(crate::RowCells::Values(cells)))
8436 },
8437 )?;
8438 Ok(Some(n))
8439 }
8440
8441 /// One row of the single-table streaming walk: the WHERE test, the
8442 /// projection, the emit. Returns whether a row was emitted.
8443 ///
8444 /// v7.39 (round 970) — factored out because the walk now has two ways
8445 /// to reach a row, the sequential scan and an index seek's candidate
8446 /// positions, and both must do IDENTICALLY this. A copy in each is how
8447 /// two paths for one job drift; this file already carries the cost of
8448 /// that lesson twice (rounds 823 and 961, both resolvers).
8449 ///
8450 /// `#[inline]` so the scan loop keeps the shape round 957 measured it
8451 /// in — a shared hot path pays for a new abstraction whether or not it
8452 /// uses it, and this one is on the scan.
8453 #[inline]
8454 #[allow(clippy::too_many_arguments)]
8455 fn stream_filter_project(
8456 row: &spg_storage::Row<'static>,
8457 where_: Option<&Expr>,
8458 // r1023 — the same WHERE, compiled once by the caller. `None` means
8459 // the expression did not qualify and `where_` is evaluated as before.
8460 compiled_where: Option<&crate::eval::CompiledExpr>,
8461 eval_stack: &mut Vec<Value<'static>>,
8462 projection: &[ProjectedItem],
8463 bound_pos: &[Option<usize>],
8464 ctx: &crate::eval::EvalContext<'_>,
8465 values: &mut Vec<Value<'static>>,
8466 ) -> Result<bool, EngineError> {
8467 // r1023 — this scan ran its predicate through the TREE INTERPRETER,
8468 // once per row, and it was the only row-returning path that did.
8469 // The aggregate path, `table_access`, and the PK walker all compile
8470 // theirs. Profiled: on `SELECT pad FROM d WHERE id % 3 = 0` the
8471 // server's live samples were `eval_expr` 99, `apply_binary` 81,
8472 // `mod_op` 29 — the interpreter, not delivery.
8473 //
8474 // The arithmetic accounted for it exactly. Over the wire, the same
8475 // filter costs 6.375 ms returning rows and 0.679 ms counting them;
8476 // the 5.70 ms difference over 50,000 scanned rows is 114 ns each,
8477 // which is what an interpreted predicate costs against the compiled
8478 // lane's 11.7. It was named "delivery after a filter" before this
8479 // profile, and it was never delivery.
8480 if let Some(c) = compiled_where {
8481 if !crate::eval::compiled::eval_compiled_pred(
8482 c,
8483 row,
8484 ctx,
8485 eval_stack,
8486 ctx.mysql_dialect,
8487 )? {
8488 return Ok(false);
8489 }
8490 } else if let Some(w) = where_ {
8491 let cond = crate::eval::eval_expr(w, row, ctx).map_err(EngineError::Eval)?;
8492 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
8493 return Ok(false);
8494 }
8495 }
8496 values.clear();
8497 for (p, bound) in projection.iter().zip(bound_pos) {
8498 values.push(match bound {
8499 Some(pos) => crate::eval::column_at(*pos, row, ctx).map_err(EngineError::Eval)?,
8500 None => crate::eval::eval_expr(&p.expr, row, ctx).map_err(EngineError::Eval)?,
8501 });
8502 }
8503 Ok(true)
8504 }
8505
8506 /// The same filter and projection, then emit. Split from
8507 /// [`Self::stream_filter_project`] so a path that has to BUFFER rows
8508 /// before it can emit them — a sort — runs the identical predicate and
8509 /// projection rather than a second copy of them.
8510 #[allow(clippy::too_many_arguments)]
8511 fn stream_project_row<F>(
8512 row: &spg_storage::Row<'static>,
8513 where_: Option<&Expr>,
8514 compiled_where: Option<&crate::eval::CompiledExpr>,
8515 eval_stack: &mut Vec<Value<'static>>,
8516 projection: &[ProjectedItem],
8517 bound_pos: &[Option<usize>],
8518 ctx: &crate::eval::EvalContext<'_>,
8519 values: &mut Vec<Value<'static>>,
8520 emit: &mut F,
8521 ) -> Result<bool, EngineError>
8522 where
8523 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
8524 {
8525 if !Self::stream_filter_project(
8526 row,
8527 where_,
8528 compiled_where,
8529 eval_stack,
8530 projection,
8531 bound_pos,
8532 ctx,
8533 values,
8534 )? {
8535 return Ok(false);
8536 }
8537 emit(crate::StreamItem::Row(crate::RowCells::Values(values)))?;
8538 Ok(true)
8539 }
8540
8541 fn try_stream_single_table<F>(
8542 &self,
8543 stmt: &SelectStatement,
8544 from: &FromClause,
8545 cancel: CancelToken<'_>,
8546 emit: &mut F,
8547 ) -> Result<Option<usize>, EngineError>
8548 where
8549 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
8550 {
8551 let Some(table) = self.active_catalog().get(&from.primary.name) else {
8552 return Ok(None);
8553 };
8554 // Cold-tier rows live outside `rows()`; the materialising fallback
8555 // covers both tiers and this walk would silently drop them.
8556 if table.has_cold_rows_fast() {
8557 return Ok(None);
8558 }
8559 let alias = from
8560 .primary
8561 .alias
8562 .as_deref()
8563 .unwrap_or(from.primary.name.as_str());
8564 let cols = table.schema().columns.clone();
8565 let sess = self.dml_session();
8566 let ctx = EvalContext::new(&cols, Some(alias))
8567 .with_catalog(self.active_catalog())
8568 .with_session(&sess);
8569 let projection = build_projection(&stmt.items, &cols, alias, self.backslash_escapes)?;
8570
8571 let columns: Vec<ColumnSchema> = projection
8572 .iter()
8573 .map(|p| {
8574 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
8575 c.user_enum_type = p.user_enum_type.clone();
8576 c.mysql_fsp = p.mysql_fsp;
8577 c
8578 })
8579 .collect();
8580 emit(crate::StreamItem::Header(&columns))?;
8581
8582 // v7.37 (round 957) — resolve each bare-column projection ONCE
8583 // instead of once per row. `find_column_pos`-style resolution is a
8584 // linear walk of the schema comparing column-name strings, and the
8585 // row loop below ran it for every cell of every row: measured at
8586 // 400k rows, binding it out of the loop took `SELECT pad` from
8587 // 16.5-17.5 ms to 10.9-11.7 ms (-41%, two windows, round 954).
8588 //
8589 // ORDER BY has bound its keys this way since round 582
8590 // (`order_by_bound_positions`); the projection never did.
8591 //
8592 // `locate_column` is the same resolution `resolve_column` performs,
8593 // returning the site instead of the value, so the two cannot drift
8594 // apart the way a second hand-written resolver would. Anything it
8595 // declines — an expression, a whole-row reference, a name that does
8596 // not resolve — binds to `None` and takes the general path below,
8597 // errors included, so an empty table still reports nothing rather
8598 // than raising at bind time.
8599 let bound_pos: Vec<Option<usize>> = projection
8600 .iter()
8601 .map(|p| match &p.expr {
8602 Expr::Column(c) => match crate::eval::locate_column(c, &ctx) {
8603 Ok(Some(pos)) => Some(pos),
8604 _ => None,
8605 },
8606 _ => None,
8607 })
8608 .collect();
8609
8610 // One snapshot for the whole scan, as the materialising path takes.
8611 let snapshot = self.current_snapshot();
8612
8613 // v7.39 (round 970) — ask the indices BEFORE walking the table.
8614 //
8615 // This walk had no index step at all, and it is preferred over the
8616 // materialising path, which does have one (`pick_indexed_rows` ->
8617 // `try_index_seek`). So a primary-key point lookup — the commonest
8618 // statement there is — read every row: measured on 500k rows,
8619 // `SELECT * FROM big WHERE id = 250000` took 14.947 ms against
8620 // PG18.4's 0.172 ms, and the cost tracked the TABLE (1k 0.315 ms,
8621 // 10k 1.660, 100k 3.518), which is not what O(log n) looks like.
8622 //
8623 // The control that named it: `... OFFSET 0` — semantically the same
8624 // query — answered in 0.159 ms, because OFFSET is one of the shape
8625 // gates that declines this walk and sends the statement to the path
8626 // that seeks. `LIMIT 1` and `GROUP BY` did the same. The three have
8627 // no semantics in common; what they share is making this function
8628 // stand down.
8629 //
8630 // The seek only NARROWS: every candidate still goes through the
8631 // full WHERE below, exactly as the mutation paths use it, so a
8632 // partial index match cannot change an answer. Positions come back
8633 // already visibility-filtered and already capped at a quarter of the
8634 // table (round 490), so a seek can never cost more than the scan it
8635 // replaces, and `None` means "walk the table" as before.
8636 //
8637 // Sorted because the scan would have produced table order and the
8638 // index produces key order. Without an ORDER BY neither is promised,
8639 // but a walk that silently reorders its answer when an index happens
8640 // to exist is a difference nobody asked for.
8641 let seek_positions: Option<Vec<usize>> = stmt.where_.as_ref().and_then(|w| {
8642 crate::index_access::try_index_seek_positions(w, &cols, table, alias, &snapshot)
8643 });
8644
8645 let mut values: Vec<Value<'static>> = Vec::with_capacity(projection.len());
8646 // r1023 — compile the predicate once for the whole scan. Same gate
8647 // every other path uses: `fully_compilable` or keep the interpreter,
8648 // so a shape the VM cannot take answers exactly as it did before.
8649 let compiled_where: Option<crate::eval::CompiledExpr> = stmt
8650 .where_
8651 .as_ref()
8652 .filter(|w| crate::eval::fully_compilable(w))
8653 .map(|w| crate::eval::compile_expr(w, &ctx));
8654 let mut eval_stack: Vec<Value<'static>> = Vec::new();
8655 let mut count: usize = 0;
8656 match seek_positions {
8657 Some(mut positions) => {
8658 positions.sort_unstable();
8659 for (n, pos) in positions.into_iter().enumerate() {
8660 if n.is_multiple_of(256) {
8661 cancel.check()?;
8662 }
8663 let Some(row) = table.rows().get(pos) else {
8664 continue;
8665 };
8666 if Self::stream_project_row(
8667 row,
8668 stmt.where_.as_ref(),
8669 compiled_where.as_ref(),
8670 &mut eval_stack,
8671 &projection,
8672 &bound_pos,
8673 &ctx,
8674 &mut values,
8675 emit,
8676 )? {
8677 count += 1;
8678 }
8679 }
8680 }
8681 None => {
8682 for (i, row) in table.scan_visible_from(0, &snapshot) {
8683 if i.is_multiple_of(256) {
8684 cancel.check()?;
8685 }
8686 if Self::stream_project_row(
8687 row,
8688 stmt.where_.as_ref(),
8689 compiled_where.as_ref(),
8690 &mut eval_stack,
8691 &projection,
8692 &bound_pos,
8693 &ctx,
8694 &mut values,
8695 emit,
8696 )? {
8697 count += 1;
8698 }
8699 }
8700 }
8701 }
8702 Ok(Some(count))
8703 }
8704
8705 pub(crate) fn try_exec_joined_streaming<F>(
8706 &self,
8707 stmt: &SelectStatement,
8708 cancel: CancelToken<'_>,
8709 emit: &mut F,
8710 ) -> Result<Option<usize>, EngineError>
8711 where
8712 F: FnMut(crate::StreamItem<'_>) -> Result<(), EngineError>,
8713 {
8714 // Shape gates — keep the streamable surface narrow on
8715 // purpose. The fall-back path still handles everything else.
8716 let Some(from) = &stmt.from else {
8717 return Ok(None);
8718 };
8719 // v7.37 (round 830) — decline anything a row-security policy binds
8720 // for this session. Policies are injected in
8721 // `exec_bare_select_cancel`, below this path, so a statement claimed
8722 // here would read the table unfiltered: measured, `SELECT val FROM
8723 // sec` returned all three rows to a session whose policy allows two,
8724 // while `SELECT upper(val) FROM sec` — declined by the shape gates
8725 // and so materialised — returned the correct two.
8726 //
8727 // Declining sends it to the path that enforces. Teaching this one to
8728 // inject the predicate itself would keep the streaming benefit for
8729 // RLS tables and is the better end state; it is not what a
8730 // correctness fix should carry, and the fall-back is exactly as
8731 // correct, only slower.
8732 if self.select_reads_policy_subject_table(stmt) {
8733 return Ok(None);
8734 }
8735 // r1058 — a WITH list this path never materialises: the CTE
8736 // name would be resolved as a physical relation and error
8737 // ("relation \"big\" does not exist" over the extended
8738 // protocol, caught by the perm-runner's wire legs). The
8739 // materialising fallback owns CTE execution.
8740 if !stmt.ctes.is_empty() {
8741 return Ok(None);
8742 }
8743 // r1058 — rewritten system catalogs (`__spg_pg_stat_user_
8744 // tables` and kin) exist only as synth arms on the
8745 // materialising path; claiming one here errored "relation
8746 // does not exist" over the extended protocol for a query the
8747 // simple protocol answered. Prefix test only — a genuinely
8748 // missing relation must keep erroring in-path.
8749 if from.primary.name.starts_with("__spg_")
8750 || from
8751 .joins
8752 .iter()
8753 .any(|j| j.table.name.starts_with("__spg_"))
8754 {
8755 return Ok(None);
8756 }
8757 // r1058 — decline partitioned / inheritance parents, same
8758 // shape of bug as the RLS decline above: this path scans the
8759 // named table's own (empty) heap, so `SELECT id, region FROM
8760 // cust` on a partition parent streamed ZERO rows over the wire
8761 // while COUNT(*) — an aggregate, materialised below — said 3.
8762 // Caught by the perm-runner's server permutations; the
8763 // materialising fallback expands children correctly.
8764 if crate::partition::has_children(self.active_catalog(), &from.primary.name)
8765 || from
8766 .joins
8767 .iter()
8768 .any(|j| crate::partition::has_children(self.active_catalog(), &j.table.name))
8769 {
8770 return Ok(None);
8771 }
8772 // v7.39 (round 790) — single-table SELECTs stream too. This
8773 // gate said "joins only" because the path was written for
8774 // mailrs's joined PROJ shape; a plain `SELECT <cols> FROM t`
8775 // fell to the materialising fallback, which builds the whole
8776 // `Vec<Row<'static>>` and only then iterates it. Measured on
8777 // 300k rows: 181 MB single-table vs 70 MB for the SAME rows
8778 // reached through a one-row JOIN — 2.6x, purely for lacking a
8779 // join. The deferred-join structure handles one source as the
8780 // degenerate stride-1 case, so the walk below is unchanged.
8781 let _single_table = from.joins.is_empty();
8782 // An ORDER BY that the bounded sort can serve streams; everything
8783 // else still falls to the materialising fallback below.
8784 // r1025 — an ordering the index already holds needs no sort at all.
8785 // Tried before the spill sort, which is the path it replaces.
8786 if !stmt.order_by.is_empty()
8787 && from.joins.is_empty()
8788 && let Some(n) = self.try_index_order_stream(stmt, from, cancel, emit)?
8789 {
8790 return Ok(Some(n));
8791 }
8792 if !stmt.order_by.is_empty()
8793 && from.joins.is_empty()
8794 && let Some(n) = self.try_spill_sorted_stream(stmt, from, cancel, emit)?
8795 {
8796 return Ok(Some(n));
8797 }
8798 // r1031 — integer keys carried inline instead of an `OrderKey`
8799 // vector per row. Tried AFTER the spill sort on purpose: this lane
8800 // buffers the whole answer, so anything the spill path would take
8801 // must keep taking it rather than be turned back into an in-memory
8802 // sort that answers with a budget error.
8803 if !stmt.order_by.is_empty()
8804 && from.joins.is_empty()
8805 && let Some(n) = self.try_int_key_sorted_stream(stmt, from, cancel, emit)?
8806 {
8807 return Ok(Some(n));
8808 }
8809 if !stmt.order_by.is_empty()
8810 || stmt.limit.is_some()
8811 || stmt.offset.is_some()
8812 || stmt.having.is_some()
8813 || stmt.group_by.is_some()
8814 || stmt.distinct
8815 || !stmt.unions.is_empty()
8816 || stmt.limit_with_ties
8817 {
8818 return Ok(None);
8819 }
8820 if aggregate::uses_aggregate(stmt) {
8821 return Ok(None);
8822 }
8823 // No window / SRF on the streaming path.
8824 if select_has_window(stmt) {
8825 return Ok(None);
8826 }
8827 if stmt
8828 .items
8829 .iter()
8830 .any(|i| matches!(i, SelectItem::Expr { expr, .. } if is_top_level_unnest(expr)))
8831 {
8832 return Ok(None);
8833 }
8834 // v7.37 (round 831) — a joinless FROM over a plain stored table
8835 // never needs the deferred structure, and building one costs the
8836 // whole table. `materialise_table_ref_filtered` clones every row
8837 // into a `Vec<Row<'static>>` before anything is filtered or
8838 // projected, so peak cost tracks the TABLE, not the result:
8839 // measured over 300k rows of 200 bytes, `SELECT id FROM big` and
8840 // `SELECT pad FROM big` both cost +107 MB over baseline, the narrow
8841 // projection saving nothing, while an arithmetic projection — which
8842 // the shape gates decline, so it materialises through the ordinary
8843 // executor — cost +21 MB.
8844 //
8845 // Scanning in batches and releasing each one is what `cursor_fill`
8846 // already does for a lazy cursor, and it is the same walk: resume
8847 // from a slot, take visible rows, evaluate, hand them over, drop
8848 // them. Round 800's finding stands and is why this reads rows OUT
8849 // rather than seeding the join by index — touching the stored
8850 // `PersistentVec` in place makes the whole table resident, which is
8851 // worse than the copy. Each batch is copied, then freed.
8852 if from.joins.is_empty()
8853 && from.primary.unnest_expr.is_none()
8854 && from.primary.lateral_subquery.is_none()
8855 && from.primary.as_of_segment.is_none()
8856 && from.primary.generate_series_args.is_none()
8857 && let Some(n) = self.try_stream_single_table(stmt, from, cancel, emit)?
8858 {
8859 return Ok(Some(n));
8860 }
8861 // Build the deferred join under the regular byte budget.
8862 let mut budget = ByteBudget::new(self.max_query_bytes);
8863 let deferred = {
8864 let mut needed = alloc::collections::BTreeSet::new();
8865 let prunable = collect_qualified_refs(stmt, &mut needed).is_some();
8866 self.build_joined_filtered_rows(
8867 from,
8868 stmt.where_.as_ref(),
8869 cancel,
8870 if prunable { Some(&needed) } else { None },
8871 &mut budget,
8872 )?
8873 };
8874 let combined_schema = &deferred.combined_schema;
8875 // v7.39 (read01 round 53) — carry the catalog (see join.rs): a
8876 // `::regclass` / enum cast in a joined projection or HAVING needs it.
8877 // v7.39 (round 525) — and the session: a joined SELECT's WHERE is
8878 // the same predicate the unjoined shape carries.
8879 let joined_sess = self.dml_session();
8880 let ctx = EvalContext::new(combined_schema, None)
8881 .with_catalog(self.active_catalog())
8882 .with_session(&joined_sess);
8883 let projection =
8884 build_projection(&stmt.items, combined_schema, "", self.backslash_escapes)?;
8885 // Every projection item must be a bound qualified column —
8886 // anything that needs `eval_expr_with_correlated` keeps the
8887 // materialising path.
8888 let bound_pos = |e: &Expr| -> Option<usize> {
8889 match e {
8890 // v7.39 (round 822) — an UNQUALIFIED column resolves here
8891 // too. The `qualifier.is_some()` guard this replaces meant
8892 // `SELECT pad FROM big` — the commonest projection there is
8893 // — never reached the streaming walk: it fell out at this
8894 // gate and re-ran on the materialising path, after the
8895 // deferred join structure had already been built and paid
8896 // for. Measured (round 821, statement_timeout=120 over 400k
8897 // rows): `big.pad` and `b.pad` streamed and cancelled at
8898 // ~65k rows in 0.14 s, while bare `pad` ran to completion in
8899 // 0.80 s with the timeout never consulted. `find_column_pos`
8900 // has always handled the unqualified case (it falls through
8901 // to a by-name match), so the guard narrowed the gate for no
8902 // reason it recorded.
8903 Expr::Column(c) => eval::find_column_pos(c, &ctx),
8904 _ => None,
8905 }
8906 };
8907 let proj_decomposed: Vec<(usize, usize)> = {
8908 let mut out = Vec::with_capacity(projection.len());
8909 for p in &projection {
8910 let Some(abs) = bound_pos(&p.expr) else {
8911 return Ok(None);
8912 };
8913 let Some(k) = deferred
8914 .offsets
8915 .partition_point(|&o| o <= abs)
8916 .checked_sub(1)
8917 else {
8918 return Ok(None);
8919 };
8920 out.push((k, abs - deferred.offsets[k]));
8921 }
8922 out
8923 };
8924 // Emit columns once.
8925 let columns: Vec<ColumnSchema> = projection
8926 .iter()
8927 // v7.39 (read01 round 54) — keep the column's enum identity through
8928 // the projection (it lives outside the DataType lattice), or a
8929 // derived table / UNION / windowed result forgets it and any outer
8930 // `ORDER BY <enum col>` silently sorts by the label's TEXT.
8931 .map(|p| {
8932 let mut c = ColumnSchema::new(p.output_name.clone(), p.ty, p.nullable);
8933 c.user_enum_type = p.user_enum_type.clone();
8934 c.mysql_fsp = p.mysql_fsp;
8935 c
8936 })
8937 .collect();
8938 emit(crate::StreamItem::Header(&columns))?;
8939 let sources_ref = &deferred.sources;
8940 let stride = deferred.stride;
8941 let survivors_ref = &deferred.survivors;
8942 let n_surv = if stride == 0 {
8943 0
8944 } else {
8945 survivors_ref.len() / stride
8946 };
8947 // Reused per-row cell-ref scratch — pushes are zero-alloc
8948 // after the first row.
8949 let null_value = Value::Null;
8950 let mut cell_refs: Vec<&Value> = Vec::with_capacity(projection.len());
8951 let mut count: usize = 0;
8952 for surv_i in 0..n_surv {
8953 if surv_i.is_multiple_of(256) {
8954 cancel.check()?;
8955 }
8956 let tuple = &survivors_ref[surv_i * stride..(surv_i + 1) * stride];
8957 cell_refs.clear();
8958 for &(k, col_in_src) in &proj_decomposed {
8959 let ri = tuple[k];
8960 let v: &Value = if ri == usize::MAX {
8961 &null_value
8962 } else {
8963 sources_ref[k]
8964 .get(ri)
8965 .and_then(|r| r.values.get(col_in_src))
8966 .unwrap_or(&null_value)
8967 };
8968 cell_refs.push(v);
8969 }
8970 emit(crate::StreamItem::Row(crate::RowCells::Refs(&cell_refs)))?;
8971 count += 1;
8972 }
8973 Ok(Some(count))
8974 }
8975
8976 fn exec_joined_select(
8977 &self,
8978 stmt: &SelectStatement,
8979 from: &FromClause,
8980 cancel: CancelToken<'_>,
8981 ) -> Result<QueryResult, EngineError> {
8982 // v7.37.x (docker-fair NOTEX attack) — short-circuit COUNT(*)
8983 // over a LEFT ANTI JOIN. The v7.37.27 NOT EXISTS pullup
8984 // rewrites `SELECT COUNT(*) FROM A WHERE NOT EXISTS (SELECT 1
8985 // FROM B WHERE B.k = A.k)` into
8986 // SELECT COUNT(*) FROM A LEFT JOIN B ON B.k = A.k
8987 // WHERE B.k IS NULL
8988 // The general join executor builds a hash, probes every outer
8989 // tuple, materialises (left_padded_with_null) for every miss,
8990 // then runs the aggregate over the result set. For COUNT(*) we
8991 // only need the count — skip the tuple materialisation. Build
8992 // a HashSet of B's unique join values, scan A's PK index, and
8993 // increment the counter on each miss. PG's Merge Anti-Join
8994 // does roughly this; ours becomes a simple HashSet probe.
8995 if let Some(out) = self.try_count_star_left_anti_join_fast(stmt, from)? {
8996 return Ok(out);
8997 }
8998 // v7.34.5 (mailrs prod #5) — walker-driven join + early stop.
8999 // When ORDER BY is on an indexed primary column, walking the
9000 // btree in the requested direction lets the streamer break
9001 // after `LIMIT + OFFSET` survivors without ever materialising
9002 // the rest of the join — the 80 ms `mailrs_prod_not_exists`
9003 // plateau is exactly this shape.
9004 if let Some(out) = self.try_streamed_inner_join_walk_topn(stmt, from, cancel)? {
9005 return Ok(out);
9006 }
9007 // v7.30.3 (mailrs round-26) — the bounded single-join path
9008 // first; peak memory scales with LIMIT instead of the table.
9009 if let Some(out) = self.try_streamed_inner_join_topn(stmt, from, cancel)? {
9010 return Ok(out);
9011 }
9012 // v7.17.0 Phase 3.P0-43 + P0-41 — delegate the join +
9013 // WHERE materialisation to the shared helper so the LATERAL
9014 // / UNNEST / regular-catalog paths route through one place.
9015 // (`build_joined_filtered_rows` carries LATERAL support as
9016 // of Phase 3.P0-41.) Downstream we still handle aggregate /
9017 // projection / ORDER BY / DISTINCT / LIMIT inline because
9018 // those depend on the SelectStatement's items list.
9019 let mut budget = ByteBudget::new(self.max_query_bytes);
9020 let deferred = {
9021 let mut needed = alloc::collections::BTreeSet::new();
9022 let prunable = collect_qualified_refs(stmt, &mut needed).is_some();
9023 self.build_joined_filtered_rows(
9024 from,
9025 stmt.where_.as_ref(),
9026 cancel,
9027 if prunable { Some(&needed) } else { None },
9028 &mut budget,
9029 )?
9030 };
9031 let combined_schema = &deferred.combined_schema;
9032 // v7.39 (read01 round 53) — carry the catalog (see join.rs): a
9033 // `::regclass` / enum cast in a joined projection or HAVING needs it.
9034 // v7.39 (round 525) — and the session: a joined SELECT's WHERE is
9035 // the same predicate the unjoined shape carries.
9036 let joined_sess = self.dml_session();
9037 let ctx = EvalContext::new(combined_schema, None)
9038 .with_catalog(self.active_catalog())
9039 .with_session(&joined_sess);
9040 // Aggregate path: handle GROUP BY / aggregate calls over the
9041 // joined+filtered rows.
9042 if aggregate::uses_aggregate(stmt) {
9043 // v7.32 (P4 borrow channel, increment 2) — borrow each
9044 // surviving join tuple as a RowRef::Tuple; the aggregate
9045 // engine reads source cells by reference (bound fast path =
9046 // zero clone) instead of consuming materialised combined
9047 // Rows. This is where the +211k materialise_tuple_vals
9048 // clones disappear for the join+aggregate shape.
9049 let refs = deferred.row_refs();
9050 // v7.29 — a per-query memo so correlated scalar
9051 // subqueries batch-evaluate once (group map) instead of
9052 // executing per group.
9053 let agg_memo = core::cell::RefCell::new(memoize::MemoizeCache::default());
9054 let agg_correlated = |e: &Expr, r: &Row<'static>, c: &EvalContext<'_>| {
9055 self.eval_expr_with_correlated(e, r, c, cancel, Some(&mut agg_memo.borrow_mut()))
9056 .map_err(|err| match err {
9057 EngineError::Eval(ev) => ev,
9058 other => eval::EvalError::TypeMismatch {
9059 detail: alloc::format!("{other}"),
9060 },
9061 })
9062 };
9063 let agg = aggregate::run(
9064 stmt,
9065 crate::join::AggRows::Refs(&refs),
9066 combined_schema,
9067 None,
9068 Some(&agg_correlated),
9069 self.parallel_runner.0.as_deref(),
9070 Some(self.active_catalog()),
9071 Some(self),
9072 )?;
9073 return self.finish_agg_result(agg, stmt, cancel);
9074 }
9075
9076 let projection =
9077 build_projection(&stmt.items, combined_schema, "", self.backslash_escapes)?;
9078 // v7.39 (round 734) — a set-returning projection over a JOIN.
9079 // This executor's projection loop treats every item as a scalar,
9080 // so `SELECT unnest(ARRAY[a.id, b.g]) FROM a JOIN b …` died with
9081 // "function unnest(integer[]) does not exist" where PG expands
9082 // it. The row-set executor already carries the full SRF pipeline
9083 // (lockstep expansion, ORDER-BY-on-expanded-rows, the round-733
9084 // sharding): materialise the joined survivors and hand over. The
9085 // WHERE is cleared — the join already applied it, and combined
9086 // columns resolve identically in both executors.
9087 if !self.srf_target_idxs(&projection).is_empty() {
9088 let refs = deferred.row_refs();
9089 let rows: Vec<Row<'static>> = refs.iter().map(|r| r.as_row().into_owned()).collect();
9090 let mut s2 = stmt.clone();
9091 s2.where_ = None;
9092 let schema = combined_schema.clone();
9093 return self.exec_select_over_rows(&s2, rows, schema, "", cancel);
9094 }
9095 // v7.33 (P4 borrow channel, increment 3) — project directly off
9096 // the deferred row-index tuples instead of materialising an
9097 // intermediate combined Row per survivor. A bound qualified
9098 // column is read by reference (`RowRef::get` → `tuple_value`) and
9099 // cloned ONCE into the output row; the old `materialise()` (a full
9100 // combined Row plus a source→intermediate clone per referenced
9101 // cell, for every survivor) is gone. A row materialises on demand
9102 // only when a projection or ORDER BY expression needs the eval
9103 // path (subquery / function / arithmetic / unqualified column).
9104 // Same bind-once classification the aggregate input fast path uses
9105 // (`accumulate_groups`), reading the same `tuple_value` mapping the
9106 // differential gate already covers.
9107 let refs = deferred.row_refs();
9108 let bound_pos = |e: &Expr| -> Option<usize> {
9109 match e {
9110 Expr::Column(c) if c.qualifier.is_some() => eval::find_column_pos(c, &ctx),
9111 _ => None,
9112 }
9113 };
9114 let proj_pos: Vec<Option<usize>> = projection.iter().map(|p| bound_pos(&p.expr)).collect();
9115 let all_proj_bound = proj_pos.iter().all(Option::is_some);
9116 // v7.36 (perf — mailrs Phase 1, PROJ SPGS 8.93 → ?) —
9117 // pre-decompose each bound projection position into
9118 // `(source_k, col_in_source)` so the per-row column read
9119 // skips the per-cell `tuple_value` partition_point + slice
9120 // walk. For PROJ_25k (5 cols × 25k rows = 125k tuple_value
9121 // calls) that walk dominated; this version reaches into
9122 // `pipe.sources[k].get(tuple[k])?.values[col]` directly.
9123 let proj_decomposed: Vec<Option<(usize, usize)>> = proj_pos
9124 .iter()
9125 .map(|p| {
9126 p.and_then(|abs| {
9127 let k = deferred
9128 .offsets
9129 .partition_point(|&o| o <= abs)
9130 .checked_sub(1)?;
9131 Some((k, abs - deferred.offsets[k]))
9132 })
9133 })
9134 .collect();
9135 // v7.39 (round 962) — which projection items are whole-row
9136 // references, and to which join source. The test is
9137 // `locate_column` declining the name, which is the SAME resolver
9138 // the evaluation path uses, so this cannot drift from it: a real
9139 // column carrying an alias's name resolves to a position and is
9140 // not reported here. The source index comes from the alias
9141 // prefix, the way the combined schema names its columns.
9142 let whole_row_src: Vec<Option<usize>> = projection
9143 .iter()
9144 .map(|p| {
9145 let Expr::Column(c) = &p.expr else {
9146 return None;
9147 };
9148 if !matches!(eval::locate_column(c, &ctx), Ok(None)) {
9149 return None;
9150 }
9151 let prefix = alloc::format!("{name}.", name = c.name);
9152 let abs = deferred
9153 .combined_schema
9154 .iter()
9155 .position(|s| s.name.starts_with(&prefix))?;
9156 deferred
9157 .offsets
9158 .partition_point(|&o| o <= abs)
9159 .checked_sub(1)
9160 })
9161 .collect();
9162 // ORDER BY (when present) still evaluates against a materialised
9163 // Row — keep the order-key encoder correct rather than fork it.
9164 let need_eval_row = !all_proj_bound || !stmt.order_by.is_empty();
9165 let mut tagged: Vec<(Vec<OrderKey>, Row<'static>)> = Vec::new();
9166 let mut proj_memo = memoize::MemoizeCache::default();
9167 let sources_ref = &deferred.sources;
9168 let stride = deferred.stride;
9169 let survivors_ref = &deferred.survivors;
9170 let n_surv = survivors_ref.len() / stride.max(1);
9171 // v7.38 (read01 B8) — streaming top-N budget (see the sibling
9172 // single-table path). Bounds this JOIN projection's accumulator
9173 // to O(keep) for `ORDER BY … LIMIT k`.
9174 let topk_stream: Option<(usize, Vec<bool>)> = if !stmt.order_by.is_empty()
9175 && !stmt.distinct
9176 && !stmt.limit_with_ties
9177 && !self.env_cfg().disable_topk
9178 {
9179 stmt.limit_literal().and_then(|l| {
9180 let keep = (l as usize).saturating_add(stmt.offset_literal().unwrap_or(0) as usize);
9181 (keep >= 1).then(|| (keep, stmt.order_by.iter().map(|o| o.desc).collect()))
9182 })
9183 } else {
9184 None
9185 };
9186 // v7.37.16 — streaming DISTINCT seen-set (see scan-path twin).
9187 let mut seen_distinct: hashbrown::HashMap<u64, crate::distinct::DistinctBucket> =
9188 hashbrown::HashMap::new();
9189 let distinct_hb = hashbrown::DefaultHashBuilder::default();
9190 for surv_i in 0..n_surv {
9191 let tuple = &survivors_ref[surv_i * stride..(surv_i + 1) * stride];
9192 let row = &refs[surv_i];
9193 let materialised: Option<Cow<'_, Row<'static>>> = if need_eval_row {
9194 Some(row.as_row())
9195 } else {
9196 None
9197 };
9198 let mut values = Vec::with_capacity(projection.len());
9199 for (i, p) in projection.iter().enumerate() {
9200 if let Some((k, col_in_src)) = proj_decomposed[i] {
9201 // v7.36 — direct (source_k, col) lookup, no
9202 // partition_point. tuple[k] is the row index in
9203 // sources[k]; LEFT-NULL slots are `usize::MAX`.
9204 let ri = tuple[k];
9205 let v: Value<'static> = if ri == usize::MAX {
9206 Value::Null
9207 } else {
9208 sources_ref[k]
9209 .get(ri)
9210 .and_then(|r| r.values.get(col_in_src))
9211 .cloned()
9212 .map(Value::into_owned)
9213 .unwrap_or(Value::Null)
9214 };
9215 values.push(v);
9216 } else if let Some(pos) = proj_pos[i] {
9217 // Bound but couldn't decompose (shouldn't normally
9218 // happen — keep as a safe path).
9219 values.push(
9220 row.get(pos)
9221 .cloned()
9222 .map(Value::into_owned)
9223 .unwrap_or(Value::Null),
9224 );
9225 } else if let Some(k) = whole_row_src[i]
9226 && tuple[k] == usize::MAX
9227 {
9228 // v7.39 (round 962) — a whole-row reference to a side
9229 // an OUTER join null-extended is NULL, not a
9230 // composite whose fields are all NULL. PG18.4 answers
9231 // `SELECT jb FROM wr LEFT JOIN jb ON <no match>` with
9232 // an empty cell; round 961 answered `(,)`.
9233 //
9234 // The evaluator below cannot tell the two apart: it
9235 // reads the MATERIALISED combined row, where a
9236 // null-extended side is indistinguishable from a real
9237 // row whose every column is NULL — and that row is
9238 // `(,)` in PG too, so guessing by "all fields NULL"
9239 // would trade one wrong answer for another. The
9240 // tuple, which is still in hand here, does know:
9241 // `usize::MAX` is the sentinel the join writes for
9242 // exactly this.
9243 values.push(Value::Null);
9244 } else {
9245 // Eval path — `materialised` is Some whenever any
9246 // projection item is non-bound (need_eval_row true).
9247 // v7.24 (round-16 B) — select-list subqueries under a
9248 // JOIN go through the correlated-aware evaluator too.
9249 let mrow = materialised.as_deref().expect("materialised for eval");
9250 values.push(self.eval_expr_with_correlated(
9251 &p.expr,
9252 mrow,
9253 &ctx,
9254 cancel,
9255 Some(&mut proj_memo),
9256 )?);
9257 }
9258 }
9259 let out_row = Row::new(values);
9260 // v7.37.16 — streaming DISTINCT (see the scan-path twin):
9261 // probe on the projected row; duplicates skip the
9262 // build_order_keys eval and never enter `tagged`.
9263 if stmt.distinct {
9264 let bucket = seen_distinct
9265 .entry(norm_hash_row(&out_row, &distinct_hb, ctx.mysql_dialect))
9266 .or_default();
9267 if bucket
9268 .iter()
9269 .any(|i| row_eq_norm(&tagged[i].1, &out_row, ctx.mysql_dialect))
9270 {
9271 continue;
9272 }
9273 bucket.push(tagged.len());
9274 }
9275 let order_keys = if stmt.order_by.is_empty() {
9276 Vec::new()
9277 } else {
9278 let mrow = materialised.as_deref().expect("materialised for order by");
9279 build_order_keys(&stmt.order_by, mrow, &ctx)?
9280 };
9281 budget.charge(approx_row_bytes(&out_row))?;
9282 tagged.push((order_keys, out_row));
9283 if let Some((k, descs)) = &topk_stream {
9284 topk_trim(&mut tagged, *k, descs);
9285 }
9286 }
9287 if !stmt.order_by.is_empty() {
9288 // v7.38 元机制 D acceptor — see other call site above.
9289 let keep = if self.env_cfg().disable_topk {
9290 None
9291 } else {
9292 stmt.limit_literal()
9293 .map(|l| l as usize + stmt.offset_literal().map_or(0, |o| o as usize))
9294 };
9295 let descs: Vec<bool> = stmt.order_by.iter().map(|o| o.desc).collect();
9296 // v7.39 (round 688) — the join's ORDER BY resolves its keys
9297 // against `ctx`, which is built from `build_combined_schema`, so
9298 // this is where a declared collation reaches the sort. There was
9299 // exactly ONE resolver call in the engine before this — the
9300 // single-table scan's — which is why every other shape sorted by
9301 // bytes no matter what the schemas carried.
9302 let colls = crate::orderby::order_by_collations(&stmt.order_by, &ctx)?;
9303 crate::orderby::partial_sort_tagged_in(&mut tagged, keep, &descs, &colls);
9304 }
9305 let mut output_rows: Vec<Row<'static>> = tagged.into_iter().map(|(_, r)| r).collect();
9306 apply_offset_and_limit(
9307 &mut output_rows,
9308 stmt.offset_literal(),
9309 stmt.limit_literal(),
9310 );
9311 let columns: Vec<ColumnSchema> = projection
9312 .into_iter()
9313 .map(|p| {
9314 let mut c = ColumnSchema::new(p.output_name, p.ty, p.nullable);
9315 c.user_enum_type = p.user_enum_type;
9316 c.collation_name = p.collation_name;
9317 c.mysql_fsp = p.mysql_fsp;
9318 c
9319 })
9320 .collect();
9321 Ok(QueryResult::Rows {
9322 columns,
9323 rows: output_rows,
9324 })
9325 }
9326}
9327
9328impl Engine {
9329 /// v6.10.2 — cold-tier time-travel scan. Resolves the segment
9330 /// by id, decodes each row body against the table's current
9331 /// schema, applies the SELECT's projection + optional WHERE +
9332 /// optional LIMIT, returns a `Rows` result. JOINs / aggregates
9333 /// / ORDER BY are unsupported on this path (STABILITY carve-
9334 /// out); operators wanting them should restore the segment
9335 /// into a regular table first.
9336 fn exec_select_as_of_segment(
9337 &self,
9338 stmt: &SelectStatement,
9339 from: &spg_sql::ast::FromClause,
9340 segment_id: u32,
9341 ) -> Result<QueryResult, EngineError> {
9342 // v6.10.2 scope: no joins, no aggregates, no ORDER BY,
9343 // no GROUP BY / HAVING / UNION / OFFSET / DISTINCT.
9344 if !from.joins.is_empty()
9345 || stmt.group_by.is_some()
9346 || stmt.having.is_some()
9347 || !stmt.unions.is_empty()
9348 || !stmt.order_by.is_empty()
9349 || stmt.offset.is_some()
9350 || stmt.distinct
9351 || aggregate::uses_aggregate(stmt)
9352 {
9353 return Err(EngineError::Unsupported(
9354 "AS OF SEGMENT supports SELECT projection + WHERE + LIMIT only \
9355 (joins / aggregates / ORDER BY are STABILITY § \"Out of v6.10\")"
9356 .into(),
9357 ));
9358 }
9359 let table = self
9360 .active_catalog()
9361 .get(&from.primary.name)
9362 .ok_or_else(|| StorageError::TableNotFound {
9363 name: from.primary.name.clone(),
9364 })?;
9365 let schema = table.schema().clone();
9366 let schema_cols = &schema.columns;
9367 let alias = from
9368 .primary
9369 .alias
9370 .as_deref()
9371 .unwrap_or(from.primary.name.as_str());
9372 let ctx = self.ev_ctx(schema_cols, Some(alias));
9373 let seg = self
9374 .active_catalog()
9375 .cold_segment(segment_id)
9376 .ok_or_else(|| {
9377 EngineError::Unsupported(alloc::format!(
9378 "AS OF SEGMENT: cold segment {segment_id} not registered"
9379 ))
9380 })?;
9381 let mut out_rows: Vec<Row<'static>> = Vec::new();
9382 let mut limit_remaining: Option<usize> =
9383 stmt.limit_literal().and_then(|n| usize::try_from(n).ok());
9384 for (_key, body) in seg.scan() {
9385 let (row, _consumed) =
9386 spg_storage::decode_row_body_dense(&body, &schema, seg.codec_version())
9387 .map_err(EngineError::Storage)?;
9388 if let Some(where_expr) = &stmt.where_ {
9389 let cond = self.eval_expr_simple(where_expr, &row, &ctx)?;
9390 if !crate::eval::predicate_is_true(&cond, "WHERE", ctx.mysql_dialect)? {
9391 continue;
9392 }
9393 }
9394 // Projection.
9395 let projected = self.project_row_simple(&row, &stmt.items, schema_cols, alias)?;
9396 out_rows.push(projected);
9397 if let Some(rem) = limit_remaining.as_mut() {
9398 if *rem == 0 {
9399 out_rows.pop();
9400 break;
9401 }
9402 *rem -= 1;
9403 }
9404 }
9405 // Output column schema: derive from SELECT items.
9406 let columns = self.derive_output_columns(&stmt.items, schema_cols, alias);
9407 Ok(QueryResult::Rows {
9408 columns,
9409 rows: out_rows,
9410 })
9411 }
9412
9413 /// v6.10.2 — simple-path WHERE eval that doesn't go through
9414 /// the correlated-subquery / Memoize machinery. AS OF SEGMENT
9415 /// scan paths predicate against a snapshot frozen segment, no
9416 /// cross-row state.
9417 fn eval_expr_simple(
9418 &self,
9419 expr: &Expr,
9420 row: &Row<'static>,
9421 ctx: &EvalContext,
9422 ) -> Result<Value<'static>, EngineError> {
9423 let cancel = CancelToken::none();
9424 self.eval_expr_with_correlated(expr, row, ctx, cancel, None)
9425 }
9426}
9427
9428// ---- SELECT result / projection / generate-series / SRF helpers (lib.rs split 12) ----
9429
9430/// One row-producing projection: an expression to evaluate, the resulting
9431/// column's user-visible name, its inferred type, and nullability.
9432#[derive(Debug, Clone)]
9433pub(crate) struct ProjectedItem {
9434 pub(crate) expr: Expr,
9435 pub(crate) output_name: String,
9436 pub(crate) ty: DataType,
9437 pub(crate) nullable: bool,
9438 /// v7.39 (read01 round 54) — a projected enum column keeps its enum
9439 /// identity. Enum-ness lives outside the DataType lattice (the value is a
9440 /// Text), so a projection that dropped this made the RESULT schema forget
9441 /// it — and a UNION's combined `ORDER BY <enum col>`, which sorts against
9442 /// that schema, silently fell back to TEXT order instead of member order.
9443 pub(crate) user_enum_type: Option<String>,
9444 /// v7.39 (round 425) — a projected MySQL temporal column keeps its
9445 /// declared fractional-seconds precision, so the renderer can pad to
9446 /// exactly that many digits (`DATETIME(3)` shows `.250`, and `.000` for
9447 /// a whole second). Like `user_enum_type` this lives outside the
9448 /// DataType lattice, so a projection that dropped it made the RESULT
9449 /// schema forget how wide the fraction should print.
9450 pub(crate) mysql_fsp: Option<u8>,
9451 /// v7.39 (round 688) — and its declared collation, the third thing to
9452 /// live outside the DataType lattice and the third to be lost the same
9453 /// way. Measured: `SELECT a.loc FROM a JOIN b … ORDER BY a.loc` over a
9454 /// column declared `COLLATE "en_US.utf8"` sorted by bytes, because the
9455 /// projection rebuilt the output column and the ORDER BY resolves
9456 /// against THAT schema.
9457 pub(crate) collation_name: Option<String>,
9458}
9459
9460/// Dedupe a row set, preserving first-seen order. `Row`'s `PartialEq` is
9461/// structural (`Vec<Value<'static>>` ⇒ pairwise `Value` equality), which gives SQL
9462/// `NULL = NULL → TRUE` and `NaN = NaN → FALSE`. The first agrees with
9463/// the spec's "two NULLs are not distinct"; the second is a tolerated
9464/// quirk for v1 (no NaN literals are reachable from the SQL surface).
9465/// v7.37 D.23 — is this expression a bare (non-window) aggregate call?
9466fn expr_is_aggregate_call(e: &Expr) -> bool {
9467 match e {
9468 Expr::FunctionCall { name, .. } => crate::aggregate::is_aggregate_name(name),
9469 Expr::AggregateOrdered { .. } => true,
9470 _ => false,
9471 }
9472}
9473
9474/// Collect distinct top-level aggregate call expressions (dedup by value). Does
9475/// not recurse into an aggregate's own args (it's hoisted whole). Reuses the same
9476/// pragmatic variant set as `rewrite_window_to_columns`; aggregates nested in
9477/// uncovered variants simply aren't hoisted (the query keeps erroring, no worse
9478/// than today — never a regression on a working query).
9479fn collect_agg_exprs(e: &Expr, out: &mut Vec<Expr>) {
9480 if expr_is_aggregate_call(e) {
9481 if !out.iter().any(|x| x == e) {
9482 out.push(e.clone());
9483 }
9484 return;
9485 }
9486 match e {
9487 Expr::Binary { lhs, rhs, .. } => {
9488 collect_agg_exprs(lhs, out);
9489 collect_agg_exprs(rhs, out);
9490 }
9491 Expr::Unary { expr, .. }
9492 | Expr::Cast { expr, .. }
9493 | Expr::IsNull { expr, .. }
9494 | Expr::BoolTest { expr, .. }
9495 | Expr::FieldAccess { base: expr, .. } => collect_agg_exprs(expr, out),
9496 Expr::FunctionCall { args, .. } => {
9497 for a in args {
9498 collect_agg_exprs(a, out);
9499 }
9500 }
9501 Expr::Like { expr, pattern, .. } => {
9502 collect_agg_exprs(expr, out);
9503 collect_agg_exprs(pattern, out);
9504 }
9505 Expr::Extract { source, .. } => collect_agg_exprs(source, out),
9506 Expr::WindowFunction {
9507 args,
9508 partition_by,
9509 order_by,
9510 ..
9511 } => {
9512 for a in args {
9513 collect_agg_exprs(a, out);
9514 }
9515 for p in partition_by {
9516 collect_agg_exprs(p, out);
9517 }
9518 for (o, _, _) in order_by {
9519 collect_agg_exprs(o, out);
9520 }
9521 }
9522 _ => {}
9523 }
9524}
9525
9526/// Replace each aggregate call in `aggs` with a `Column(__aggN)` reference.
9527fn replace_agg_exprs(e: &mut Expr, aggs: &[Expr]) {
9528 if expr_is_aggregate_call(e) {
9529 if let Some(idx) = aggs.iter().position(|x| x == e) {
9530 *e = Expr::Column(ColumnName {
9531 qualifier: None,
9532 name: alloc::format!("__agg{idx}"),
9533 });
9534 }
9535 return;
9536 }
9537 match e {
9538 Expr::Binary { lhs, rhs, .. } => {
9539 replace_agg_exprs(lhs, aggs);
9540 replace_agg_exprs(rhs, aggs);
9541 }
9542 Expr::Unary { expr, .. }
9543 | Expr::Cast { expr, .. }
9544 | Expr::IsNull { expr, .. }
9545 | Expr::BoolTest { expr, .. }
9546 | Expr::FieldAccess { base: expr, .. } => replace_agg_exprs(expr, aggs),
9547 Expr::FunctionCall { args, .. } => {
9548 for a in args {
9549 replace_agg_exprs(a, aggs);
9550 }
9551 }
9552 Expr::Like { expr, pattern, .. } => {
9553 replace_agg_exprs(expr, aggs);
9554 replace_agg_exprs(pattern, aggs);
9555 }
9556 Expr::Extract { source, .. } => replace_agg_exprs(source, aggs),
9557 Expr::WindowFunction {
9558 args,
9559 partition_by,
9560 order_by,
9561 ..
9562 } => {
9563 for a in args {
9564 replace_agg_exprs(a, aggs);
9565 }
9566 for p in partition_by {
9567 replace_agg_exprs(p, aggs);
9568 }
9569 for (o, _, _) in order_by {
9570 replace_agg_exprs(o, aggs);
9571 }
9572 }
9573 _ => {}
9574 }
9575}
9576
9577/// v7.37 D.23 — window functions run AFTER GROUP BY aggregation. Rewrite
9578/// `SELECT g, sum(v), rank() OVER (ORDER BY sum(v)) FROM t GROUP BY g` into an
9579/// aggregate derived subquery (`SELECT g, sum(v) AS __agg0 FROM t GROUP BY g`) +
9580/// an outer window query over it (`SELECT g, __agg0, rank() OVER (ORDER BY
9581/// __agg0) FROM (...) __aggwin`), which the window-over-derived path (D.13) runs.
9582/// Returns None outside the bounded subset (leaves current behaviour). Only fires
9583/// on the currently-erroring agg+window+GROUP BY shape → cannot regress working
9584/// window-only / aggregate-only queries.
9585fn rewrite_agg_before_window(stmt: &SelectStatement) -> Option<SelectStatement> {
9586 if !(crate::aggregate::uses_aggregate(stmt) || stmt.group_by.is_some()) {
9587 return None;
9588 }
9589 // Bounded subset: no set-ops; GROUP BY keys must be simple columns.
9590 if !stmt.unions.is_empty() {
9591 return None;
9592 }
9593 let group_cols: Vec<Expr> = stmt.group_by.clone().unwrap_or_default();
9594 if group_cols.iter().any(|g| !matches!(g, Expr::Column(_))) {
9595 return None;
9596 }
9597 stmt.from.as_ref()?;
9598 // Collect the aggregate calls to hoist from projection + outer ORDER BY.
9599 let mut aggs: Vec<Expr> = Vec::new();
9600 for item in &stmt.items {
9601 if let SelectItem::Expr { expr, .. } = item {
9602 collect_agg_exprs(expr, &mut aggs);
9603 }
9604 }
9605 for ob in &stmt.order_by {
9606 collect_agg_exprs(&ob.expr, &mut aggs);
9607 }
9608 // Inner aggregate subquery: group cols (by name) + each aggregate as __aggN.
9609 let mut inner_items: Vec<SelectItem> = Vec::new();
9610 for g in &group_cols {
9611 inner_items.push(SelectItem::Expr {
9612 expr: g.clone(),
9613 alias: None,
9614 });
9615 }
9616 for (i, a) in aggs.iter().enumerate() {
9617 inner_items.push(SelectItem::Expr {
9618 expr: a.clone(),
9619 alias: Some(alloc::format!("__agg{i}")),
9620 });
9621 }
9622 let inner = SelectStatement {
9623 items: inner_items,
9624 distinct: false,
9625 distinct_on: Vec::new(),
9626 unions: Vec::new(),
9627 order_by: Vec::new(),
9628 limit: None,
9629 offset: None,
9630 limit_with_ties: false,
9631 window_check_exprs: Vec::new(),
9632 ..stmt.clone()
9633 };
9634 let derived = TableRef {
9635 name: "__aggwin".into(),
9636 alias: Some("__aggwin".into()),
9637 only: false,
9638 as_of_segment: None,
9639 unnest_expr: None,
9640 unnest_column_aliases: Vec::new(),
9641 with_ordinality: false,
9642 generate_series_args: None,
9643 lateral_subquery: Some(alloc::boxed::Box::new(inner)),
9644 jsonb_each_text_arg: None,
9645 table_fn_call: None,
9646 rows_from: None,
9647 json_table: None,
9648 scalar_fn_item: false,
9649 };
9650 // Outer window query over the derived rows: aggregates → __aggN column refs.
9651 let mut outer_items = stmt.items.clone();
9652 for item in &mut outer_items {
9653 if let SelectItem::Expr { expr, alias } = item {
9654 // Preserve PG's column label for a bare aggregate projection.
9655 if alias.is_none()
9656 && let Expr::FunctionCall { name, .. } = expr
9657 && crate::aggregate::is_aggregate_name(name)
9658 {
9659 *alias = Some(name.to_ascii_lowercase());
9660 }
9661 replace_agg_exprs(expr, &aggs);
9662 }
9663 }
9664 let mut outer_order = stmt.order_by.clone();
9665 for ob in &mut outer_order {
9666 replace_agg_exprs(&mut ob.expr, &aggs);
9667 }
9668 let mut outer_distinct_on = stmt.distinct_on.clone();
9669 for e in &mut outer_distinct_on {
9670 replace_agg_exprs(e, &aggs);
9671 }
9672 Some(SelectStatement {
9673 locking: None,
9674 ctes: Vec::new(),
9675 distinct: stmt.distinct,
9676 distinct_on: outer_distinct_on,
9677 items: outer_items,
9678 from: Some(FromClause {
9679 primary: derived,
9680 joins: Vec::new(),
9681 }),
9682 where_: None,
9683 group_by: None,
9684 group_by_all: false,
9685 having: None,
9686 unions: Vec::new(),
9687 order_by: outer_order,
9688 limit: stmt.limit.clone(),
9689 offset: stmt.offset.clone(),
9690 limit_with_ties: stmt.limit_with_ties,
9691 window_check_exprs: Vec::new(),
9692 })
9693}
9694
9695/// v7.39 (round 591) — the right-hand side of a set operation, bucketed for
9696/// membership.
9697///
9698/// INTERSECT, EXCEPT and their ALL forms all ask "is this left row over
9699/// there?", and all four answered by scanning the whole right side once per
9700/// left row. The cost was (left rows x right rows), which is why
9701/// `500k INTERSECT 1000` took 1.67 s while the same two inputs the other way
9702/// round took 20 ms: a left row that MATCHES stops the scan early, and a left
9703/// row that does not pays for all of it. Over 100k left rows, raising the
9704/// right side from 100 to 10,000 took 35 ms to 2848.
9705///
9706/// This is the shape round 485 already solved for DISTINCT, and it reuses
9707/// that machinery: bucket by `norm_hash_row`, whose only guarantee is the one
9708/// needed here — rows `row_eq_norm` calls equal hash the same — and settle
9709/// every bucket with the exact comparator, so a collision costs time and
9710/// never an answer.
9711struct PeerIndex<'r> {
9712 bh: hashbrown::DefaultHashBuilder,
9713 buckets: hashbrown::HashMap<u64, Vec<usize>>,
9714 rows: &'r [Row<'static>],
9715 mysql: bool,
9716}
9717
9718impl<'r> PeerIndex<'r> {
9719 fn build(rows: &'r [Row<'static>], mysql: bool) -> Self {
9720 // ONE hasher for the whole pass: the default builder is seeded per
9721 // instance, so a fresh one per row would put equal rows in different
9722 // buckets.
9723 let bh = hashbrown::DefaultHashBuilder::default();
9724 let mut buckets: hashbrown::HashMap<u64, Vec<usize>> =
9725 hashbrown::HashMap::with_capacity(rows.len());
9726 for (i, r) in rows.iter().enumerate() {
9727 buckets
9728 .entry(norm_hash_row(r, &bh, mysql))
9729 .or_default()
9730 .push(i);
9731 }
9732 Self {
9733 bh,
9734 buckets,
9735 rows,
9736 mysql,
9737 }
9738 }
9739
9740 fn contains(&self, r: &Row<'static>) -> bool {
9741 let h = norm_hash_row(r, &self.bh, self.mysql);
9742 self.buckets
9743 .get(&h)
9744 .is_some_and(|b| b.iter().any(|&i| row_eq_norm(&self.rows[i], r, self.mysql)))
9745 }
9746
9747 /// Remove ONE occurrence, so the multiset forms cancel row for row the
9748 /// way the pool they replaced did.
9749 fn take_one(&mut self, r: &Row<'static>) -> bool {
9750 let h = norm_hash_row(r, &self.bh, self.mysql);
9751 let Some(b) = self.buckets.get_mut(&h) else {
9752 return false;
9753 };
9754 let Some(pos) = b
9755 .iter()
9756 .position(|&i| row_eq_norm(&self.rows[i], r, self.mysql))
9757 else {
9758 return false;
9759 };
9760 b.swap_remove(pos);
9761 true
9762 }
9763}
9764
9765pub(crate) fn dedup_rows(rows: Vec<Row<'static>>, mysql: bool) -> Vec<Row<'static>> {
9766 dedup_by_row(rows, |r| r, mysql)
9767}
9768
9769/// v7.37.16 — hash-bucketed DISTINCT. The old `out.iter().any(row_eq_norm)`
9770/// was O(n·u) — `SELECT DISTINCT v` over 50 k rows with ~39 k unique values
9771/// ran 4 SECONDS (80 µs/row) vs PG's ~5 ms. Bucket rows by `norm_hash_row`
9772/// and run the exact `row_eq_norm` only within a bucket: first-occurrence
9773/// order is preserved, and correctness needs only the one-way guarantee
9774/// "row_eq_norm-Equal ⇒ equal hash" (collisions are re-checked exactly).
9775/// Small inputs keep the linear scan — no hasher setup for a 10-row page.
9776fn dedup_by_row<T>(items: Vec<T>, row_of: impl Fn(&T) -> &Row<'static>, mysql: bool) -> Vec<T> {
9777 if items.len() <= 32 {
9778 let mut out: Vec<T> = Vec::with_capacity(items.len());
9779 for it in items {
9780 if !out
9781 .iter()
9782 .any(|seen| row_eq_norm(row_of(seen), row_of(&it), mysql))
9783 {
9784 out.push(it);
9785 }
9786 }
9787 return out;
9788 }
9789 // ONE BuildHasher instance for the whole pass — the default builder
9790 // is randomly seeded PER INSTANCE, so a fresh one per row would give
9791 // equal rows different hashes and never dedup.
9792 let bh = hashbrown::DefaultHashBuilder::default();
9793 let mut out: Vec<T> = Vec::with_capacity(items.len().min(1024));
9794 let mut buckets: hashbrown::HashMap<u64, crate::distinct::DistinctBucket> =
9795 hashbrown::HashMap::with_capacity(items.len());
9796 for it in items {
9797 let h = norm_hash_row(row_of(&it), &bh, mysql);
9798 let bucket = buckets.entry(h).or_default();
9799 if !bucket
9800 .iter()
9801 .any(|i| row_eq_norm(row_of(&out[i]), row_of(&it), mysql))
9802 {
9803 bucket.push(out.len());
9804 out.push(it);
9805 }
9806 }
9807 out
9808}
9809
9810/// Hash companion to [`row_eq_norm`]. Guarantees only the direction dedup
9811/// needs: rows that `row_eq_norm` deems Equal hash identically; DISTINCT
9812/// rows may collide (buckets are re-checked with the exact comparator).
9813///
9814/// Domain design mirrors `value_cmp`'s equivalence classes:
9815/// - The numeric family (SmallInt/Int/BigInt/Float/Numeric/NumericBig)
9816/// shares one domain: a value that is an integer fitting i64 hashes the
9817/// i64 (so `Int(1)`, `BigInt(1)`, `Float(1.0)`, `Numeric(1.00)` agree);
9818/// anything else hashes the f64 approximation computed by THE SAME
9819/// formula the value_cmp float arms use (`numeric_to_f64`), so
9820/// `Numeric(0.5) == Float(0.5)` agree bit-for-bit. NaN (any family)
9821/// hashes a constant; ±Inf hash their f64 bits; -0.0 folds into 0.0.
9822/// Known un-closable corner: an integer in [2^53, 2^63) can compare
9823/// Equal to a float via value_cmp's lossy f64 arm while hashing in the
9824/// exact-i64 domain — mixed int/float rows at that magnitude may miss a
9825/// dedup (PG itself compares int8↔float8 in the lossy float8 domain).
9826/// - Text and BpChar share a trailing-blank-trimmed byte domain (value_cmp
9827/// compares them blank-insensitively; plain Text pairs that differ only
9828/// in trailing blanks merely collide and are separated exactly).
9829/// - Families value_cmp compares exactly (Bool/Date/Time/Timestamp/…)
9830/// hash their fields under a distinct tag.
9831/// - Everything value_cmp falls back to debug-format ordering for
9832/// (Json, arrays, vectors, geometry, ranges, …) shares one constant
9833/// bucket — degrades to the exact linear scan, never wrong.
9834fn norm_hash_row(row: &Row<'static>, bh: &hashbrown::DefaultHashBuilder, mysql: bool) -> u64 {
9835 norm_hash_values(&row.values, bh, mysql)
9836}
9837
9838/// v7.39 (round 485) — the same hash over a bare value slice, so the
9839/// DISTINCT probe can run against a reused buffer instead of demanding a
9840/// `Row` that has to be allocated first (see `values_eq_norm`).
9841fn norm_hash_values(
9842 values: &[Value<'static>],
9843 bh: &hashbrown::DefaultHashBuilder,
9844 mysql: bool,
9845) -> u64 {
9846 use core::hash::{BuildHasher, Hash, Hasher};
9847 let mut h = bh.build_hasher();
9848 for v in values {
9849 // v7.39 (round 410) — hash the folded key when the MySQL collation
9850 // deduplicates a text value, so `row_eq_norm`-equal rows (`'a'` vs
9851 // `'A'` vs `'a '`) share a hash bucket.
9852 if mysql {
9853 if let Some(folded) = mysql_dedup_fold(v) {
9854 folded.hash(&mut h);
9855 continue;
9856 }
9857 }
9858 norm_hash_value(v, &mut h);
9859 }
9860 h.finish()
9861}
9862
9863/// r1044 — `10^p` as an `i128`, or `None` past what one holds.
9864///
9865/// `i128::MAX` is about 1.7e38, so 10^38 is the last power that fits.
9866const fn pow10_i128(p: u16) -> Option<i128> {
9867 const P: [i128; 39] = {
9868 let mut t = [1i128; 39];
9869 let mut i = 1;
9870 while i < 39 {
9871 t[i] = t[i - 1] * 10;
9872 i += 1;
9873 }
9874 t
9875 };
9876 if (p as usize) < P.len() {
9877 Some(P[p as usize])
9878 } else {
9879 None
9880 }
9881}
9882
9883fn norm_hash_value<H: core::hash::Hasher>(v: &Value<'static>, h: &mut H) {
9884 const TAG_NULL: u8 = 0;
9885 const TAG_BOOL: u8 = 1;
9886 const TAG_NUM_I64: u8 = 2;
9887 const TAG_NUM_F64: u8 = 3;
9888 const TAG_TEXT: u8 = 4;
9889 const TAG_DATE: u8 = 6;
9890 const TAG_TIME: u8 = 7;
9891 const TAG_TIMESTAMP: u8 = 8;
9892 const TAG_TIMETZ: u8 = 10;
9893 const TAG_UUID: u8 = 11;
9894 const TAG_MONEY: u8 = 12;
9895 const TAG_BYTES: u8 = 13;
9896 const TAG_INTERVAL: u8 = 14;
9897 const TAG_CHAR1: u8 = 15;
9898 const TAG_OPAQUE: u8 = 255;
9899 // One shared writer for the numeric family: an integer value
9900 // representable as i64 goes exact (round-trip probe — no_std, so no
9901 // f64::trunc); otherwise the f64 approximation. -0.0 round-trips
9902 // through 0i64, folding it into 0.0 as value_cmp requires.
9903 let num_f64 = |h: &mut H, x: f64| {
9904 if x.is_nan() {
9905 h.write_u8(TAG_NUM_F64);
9906 h.write_u64(0x7ff8_dead_beef_0001); // one bucket for every NaN
9907 return;
9908 }
9909 const TWO63: f64 = 9_223_372_036_854_775_808.0;
9910 if (-TWO63..TWO63).contains(&x) {
9911 #[allow(clippy::cast_possible_truncation)]
9912 let n = x as i64;
9913 #[allow(clippy::cast_precision_loss)]
9914 if (n as f64) == x {
9915 h.write_u8(TAG_NUM_I64);
9916 h.write_i64(n);
9917 return;
9918 }
9919 }
9920 h.write_u8(TAG_NUM_F64);
9921 h.write_u64(x.to_bits());
9922 };
9923 match v {
9924 Value::Null => h.write_u8(TAG_NULL),
9925 Value::Bool(b) => {
9926 h.write_u8(TAG_BOOL);
9927 h.write_u8(u8::from(*b));
9928 }
9929 Value::SmallInt(n) => {
9930 h.write_u8(TAG_NUM_I64);
9931 h.write_i64(i64::from(*n));
9932 }
9933 Value::Int(n) => {
9934 h.write_u8(TAG_NUM_I64);
9935 h.write_i64(i64::from(*n));
9936 }
9937 Value::BigInt(n) => {
9938 h.write_u8(TAG_NUM_I64);
9939 h.write_i64(*n);
9940 }
9941 Value::Float(x) => num_f64(h, *x),
9942 Value::Numeric {
9943 scaled,
9944 scale,
9945 kind,
9946 } => match kind {
9947 spg_storage::NumericKind::NaN => num_f64(h, f64::NAN),
9948 spg_storage::NumericKind::PosInf => num_f64(h, f64::INFINITY),
9949 spg_storage::NumericKind::NegInf => num_f64(h, f64::NEG_INFINITY),
9950 spg_storage::NumericKind::Finite => {
9951 // Reduce trailing fractional zeros so 1.50 and 1.5 share a
9952 // representation, then: exact integers fitting i64 go to the
9953 // i64 domain; everything else uses numeric_to_f64 — the SAME
9954 // formula value_cmp's Numeric↔Float arm compares with.
9955 // r1044 — the reduction is required (`1.5` and `1.50` are
9956 // one value and must land in one bucket) and it used to
9957 // walk one digit at a time. That is O(scale), and scale
9958 // is not small in practice: `n / 100` on a NUMERIC
9959 // column stores `9.1900000000000000`, scale 16, so the
9960 // loop ran fourteen times PER ROW.
9961 //
9962 // Priced by ablation rather than guessed at — removing
9963 // the loop entirely took `SELECT DISTINCT n FROM t ORDER
9964 // BY n` over 400,000 rows from 52 ms to 14.8, against
9965 // PostgreSQL's 12.2-13.8. Two `pow10` lookup tables
9966 // tried first moved it not at all, which is why this one
9967 // was measured before it was written.
9968 //
9969 // Binary search over the same powers finds the whole
9970 // run of trailing zeros in at most six tests and one
9971 // division, instead of one test and one division per
9972 // digit.
9973 let (mut s, mut sc) = (*scaled, *scale);
9974 if sc > 0 && s != 0 {
9975 let mut lo: u16 = 0;
9976 let mut hi: u16 = sc;
9977 while lo < hi {
9978 let mid = (lo + hi).div_ceil(2);
9979 match pow10_i128(mid) {
9980 Some(p) if s % p == 0 => lo = mid,
9981 _ => hi = mid - 1,
9982 }
9983 }
9984 if lo > 0 {
9985 if let Some(p) = pow10_i128(lo) {
9986 s /= p;
9987 sc -= lo;
9988 }
9989 }
9990 }
9991 if sc == 0 {
9992 if let Ok(n) = i64::try_from(s) {
9993 h.write_u8(TAG_NUM_I64);
9994 h.write_i64(n);
9995 } else {
9996 num_f64(h, crate::orderby::numeric_to_f64(s, 0));
9997 }
9998 } else {
9999 num_f64(h, crate::orderby::numeric_to_f64(s, sc));
10000 }
10001 }
10002 },
10003 // Beyond-i128 NUMERIC compares exactly via numeric_bignum_cmp; a
10004 // value that also fits i128 reuses the Numeric path above so
10005 // Big(5) and Numeric(5) agree. A genuinely huge one can't equal
10006 // any i128-representable value — constant bucket is safe.
10007 Value::NumericBig(b) => match b.to_i128() {
10008 Some(s) => norm_hash_value(
10009 &Value::Numeric {
10010 scaled: s,
10011 scale: b.scale(),
10012 kind: spg_storage::NumericKind::Finite,
10013 },
10014 h,
10015 ),
10016 None => h.write_u8(TAG_OPAQUE),
10017 },
10018 // value_cmp compares Text↔BpChar blank-insensitively (both sides
10019 // trimmed), so both hash the trimmed bytes. Text pairs differing
10020 // only in trailing blanks collide and are split exactly in-bucket.
10021 Value::Text(s) | Value::BpChar(s) => {
10022 h.write_u8(TAG_TEXT);
10023 h.write(s.trim_end_matches(' ').as_bytes());
10024 }
10025 Value::Char1(c) => {
10026 h.write_u8(TAG_CHAR1);
10027 h.write_u8(*c);
10028 }
10029 Value::Date(d) => {
10030 h.write_u8(TAG_DATE);
10031 h.write_i32(*d);
10032 }
10033 Value::Time(t) => {
10034 h.write_u8(TAG_TIME);
10035 h.write_i64(*t);
10036 }
10037 Value::Timestamp(t) => {
10038 h.write_u8(TAG_TIMESTAMP);
10039 h.write_i64(*t);
10040 }
10041 Value::TimeTz { us, offset_secs } => {
10042 h.write_u8(TAG_TIMETZ);
10043 h.write_i64(*us);
10044 h.write_i32(*offset_secs);
10045 }
10046 Value::Uuid(u) => {
10047 h.write_u8(TAG_UUID);
10048 h.write(u);
10049 }
10050 Value::Money(c) => {
10051 h.write_u8(TAG_MONEY);
10052 h.write_i64(*c);
10053 }
10054 Value::Bytes(b) => {
10055 h.write_u8(TAG_BYTES);
10056 h.write(b.as_ref());
10057 }
10058 Value::Interval {
10059 months,
10060 days,
10061 micros,
10062 } => {
10063 h.write_u8(TAG_INTERVAL);
10064 h.write_i32(*months);
10065 h.write_i32(*days);
10066 h.write_i64(*micros);
10067 }
10068 // v7.37.16 — REAL joined the numeric value_cmp family (widened
10069 // to f64, same formulas as the arms), so it hashes in the shared
10070 // numeric domain: Real(1.5) must agree with Float(1.5)/Int/…
10071 // f32→f64 is exact, so equal-under-cmp implies equal bits here.
10072 Value::Real(x) => num_f64(h, f64::from(*x)),
10073 // Json (structural equality), vector families (float rendering),
10074 // arrays / geometry / net / ranges / composites (debug-format
10075 // fallback): one constant bucket — exact linear within.
10076 _ => h.write_u8(TAG_OPAQUE),
10077 }
10078}
10079
10080/// v7.38 (read01) — row equality for DISTINCT / UNION / INTERSECT / EXCEPT that
10081/// treats numerically-equal exact values as one regardless of type or scale
10082/// (`1 = 1.0 = 1.00`), matching PG (and GROUP BY). Uses the scale-aware
10083/// `orderby::value_cmp`, so `Int(1)` and `Numeric{10,1}` compare Equal; plain
10084/// `Row` `==` would keep them distinct.
10085/// v7.39 (round 410) — under the MySQL dialect a set operation / DISTINCT
10086/// deduplicates by the session collation (`utf8mb4_uca1400_ai_ci`, which is
10087/// case- and accent-insensitive and PAD SPACE): `'a'`, `'A'`, and `'a '`
10088/// collapse to one row, exactly as GROUP BY already folds its keys. Returns
10089/// the folded comparison key for a text value, None for anything else (which
10090/// keeps the byte-exact `value_cmp` path).
10091fn mysql_dedup_fold(v: &Value) -> Option<String> {
10092 match v {
10093 Value::Text(s) | Value::BpChar(s) => {
10094 Some(spg_storage::mysql_ci_fold(s.trim_end_matches(' ')))
10095 }
10096 _ => None,
10097 }
10098}
10099
10100/// v7.39 (round 485) — how many projected rows the single-table scan
10101/// builds, and how many of those the DISTINCT probe throws away again.
10102///
10103/// The round-485 profile of `SELECT DISTINCT g FROM h ORDER BY g` put
10104/// 21 % of all samples in malloc/free called straight from the scan
10105/// closure. The closure's one per-row allocation is the projected
10106/// `Vec<Value>`, and under DISTINCT most of those are discarded a few
10107/// instructions later — but "most" is a guess until it is a number, so
10108/// these count it. (Round 480 was spent acting on an inference about a
10109/// branch that turned out never to run.)
10110/// v7.39 (round 488) — reachability counters for round 487's projection
10111/// binding. The interleaved panel says round 487 costs `group_500k` 13 %,
10112/// and a never-called-function probe rules out code layout — so the
10113/// question is whether that shape reaches this code at all, which is a
10114/// number, not an inference.
10115pub static SCAN_PATH_ENTERED: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0);
10116pub static PROJ_DIRECT_FIRE: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0);
10117
10118pub static PROJ_ROW_BUILT: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0);
10119pub static DISTINCT_DUP_DROPPED: core::sync::atomic::AtomicU64 =
10120 core::sync::atomic::AtomicU64::new(0);
10121
10122pub(crate) fn row_eq_norm(a: &Row<'static>, b: &Row<'static>, mysql: bool) -> bool {
10123 values_eq_norm(&a.values, &b.values, mysql)
10124}
10125
10126/// v7.39 (round 485) — `row_eq_norm` over bare value slices, so the
10127/// DISTINCT probe can compare a reused projection buffer against a kept
10128/// row without building a `Row` for it.
10129pub(crate) fn values_eq_norm(a: &[Value<'static>], b: &[Value<'static>], mysql: bool) -> bool {
10130 a.len() == b.len()
10131 && a.iter().zip(b).all(|(x, y)| {
10132 if mysql {
10133 if let (Some(fx), Some(fy)) = (mysql_dedup_fold(x), mysql_dedup_fold(y)) {
10134 return fx == fy;
10135 }
10136 }
10137 crate::orderby::value_cmp(x, y) == core::cmp::Ordering::Equal
10138 })
10139}
10140
10141/// Coerce a `Value` to an `f64` sort key for ORDER BY. Numbers map directly;
10142/// NULL sorts last (treated as `+∞`); booleans are 0.0 / 1.0; text uses lex
10143/// order via the byte values; vectors are not sortable.
10144pub(crate) fn value_to_order_key(v: &Value) -> Result<OrderKey, EngineError> {
10145 // v7.37.16 — TEXT rides a FULL-precision key: carry the whole string
10146 // so values sharing a ≥6-byte common prefix (`product_001` vs
10147 // `product_002`, ISO timestamps stored as text, prefixed IDs / SKUs)
10148 // order by their exact bytes instead of the old lossy f64 coarse key.
10149 // Comparison is byte-lexicographic (see `order_key_elem_cmp`), which
10150 // matches PG's default C / binary text collation. Every other type
10151 // keeps the lossless-enough `f64` fast path below.
10152 if let Value::Text(s) = v {
10153 return Ok(OrderKey::Text(s.as_ref().into()));
10154 }
10155 // v7.39 (bpchar epic) — bpchar sorts by its blank-stripped form then
10156 // byte order (PG bpcharcmp under C collation), so mixed-pad values of
10157 // the same logical string order equal.
10158 if let Value::BpChar(s) = v {
10159 return Ok(OrderKey::Text(s.trim_end_matches(' ').into()));
10160 }
10161 // v7.38 (read01 P6.24) — jsonb sorts by PG's type-aware total order, so
10162 // carry the parsed value and compare it structurally (see
10163 // `order_key_elem_cmp`). Unparseable text falls back to a Text key.
10164 if let Value::Json(s) = v {
10165 return Ok(match crate::json::parse(s) {
10166 Ok(jv) => OrderKey::Json(jv),
10167 Err(_) => OrderKey::Text(s.as_ref().into()),
10168 });
10169 }
10170 // v7.37 — byte-orderable types PG sorts byte-wise but that have no
10171 // meaningful f64 projection. bytea/uuid/macaddr sort by their raw bytes;
10172 // inet/cidr by `[family, addr.., bits]` (family, then address, then mask),
10173 // matching PG's network ordering.
10174 match v {
10175 Value::Bytes(b) => return Ok(OrderKey::Bytes(b.as_ref().to_vec())),
10176 // v7.38 (read01, T3.C3) — arbitrary-precision NUMERIC sorts by exact value.
10177 Value::NumericBig(b) => {
10178 return Ok(OrderKey::Numeric(alloc::boxed::Box::new(
10179 spg_storage::NumericKey::from_big(b),
10180 )));
10181 }
10182 Value::Uuid(u) => return Ok(OrderKey::Bytes(u.to_vec())),
10183 Value::Macaddr(m) => return Ok(OrderKey::Bytes(m.to_vec())),
10184 Value::Macaddr8(m) => return Ok(OrderKey::Bytes(m.to_vec())),
10185 Value::PgLsn(l) => return Ok(OrderKey::Bytes(l.to_be_bytes().to_vec())),
10186 Value::Inet { family, bits, addr } | Value::Cidr { family, bits, addr } => {
10187 let mut key = alloc::vec::Vec::with_capacity(18);
10188 key.push(*family);
10189 key.extend_from_slice(addr);
10190 key.push(*bits);
10191 return Ok(OrderKey::Bytes(key));
10192 }
10193 _ => {}
10194 }
10195 // v7.38 (read01, U16) — one-dimensional arrays sort element-wise, then
10196 // shorter-first (PG: `{1} < {1,2} < {2} < {10}`). Each element carries its
10197 // own OrderKey so integer arrays sort numerically; a NULL element rides to
10198 // the end via the +INF sentinel.
10199 let inf = || OrderKey::NullBig;
10200 let arr = match v {
10201 Value::IntArray(a) => Some(
10202 a.iter()
10203 .map(|o| o.map_or_else(inf, |n| OrderKey::Int(i128::from(n))))
10204 .collect(),
10205 ),
10206 Value::SmallIntArray(a) => Some(
10207 a.iter()
10208 .map(|o| o.map_or_else(inf, |n| OrderKey::Int(i128::from(n))))
10209 .collect(),
10210 ),
10211 Value::BigIntArray(a) => Some(
10212 a.iter()
10213 .map(|o| o.map_or_else(inf, |n| OrderKey::Int(i128::from(n))))
10214 .collect(),
10215 ),
10216 Value::BoolArray(a) => Some(
10217 a.iter()
10218 .map(|o| o.map_or_else(inf, |b| OrderKey::Int(i128::from(b))))
10219 .collect(),
10220 ),
10221 Value::TextArray(a) => Some(
10222 a.iter()
10223 .map(|o| o.as_ref().map_or_else(inf, |s| OrderKey::Text(s.clone())))
10224 .collect(),
10225 ),
10226 #[allow(clippy::cast_precision_loss)]
10227 Value::FloatArray(a) => Some(
10228 a.iter()
10229 .map(|o| o.map_or(OrderKey::NullBig, OrderKey::Num))
10230 .collect(),
10231 ),
10232 // r1040 — array elements take the same exact key their scalar
10233 // form does; an f64 projection here would order `{0.1}` against
10234 // `{0.1000000000000000001}` by luck.
10235 Value::NumericArray(a) => Some(
10236 a.iter()
10237 .map(|o| {
10238 o.map_or_else(inf, |(m, s)| {
10239 OrderKey::Numeric(alloc::boxed::Box::new(
10240 spg_storage::NumericKey::from_numeric(
10241 m,
10242 s,
10243 spg_storage::NumericKind::Finite,
10244 ),
10245 ))
10246 })
10247 })
10248 .collect(),
10249 ),
10250 Value::DateArray(a) => Some(
10251 a.iter()
10252 .map(|o| o.map_or_else(inf, |n| OrderKey::Int(i128::from(n))))
10253 .collect(),
10254 ),
10255 _ => None,
10256 };
10257 if let Some(elements) = arr {
10258 return Ok(OrderKey::Array(elements));
10259 }
10260 // v7.39 (read01 round 56) — a COMPOSITE sorts field by field, left to
10261 // right, which is exactly the lexicographic element order an Array key
10262 // already gives: `(2,'b') < (9,'a')` because the leading field decides.
10263 if let Value::Composite(fields) = v {
10264 let elements = fields
10265 .iter()
10266 .map(|(_, fv)| value_to_order_key(fv))
10267 .collect::<Result<alloc::vec::Vec<_>, _>>()?;
10268 return Ok(OrderKey::Array(elements));
10269 }
10270 // v7.38 (read01 U31) — the integer-valued types carry an EXACT i128 key.
10271 // Projecting these to f64 (the historic path) silently collapses BigInt /
10272 // Timestamp / Time / TimeTz / Money values past 2^53, so `ORDER BY` gave
10273 // the wrong order for large ids and microsecond timestamps.
10274 match v {
10275 Value::SmallInt(n) => return Ok(OrderKey::Int(i128::from(*n))),
10276 Value::Int(n) => return Ok(OrderKey::Int(i128::from(*n))),
10277 Value::BigInt(n) => return Ok(OrderKey::Int(i128::from(*n))),
10278 // PG TIME/TIMESTAMP/DATE/MONEY/YEAR are ordered by their underlying
10279 // integer (days / micros / cents / calendar year); TIMETZ by the
10280 // UTC-equivalent micros (local wall - offset) so the same physical
10281 // instant in different zones sorts equal.
10282 Value::Date(d) => return Ok(OrderKey::Int(i128::from(*d))),
10283 Value::Timestamp(t) => return Ok(OrderKey::Int(i128::from(*t))),
10284 Value::Time(us) => return Ok(OrderKey::Int(i128::from(*us))),
10285 Value::Year(y) => return Ok(OrderKey::Int(i128::from(*y))),
10286 Value::TimeTz { us, offset_secs } => {
10287 return Ok(OrderKey::Int(
10288 i128::from(*us) - i128::from(*offset_secs) * 1_000_000,
10289 ));
10290 }
10291 Value::Money(c) => return Ok(OrderKey::Int(i128::from(*c))),
10292 _ => {}
10293 }
10294 let num = match v {
10295 // Callers without NULLS FIRST/LAST context (array elements,
10296 // histogram sampling) put NULL last, as before.
10297 Value::Null => return Ok(OrderKey::NullBig),
10298 // v7.17.0 Phase 3.P0-38 — range ordering is not supported
10299 // in v7.17.0 (needs lex-then-inclusivity tiebreak).
10300 Value::Range { .. } => {
10301 return Err(EngineError::Unsupported(
10302 "ORDER BY of a range value is not supported in v7.17.0".into(),
10303 ));
10304 }
10305 // v7.17.0 Phase 3.P0-39 — hstore is not orderable.
10306 Value::Hstore(_) => {
10307 return Err(EngineError::Unsupported(
10308 "ORDER BY of a hstore value is not supported".into(),
10309 ));
10310 }
10311 // v7.17.0 Phase 3.P0-40 — 2D arrays not orderable.
10312 Value::IntArray2D(_) | Value::BigIntArray2D(_) | Value::TextArray2D(_) => {
10313 return Err(EngineError::Unsupported(
10314 "ORDER BY of a 2D array is not supported in v7.17.0".into(),
10315 ));
10316 }
10317 // r1039/r1040 — the exact canonical key, not an f64 projection.
10318 //
10319 // r1039 fixed the three specials, which carry a canonical zero in
10320 // `scaled` and so all sorted as the number 0. The projection
10321 // itself was the rest of the defect: "precision losses here only
10322 // matter for tie-breaks well past 15 significant digits" was the
10323 // comment, and the measurement disagreed — f64 called
10324 // `0.1` and `0.1000000000000000001` Equal, and a stable sort then
10325 // returned them in insertion order. Three of ten values came back
10326 // in the wrong place against PG18.4.
10327 Value::Numeric {
10328 scaled,
10329 scale,
10330 kind,
10331 } => {
10332 return Ok(OrderKey::Numeric(alloc::boxed::Box::new(
10333 spg_storage::NumericKey::from_numeric(*scaled, *scale, *kind),
10334 )));
10335 }
10336 Value::Float(x) => *x,
10337 // v7.37.16 — REAL sorts by its exact f64 widening (it had no
10338 // arm and fell through to the unsupported error).
10339 Value::Real(x) => f64::from(*x),
10340 Value::Bool(b) => {
10341 if *b {
10342 1.0
10343 } else {
10344 0.0
10345 }
10346 }
10347 Value::Vector(_) | Value::Sq8Vector(_) | Value::HalfVector(_) => {
10348 return Err(EngineError::Unsupported(
10349 "ORDER BY of a raw vector column is not meaningful — use `<->`".into(),
10350 ));
10351 }
10352 // v7.37 — PG orders INTERVAL by its total time, treating a month as
10353 // 30 days (`1 hour < 90 min < 1 day < 1 mon`). Project to total micros;
10354 // f64 is exact for any interval under ~285 years, and only ORDER BY
10355 // tie-breaks past that magnitude lose precision. Matches the
10356 // min/max(interval) comparator in aggregate.rs.
10357 #[allow(clippy::cast_precision_loss)]
10358 Value::Interval {
10359 months,
10360 days,
10361 micros,
10362 } => {
10363 let total = i128::from(*months) * 30 * 86_400_000_000
10364 + i128::from(*days) * 86_400_000_000
10365 + i128::from(*micros);
10366 total as f64
10367 }
10368 Value::Json(_) => {
10369 return Err(EngineError::Unsupported(
10370 "ORDER BY of a JSON value is not supported — cast the document to text first"
10371 .into(),
10372 ));
10373 }
10374 // v7.5.0 — Value is #[non_exhaustive]; future variants need
10375 // an explicit ORDER BY mapping. Surface as Unsupported until
10376 // engine support is added.
10377 _ => {
10378 return Err(EngineError::Unsupported(
10379 "ORDER BY of this value type is not supported".into(),
10380 ));
10381 }
10382 };
10383 Ok(OrderKey::Num(num))
10384}
10385
10386/// Find the schema entry that a SELECT-list `Expr::Column` refers to.
10387/// Mirrors `resolve_column` in `eval.rs`, but returns a proper
10388/// `EngineError` so the projection-build path keeps `UnknownQualifier`
10389/// vs `ColumnNotFound` distinct.
10390/// PG's name for the physical row identity. It is reserved there — no table
10391/// can have a column called this — which is what lets `*` skip it by name.
10392pub(crate) const CTID_COLUMN: &str = "ctid";
10393
10394/// v7.39 (round 512) — PG's system columns, in the order they are appended.
10395/// All six are reserved names there, which is what lets `*` skip them and
10396/// lets a scan tell them from a user column without a flag.
10397pub(crate) const SYSTEM_COLUMNS: [&str; 6] = ["ctid", "xmin", "xmax", "cmin", "cmax", "tableoid"];
10398
10399/// Is this name one of them?
10400pub(crate) fn is_system_column(name: &str) -> bool {
10401 SYSTEM_COLUMNS.iter().any(|s| name.eq_ignore_ascii_case(s))
10402}
10403
10404/// Where the scan's appended system columns begin, if this schema carries
10405/// them: the trailing six, named in order. A catalog view with a column of
10406/// its own called `xmin` does not match, which is the point.
10407fn system_column_tail_start(cols: &[ColumnSchema]) -> Option<usize> {
10408 let start = cols.len().checked_sub(SYSTEM_COLUMNS.len())?;
10409 cols[start..]
10410 .iter()
10411 .zip(SYSTEM_COLUMNS)
10412 .all(|(c, name)| c.name.eq_ignore_ascii_case(name))
10413 .then_some(start)
10414}
10415
10416/// v7.39 (round 540) — which positions `*` must skip.
10417///
10418/// The rule stays round 512's — the synthetic columns are the trailing
10419/// six of a relation's block, matched by POSITION so a genuine `xmin`
10420/// column is not lost — but a JOINED schema names its columns
10421/// `alias.column` and lays the peers out end to end, so a peer's six sit
10422/// in the MIDDLE of the whole list. Grouping by qualifier first puts the
10423/// "trailing six" test back on the block it was written for.
10424fn synthetic_system_positions(cols: &[ColumnSchema]) -> alloc::vec::Vec<bool> {
10425 let mut skip = alloc::vec![false; cols.len()];
10426 fn qualifier(n: &str) -> Option<&str> {
10427 n.rsplit_once('.').map(|(q, _)| q)
10428 }
10429 fn bare(n: &str) -> &str {
10430 n.rsplit('.').next().unwrap_or(n)
10431 }
10432 let mut i = 0;
10433 while i < cols.len() {
10434 let q = qualifier(&cols[i].name);
10435 let mut end = i;
10436 while end < cols.len() && qualifier(&cols[end].name) == q {
10437 end += 1;
10438 }
10439 if let Some(start) = (end - i)
10440 .checked_sub(SYSTEM_COLUMNS.len())
10441 .map(|off| i + off)
10442 && cols[start..end]
10443 .iter()
10444 .zip(SYSTEM_COLUMNS)
10445 .all(|(c, name)| bare(&c.name).eq_ignore_ascii_case(name))
10446 {
10447 for s in skip.iter_mut().take(end).skip(start) {
10448 *s = true;
10449 }
10450 }
10451 i = end;
10452 }
10453 skip
10454}
10455
10456/// v7.39 (round 511) — does this statement name `ctid` anywhere it would be
10457/// read? Only then is the column materialised.
10458pub(crate) fn expr_references_ctid(e: &Expr) -> bool {
10459 let mut found = false;
10460 crate::expr_analysis::visit_expr_columns_and_subqueries(
10461 e,
10462 &mut |c| {
10463 if is_system_column(&c.name) {
10464 found = true;
10465 }
10466 },
10467 &mut |_| {},
10468 );
10469 found
10470}
10471
10472fn references_ctid(stmt: &SelectStatement) -> bool {
10473 let in_expr = expr_references_ctid;
10474 stmt.items.iter().any(|i| match i {
10475 SelectItem::Expr { expr, .. } => in_expr(expr),
10476 _ => false,
10477 }) || stmt.where_.as_ref().is_some_and(in_expr)
10478 || stmt.order_by.iter().any(|o| in_expr(&o.expr))
10479 || stmt
10480 .group_by
10481 .as_ref()
10482 .is_some_and(|g| g.iter().any(in_expr))
10483 || stmt.having.as_ref().is_some_and(in_expr)
10484}
10485
10486/// v7.39 (round 961) — the whole-row schema for `SELECT t FROM t`, which
10487/// is a name the projection has to TYPE before any row exists.
10488///
10489/// Evaluation has answered this since round T9 (`resolve_column` builds a
10490/// `Value::Composite` of every column), but the typing side below had no
10491/// such branch and raised `column "t" does not exist` first — so the
10492/// feature was unreachable through a projection. Measured against PG18.4:
10493/// `SELECT wr FROM wr` answers `(7,z)` there and errored here.
10494///
10495/// The type is `Jsonb` + a composite marker, which is exactly how a
10496/// column DECLARED as a composite type is described (`ddl.rs`, round 56):
10497/// the value travels as a `Value::Composite` and renders in the canonical
10498/// `(7,z)` form. SPG has no catalog entry for a table's implicit row type,
10499/// so the marker names the alias and no rehydration keys off it — the
10500/// value arrives already built.
10501fn whole_row_projection_schema(alias: &str) -> ColumnSchema {
10502 let mut s = ColumnSchema::new(
10503 alloc::string::String::from(alias),
10504 spg_storage::DataType::Jsonb,
10505 true,
10506 );
10507 s.user_composite_type = Some(alloc::string::String::from(alias));
10508 s
10509}
10510
10511pub(crate) fn resolve_projection_column<'a>(
10512 c: &ColumnName,
10513 schema_cols: &'a [ColumnSchema],
10514 table_alias: &str,
10515) -> Result<Cow<'a, ColumnSchema>, EngineError> {
10516 if let Some(q) = &c.qualifier {
10517 let composite = alloc::format!("{q}.{name}", name = c.name);
10518 if let Some(s) = schema_cols.iter().find(|s| s.name == composite) {
10519 return Ok(Cow::Borrowed(s));
10520 }
10521 // Single-table case: the qualifier may equal the active alias —
10522 // then look for the bare column name.
10523 if q == table_alias
10524 && let Some(s) = schema_cols.iter().find(|s| s.name == c.name)
10525 {
10526 return Ok(Cow::Borrowed(s));
10527 }
10528 // For multi-table schemas the qualifier is unknown only if no
10529 // column bears the "<q>." prefix. For single-table, the alias
10530 // mismatch alone is enough.
10531 let prefix = alloc::format!("{q}.");
10532 let qualifier_known =
10533 q == table_alias || schema_cols.iter().any(|s| s.name.starts_with(&prefix));
10534 if !qualifier_known {
10535 return Err(EngineError::Eval(EvalError::UnknownQualifier {
10536 qualifier: q.clone(),
10537 }));
10538 }
10539 return Err(EngineError::Eval(EvalError::ColumnNotFound {
10540 name: c.name.clone(),
10541 }));
10542 }
10543 if let Some(s) = schema_cols.iter().find(|s| s.name == c.name) {
10544 return Ok(Cow::Borrowed(s));
10545 }
10546 let suffix = alloc::format!(".{name}", name = c.name);
10547 let mut matches = schema_cols.iter().filter(|s| s.name.ends_with(&suffix));
10548 let first = matches.next();
10549 let extra = matches.next();
10550 match (first, extra) {
10551 (Some(s), None) => Ok(Cow::Borrowed(s)),
10552 (Some(_), Some(_)) => Err(EngineError::Eval(EvalError::TypeMismatch {
10553 detail: alloc::format!("column reference \"{}\" is ambiguous", c.name),
10554 })),
10555 // The whole-row reference, checked LAST so a real column carrying
10556 // the alias's name still wins — the same precedence
10557 // `resolve_column` applies on the evaluation side.
10558 //
10559 // Two schema shapes reach here. A single-table (or subquery, or
10560 // CTE) scan carries its alias and bare column names, so the name
10561 // has to equal the alias. A JOIN's combined schema carries no
10562 // alias at all and qualifies every column `alias.col`, so the
10563 // alias is identified by the prefix instead — which is exactly
10564 // how `whole_row_composite` picks the fields out on the
10565 // evaluation side. Measured: `SELECT wr FROM wr JOIN jb ON …`
10566 // answers `(7,z)` on PG18.4 and errored here until this arm
10567 // covered the joined shape too.
10568 _ if !table_alias.is_empty() && c.name == table_alias => {
10569 Ok(Cow::Owned(whole_row_projection_schema(table_alias)))
10570 }
10571 _ if table_alias.is_empty() && {
10572 let prefix = alloc::format!("{name}.", name = c.name);
10573 schema_cols.iter().any(|s| s.name.starts_with(&prefix))
10574 } =>
10575 {
10576 Ok(Cow::Owned(whole_row_projection_schema(&c.name)))
10577 }
10578 _ => Err(EngineError::Eval(EvalError::ColumnNotFound {
10579 name: c.name.clone(),
10580 })),
10581 }
10582}
10583
10584/// v7.39 (round 135) — drop the synthetic `__grp_ord_*` columns injected by the
10585/// parser to carry per-branch GROUPING() masks into a grouping-set query's
10586/// ORDER BY. They must never reach the output. No-op unless such a column is
10587/// present, so the common path is untouched.
10588/// v7.39 (round 529) — the LIMIT / OFFSET that DISTINCT ON deferred.
10589///
10590/// PG limits what the dedup LEFT, not what fed it; SPG limited first, so
10591/// a `LIMIT 2` that should have answered two groups answered one.
10592fn apply_deferred_limit(
10593 rows: alloc::vec::Vec<Row<'static>>,
10594 deferred: &(
10595 Option<spg_sql::ast::LimitExpr>,
10596 Option<spg_sql::ast::LimitExpr>,
10597 ),
10598) -> alloc::vec::Vec<Row<'static>> {
10599 let count = |e: &Option<spg_sql::ast::LimitExpr>| match e {
10600 Some(spg_sql::ast::LimitExpr::Literal(n)) => Some(*n as usize),
10601 _ => None,
10602 };
10603 let mut rows = rows;
10604 if let Some(off) = count(&deferred.1) {
10605 rows = rows.split_off(off.min(rows.len()));
10606 }
10607 if let Some(lim) = count(&deferred.0) {
10608 rows.truncate(lim);
10609 }
10610 rows
10611}
10612
10613fn strip_synthetic_order_cols(result: QueryResult) -> QueryResult {
10614 let QueryResult::Rows { columns, rows } = result else {
10615 return result;
10616 };
10617 if !columns.iter().any(|c| c.name.starts_with("__grp_ord_")) {
10618 return QueryResult::Rows { columns, rows };
10619 }
10620 let keep: Vec<usize> = columns
10621 .iter()
10622 .enumerate()
10623 .filter(|(_, c)| !c.name.starts_with("__grp_ord_"))
10624 .map(|(i, _)| i)
10625 .collect();
10626 let new_cols: Vec<ColumnSchema> = keep.iter().map(|&i| columns[i].clone()).collect();
10627 let new_rows: Vec<Row<'static>> = rows
10628 .into_iter()
10629 .map(|r| Row::new(keep.iter().map(|&i| r.values[i].clone()).collect()))
10630 .collect();
10631 QueryResult::Rows {
10632 columns: new_cols,
10633 rows: new_rows,
10634 }
10635}
10636
10637/// v7.39 (round 487) — bind every projection item that is a bare column
10638/// reference to its position, once per query.
10639///
10640/// `#[inline(never)]` and out of line on purpose. Round 486 established
10641/// that adding code inside these scan bodies moves neighbouring hot
10642/// functions around under fat LTO: the first version of this had the loop
10643/// inline in `run_single_table_scan` and four aggregate shapes that never
10644/// touch that function — `full_agg`, `join_agg`, `group_500k`,
10645/// `filter_agg` — went up ~5 %, reproduced against the parent commit on
10646/// the same machine. Keeping it out of line kept them still.
10647#[inline(never)]
10648fn bind_direct_columns(
10649 projection: &[ProjectedItem],
10650 ctx: &eval::EvalContext<'_>,
10651) -> Vec<Option<usize>> {
10652 projection
10653 .iter()
10654 .map(|p| match &p.expr {
10655 Expr::Column(c) => eval::compile_column_pos(c, ctx).filter(|pos| {
10656 // Same exclusion `compile_into` makes: a composite column
10657 // has to be rehydrated from stored JSON, which is not a
10658 // cell read.
10659 ctx.columns
10660 .get(*pos)
10661 .is_none_or(|sc| sc.user_composite_type.is_none())
10662 }),
10663 _ => None,
10664 })
10665 .collect()
10666}
10667
10668/// v7.39 (round 505) — the name an un-aliased projected expression reports.
10669///
10670/// PG18 names a call for its function and everything else `?column?`;
10671/// measured with `\gdesc`. SPG used to print the parsed expression back
10672/// out for both dialects, so `SELECT upper(s)` reported `upper(s)` and
10673/// name-keyed row access found nothing under `upper`.
10674///
10675/// The MySQL half is NOT this rule and is deliberately left alone here:
10676/// MariaDB echoes the item's SOURCE TEXT verbatim (`a+b`, spacing and all),
10677/// which needs the parser to hand over spans the AST does not carry yet.
10678/// Until it does, a MySQL session keeps the printed form — closer to what
10679/// MariaDB answers than `?column?` would be.
10680pub(crate) fn default_output_name(expr: &Expr, mysql: bool) -> String {
10681 if mysql {
10682 return expr.to_string();
10683 }
10684 spg_sql::ast::figure_column_name(expr).unwrap_or_else(|| "?column?".to_string())
10685}
10686
10687pub(crate) fn build_projection(
10688 items: &[SelectItem],
10689 schema_cols: &[ColumnSchema],
10690 table_alias: &str,
10691 mysql: bool,
10692) -> Result<Vec<ProjectedItem>, EngineError> {
10693 build_projection_hiding_tail(items, schema_cols, table_alias, mysql, 0)
10694}
10695
10696/// v7.39 (round 592) — `build_projection` with the last `hidden_tail` columns
10697/// invisible to `*`.
10698///
10699/// The windowed-SELECT path appends a synthetic `__win_N` column per window
10700/// function so the rewritten projection can reference the computed values as
10701/// ordinary columns. `*` then expanded them too, and
10702/// `SELECT wr.*, row_number() OVER (ORDER BY id) FROM wr` came back with an
10703/// EXTRA column — the internal name's value, repeated. A wrong answer, and a
10704/// silent one: the row simply had one more field than the client asked for.
10705///
10706/// Hidden by POSITION rather than by name, for the reason round 512 recorded
10707/// about the system columns: a name test looks safe until a real column
10708/// happens to carry the name. These are appended last, so the count is what
10709/// identifies them.
10710pub(crate) fn build_projection_hiding_tail(
10711 items: &[SelectItem],
10712 schema_cols: &[ColumnSchema],
10713 table_alias: &str,
10714 mysql: bool,
10715 hidden_tail: usize,
10716) -> Result<Vec<ProjectedItem>, EngineError> {
10717 let visible = schema_cols.len().saturating_sub(hidden_tail);
10718 // v7.39 (round 462) — a join's combined schema qualifies every column
10719 // `alias.col` so the deferred-join cell lookups resolve by composite
10720 // name. That is an internal convention, and `*` was handing it to the
10721 // client: PG18 answers `SELECT * FROM a JOIN b` with the BARE names
10722 // (`id, g, id, h` — duplicates and all), SPG answered `a.id, a.g,
10723 // b.id, b.h`, so name-keyed row access found nothing. Round 128 had
10724 // already learned this for `q.*`; plain `*` never got the same rule.
10725 //
10726 // The signal is the schema itself, not the call site: only a combined
10727 // join schema arrives with no table alias AND every column qualified.
10728 // A single-table schema carries its alias, an empty schema has nothing
10729 // to strip, and a synthetic schema's names carry no dot.
10730 let joined_schema = table_alias.is_empty()
10731 && !schema_cols.is_empty()
10732 && schema_cols.iter().all(|c| c.name.contains('.'));
10733 let bare_name = |name: &str| -> String {
10734 if !joined_schema {
10735 return name.to_string();
10736 }
10737 match name.split_once('.') {
10738 Some((_, rest)) if !rest.is_empty() => rest.to_string(),
10739 _ => name.to_string(),
10740 }
10741 };
10742 let mut out = Vec::new();
10743 for item in items {
10744 match item {
10745 SelectItem::Wildcard => {
10746 // v7.39 (round 511) — `*` never expands a system column, as
10747 // PG's does not. They join the schema only when the statement
10748 // asked for them, so this matters for the mixed shape
10749 // `SELECT *, ctid FROM t`.
10750 //
10751 // v7.39 (round 512) — by POSITION, not by name. Matching on
10752 // the name alone looked safe because PG reserves them, and it
10753 // is not: `pg_replication_slots` genuinely has a column called
10754 // `xmin`, and `SELECT * FROM pg_replication_slots` lost it.
10755 // Only the trailing six, in the order the scan appends them,
10756 // are the synthetic ones.
10757 let sys_skip = synthetic_system_positions(schema_cols);
10758 for (idx, col) in schema_cols.iter().enumerate() {
10759 if sys_skip[idx] || idx >= visible {
10760 continue;
10761 }
10762 out.push(ProjectedItem {
10763 expr: Expr::Column(ColumnName {
10764 qualifier: None,
10765 name: col.name.clone(),
10766 }),
10767 output_name: bare_name(&col.name),
10768 ty: col.ty,
10769 nullable: col.nullable,
10770 user_enum_type: col.user_enum_type.clone(),
10771 mysql_fsp: col.mysql_fsp,
10772 collation_name: col.collation_name.clone(),
10773 });
10774 }
10775 }
10776 // v7.39 (round 128) — `q.*` expands to every column belonging to
10777 // the qualifier `q`. Single-table schemas carry bare column names
10778 // reachable via `table_alias`; a join's combined schema carries
10779 // `alias.col` names, so a column belongs to `q` when its name has
10780 // the `q.` prefix. PG labels the expanded columns by their bare
10781 // name, so the `alias.` prefix is stripped from the output name.
10782 SelectItem::QualifiedWildcard(q) => {
10783 let prefix = alloc::format!("{q}.");
10784 let single_table = !table_alias.is_empty() && q == table_alias;
10785 let mut matched = 0usize;
10786 for col in &schema_cols[..visible] {
10787 let belongs =
10788 col.name.starts_with(&prefix) || (single_table && !col.name.contains('.'));
10789 if !belongs {
10790 continue;
10791 }
10792 matched += 1;
10793 let output_name = col
10794 .name
10795 .strip_prefix(&prefix)
10796 .unwrap_or(&col.name)
10797 .to_string();
10798 out.push(ProjectedItem {
10799 expr: Expr::Column(ColumnName {
10800 qualifier: None,
10801 name: col.name.clone(),
10802 }),
10803 output_name,
10804 ty: col.ty,
10805 nullable: col.nullable,
10806 user_enum_type: col.user_enum_type.clone(),
10807 mysql_fsp: col.mysql_fsp,
10808 collation_name: col.collation_name.clone(),
10809 });
10810 }
10811 if matched == 0 {
10812 return Err(EngineError::Eval(EvalError::UnknownQualifier {
10813 qualifier: q.clone(),
10814 }));
10815 }
10816 }
10817 SelectItem::Expr { expr, alias } => {
10818 // Plain column ref keeps full schema info (real type +
10819 // nullability). For compound expressions try the
10820 // describe-side function-return-type table first
10821 // (e.g. `SELECT now()` → Timestamptz, `SELECT
10822 // concat(…)` → Text). Falls back to nullable Text
10823 // for shapes the describe path can't resolve.
10824 if let Expr::Column(c) = expr {
10825 let sch = resolve_projection_column(c, schema_cols, table_alias)?;
10826 let output_name = alias.clone().unwrap_or_else(|| c.name.clone());
10827 out.push(ProjectedItem {
10828 expr: expr.clone(),
10829 output_name,
10830 ty: sch.ty,
10831 nullable: sch.nullable,
10832 // v7.39 (read01 round 54) — a bare enum column keeps
10833 // its enum identity through the projection.
10834 user_enum_type: sch.user_enum_type.clone(),
10835 mysql_fsp: sch.mysql_fsp,
10836 collation_name: sch.collation_name.clone(),
10837 });
10838 } else if let Some(shape) = describe::describe_expr(expr, schema_cols) {
10839 let output_name = alias
10840 .clone()
10841 .unwrap_or_else(|| default_output_name(expr, mysql));
10842 out.push(ProjectedItem {
10843 expr: expr.clone(),
10844 output_name,
10845 ty: shape.ty,
10846 // v7.39 (round 258) — a projected EXPRESSION keeps its
10847 // enum identity too, not just a bare column. `FROM
10848 // (VALUES ('happy'::mood), …) t(m)` lowers to constant
10849 // SELECTs, so the derived column arrived here as a cast
10850 // and lost the enum — making the outer ORDER BY / min /
10851 // max / array_agg sort by the label's TEXT.
10852 nullable: shape.nullable,
10853 user_enum_type: None,
10854 mysql_fsp: crate::eval::expr_mysql_fsp(expr, schema_cols),
10855 // A bare column reference keeps its collation; any
10856 // other expression produces a new value and has none.
10857 collation_name: match expr {
10858 Expr::Column(c) => schema_cols
10859 .iter()
10860 .find(|sc| sc.name.eq_ignore_ascii_case(&c.name))
10861 .and_then(|sc| sc.collation_name.clone()),
10862 _ => None,
10863 },
10864 });
10865 } else {
10866 let output_name = alias
10867 .clone()
10868 .unwrap_or_else(|| default_output_name(expr, mysql));
10869 out.push(ProjectedItem {
10870 expr: expr.clone(),
10871 output_name,
10872 // A user ENUM has no DataType of its own, so
10873 // `describe_expr` cannot type `'ok'::mood` and the
10874 // item lands HERE, defaulting to text — which is why
10875 // pg_typeof answered `text` and a derived table sorted
10876 // enum values by their label.
10877 ty: DataType::Text,
10878 nullable: true,
10879 user_enum_type: crate::eval::expr_enum_type_name_pub(expr, schema_cols)
10880 .map(alloc::string::String::from),
10881 mysql_fsp: crate::eval::expr_mysql_fsp(expr, schema_cols),
10882 collation_name: match expr {
10883 Expr::Column(c) => schema_cols
10884 .iter()
10885 .find(|sc| sc.name.eq_ignore_ascii_case(&c.name))
10886 .and_then(|sc| sc.collation_name.clone()),
10887 _ => None,
10888 },
10889 });
10890 }
10891 }
10892 }
10893 }
10894 Ok(out)
10895}
10896
10897// ---- v4.12 window-function helpers ----
10898// The (partition-key, order-key, original-index) tuple shape used
10899// across these helpers is intrinsic to the planner. Factoring it
10900// into a typedef adds indirection without making the code clearer,
10901// so several lints are allowed inline on the affected functions
10902// rather than module-wide.
10903
10904/// v4.22: pick more specific column types from observed rows when
10905/// the projection builder defaulted to Text (the v1.x behavior for
10906/// non-column expressions). Lets `WITH t(n) AS (SELECT 1 ...)`
10907/// land an Int column in the CTE storage table rather than failing
10908/// the insert with "expected TEXT, got INT".
10909pub(crate) fn infer_column_types(
10910 columns: &[ColumnSchema],
10911 rows: &[Row<'static>],
10912) -> Vec<ColumnSchema> {
10913 let mut out = columns.to_vec();
10914 for (col_idx, col) in out.iter_mut().enumerate() {
10915 if col.ty != DataType::Text {
10916 continue;
10917 }
10918 let mut inferred: Option<DataType> = None;
10919 let mut all_null = true;
10920 for row in rows {
10921 let Some(v) = row.values.get(col_idx) else {
10922 continue;
10923 };
10924 let ty = match v {
10925 Value::Null => continue,
10926 Value::SmallInt(_) => DataType::SmallInt,
10927 Value::Int(_) => DataType::Int,
10928 Value::BigInt(_) => DataType::BigInt,
10929 Value::Float(_) => DataType::Float,
10930 Value::Bool(_) => DataType::Bool,
10931 Value::Vector(_) => DataType::Vector {
10932 dim: 0,
10933 encoding: VecEncoding::F32,
10934 },
10935 // v7.38 (read01 U16) — carry array values through with an
10936 // array type so a recursive CTE that projects an array
10937 // (e.g. a SEARCH/CYCLE ord / path column) types the working
10938 // column as an array, not Text.
10939 Value::TextArray(_) => DataType::TextArray,
10940 Value::IntArray(_) => DataType::IntArray,
10941 Value::BigIntArray(_) => DataType::BigIntArray,
10942 Value::SmallIntArray(_) => DataType::SmallIntArray,
10943 Value::FloatArray(_) => DataType::FloatArray,
10944 Value::BoolArray(_) => DataType::BoolArray,
10945 // v7.39 (GUC knife 2) — an interval projection describes
10946 // as INTERVAL (typed drivers read the RowDescription OID).
10947 Value::Interval { .. } => DataType::Interval,
10948 _ => DataType::Text,
10949 };
10950 all_null = false;
10951 inferred = Some(match inferred {
10952 None => ty,
10953 Some(prev) if prev == ty => prev,
10954 Some(_) => DataType::Text,
10955 });
10956 }
10957 if let Some(t) = inferred {
10958 col.ty = t;
10959 col.nullable = true;
10960 } else if all_null {
10961 col.nullable = true;
10962 }
10963 }
10964 out
10965}
10966
10967/// Numeric widening rank for UNION type resolution (higher = wider).
10968fn numeric_rank(t: DataType) -> Option<u8> {
10969 match t {
10970 DataType::SmallInt => Some(1),
10971 DataType::Int => Some(2),
10972 DataType::BigInt => Some(3),
10973 DataType::Numeric { .. } => Some(4),
10974 DataType::Float => Some(5),
10975 _ => None,
10976 }
10977}
10978
10979/// Resolve the common result type for a UNION / VALUES column from the
10980/// set of concrete (non-NULL) branch types, following the safe subset
10981/// of PG's type resolution:
10982/// * all-numeric → the widest numeric (int ∪ bigint → bigint, … ∪
10983/// numeric → numeric, … ∪ float → float);
10984/// * DATE ∪ TIMESTAMP → TIMESTAMP;
10985/// * exactly one concrete non-TEXT type mixed with TEXT literals →
10986/// that concrete type (the TEXT cells get parsed into it).
10987/// Returns `None` for anything ambiguous, so the caller leaves the
10988/// column untouched rather than risk a wrong or failing coercion.
10989fn resolve_union_common_type(types: &[DataType]) -> Option<DataType> {
10990 // NB: types are collected from RUNTIME values, which are coarser
10991 // than the schema (e.g. a timestamptz cell is Value::Timestamp), so
10992 // a single-concrete-type fast path must NOT overwrite the column
10993 // type — it would downgrade tstz to ts. NULL-only unification (PG:
10994 // `VALUES (NULL),(1.5)` types the column numeric even on the NULL
10995 // row's pg_typeof) needs schema-level resolution — recorded, not
10996 // attempted here.
10997 if types.len() < 2 {
10998 return None;
10999 }
11000 if types.iter().all(|t| numeric_rank(*t).is_some()) {
11001 return types
11002 .iter()
11003 .max_by_key(|t| numeric_rank(**t).unwrap_or(0))
11004 .copied();
11005 }
11006 let non_text: Vec<&DataType> = types
11007 .iter()
11008 .filter(|t| !matches!(t, DataType::Text))
11009 .collect();
11010 // v7.38 (T-tstz Phase 1) — temporal common type, per PG18.4: if any branch
11011 // is timestamptz the result is timestamptz (tstz ∪ ts, tstz ∪ date), else
11012 // if any is timestamp the result is timestamp (ts ∪ date). All values are
11013 // the same UTC-micros instant, so widening date/ts to tstz is lossless.
11014 if non_text.iter().all(|t| {
11015 matches!(
11016 t,
11017 DataType::Date | DataType::Timestamp | DataType::Timestamptz
11018 )
11019 }) && non_text
11020 .iter()
11021 .any(|t| matches!(t, DataType::Timestamp | DataType::Timestamptz))
11022 {
11023 if non_text.iter().any(|t| matches!(t, DataType::Timestamptz)) {
11024 return Some(DataType::Timestamptz);
11025 }
11026 return Some(DataType::Timestamp);
11027 }
11028 // A single concrete non-TEXT type mixed with TEXT literals.
11029 if non_text.len() == 1 {
11030 return Some(*non_text[0]);
11031 }
11032 // v7.37.16 — SEVERAL concrete types mixed with TEXT literals
11033 // (`VALUES ('NaN'::float8),(1.0),('NaN')` → float8 ∪ numeric ∪
11034 // text): resolve the concrete set first (PG treats the unknown-
11035 // typed string literals as castable to whatever the knowns
11036 // resolve to), then the TEXT cells parse into that target — the
11037 // caller's coercion dry-run still abandons the column if any
11038 // literal doesn't parse.
11039 if !non_text.is_empty() && non_text.len() < types.len() {
11040 let concrete: Vec<DataType> = non_text.iter().map(|t| **t).collect();
11041 return resolve_union_common_type(&concrete);
11042 }
11043 None
11044}
11045
11046/// Coerce every cell of a UNION / VALUES result column to one common
11047/// type (see [`resolve_union_common_type`]). Conservative: a column
11048/// whose branches already agree, or whose types don't resolve, or where
11049/// any cell fails to coerce, is left exactly as it was — this never
11050/// turns a previously-working query into an error.
11051fn unify_union_columns(columns: &mut [ColumnSchema], rows: &mut [Row<'static>]) {
11052 for col_idx in 0..columns.len() {
11053 let mut seen: Vec<DataType> = Vec::new();
11054 for row in rows.iter() {
11055 if let Some(dt) = row.values.get(col_idx).and_then(Value::data_type) {
11056 if !seen.contains(&dt) {
11057 seen.push(dt);
11058 }
11059 }
11060 }
11061 // v7.37.16 — a single concrete runtime type under a TEXT-typed
11062 // column means the column type came off a NULL (or unknown-text)
11063 // branch: NULL literals describe as TEXT (`L::Null → Text`), so
11064 // `VALUES (NULL),(1.5)` left the column "text" while every
11065 // non-NULL cell is numeric. Adopt the concrete type — schema
11066 // only, no cell changes. tstz-safe by construction: a real
11067 // timestamptz column's schema type is Timestamptz, not Text, so
11068 // the coarser runtime type (Value::Timestamp) can't downgrade it
11069 // through this arm; and a real text column's non-NULL cells are
11070 // Text, which keeps seen == [Text] and skips it.
11071 if seen.len() == 1
11072 && matches!(columns[col_idx].ty, DataType::Text)
11073 && !matches!(seen[0], DataType::Text)
11074 {
11075 columns[col_idx].ty = seen[0];
11076 continue;
11077 }
11078 let Some(target) = resolve_union_common_type(&seen) else {
11079 continue;
11080 };
11081 // v7.38 (read01) — an unconstrained NUMERIC result column keeps each
11082 // value's own scale in PG (`VALUES (1.0),(1.00)` renders `1.0` / `1.00`,
11083 // not `1.00` / `1.00`). So when the common type is NUMERIC, leave an
11084 // existing numeric cell untouched and only promote integers (to scale 0)
11085 // rather than rescaling everything to the widest scale.
11086 let scale_preserving_numeric = matches!(target, DataType::Numeric { .. });
11087 // Dry-run the coercion; abandon the whole column if any fails.
11088 let mut coerced: Vec<Option<Value<'static>>> = Vec::with_capacity(rows.len());
11089 let mut ok = true;
11090 for row in rows.iter() {
11091 match row.values.get(col_idx) {
11092 Some(Value::Numeric { .. }) if scale_preserving_numeric => {
11093 coerced.push(Some(row.values[col_idx].clone()));
11094 }
11095 Some(v) => {
11096 let cell_target = if scale_preserving_numeric {
11097 DataType::Numeric {
11098 precision: 0,
11099 scale: 0,
11100 }
11101 } else {
11102 target
11103 };
11104 match crate::conversions::coerce_value(
11105 v.clone(),
11106 cell_target,
11107 &columns[col_idx].name,
11108 col_idx,
11109 ) {
11110 Ok(cv) => coerced.push(Some(cv)),
11111 Err(_) => {
11112 ok = false;
11113 break;
11114 }
11115 }
11116 }
11117 None => coerced.push(None),
11118 }
11119 }
11120 if !ok {
11121 continue;
11122 }
11123 for (row, cv) in rows.iter_mut().zip(coerced) {
11124 if let (Some(slot), Some(nv)) = (row.values.get_mut(col_idx), cv) {
11125 *slot = nv;
11126 }
11127 }
11128 columns[col_idx].ty = target;
11129 }
11130}
11131
11132/// v4.22: encode a Row to a comparable byte key for UNION-DISTINCT
11133/// dedup inside the recursive iteration. Crude but deterministic
11134/// — Debug prints embed type discriminants so NULL ≠ "" ≠ 0.
11135fn encode_row_key(row: &Row<'static>) -> Vec<u8> {
11136 let mut out = Vec::new();
11137 for v in &row.values {
11138 // v7.38 (read01) — UNION / DISTINCT dedup must treat numerically-equal
11139 // exact values as one, regardless of type or scale (`1 = 1.0 = 1.00`),
11140 // like PG (and like GROUP BY, which already normalizes). The old
11141 // `{v:?}` key made `Numeric{10,1}` differ from `Numeric{100,2}`. Encode
11142 // the exact-decimal family through one scale-stripped canonical form.
11143 match v {
11144 Value::SmallInt(n) => encode_numeric_key(&mut out, i128::from(*n), 0),
11145 Value::Int(n) => encode_numeric_key(&mut out, i128::from(*n), 0),
11146 Value::BigInt(n) => encode_numeric_key(&mut out, i128::from(*n), 0),
11147 Value::Numeric { scaled, scale, .. } => encode_numeric_key(&mut out, *scaled, *scale),
11148 other => {
11149 let s = alloc::format!("{other:?}|");
11150 out.extend_from_slice(s.as_bytes());
11151 }
11152 }
11153 }
11154 out
11155}
11156
11157/// Append a scale-independent canonical key for an exact-decimal value: strip
11158/// trailing fractional zeros so `1`, `1.0`, `1.00` all key the same. The `\x01`
11159/// tag keeps a numeric key from colliding with a text value's `{v:?}` form.
11160fn encode_numeric_key(out: &mut Vec<u8>, mut scaled: i128, mut scale: u16) {
11161 while scale > 0 && scaled % 10 == 0 {
11162 scaled /= 10;
11163 scale -= 1;
11164 }
11165 let s = alloc::format!("\u{1}{scaled}e-{scale}|");
11166 out.extend_from_slice(s.as_bytes());
11167}
11168
11169/// Multi-arg `unnest(a, b, …)` — evaluate each array argument
11170/// (uncorrelated; outer refs were substituted upstream), then zip
11171/// them in parallel, NULL-padding shorter arrays to the longest
11172/// (PG's ROWS FROM shorthand). Shared by the primary-position
11173/// executor and the join-position materialiser, which both detect
11174/// the parser's `__unnest_zip` marker call.
11175pub(crate) fn unnest_zip_rows(
11176 args: &[Expr],
11177) -> Result<(alloc::vec::Vec<DataType>, alloc::vec::Vec<Row<'static>>), EngineError> {
11178 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
11179 let ctx = EvalContext::new(&empty_schema, None);
11180 let dummy_row = Row::new(alloc::vec::Vec::new());
11181 let mut dtypes: alloc::vec::Vec<DataType> = alloc::vec::Vec::with_capacity(args.len());
11182 let mut columns: alloc::vec::Vec<alloc::vec::Vec<Value<'static>>> =
11183 alloc::vec::Vec::with_capacity(args.len());
11184 for a in args {
11185 let v = eval::eval_expr(a, &dummy_row, &ctx).map_err(EngineError::Eval)?;
11186 let (dt, items): (DataType, alloc::vec::Vec<Value<'static>>) = match v {
11187 Value::Null => (DataType::Text, alloc::vec::Vec::new()),
11188 Value::TextArray(xs) => (
11189 DataType::Text,
11190 xs.into_iter()
11191 .map(|x| x.map(Value::text).unwrap_or(Value::Null))
11192 .collect(),
11193 ),
11194 Value::IntArray(xs) => (
11195 DataType::Int,
11196 xs.into_iter()
11197 .map(|x| x.map(Value::Int).unwrap_or(Value::Null))
11198 .collect(),
11199 ),
11200 Value::BigIntArray(xs) => (
11201 DataType::BigInt,
11202 xs.into_iter()
11203 .map(|x| x.map(Value::BigInt).unwrap_or(Value::Null))
11204 .collect(),
11205 ),
11206 other => {
11207 return Err(EngineError::Unsupported(alloc::format!(
11208 "unnest() expects array arguments, got {}",
11209 crate::conversions::pg_type_name_for_error_opt(other.data_type())
11210 )));
11211 }
11212 };
11213 dtypes.push(dt);
11214 columns.push(items);
11215 }
11216 let max_len = columns.iter().map(|c| c.len()).max().unwrap_or(0);
11217 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::with_capacity(max_len);
11218 for i in 0..max_len {
11219 let vals: alloc::vec::Vec<Value<'static>> = columns
11220 .iter()
11221 .map(|c| c.get(i).cloned().unwrap_or(Value::Null))
11222 .collect();
11223 rows.push(Row::new(vals));
11224 }
11225 Ok((dtypes, rows))
11226}
11227
11228/// Detect the parser's multi-arg unnest marker on an unnest_expr.
11229pub(crate) fn unnest_zip_args(expr: &Expr) -> Option<&[Expr]> {
11230 match expr {
11231 Expr::FunctionCall { name, args } if name == "__unnest_zip" => Some(args.as_slice()),
11232 _ => None,
11233 }
11234}
11235
11236/// Evaluate generate_series arguments (uncorrelated — outer refs
11237/// were substituted upstream where applicable) and build the row
11238/// stream. Dispatches on the start value's shape and rejects
11239/// mixed-shape calls early (e.g. start = timestamp, stop =
11240/// integer) so the caller gets a clean error rather than a panic.
11241/// Shared by the primary-position executor and the join-position
11242/// materialiser.
11243pub(crate) fn generate_series_rows(
11244 args: &[Expr],
11245 cancel: &CancelToken<'_>,
11246) -> Result<(DataType, alloc::vec::Vec<Row<'static>>), EngineError> {
11247 let empty_schema: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
11248 let ctx = EvalContext::new(&empty_schema, None);
11249 let dummy_row = Row::new(alloc::vec::Vec::new());
11250 let mut arg_values: alloc::vec::Vec<Value<'static>> =
11251 alloc::vec::Vec::with_capacity(args.len());
11252 for a in args {
11253 arg_values.push(eval::eval_expr(a, &dummy_row, &ctx).map_err(EngineError::Eval)?);
11254 }
11255 generate_series_from_values(arg_values, args, cancel)
11256}
11257
11258/// v7.39 (read01 round 96) — the value-producing core of `generate_series`,
11259/// split out so the SELECT-list SRF path (`top_level_srf_output`) shares the
11260/// full integer / numeric / timestamp overload set with the FROM-clause path.
11261/// Before this split the target-list arm reimplemented only the integer case,
11262/// so `SELECT generate_series(1,2), generate_series(ts, ts, interval)` yielded
11263/// NULL for the timestamp column instead of the series. `arg_values` are the
11264/// already-evaluated arguments; `args` is kept only for the timestamptz-vs-
11265/// timestamp type resolution (it inspects the argument expressions' types).
11266pub(crate) fn generate_series_from_values(
11267 mut arg_values: alloc::vec::Vec<Value<'static>>,
11268 args: &[Expr],
11269 cancel: &CancelToken<'_>,
11270) -> Result<(DataType, alloc::vec::Vec<Row<'static>>), EngineError> {
11271 // PG: a NULL bound or step yields zero rows (also keeps the
11272 // NULL-padded lateral probe alive — schema without data).
11273 if arg_values.iter().any(|v| matches!(v, Value::Null)) {
11274 return Ok((DataType::BigInt, alloc::vec::Vec::new()));
11275 }
11276 // PG resolves `generate_series(date, date, interval)` to the
11277 // timestamp/timestamptz overload by implicitly casting each date
11278 // bound up to a timestamp at midnight (verified vs live PG18.4:
11279 // date args yield rows anchored at 00:00:00). SPG's TZ-naive
11280 // timestamp model renders the same instants, so fold any Date
11281 // bound to its midnight Timestamp (canonical `days *
11282 // 86_400_000_000`, matching cast.rs `cast_to_timestamp`) before
11283 // the shape match so the existing timestamp arm drives the walk.
11284 // v7.39 (read01 round 76) — WHICH timestamp overload PG picks matters:
11285 // `generate_series(date, date, interval)` has no date overload, and among
11286 // the two candidates PG prefers the timestamptz one (timestamptz is the
11287 // preferred type of the datetime category), so the column comes back
11288 // `timestamp with time zone` — the rows render with a `+00` offset. A
11289 // timestamptz bound obviously lands there too. Only genuinely
11290 // timestamp-typed bounds keep the TZ-naive result type.
11291 let empty_cols: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
11292 let tz = arg_values.iter().any(|v| matches!(v, Value::Date(_)))
11293 || args.iter().any(|a| {
11294 crate::describe::describe_expr(a, &empty_cols)
11295 .is_some_and(|s| matches!(s.ty, DataType::Timestamptz))
11296 });
11297 for v in &mut arg_values {
11298 if let Value::Date(d) = *v {
11299 *v = Value::Timestamp(crate::conversions::date_days_to_micros(d));
11300 }
11301 }
11302 match arg_values.as_slice() {
11303 [Value::Timestamp(start), Value::Timestamp(stop), step] => {
11304 let interval_step = match step {
11305 Value::Interval { .. } => step.clone(),
11306 // v7.38 (read01) — PG resolves an unknown-type string step
11307 // (`generate_series(date, date, '2 days')`) to INTERVAL; accept
11308 // a bare text step by parsing it the same way `::interval` does.
11309 Value::Text(s) => crate::conversions::coerce_value(
11310 Value::text(s.as_ref()),
11311 DataType::Interval,
11312 "",
11313 0,
11314 )
11315 .map_err(|_| {
11316 EngineError::Unsupported(alloc::format!(
11317 "generate_series(timestamp, timestamp, …): \
11318 could not parse step {s:?} as INTERVAL"
11319 ))
11320 })?,
11321 other => {
11322 return Err(EngineError::Unsupported(alloc::format!(
11323 "generate_series(timestamp, timestamp, …): \
11324 step must be INTERVAL, got {}",
11325 crate::conversions::pg_type_name_for_error_opt(other.data_type())
11326 )));
11327 }
11328 };
11329 let rows = generate_series_timestamps(*start, *stop, interval_step, cancel)?;
11330 Ok((
11331 if tz {
11332 DataType::Timestamptz
11333 } else {
11334 DataType::Timestamp
11335 },
11336 rows,
11337 ))
11338 }
11339 [start, stop, step]
11340 if value_is_integer(start) && value_is_integer(stop) && value_is_integer(step) =>
11341 {
11342 let s = value_to_i64(start);
11343 let e = value_to_i64(stop);
11344 let st = value_to_i64(step);
11345 // PG types the series by the argument type: int4 args → int4
11346 // elements, int8 (bigint) args → int8. Any BigInt operand widens.
11347 let wide = value_is_bigint(start) || value_is_bigint(stop) || value_is_bigint(step);
11348 let rows = generate_series_integers(s, e, st, wide, cancel)?;
11349 Ok((
11350 if wide {
11351 DataType::BigInt
11352 } else {
11353 DataType::Int
11354 },
11355 rows,
11356 ))
11357 }
11358 [start, stop] if value_is_integer(start) && value_is_integer(stop) => {
11359 let s = value_to_i64(start);
11360 let e = value_to_i64(stop);
11361 let wide = value_is_bigint(start) || value_is_bigint(stop);
11362 let rows = generate_series_integers(s, e, 1, wide, cancel)?;
11363 Ok((
11364 if wide {
11365 DataType::BigInt
11366 } else {
11367 DataType::Int
11368 },
11369 rows,
11370 ))
11371 }
11372 // v7.39 (read01 numeric.c) — the NUMERIC overload. PG walks the
11373 // series in exact numeric arithmetic; NaN / infinity bounds and a
11374 // zero step get dedicated wordings, and a mixed int/numeric call
11375 // resolves here via the implicit int→numeric cast.
11376 [_, _] | [_, _, _]
11377 if arg_values
11378 .iter()
11379 .any(|v| matches!(v, Value::Numeric { .. } | Value::NumericBig(_)))
11380 && arg_values.iter().all(|v| {
11381 matches!(v, Value::Numeric { .. } | Value::NumericBig(_)) || value_is_integer(v)
11382 }) =>
11383 {
11384 use spg_storage::NumericKind as K;
11385 let words: [(&str, &str); 3] = [
11386 (
11387 "start value cannot be NaN",
11388 "start value cannot be infinity",
11389 ),
11390 ("stop value cannot be NaN", "stop value cannot be infinity"),
11391 ("step size cannot be NaN", "step size cannot be infinity"),
11392 ];
11393 for (i, v) in arg_values.iter().enumerate() {
11394 if let Value::Numeric { kind, .. } = v {
11395 if *kind != K::Finite {
11396 let (nan_w, inf_w) = words[i];
11397 return Err(EngineError::Unsupported(
11398 if *kind == K::NaN { nan_w } else { inf_w }.into(),
11399 ));
11400 }
11401 }
11402 }
11403 let big =
11404 |v: &Value<'_>| eval::binop::value_to_bignum(v).expect("finite numeric or integer");
11405 let start = big(&arg_values[0]);
11406 let stop = big(&arg_values[1]);
11407 let step = if arg_values.len() == 3 {
11408 big(&arg_values[2])
11409 } else {
11410 spg_storage::bignum::BigNumeric::from_i128(1, 0)
11411 };
11412 if step.is_zero() {
11413 return Err(EngineError::Unsupported(
11414 "step size cannot equal zero".into(),
11415 ));
11416 }
11417 let descending = step.parts().0;
11418 let mut rows = alloc::vec::Vec::new();
11419 let mut cur = start;
11420 const MAX_ROWS: usize = 10_000_000;
11421 loop {
11422 cancel.check()?;
11423 let c = cur.cmp(&stop);
11424 if descending {
11425 if c == core::cmp::Ordering::Less {
11426 break;
11427 }
11428 } else if c == core::cmp::Ordering::Greater {
11429 break;
11430 }
11431 if rows.len() >= MAX_ROWS {
11432 return Err(EngineError::Unsupported(alloc::format!(
11433 "generate_series() result exceeds {MAX_ROWS} rows"
11434 )));
11435 }
11436 rows.push(Row::new(alloc::vec![eval::binop::bignum_to_value(
11437 cur.clone()
11438 )]));
11439 cur = cur.add(&step);
11440 }
11441 Ok((
11442 DataType::Numeric {
11443 precision: 0,
11444 scale: 0,
11445 },
11446 rows,
11447 ))
11448 }
11449 _ => Err(EngineError::Unsupported(alloc::format!(
11450 "generate_series(): v7.17 supports integer or (timestamp, timestamp, interval) \
11451 argument shapes; got {}",
11452 arg_values
11453 .iter()
11454 .map(|v| crate::conversions::pg_type_name_for_error_opt(v.data_type()))
11455 .collect::<alloc::vec::Vec<_>>()
11456 .join(", ")
11457 ))),
11458 }
11459}
11460
11461/// v7.17.0 Phase 3.10 — integer-mode generate_series materialiser.
11462/// Step direction follows the sign: positive step iterates upward
11463/// (stops when current > stop); negative iterates downward; zero
11464/// errors. Caller-facing row stream is `BigInt`-typed so a single
11465/// projection schema covers SmallInt / Int / BigInt callers.
11466fn generate_series_integers(
11467 start: i64,
11468 stop: i64,
11469 step: i64,
11470 wide: bool,
11471 cancel: &CancelToken<'_>,
11472) -> Result<alloc::vec::Vec<Row<'static>>, EngineError> {
11473 if step == 0 {
11474 return Err(EngineError::Unsupported(
11475 "step size cannot equal zero".into(),
11476 ));
11477 }
11478 let mut out = alloc::vec::Vec::new();
11479 let mut cur = start;
11480 // Hard cap to keep a runaway call from eating all memory. PG
11481 // has no such cap but does honour query timeout; SPG's cancel
11482 // token will fire too — this is a defense-in-depth backstop.
11483 const MAX_ROWS: usize = 10_000_000;
11484 loop {
11485 cancel.check()?;
11486 if step > 0 && cur > stop {
11487 break;
11488 }
11489 if step < 0 && cur < stop {
11490 break;
11491 }
11492 out.push(Row::new(alloc::vec![if wide {
11493 Value::BigInt(cur)
11494 } else {
11495 Value::Int(cur as i32)
11496 }]));
11497 if out.len() > MAX_ROWS {
11498 return Err(EngineError::Unsupported(alloc::format!(
11499 "generate_series(): exceeded {MAX_ROWS} rows; \
11500 narrow start/stop or use a larger step"
11501 )));
11502 }
11503 cur = match cur.checked_add(step) {
11504 Some(n) => n,
11505 None => break,
11506 };
11507 }
11508 Ok(out)
11509}
11510
11511/// v7.17.0 Phase 3.10 — timestamp-mode generate_series. step is a
11512/// `Value::Interval { months, micros }` per the caller's guard;
11513/// each iteration adds the interval via `apply_binary_interval`
11514/// so month-shifting handles short-month rollover (PG semantics).
11515fn generate_series_timestamps(
11516 start: i64,
11517 stop: i64,
11518 step: Value,
11519 cancel: &CancelToken<'_>,
11520) -> Result<alloc::vec::Vec<Row<'static>>, EngineError> {
11521 let (months, days, micros) = match &step {
11522 Value::Interval {
11523 months,
11524 days,
11525 micros,
11526 } => (*months, *days, *micros),
11527 _ => unreachable!("caller guards step.is_interval"),
11528 };
11529 if months == 0 && days == 0 && micros == 0 {
11530 return Err(EngineError::Unsupported(
11531 "generate_series(): INTERVAL step cannot be zero".into(),
11532 ));
11533 }
11534 let ascending = months > 0 || days > 0 || micros > 0;
11535 let mut out = alloc::vec::Vec::new();
11536 let mut cur = Value::Timestamp(start);
11537 const MAX_ROWS: usize = 10_000_000;
11538 loop {
11539 cancel.check()?;
11540 let cur_t = match cur {
11541 Value::Timestamp(t) => t,
11542 _ => unreachable!("loop invariant: cur is Timestamp"),
11543 };
11544 if ascending && cur_t > stop {
11545 break;
11546 }
11547 if !ascending && cur_t < stop {
11548 break;
11549 }
11550 out.push(Row::new(alloc::vec![Value::Timestamp(cur_t)]));
11551 if out.len() > MAX_ROWS {
11552 return Err(EngineError::Unsupported(alloc::format!(
11553 "generate_series(): exceeded {MAX_ROWS} rows; \
11554 narrow start/stop or use a larger step"
11555 )));
11556 }
11557 let next = eval::apply_binary_interval(
11558 spg_sql::ast::BinOp::Add,
11559 &cur,
11560 &Value::Interval {
11561 months,
11562 days,
11563 micros,
11564 },
11565 )
11566 .map_err(EngineError::Eval)?;
11567 cur = match next {
11568 Some(v) => v,
11569 None => break,
11570 };
11571 }
11572 Ok(out)
11573}
11574
11575/// v7.17.0 Phase 3.P0-49 — PG-canonical: `FETCH FIRST <n> ROWS
11576/// WITH TIES` requires an `ORDER BY`. Without one, there's no
11577/// way to identify "ties" deterministically, so PG errors at
11578/// plan time. SPG mirrors that surface so the same DDL / app
11579/// behaviour holds on cutover.
11580fn check_with_ties_requires_order_by(stmt: &SelectStatement) -> Result<(), EngineError> {
11581 if stmt.limit_with_ties && stmt.order_by.is_empty() {
11582 return Err(EngineError::Unsupported(alloc::string::String::from(
11583 "WITH TIES cannot be specified without ORDER BY clause",
11584 )));
11585 }
11586 Ok(())
11587}
11588
11589/// v7.19 P5 — true iff `expr` is `unnest(arg)` at the top level
11590/// (case-insensitive). Used by `exec_select_cancel`'s
11591/// projection loop to detect Set-Returning-Function rows that
11592/// need per-row expansion. Only the top-level call counts —
11593/// `coalesce(unnest(arr), 'x')` is NOT a SRF row from the
11594/// projection's perspective; it would surface as an "unknown
11595/// function" mismatch downstream, which is what we want
11596/// (multi-SRF / nested SRF is documented carve-out for v7.19).
11597fn is_top_level_unnest(expr: &spg_sql::ast::Expr) -> bool {
11598 top_level_srf_kind(expr).is_some()
11599}
11600
11601/// v7.38 (read01, T15) — which set-returning function a top-level SELECT-list
11602/// call is, if any. Matching is allocation-free (`eq_ignore_ascii_case`, no
11603/// `to_ascii_lowercase`) because `top_level_srf_output` classifies once per
11604/// source row.
11605#[derive(Clone, Copy, PartialEq, Eq)]
11606pub(crate) enum SrfKind {
11607 Unnest,
11608 /// v7.39 (read01 round 67) — `generate_series(a, b[, step])` in the target
11609 /// list. It used to be handled ONLY by the parser's lift into FROM, so a
11610 /// second one in the same list came back as "unknown function".
11611 GenerateSeries,
11612 GenerateSubscripts,
11613 /// `_text` variants unwrap scalars to their lexeme; the plain forms render
11614 /// every value as compact JSON text.
11615 ArrayElements {
11616 as_text: bool,
11617 },
11618 PathQuery,
11619 RegexpMatches,
11620 Each {
11621 as_text: bool,
11622 },
11623 ObjectKeys,
11624}
11625
11626/// Case-insensitive match against any of `names`.
11627fn name_is(name: &str, names: &[&str]) -> bool {
11628 names.iter().any(|n| name.eq_ignore_ascii_case(n))
11629}
11630
11631pub(crate) fn top_level_srf_kind(expr: &spg_sql::ast::Expr) -> Option<SrfKind> {
11632 let spg_sql::ast::Expr::FunctionCall { name, args } = expr else {
11633 return None;
11634 };
11635 let n = args.len();
11636 // v7.38 (read01) — generate_subscripts(arr, dim) is set-returning in the
11637 // SELECT list (it returned an array there before) and shares the unnest
11638 // expansion machinery.
11639 if n == 1 && name.eq_ignore_ascii_case("unnest") {
11640 return Some(SrfKind::Unnest);
11641 }
11642 if (2..=3).contains(&n) && name.eq_ignore_ascii_case("generate_series") {
11643 return Some(SrfKind::GenerateSeries);
11644 }
11645 if n == 2 && name.eq_ignore_ascii_case("generate_subscripts") {
11646 return Some(SrfKind::GenerateSubscripts);
11647 }
11648 // v7.38 (read01, T15) — the jsonb/json SRF family and regexp_matches expand
11649 // per element / match in the SELECT list; they collapsed to a single row
11650 // (a TextArray, or an "unknown function" error for `each`) before.
11651 if n == 1 && name_is(name, &["jsonb_array_elements", "json_array_elements"]) {
11652 return Some(SrfKind::ArrayElements { as_text: false });
11653 }
11654 if n == 1
11655 && name_is(
11656 name,
11657 &["jsonb_array_elements_text", "json_array_elements_text"],
11658 )
11659 {
11660 return Some(SrfKind::ArrayElements { as_text: true });
11661 }
11662 // v7.39 (jsonpath depth) — 3rd arg = vars, 4th = silent.
11663 if (2..=4).contains(&n) && name_is(name, &["jsonb_path_query", "json_path_query"]) {
11664 return Some(SrfKind::PathQuery);
11665 }
11666 if (2..=3).contains(&n) && name.eq_ignore_ascii_case("regexp_matches") {
11667 return Some(SrfKind::RegexpMatches);
11668 }
11669 if n == 1 && name_is(name, &["jsonb_each", "json_each"]) {
11670 return Some(SrfKind::Each { as_text: false });
11671 }
11672 if n == 1 && name_is(name, &["jsonb_each_text", "json_each_text"]) {
11673 return Some(SrfKind::Each { as_text: true });
11674 }
11675 if n == 1 && name_is(name, &["jsonb_object_keys", "json_object_keys"]) {
11676 return Some(SrfKind::ObjectKeys);
11677 }
11678 None
11679}
11680
11681/// v7.38 (read01) — the row-set a top-level SELECT-list SRF emits: the elements
11682/// for `unnest(arr)`, or the 1-based subscripts `1..=length` for
11683/// `generate_subscripts(arr, 1)` (a non-1 dimension over a 1-D array yields no
11684/// rows, as in PG).
11685pub(crate) fn top_level_srf_output(
11686 expr: &spg_sql::ast::Expr,
11687 row: &Row<'static>,
11688 ctx: &EvalContext<'_>,
11689) -> Result<Vec<Value<'static>>, EngineError> {
11690 let (Some(kind), spg_sql::ast::Expr::FunctionCall { name, args }) =
11691 (top_level_srf_kind(expr), expr)
11692 else {
11693 return Err(EngineError::Unsupported(
11694 "expected a SELECT-list SRF call".into(),
11695 ));
11696 };
11697 match kind {
11698 SrfKind::Unnest => {
11699 // v7.39 (round 743) — `unnest(ARRAY[e1, …, ek])` evaluates
11700 // the elements DIRECTLY: the old path built the whole
11701 // Value::Array (one eval + a clone per element) only for
11702 // array_value_to_elements to clone every element back out.
11703 // Any other argument shape (a column, a function result)
11704 // keeps the build-then-split path.
11705 if let spg_sql::ast::Expr::Array(items) = &args[0] {
11706 return items
11707 .iter()
11708 .map(|e| eval::eval_expr(e, row, ctx).map_err(EngineError::Eval))
11709 .collect();
11710 }
11711 let arr = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11712 array_value_to_elements(&arr)
11713 }
11714 SrfKind::GenerateSeries => {
11715 // v7.39 (read01 round 96) — evaluate the args against the actual
11716 // row, then hand off to the shared core so the numeric and
11717 // timestamp/timestamptz overloads work here too (this arm used to
11718 // handle only integers, silently NULLing a temporal/numeric series
11719 // when it shared a target list with another SRF).
11720 let mut arg_values: Vec<Value<'static>> = Vec::with_capacity(args.len());
11721 for a in args {
11722 arg_values.push(eval::eval_expr(a, row, ctx).map_err(EngineError::Eval)?);
11723 }
11724 let (_, rows) = generate_series_from_values(arg_values, args, &CancelToken::none())?;
11725 Ok(rows
11726 .into_iter()
11727 .map(|r| r.values.into_iter().next().unwrap_or(Value::Null))
11728 .collect())
11729 }
11730 SrfKind::GenerateSubscripts => {
11731 let arr = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11732 let dim = eval::eval_expr(&args[1], row, ctx).map_err(EngineError::Eval)?;
11733 if !matches!(dim, Value::Int(1) | Value::BigInt(1) | Value::SmallInt(1)) {
11734 return Ok(Vec::new());
11735 }
11736 let len = array_value_to_elements(&arr)?.len();
11737 Ok((1..=len).map(|i| Value::Int(i as i32)).collect())
11738 }
11739 // One Value per array element (`_text` → text / SQL NULL, plain → the
11740 // element's compact JSON text) — the element list the FROM-clause form
11741 // materialises.
11742 SrfKind::ArrayElements { as_text } => {
11743 let arg = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11744 if matches!(arg, Value::Null) {
11745 return Ok(Vec::new());
11746 }
11747 let items =
11748 crate::json::array_element_rows(&arg, as_text, name).map_err(EngineError::Eval)?;
11749 Ok(items
11750 .into_iter()
11751 .map(|opt| opt.map(Value::text).unwrap_or(Value::Null))
11752 .collect())
11753 }
11754 // The scalar form already yields a TextArray of the keys (or errors on
11755 // a non-object, like PG); expand it into rows.
11756 SrfKind::ObjectKeys => {
11757 let v = eval::eval_expr(expr, row, ctx).map_err(EngineError::Eval)?;
11758 array_value_to_elements(&v)
11759 }
11760 // One row per match, each a text[] of the pattern's capture groups.
11761 SrfKind::RegexpMatches => {
11762 let vals: Vec<Value<'static>> = args
11763 .iter()
11764 .map(|a| eval::eval_expr(a, row, ctx).map_err(EngineError::Eval))
11765 .collect::<Result<_, _>>()?;
11766 crate::eval::regexp_matches_rows(&vals).map_err(EngineError::Eval)
11767 }
11768 // One composite `(key, value)` row per object member (plain → jsonb
11769 // value, `_text` → text / SQL NULL).
11770 SrfKind::Each { as_text } => {
11771 let arg = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11772 if matches!(arg, Value::Null) {
11773 return Ok(Vec::new());
11774 }
11775 let pairs = crate::json::each_rows(&arg, as_text, name).map_err(EngineError::Eval)?;
11776 Ok(pairs
11777 .into_iter()
11778 .map(|(k, v)| {
11779 let val = if as_text {
11780 v.map(Value::text).unwrap_or(Value::Null)
11781 } else {
11782 v.map(Value::json).unwrap_or(Value::Null)
11783 };
11784 Value::Composite(alloc::vec![
11785 ("key".to_string(), Value::text(k)),
11786 ("value".to_string(), val),
11787 ])
11788 })
11789 .collect())
11790 }
11791 // One Value per matched JSON value.
11792 SrfKind::PathQuery => {
11793 let doc = eval::eval_expr(&args[0], row, ctx).map_err(EngineError::Eval)?;
11794 let path = eval::eval_expr(&args[1], row, ctx).map_err(EngineError::Eval)?;
11795 // v7.39 — optional vars document (3rd arg).
11796 let vars = match args.get(2) {
11797 Some(a) => {
11798 let v = eval::eval_expr(a, row, ctx).map_err(EngineError::Eval)?;
11799 crate::json::parse_path_vars(&v).map_err(EngineError::Eval)?
11800 }
11801 None => None,
11802 };
11803 match crate::json::path_query_vars(&doc, &path, vars.as_ref())
11804 .map_err(EngineError::Eval)?
11805 {
11806 Value::Null => Ok(Vec::new()),
11807 Value::TextArray(items) => Ok(items
11808 .into_iter()
11809 .map(|opt| opt.map(Value::text).unwrap_or(Value::Null))
11810 .collect()),
11811 other => Ok(alloc::vec![other]),
11812 }
11813 }
11814 }
11815}
11816
11817/// v7.19 P5 — turn an array-typed `Value` into the element list
11818/// `unnest()` projection emits. NULL → empty list (PG: `unnest(NULL)
11819/// = (no rows)`). Non-array values fall through to a type-mismatch
11820/// error.
11821pub(crate) fn array_value_to_elements(v: &Value) -> Result<Vec<Value<'static>>, EngineError> {
11822 // v7.39 (round 236) — PG unnests a multidimensional array into its
11823 // elements in row-major order (`unnest(ARRAY[[1,2],[3,4]])` is four
11824 // rows). SPG stores 2-D arrays as their own variants, which fell
11825 // through to the type-mismatch arm below.
11826 if let Some(flat) = crate::eval::values::flatten_2d(v) {
11827 return array_value_to_elements(&flat);
11828 }
11829 match v {
11830 Value::Null => Ok(Vec::new()),
11831 Value::TextArray(items) => Ok(items
11832 .iter()
11833 .map(|opt| {
11834 opt.as_ref()
11835 .map(|s| Value::text(s.clone()))
11836 .unwrap_or(Value::Null)
11837 })
11838 .collect()),
11839 Value::IntArray(items) => Ok(items
11840 .iter()
11841 .map(|opt| opt.map(Value::Int).unwrap_or(Value::Null))
11842 .collect()),
11843 Value::BigIntArray(items) => Ok(items
11844 .iter()
11845 .map(|opt| opt.map(Value::BigInt).unwrap_or(Value::Null))
11846 .collect()),
11847 // v7.39 (read01 multirangetypes.c) — unnest(anymultirange): one
11848 // range per canonical span.
11849 Value::Multirange { kind, ranges } => Ok(ranges
11850 .iter()
11851 .map(|s| Value::Range {
11852 kind: *kind,
11853 lower: s.lower.clone(),
11854 upper: s.upper.clone(),
11855 lower_inc: s.lower_inc,
11856 upper_inc: s.upper_inc,
11857 empty: false,
11858 })
11859 .collect()),
11860 other => Err(EngineError::Eval(EvalError::TypeMismatch {
11861 detail: alloc::format!(
11862 "unnest() expects an array argument, got {}",
11863 crate::conversions::pg_type_name_for_error_opt(other.data_type())
11864 ),
11865 })),
11866 }
11867}
11868
11869impl Engine {
11870 /// v7.17.0 Phase 1.2 — find every catalog VIEW referenced in
11871 /// the SELECT's FROM / JOIN graph, re-parse each view's body
11872 /// source, and prepend it as a synthetic CTE on the
11873 /// returned SelectStatement. Returns `None` when no view
11874 /// references are found (caller proceeds with the original
11875 /// statement); returns `Some(rewritten)` otherwise (caller
11876 /// re-runs exec_select_cancel on the rewritten form so the
11877 /// regular CTE materialiser handles it).
11878 fn expand_views_in_select(
11879 &self,
11880 stmt: &SelectStatement,
11881 ) -> Result<Option<SelectStatement>, EngineError> {
11882 let cat = self.active_catalog();
11883 let mut referenced: Vec<String> = Vec::new();
11884 if let Some(from) = &stmt.from {
11885 collect_view_refs(&from.primary, cat, &mut referenced);
11886 for j in &from.joins {
11887 collect_view_refs(&j.table, cat, &mut referenced);
11888 }
11889 }
11890 // Don't expand a view name that's already shadowed by a
11891 // CTE on the same SELECT — the CTE wins per PG.
11892 referenced.retain(|n| !stmt.ctes.iter().any(|c| c.name == *n));
11893 if referenced.is_empty() {
11894 return Ok(None);
11895 }
11896 let mut new_ctes: Vec<spg_sql::ast::Cte> = Vec::with_capacity(referenced.len());
11897 for name in &referenced {
11898 let view = cat.view(name).ok_or_else(|| {
11899 EngineError::Storage(spg_storage::StorageError::Corrupt(alloc::format!(
11900 "view {name:?} disappeared mid-expansion"
11901 )))
11902 })?;
11903 let parsed = spg_sql::parser::parse_statement(&view.body).map_err(|e| {
11904 EngineError::Unsupported(alloc::format!("view {name:?} body re-parse failed: {e}"))
11905 })?;
11906 let Statement::Select(body) = parsed else {
11907 return Err(EngineError::Unsupported(alloc::format!(
11908 "view {name:?} body is not a SELECT (catalog corruption)"
11909 )));
11910 };
11911 new_ctes.push(spg_sql::ast::Cte {
11912 name: name.clone(),
11913 body: spg_sql::ast::CteBody::Select(body),
11914 recursive: false,
11915 column_overrides: view.columns.clone(),
11916 search: None,
11917 cycle: None,
11918 });
11919 }
11920 let mut out = stmt.clone();
11921 // Prepend so view CTEs are visible to caller-supplied CTEs.
11922 new_ctes.extend(out.ctes);
11923 out.ctes = new_ctes;
11924 Ok(Some(out))
11925 }
11926
11927 /// v7.37.6-B(sentori Epic 2 P0)— if `stmt`'s FROM-clause references
11928 /// any partition-parent table, rewrite the SELECT so each parent
11929 /// reference resolves to a CTE whose body is a `UNION ALL` over the
11930 /// children that pass the WHERE-derived partition-key range. Returns
11931 /// `None`(no rewrite needed)when no parent is referenced or all
11932 /// references are shadowed by a same-name CTE.
11933 ///
11934 /// Pruning vocabulary at v7.37.6-B:
11935 /// * Flat `AND` chain over `<key> {>= | > | < | <= | =} literal`
11936 /// and `<key> BETWEEN literal AND literal`.
11937 /// * Anything outside that(OR / nested IN / function call on the
11938 /// key)defaults to "no pruning" — every child + DEFAULT lands
11939 /// in the UNION. Correctness is preserved; only the plan size
11940 /// widens.
11941 fn expand_partition_parents_in_select(
11942 &self,
11943 stmt: &SelectStatement,
11944 ) -> Result<Option<SelectStatement>, EngineError> {
11945 let cat = self.active_catalog();
11946 let Some(from) = &stmt.from else {
11947 return Ok(None);
11948 };
11949 let mut parent_refs: Vec<String> = Vec::new();
11950 collect_partition_parent_refs(&from.primary, cat, &mut parent_refs);
11951 for j in &from.joins {
11952 collect_partition_parent_refs(&j.table, cat, &mut parent_refs);
11953 }
11954 // Drop names shadowed by a CTE on the same SELECT(PG semantics
11955 // — same as view expansion above).
11956 parent_refs.retain(|n| !stmt.ctes.iter().any(|c| c.name.eq_ignore_ascii_case(n)));
11957 if parent_refs.is_empty() {
11958 return Ok(None);
11959 }
11960 // Synthesise a CTE name per parent so the existing
11961 // "CTE shadows a real table" guard doesn't fire (the parent
11962 // IS a real table in the catalog, unlike VIEW expansion's
11963 // case). The FROM-clause TableRef walker below rewrites
11964 // every parent reference to point at the synthetic CTE.
11965 let synth_name = |p: &str| alloc::format!("__spg_partition_{p}");
11966 let mut new_ctes: Vec<spg_sql::ast::Cte> = Vec::with_capacity(parent_refs.len());
11967 let mut expanded_parents: Vec<alloc::string::String> = Vec::new();
11968 for parent_name in &parent_refs {
11969 // No children = no rewrite. The parent itself is a real
11970 // (empty-rows) table — the regular FROM-resolution path
11971 // will scan it and return 0 rows, matching the
11972 // "partition parent with no children" plan. Skipping the
11973 // CTE here also avoids `SELECT * FROM parent` re-entering
11974 // this rewrite on the synthetic body (infinite recursion).
11975 let Some(body) = self.build_partition_parent_union_body(parent_name, stmt)? else {
11976 continue;
11977 };
11978 new_ctes.push(spg_sql::ast::Cte {
11979 name: synth_name(parent_name),
11980 body: spg_sql::ast::CteBody::Select(body),
11981 recursive: false,
11982 column_overrides: Vec::new(),
11983 search: None,
11984 cycle: None,
11985 });
11986 expanded_parents.push(parent_name.clone());
11987 }
11988 if expanded_parents.is_empty() {
11989 return Ok(None);
11990 }
11991 let mut out = stmt.clone();
11992 if let Some(from) = out.from.as_mut() {
11993 rewrite_partition_parent_table_ref(&mut from.primary, &expanded_parents, &synth_name);
11994 for j in &mut from.joins {
11995 rewrite_partition_parent_table_ref(&mut j.table, &expanded_parents, &synth_name);
11996 }
11997 }
11998 new_ctes.extend(out.ctes);
11999 out.ctes = new_ctes;
12000 Ok(Some(out))
12001 }
12002
12003 /// Build the `SELECT * FROM child1 UNION ALL …` body for one parent.
12004 /// Children include every overlap-hit `Range` plus(always)the
12005 /// `Default` child(if any). Returns `Ok(None)` when no children
12006 /// would survive — caller skips the CTE injection and lets the
12007 /// parent fall through to the regular(empty-rows)scan path,
12008 /// avoiding the infinite recursion that an empty-body CTE
12009 /// referencing the parent name would trigger.
12010 /// v7.37.16 (16.10) — public helper invoked from explain.rs to
12011 /// surface "which children survive the WHERE-clause prune" in
12012 /// EXPLAIN output. Returns `None` when `parent_name` isn't
12013 /// actually a partition parent; otherwise returns the list of
12014 /// children the planner would scan (same algorithm as
12015 /// [`Self::build_partition_parent_union_body`] but without the
12016 /// SQL re-parse).
12017 /// v7.39 (round 224) — the kept-children prune keyed off a bare WHERE
12018 /// expression (the PG-shaped EXPLAIN's scan builder has no full
12019 /// SelectStatement in hand). Wraps the original by synthesising a
12020 /// minimal statement carrying just the predicate.
12021 pub(crate) fn explain_partition_kept_children_by_where(
12022 &self,
12023 parent_name: &str,
12024 where_: Option<&spg_sql::ast::Expr>,
12025 ) -> Option<Vec<alloc::string::String>> {
12026 let mut synth = SelectStatement::default();
12027 synth.where_ = where_.cloned();
12028 self.explain_partition_kept_children(parent_name, &synth)
12029 }
12030
12031 pub(crate) fn explain_partition_kept_children(
12032 &self,
12033 parent_name: &str,
12034 outer: &SelectStatement,
12035 ) -> Option<Vec<alloc::string::String>> {
12036 use spg_storage::PartitionRole;
12037 let cat = self.active_catalog();
12038 let parent = cat.get(parent_name)?;
12039 let (key_position, parent_kind) = match &parent.schema().partition_role {
12040 Some(PartitionRole::Parent {
12041 key_column_positions,
12042 kind,
12043 ..
12044 }) => (*key_column_positions.first().unwrap_or(&0), *kind),
12045 _ => return None,
12046 };
12047 let key_col_name = parent.schema().columns[key_position].name.clone();
12048 let (lo_bound, hi_bound) = match outer.where_.as_ref() {
12049 Some(expr) => extract_key_range(expr, &key_col_name),
12050 None => (None, None),
12051 };
12052 let eq_value: Option<spg_storage::Value<'static>> = match outer.where_.as_ref() {
12053 Some(expr) => extract_key_eq_value(expr, &key_col_name),
12054 None => None,
12055 };
12056 let children = crate::partition::children_of_parent(cat, parent_name);
12057 let mut kept: Vec<alloc::string::String> = Vec::new();
12058 let mut default_child: Option<alloc::string::String> = None;
12059 for child_name in &children {
12060 let Some(child) = cat.get(child_name) else {
12061 continue;
12062 };
12063 match &child.schema().partition_role {
12064 Some(PartitionRole::Range { lower, upper, .. }) => {
12065 if range_satisfies_filter(lower, upper, lo_bound.as_ref(), hi_bound.as_ref()) {
12066 kept.push(child_name.clone());
12067 }
12068 }
12069 Some(PartitionRole::List { values, .. }) => match &eq_value {
12070 Some(v) => {
12071 if values.iter().any(|b| b.equals_value(v)) {
12072 kept.push(child_name.clone());
12073 }
12074 }
12075 None => kept.push(child_name.clone()),
12076 },
12077 Some(PartitionRole::Hash {
12078 modulus, remainder, ..
12079 }) => match &eq_value {
12080 Some(v) => {
12081 let h = crate::partition::pg_compatible_hash(v);
12082 if h.rem_euclid(u64::from(*modulus)) == u64::from(*remainder) {
12083 kept.push(child_name.clone());
12084 }
12085 }
12086 None => kept.push(child_name.clone()),
12087 },
12088 Some(PartitionRole::Default { .. }) => {
12089 default_child = Some(child_name.clone());
12090 }
12091 _ => {}
12092 }
12093 }
12094 let _ = parent_kind;
12095 if let Some(d) = default_child {
12096 if kept.is_empty() || eq_value.is_none() {
12097 kept.push(d);
12098 }
12099 }
12100 Some(kept)
12101 }
12102
12103 fn build_partition_parent_union_body(
12104 &self,
12105 parent_name: &str,
12106 outer: &SelectStatement,
12107 ) -> Result<Option<SelectStatement>, EngineError> {
12108 use spg_storage::PartitionRole;
12109 let cat = self.active_catalog();
12110 let parent = cat.get(parent_name).ok_or_else(|| {
12111 EngineError::Storage(spg_storage::StorageError::Corrupt(alloc::format!(
12112 "partition parent {parent_name:?} disappeared mid-expansion"
12113 )))
12114 })?;
12115 let (key_position, parent_kind) = match &parent.schema().partition_role {
12116 Some(PartitionRole::Parent {
12117 key_column_positions,
12118 kind,
12119 ..
12120 }) => (*key_column_positions.first().unwrap_or(&0), *kind),
12121 // v7.39 (round 645) — an INHERITANCE parent, which has no
12122 // role of its own: the relationship is recorded only in the
12123 // children. Three things differ from a partition parent and
12124 // all three are in this body.
12125 //
12126 // * The parent HOLDS ROWS, so it is a term of the union —
12127 // `FROM ONLY`, or expanding it would recurse.
12128 // * There is no partition key, so there is nothing to
12129 // prune: every child is a term.
12130 // * A child may declare columns of its own, so the terms
12131 // name the PARENT's columns rather than `*`. PG's
12132 // `SELECT * FROM parent` returns the parent's shape.
12133 //
12134 // Answered from this match rather than a branch before it —
12135 // round 644 measured what an extra early return beside an
12136 // existing test costs in this file.
12137 _ if crate::partition::has_inheritance_children(cat, parent_name) => {
12138 let cols = parent
12139 .schema()
12140 .columns
12141 .iter()
12142 .map(|c| quote_ident_for_sql(&c.name))
12143 .collect::<Vec<_>>()
12144 .join(", ");
12145 let carry_sys = references_ctid(outer);
12146 let sys = if carry_sys {
12147 let mut t = alloc::string::String::new();
12148 for s in SYSTEM_COLUMNS {
12149 t.push_str(", ");
12150 t.push_str(s);
12151 }
12152 t
12153 } else {
12154 alloc::string::String::new()
12155 };
12156 let mut body = alloc::format!(
12157 "SELECT {cols}{sys} FROM ONLY {}",
12158 quote_ident_for_sql(parent_name)
12159 );
12160 for child in crate::partition::children_of_parent(cat, parent_name) {
12161 body.push_str(&alloc::format!(
12162 " UNION ALL SELECT {cols}{sys} FROM {}",
12163 quote_ident_for_sql(&child)
12164 ));
12165 }
12166 return parse_select_or_corrupt(&body).map(Some);
12167 }
12168 _ => {
12169 return Err(EngineError::Unsupported(alloc::format!(
12170 "partition expansion: {parent_name:?} is not a parent"
12171 )));
12172 }
12173 };
12174 let key_col_name = parent.schema().columns[key_position].name.clone();
12175 // v7.37.16 (16.7) — for RANGE we extract a (lo, hi) interval
12176 // off the WHERE; for LIST / HASH we extract a single `=`
12177 // literal (and the rest of the planner falls back to "keep
12178 // every child" — same conservative path as 16.1/16.2).
12179 let (lo_bound, hi_bound) = match outer.where_.as_ref() {
12180 Some(expr) => extract_key_range(expr, &key_col_name),
12181 None => (None, None),
12182 };
12183 let eq_value: Option<spg_storage::Value<'static>> = match outer.where_.as_ref() {
12184 Some(expr) => extract_key_eq_value(expr, &key_col_name),
12185 None => None,
12186 };
12187 let children = crate::partition::children_of_parent(cat, parent_name);
12188 let mut kept: Vec<String> = Vec::new();
12189 let mut default_child: Option<String> = None;
12190 // First pass — apply per-strategy gates, defer DEFAULT until
12191 // we know whether some non-DEFAULT child matched.
12192 for child_name in &children {
12193 let Some(child) = cat.get(child_name) else {
12194 continue;
12195 };
12196 match &child.schema().partition_role {
12197 Some(PartitionRole::Range { lower, upper, .. }) => {
12198 if range_satisfies_filter(lower, upper, lo_bound.as_ref(), hi_bound.as_ref()) {
12199 kept.push(child_name.clone());
12200 }
12201 }
12202 // v7.37.16 (16.7) — LIST pruning: if WHERE has `key
12203 // = <lit>`, only the child whose values contain that
12204 // literal survives. Otherwise (no equality predicate
12205 // or planner couldn't extract one) keep the child
12206 // conservatively.
12207 Some(PartitionRole::List { values, .. }) => match &eq_value {
12208 Some(v) => {
12209 if values.iter().any(|b| b.equals_value(v)) {
12210 kept.push(child_name.clone());
12211 }
12212 }
12213 None => kept.push(child_name.clone()),
12214 },
12215 // v7.37.16 (16.7) — HASH pruning: with `key = <lit>`
12216 // we know the residue class deterministically, so
12217 // only the matching REMAINDER child survives.
12218 Some(PartitionRole::Hash {
12219 modulus, remainder, ..
12220 }) => match &eq_value {
12221 Some(v) => {
12222 let h = crate::partition::pg_compatible_hash(v);
12223 if h.rem_euclid(u64::from(*modulus)) == u64::from(*remainder) {
12224 kept.push(child_name.clone());
12225 }
12226 }
12227 None => kept.push(child_name.clone()),
12228 },
12229 Some(PartitionRole::Default { .. }) => {
12230 default_child = Some(child_name.clone());
12231 }
12232 _ => {}
12233 }
12234 }
12235 // PG-style DEFAULT semantics: the DEFAULT child must be
12236 // scanned iff some row could fall outside every concrete
12237 // child's bound predicate. We approximate that as "no
12238 // concrete child matched" (== full prune) — strictly
12239 // conservative for LIST / HASH (DEFAULT also catches rows
12240 // outside the union of value-sets / residues), and matches
12241 // PG for the equality case where we *do* know the routing
12242 // outcome.
12243 let _ = parent_kind; // used to silence dead-code lint while 16.8-9 lands.
12244 if let Some(d) = default_child {
12245 if kept.is_empty() {
12246 kept.push(d);
12247 } else if eq_value.is_none() {
12248 // Without an equality literal, the DEFAULT child may
12249 // still hold matching rows (e.g. LIKE on TEXT keys
12250 // for which a LIST partition exists). Keep it.
12251 kept.push(d);
12252 }
12253 }
12254 // Build the UNION ALL body text and re-parse — keeps the
12255 // rewrite expressible in surface SQL so the engine's existing
12256 // parser path handles the AST shape uniformly.
12257 if kept.is_empty() {
12258 // No children survive — caller falls back to scanning the
12259 // (empty) parent table. Returning None here is what
12260 // prevents the synthetic CTE from referring back to the
12261 // parent name and re-entering this rewrite pass.
12262 let _ = parent_name;
12263 return Ok(None);
12264 }
12265 // v7.39 (round 622, S05a) — the system columns of the CHILD the row
12266 // actually lives in.
12267 //
12268 // The parent is read through a synthetic CTE, so a `tableoid` on it
12269 // resolved against that CTE: every row of every child reported
12270 // `__spg_partition_pm`, an internal name no user ever typed, where
12271 // PG reports `pm_a` / `pm_b`. That is not only a leak — it silently
12272 // empties `WHERE tableoid::regclass::TEXT = 'pm_a'`, which is how
12273 // one asks "which partition is this row in", answering 0 rows where
12274 // PG answers 1. `ctid` had the same shape: it numbered the CTE's
12275 // output, so rows in different children got distinct ctids instead
12276 // of each child's own physical position.
12277 //
12278 // Naming them in the term is what carries them: the child scan
12279 // materialises its own six because the statement now references
12280 // them, and they land in SYSTEM_COLUMNS order right after the user
12281 // columns — the exact layout the positional `*` skip already
12282 // expects. Only done when the outer statement asks for one, so a
12283 // plain `SELECT * FROM parent` scans exactly what it scanned.
12284 let carry_sys = references_ctid(outer);
12285 let mut body = alloc::string::String::new();
12286 for (i, child_name) in kept.iter().enumerate() {
12287 if i > 0 {
12288 body.push_str(" UNION ALL ");
12289 }
12290 body.push_str("SELECT *");
12291 if carry_sys {
12292 for sys in SYSTEM_COLUMNS {
12293 body.push_str(", ");
12294 body.push_str(sys);
12295 }
12296 }
12297 body.push_str(" FROM ");
12298 body.push_str("e_ident_for_sql(child_name));
12299 }
12300 parse_select_or_corrupt(&body).map(Some)
12301 }
12302}
12303
12304/// Rewrite a `TableRef` pointing at a partition parent so it
12305/// references the synthetic CTE created by the expansion. If the
12306/// original ref had no alias, preserve the parent name as an alias
12307/// so column references like `events_partitioned.received_at`
12308/// keep resolving.
12309fn rewrite_partition_parent_table_ref(
12310 t: &mut spg_sql::ast::TableRef,
12311 parents: &[alloc::string::String],
12312 synth_name: &impl Fn(&str) -> alloc::string::String,
12313) {
12314 if t.lateral_subquery.is_some() || t.unnest_expr.is_some() || t.generate_series_args.is_some() {
12315 return;
12316 }
12317 // v7.39 (round 644) — an ONLY reference stays pointed at the parent
12318 // itself. The rewrite is keyed on the NAME, so in
12319 // `FROM ONLY po a JOIN po b` the un-qualified `b` put `po` on the
12320 // parent list and this then rewrote BOTH — including the one that
12321 // asked not to descend. PG answers 0 for that join; SPG answered 2.
12322 // Folded into the existing test — see the note in
12323 // `collect_partition_parent_refs` for what a separate one cost.
12324 if t.only || !parents.iter().any(|p| p == &t.name) {
12325 return;
12326 }
12327 if t.alias.is_none() {
12328 t.alias = Some(t.name.clone());
12329 }
12330 t.name = synth_name(&t.name);
12331}
12332
12333/// Walk a `TableRef` and push its `name` if it resolves to a partition
12334/// parent in `cat`. Skips `lateral_subquery` / `unnest_expr` /
12335/// `generate_series_args` references — those aren't catalog tables.
12336fn collect_partition_parent_refs(
12337 t: &spg_sql::ast::TableRef,
12338 cat: &spg_storage::Catalog,
12339 out: &mut Vec<alloc::string::String>,
12340) {
12341 if t.lateral_subquery.is_some() || t.unnest_expr.is_some() || t.generate_series_args.is_some() {
12342 return;
12343 }
12344 // v7.39 (round 644) — `FROM ONLY <parent>` scans the parent alone.
12345 // The keyword used to be absorbed at parse time, so this fanned out
12346 // anyway and `SELECT count(*) FROM ONLY <partitioned parent>`
12347 // answered 2 where PG answers 0.
12348 //
12349 // Folded into the existing test rather than given an early return of
12350 // its own: as two extra lines in this function's body it cost
12351 // `WHERE g BETWEEN 10 AND 20` **26x**, 5.9 ms to 155 ms, measured
12352 // outside the panel. Rounds 641 and 643 met the same wall from the
12353 // other two directions — adding to a hot function and taking away
12354 // from a cold one. What goes in a body near the row loop is a
12355 // codegen decision whatever its shape.
12356 if !t.only && crate::partition::has_children(cat, &t.name) {
12357 out.push(t.name.clone());
12358 }
12359}
12360
12361/// v7.37.6-B partition-key range derived from a WHERE expression.
12362/// `i64` microseconds since epoch with the same sign convention as
12363/// `Value::Timestamp`. Inclusive bool: `true` ⇒ inclusive(`>=` / `<=`
12364/// / `=`),`false` ⇒ exclusive(`>` / `<`).
12365#[derive(Debug, Clone, Copy)]
12366pub(crate) struct PartitionFilterBound {
12367 pub micros: i64,
12368 pub inclusive: bool,
12369}
12370
12371/// Walk a flat AND chain looking for `<key> <op> <timestamptz-literal>`
12372/// shapes; tighten the running lo / hi as we go. Anything outside that
12373/// (OR / nested calls / non-key columns)is ignored — caller treats
12374/// `None` as "no constraint on that side."
12375fn extract_key_range(
12376 expr: &spg_sql::ast::Expr,
12377 key_col: &str,
12378) -> (Option<PartitionFilterBound>, Option<PartitionFilterBound>) {
12379 let mut lo: Option<PartitionFilterBound> = None;
12380 let mut hi: Option<PartitionFilterBound> = None;
12381 let mut stack: Vec<&spg_sql::ast::Expr> = alloc::vec![expr];
12382 while let Some(e) = stack.pop() {
12383 match e {
12384 spg_sql::ast::Expr::Binary {
12385 lhs,
12386 op: spg_sql::ast::BinOp::And,
12387 rhs,
12388 } => {
12389 stack.push(lhs);
12390 stack.push(rhs);
12391 }
12392 // BETWEEN is desugared at parse time into `lhs >= low AND
12393 // lhs <= high`, so it lands here as two regular Binary
12394 // arms via the AND walker above.
12395 spg_sql::ast::Expr::Binary { lhs, op, rhs } => {
12396 let (col_ref, lit_side, swapped) = if is_column_ref(lhs, key_col) {
12397 (Some(lhs.as_ref()), rhs.as_ref(), false)
12398 } else if is_column_ref(rhs, key_col) {
12399 (Some(rhs.as_ref()), lhs.as_ref(), true)
12400 } else {
12401 (None, lhs.as_ref(), false)
12402 };
12403 if col_ref.is_none() {
12404 continue;
12405 }
12406 let Some(lit) = literal_to_micros(lit_side) else {
12407 continue;
12408 };
12409 use spg_sql::ast::BinOp::{Eq, Gt, GtEq, Lt, LtEq};
12410 let effective_op = if swapped {
12411 match op {
12412 Lt => Gt,
12413 LtEq => GtEq,
12414 Gt => Lt,
12415 GtEq => LtEq,
12416 other => *other,
12417 }
12418 } else {
12419 *op
12420 };
12421 match effective_op {
12422 Eq => {
12423 tighten_lo(
12424 &mut lo,
12425 PartitionFilterBound {
12426 micros: lit,
12427 inclusive: true,
12428 },
12429 );
12430 tighten_hi(
12431 &mut hi,
12432 PartitionFilterBound {
12433 micros: lit,
12434 inclusive: true,
12435 },
12436 );
12437 }
12438 GtEq => {
12439 tighten_lo(
12440 &mut lo,
12441 PartitionFilterBound {
12442 micros: lit,
12443 inclusive: true,
12444 },
12445 );
12446 }
12447 Gt => {
12448 tighten_lo(
12449 &mut lo,
12450 PartitionFilterBound {
12451 micros: lit,
12452 inclusive: false,
12453 },
12454 );
12455 }
12456 LtEq => {
12457 tighten_hi(
12458 &mut hi,
12459 PartitionFilterBound {
12460 micros: lit,
12461 inclusive: true,
12462 },
12463 );
12464 }
12465 Lt => {
12466 tighten_hi(
12467 &mut hi,
12468 PartitionFilterBound {
12469 micros: lit,
12470 inclusive: false,
12471 },
12472 );
12473 }
12474 _ => {}
12475 }
12476 }
12477 _ => {}
12478 }
12479 }
12480 (lo, hi)
12481}
12482
12483fn tighten_lo(slot: &mut Option<PartitionFilterBound>, new: PartitionFilterBound) {
12484 match slot {
12485 None => *slot = Some(new),
12486 Some(cur) => {
12487 if new.micros > cur.micros
12488 || (new.micros == cur.micros && !new.inclusive && cur.inclusive)
12489 {
12490 *slot = Some(new);
12491 }
12492 }
12493 }
12494}
12495
12496fn tighten_hi(slot: &mut Option<PartitionFilterBound>, new: PartitionFilterBound) {
12497 match slot {
12498 None => *slot = Some(new),
12499 Some(cur) => {
12500 if new.micros < cur.micros
12501 || (new.micros == cur.micros && !new.inclusive && cur.inclusive)
12502 {
12503 *slot = Some(new);
12504 }
12505 }
12506 }
12507}
12508
12509fn is_column_ref(e: &spg_sql::ast::Expr, key_col: &str) -> bool {
12510 if let spg_sql::ast::Expr::Column(c) = e {
12511 c.name.eq_ignore_ascii_case(key_col)
12512 } else {
12513 false
12514 }
12515}
12516
12517/// v7.37.16 (16.7) — walk an AND-chain WHERE and pull a single
12518/// `key_col = <literal>` predicate out for LIST/HASH partition
12519/// pruning. Returns `None` when no equality literal can be lifted
12520/// (planner then keeps every child — correctness preserved). The
12521/// returned `Value<'static>` is an owned coercion so the caller can
12522/// outlive any AST node it was extracted from.
12523pub(crate) fn extract_key_eq_value(
12524 expr: &spg_sql::ast::Expr,
12525 key_col: &str,
12526) -> Option<spg_storage::Value<'static>> {
12527 let mut stack: Vec<&spg_sql::ast::Expr> = alloc::vec![expr];
12528 while let Some(e) = stack.pop() {
12529 match e {
12530 spg_sql::ast::Expr::Binary {
12531 lhs,
12532 op: spg_sql::ast::BinOp::And,
12533 rhs,
12534 } => {
12535 stack.push(lhs);
12536 stack.push(rhs);
12537 }
12538 spg_sql::ast::Expr::Binary {
12539 lhs,
12540 op: spg_sql::ast::BinOp::Eq,
12541 rhs,
12542 } => {
12543 let lit_side = if is_column_ref(lhs, key_col) {
12544 rhs.as_ref()
12545 } else if is_column_ref(rhs, key_col) {
12546 lhs.as_ref()
12547 } else {
12548 continue;
12549 };
12550 let cloned = lit_side.clone();
12551 let Ok(v) = crate::conversions::literal_expr_to_value(cloned) else {
12552 continue;
12553 };
12554 // Coerce to an owned Value<'static> so the caller
12555 // can hold it past the WHERE expression's lifetime.
12556 let owned: spg_storage::Value<'static> = match v {
12557 spg_storage::Value::Text(s) => {
12558 spg_storage::Value::Text(alloc::borrow::Cow::Owned(s.into_owned()))
12559 }
12560 spg_storage::Value::SmallInt(n) => spg_storage::Value::SmallInt(n),
12561 spg_storage::Value::Int(n) => spg_storage::Value::Int(n),
12562 spg_storage::Value::BigInt(n) => spg_storage::Value::BigInt(n),
12563 spg_storage::Value::Date(d) => spg_storage::Value::Date(d),
12564 spg_storage::Value::Timestamp(t) => spg_storage::Value::Timestamp(t),
12565 spg_storage::Value::Bool(b) => spg_storage::Value::Bool(b),
12566 spg_storage::Value::Null => spg_storage::Value::Null,
12567 // Anything else (Vector / Json / Bytes / Numeric /
12568 // arrays / interval / …) isn't a current partition
12569 // key type; skip without pruning.
12570 _ => continue,
12571 };
12572 return Some(owned);
12573 }
12574 _ => {}
12575 }
12576 }
12577 None
12578}
12579
12580/// Coerce a literal Expr(after the parser folded sequence calls etc.)
12581/// to i64 microseconds. Mirrors `evaluate_partition_bound`'s shape so
12582/// pruning and routing agree on the literal vocabulary. Returns
12583/// `None` when the literal isn't recognised(planner then skips
12584/// pruning on that branch — correctness preserved).
12585fn literal_to_micros(e: &spg_sql::ast::Expr) -> Option<i64> {
12586 let cloned = e.clone();
12587 let value = crate::conversions::literal_expr_to_value(cloned).ok()?;
12588 match value {
12589 spg_storage::Value::Timestamp(m) => Some(m),
12590 spg_storage::Value::Date(days) => Some(i64::from(days) * 86_400i64 * 1_000_000i64),
12591 spg_storage::Value::Text(s) => crate::eval::parse_timestamp_literal(&s),
12592 _ => None,
12593 }
12594}
12595
12596/// `[range_lo, range_hi)` of a child is kept iff it can hold any row
12597/// satisfying the WHERE-derived filter range. PG-style half-open:
12598/// child upper exclusive. Filter inclusivity is honoured per-bound.
12599fn range_satisfies_filter(
12600 range_lo: &spg_storage::PartitionBound,
12601 range_hi: &spg_storage::PartitionBound,
12602 filter_lo: Option<&PartitionFilterBound>,
12603 filter_hi: Option<&PartitionFilterBound>,
12604) -> bool {
12605 use spg_storage::PartitionBound;
12606 // For each filter side, reject children that can't host any row
12607 // matching the predicate.
12608 if let Some(lo) = filter_lo {
12609 // child upper bound vs filter lower:
12610 // if filter is x >= L, child rejects iff child.hi <= L
12611 // if filter is x > L, child rejects iff child.hi <= L
12612 // (child.hi exclusive, so equality with L still rejects)
12613 match range_hi {
12614 PartitionBound::MinValue => return false,
12615 PartitionBound::MaxValue => {}
12616 PartitionBound::TimestampTz(hi) => {
12617 if *hi <= lo.micros {
12618 return false;
12619 }
12620 }
12621 // v7.37.16 (16.6) — non-TIMESTAMPTZ bounds aren't
12622 // matched against TIMESTAMPTZ filters here; keep child
12623 // (conservative: don't prune).
12624 PartitionBound::BigInt(_)
12625 | PartitionBound::Int(_)
12626 | PartitionBound::SmallInt(_)
12627 | PartitionBound::Date(_)
12628 | PartitionBound::Text(_) => {}
12629 }
12630 }
12631 if let Some(hi) = filter_hi {
12632 // child lower bound vs filter upper:
12633 // if filter is x <= U, child rejects iff child.lo > U
12634 // if filter is x < U, child rejects iff child.lo >= U
12635 match range_lo {
12636 PartitionBound::MaxValue => return false,
12637 PartitionBound::MinValue => {}
12638 PartitionBound::TimestampTz(lo) => {
12639 let rejects = if hi.inclusive {
12640 *lo > hi.micros
12641 } else {
12642 *lo >= hi.micros
12643 };
12644 if rejects {
12645 return false;
12646 }
12647 }
12648 PartitionBound::BigInt(_)
12649 | PartitionBound::Int(_)
12650 | PartitionBound::SmallInt(_)
12651 | PartitionBound::Date(_)
12652 | PartitionBound::Text(_) => {}
12653 }
12654 }
12655 true
12656}
12657
12658fn quote_ident_for_sql(name: &str) -> alloc::string::String {
12659 // Match spg-sql's quoting rule(unquoted when ASCII-lowercase
12660 // identifier, otherwise quoted). Conservative: always quote so
12661 // children with reserved names round-trip safely through the
12662 // CTE-body parse.
12663 let mut out = alloc::string::String::with_capacity(name.len() + 2);
12664 out.push('"');
12665 for c in name.chars() {
12666 if c == '"' {
12667 out.push('"');
12668 }
12669 out.push(c);
12670 }
12671 out.push('"');
12672 out
12673}
12674
12675fn parse_select_or_corrupt(sql: &str) -> Result<SelectStatement, EngineError> {
12676 let parsed = spg_sql::parser::parse_statement(sql).map_err(|e| {
12677 EngineError::Unsupported(alloc::format!(
12678 "partition expansion: generated SQL {sql:?} failed to re-parse: {e}"
12679 ))
12680 })?;
12681 let Statement::Select(body) = parsed else {
12682 return Err(EngineError::Unsupported(alloc::format!(
12683 "partition expansion: generated SQL {sql:?} is not a SELECT"
12684 )));
12685 };
12686 Ok(body)
12687}
12688
12689/// v7.39 (read01 round 65/66) — the column shape a set-returning function
12690/// exposes. `RETURNS TABLE(id int, v text)` names them; a `SETOF <scalar>`
12691/// yields ONE column named after the call's alias when there is one (`FROM
12692/// odds() AS x` → `x`), else after the function. Get this wrong and the alias
12693/// resolves to the whole ROW: `SELECT x::text FROM odds() AS x` renders `(1)`.
12694fn setof_column_shape_from(
12695 declared: &str,
12696 name: &str,
12697 alias: Option<&str>,
12698 got: &[ColumnSchema],
12699) -> alloc::vec::Vec<ColumnSchema> {
12700 let upper = declared.to_ascii_uppercase();
12701 if upper.starts_with("TABLE(") {
12702 let raw = &declared["TABLE(".len()..declared.len() - 1];
12703 return raw
12704 .split(',')
12705 .zip(got.iter())
12706 .map(|(decl, g)| {
12707 let cname = decl.split_whitespace().next().unwrap_or(g.name.as_str());
12708 ColumnSchema::new(cname.to_string(), g.ty, true)
12709 })
12710 .collect();
12711 }
12712 let cname = alias.unwrap_or(name);
12713 got.first()
12714 .map(|c| alloc::vec![ColumnSchema::new(cname.to_string(), c.ty, true)])
12715 .unwrap_or_default()
12716}
12717
12718/// The plpgsql twin: the interpreter hands back raw value rows, so the types
12719/// come off the first row.
12720fn setof_column_shape(
12721 declared: &str,
12722 name: &str,
12723 alias: Option<&str>,
12724 first_row: Option<&alloc::vec::Vec<Value<'static>>>,
12725) -> alloc::vec::Vec<ColumnSchema> {
12726 let got: alloc::vec::Vec<ColumnSchema> = first_row
12727 .map(|r| {
12728 r.iter()
12729 .enumerate()
12730 .map(|(i, v)| {
12731 ColumnSchema::new(
12732 alloc::format!("col{i}"),
12733 v.data_type().unwrap_or(DataType::Text),
12734 true,
12735 )
12736 })
12737 .collect()
12738 })
12739 .unwrap_or_default();
12740 setof_column_shape_from(declared, name, alias, &got)
12741}
12742
12743/// v7.39 (read01 round 67) — expand every set-returning call in a target list
12744/// for ONE input row, PG's ProjectSet semantics.
12745///
12746/// Several SRFs in one list run in **LOCKSTEP**, not as a cross product: the
12747/// output has as many rows as the LONGEST of them, and a shorter one is padded
12748/// with NULLs. (`SELECT generate_series(1,3), generate_series(10,11)` →
12749/// `1/10, 2/11, 3/NULL`.) A single SRF is the degenerate case of that, and an
12750/// SRF that yields no rows at all contributes none — `SELECT unnest('{}'::int[])`
12751/// is zero rows, not one NULL row.
12752///
12753/// Non-SRF items repeat, evaluated once per output row from the same input row.
12754/// v7.39 (read01 round 79) — where an aggregate may NOT appear. Both of these
12755/// used to reach the scalar function dispatcher, which reported the aggregate as
12756/// an *unknown function* — the same "symptom two layers above the cause" shape
12757/// round 78 found with SRFs. Neither can be diagnosed down there: the dispatcher
12758/// sees a call, not the clause it came from. The statement knows.
12759/// v7.39 (round 294, E3 Phase 1b) — PG's rules on WHERE a row-locking
12760/// clause may appear.
12761///
12762/// PG rejects `FOR UPDATE` on exactly the shapes that have no
12763/// identifiable base row to lock, each with its own wording. SPG
12764/// accepted all of them and locked nothing, so a query that PG refuses
12765/// outright came back looking like it had taken locks.
12766///
12767/// Every wording read off live PG 18.4.
12768fn validate_locking_clause(stmt: &SelectStatement) -> Result<(), EngineError> {
12769 let Some(lock) = &stmt.locking else {
12770 return Ok(());
12771 };
12772 let verb = lock_clause_verb(lock.strength);
12773 let refuse = |what: &str| {
12774 Err(EngineError::Unsupported(alloc::format!(
12775 "{verb} is not allowed with {what}"
12776 )))
12777 };
12778 if !stmt.unions.is_empty() {
12779 return refuse("UNION/INTERSECT/EXCEPT");
12780 }
12781 if stmt.distinct || !stmt.distinct_on.is_empty() {
12782 return refuse("DISTINCT clause");
12783 }
12784 if stmt.group_by.is_some() || stmt.group_by_all {
12785 return refuse("GROUP BY clause");
12786 }
12787 let has_agg = stmt.items.iter().any(|it| match it {
12788 spg_sql::ast::SelectItem::Expr { expr, .. } => crate::aggregate::contains_aggregate(expr),
12789 _ => false,
12790 });
12791 if has_agg {
12792 return refuse("aggregate functions");
12793 }
12794 // `FOR UPDATE OF t` must name a relation that is actually in FROM.
12795 for want in &lock.of_tables {
12796 if !locking_from_names(stmt)
12797 .iter()
12798 .any(|n| n.eq_ignore_ascii_case(want))
12799 {
12800 return Err(EngineError::Unsupported(alloc::format!(
12801 "relation \"{want}\" in {verb} clause not found in FROM clause"
12802 )));
12803 }
12804 }
12805 Ok(())
12806}
12807
12808/// How PG names the clause in its diagnostics.
12809const fn lock_clause_verb(s: spg_sql::ast::LockStrength) -> &'static str {
12810 use spg_sql::ast::LockStrength as LS;
12811 match s {
12812 LS::Update => "FOR UPDATE",
12813 LS::NoKeyUpdate => "FOR NO KEY UPDATE",
12814 LS::Share => "FOR SHARE",
12815 LS::KeyShare => "FOR KEY SHARE",
12816 }
12817}
12818
12819/// Every relation name (or alias) the FROM clause exposes.
12820fn locking_from_names(stmt: &SelectStatement) -> alloc::vec::Vec<String> {
12821 let mut out = alloc::vec::Vec::new();
12822 if let Some(f) = &stmt.from {
12823 let mut push = |t: &spg_sql::ast::TableRef| {
12824 if let Some(a) = &t.alias {
12825 out.push(a.clone());
12826 }
12827 out.push(t.name.clone());
12828 };
12829 push(&f.primary);
12830 for j in &f.joins {
12831 push(&j.table);
12832 }
12833 }
12834 out
12835}
12836
12837fn validate_aggregate_placement(stmt: &SelectStatement) -> Result<(), EngineError> {
12838 use spg_sql::ast::Expr;
12839 if let Some(w) = &stmt.where_
12840 && aggregate::contains_aggregate(w)
12841 {
12842 return Err(EngineError::Unsupported(
12843 "aggregate functions are not allowed in WHERE".into(),
12844 ));
12845 }
12846 let mut nested = false;
12847 let mut check = |e: &Expr| {
12848 let mut probe = e.clone();
12849 crate::expr_analysis::rewrite_nodes_mut(&mut probe, &mut |n| {
12850 let args = match n {
12851 Expr::FunctionCall { name, args } if aggregate::is_aggregate_name(name) => args,
12852 _ => return false,
12853 };
12854 if args.iter().any(aggregate::contains_aggregate) {
12855 nested = true;
12856 }
12857 false
12858 });
12859 };
12860 for it in &stmt.items {
12861 if let spg_sql::ast::SelectItem::Expr { expr, .. } = it {
12862 check(expr);
12863 }
12864 }
12865 if let Some(h) = &stmt.having {
12866 check(h);
12867 }
12868 for o in &stmt.order_by {
12869 check(&o.expr);
12870 }
12871 if nested {
12872 return Err(EngineError::Unsupported(
12873 "aggregate function calls cannot be nested".into(),
12874 ));
12875 }
12876 Ok(())
12877}
12878
12879/// v7.39 (read01 round 78) — an SRF may sit ANYWHERE inside a target-list
12880/// expression, not only as the whole item: `upper(unnest(a))`, `unnest(a) + 10`,
12881/// `'x:' || unnest(a)`, `(regexp_matches(s, p, 'g'))::text`. PG evaluates the SRF
12882/// to a set and then applies the enclosing expression once per element. SPG only
12883/// ever recognised an SRF that WAS the item, so everything above died on
12884/// "unknown function unnest" — the set-returning call, wrapped in anything at
12885/// all, fell through to the scalar function dispatcher which has no such name.
12886///
12887/// Each SRF node is lifted out into a synthetic column (`__srf_k`), the tree is
12888/// rewritten to read that column, and the rewritten expression is evaluated once
12889/// per output row against the input row extended with the lifted values. The
12890/// lift is by VALUE, not by literal: a text[] or a jsonb keeps its type exactly.
12891/// v7.39 (read01 round 80) — `ORDER BY <n>` names the Nth OUTPUT column. Three
12892/// executors (the single-table scan, the synthetic-table pipeline, and the
12893/// unnest FROM path) each evaluated the key as an ordinary expression, where the
12894/// literal `n` is just the constant n — the same sort key for every row. The
12895/// sort therefore ran and changed nothing, which is why nobody noticed: rows came
12896/// back in input order, not in a wrong order. Statement prep resolves the common
12897/// case, but only when the SELECT item is an expression — a `*` is not one, and
12898/// `SELECT unnest(a) x` becomes `SELECT * FROM unnest(a) x`, so the everyday
12899/// spelling landed on exactly the shape prep could not resolve.
12900///
12901/// A set-returning item is left alone: copying it into ORDER BY would make the
12902/// key "the whole set", evaluated once per INPUT row.
12903fn resolve_positional_order_by(
12904 order_by: &[spg_sql::ast::OrderBy],
12905 projection: &[ProjectedItem],
12906) -> alloc::vec::Vec<spg_sql::ast::OrderBy> {
12907 order_by
12908 .iter()
12909 .filter_map(|o| {
12910 let mut o = o.clone();
12911 if let Expr::Literal(spg_sql::ast::Literal::Integer(n)) = &o.expr
12912 && *n >= 1
12913 && let Ok(idx) = usize::try_from(*n - 1)
12914 && let Some(item) = projection.get(idx)
12915 && !expr_contains_builtin_srf(&item.expr)
12916 {
12917 // 7.38.1 S6.1 (gendiff fourth leg) — an ordinal whose
12918 // item is itself an integer LITERAL must not be
12919 // substituted textually: the literal would read as an
12920 // ordinal again downstream, and `SELECT 10 … ORDER BY
12921 // 1` died with "position 10 is not in select list"
12922 // where PG happily returns the rows. Ordering by a
12923 // constant orders nothing, so the key drops.
12924 if matches!(item.expr, Expr::Literal(spg_sql::ast::Literal::Integer(_))) {
12925 return None;
12926 }
12927 o.expr = item.expr.clone();
12928 }
12929 Some(o)
12930 })
12931 .collect()
12932}
12933
12934/// v7.39 (read01 round 80) — does a BUILTIN set-returning call appear anywhere in
12935/// this expression? Statement preparation (`resolve_order_by_position`) runs
12936/// before any catalog is in hand, and it only needs to know "is this item's value
12937/// a set", which the builtin SRFs answer syntactically.
12938pub(crate) fn expr_contains_builtin_srf(e: &spg_sql::ast::Expr) -> bool {
12939 let mut found = false;
12940 let mut probe = e.clone();
12941 crate::expr_analysis::rewrite_nodes_mut(&mut probe, &mut |n| {
12942 if is_top_level_unnest(n) {
12943 found = true;
12944 return true;
12945 }
12946 false
12947 });
12948 found
12949}
12950
12951/// v7.39 (round 599) — everything about a target-list SRF that does not
12952/// depend on the row.
12953///
12954/// `expand_srf_row` derived all of this again for EVERY input row: it cloned
12955/// each SRF-bearing projection expression, walked and rewrote the tree,
12956/// formatted a `__srf_N` name per node, and copied the whole column schema.
12957/// A counting allocator put the path at 24 allocations per input row for a
12958/// single-element `unnest`, against 0 for the same scan without one — 211 MB
12959/// where the plain scan took 4.3 — and the shape held whatever the array
12960/// contained, which is what invariant work looks like.
12961struct SrfPlan {
12962 /// The lifted SRF calls, in slot order.
12963 nodes: alloc::vec::Vec<spg_sql::ast::Expr>,
12964 /// Per projection position, the expression with its SRF calls replaced
12965 /// by `__srf_N` column references. `None` means the item has none.
12966 rewritten: alloc::vec::Vec<Option<spg_sql::ast::Expr>>,
12967 /// The input schema followed by one column per slot. Only the slots'
12968 /// TYPES vary per row, and they are patched in place.
12969 ext_cols: alloc::vec::Vec<ColumnSchema>,
12970 /// v7.39 (round 743) — the rewritten projection COMPILED against the
12971 /// extended schema, once per plan. The per-output-row evaluation ran
12972 /// the interpreter (~560 ns/row on the unnest panel cell); the Step
12973 /// VM reads the `__srf_N` slots as plain columns. `None` = that item
12974 /// is not fully compilable and keeps the interpreter.
12975 compiled: alloc::vec::Vec<Option<eval::CompiledExpr>>,
12976 base_cols: usize,
12977}
12978
12979fn build_srf_plan(
12980 engine: &Engine,
12981 projection: &[ProjectedItem],
12982 srf_idxs: &[usize],
12983 ctx: &EvalContext<'_>,
12984) -> Result<SrfPlan, EngineError> {
12985 // Lift every SRF node out of every item that contains one.
12986 let mut nodes: Vec<spg_sql::ast::Expr> = Vec::new();
12987 let mut rewritten: Vec<Option<spg_sql::ast::Expr>> = alloc::vec![None; projection.len()];
12988 let mut reject: Option<EngineError> = None;
12989 for &i in srf_idxs {
12990 let mut e = projection[i].expr.clone();
12991 crate::expr_analysis::rewrite_nodes_mut(&mut e, &mut |n| {
12992 if reject.is_some() {
12993 return true;
12994 }
12995 // PG refuses a set-returning function inside a conditional: the set
12996 // would have to be produced before anyone knows whether the branch
12997 // is even taken.
12998 let conditional = match n {
12999 spg_sql::ast::Expr::Case { .. } => Some("CASE"),
13000 spg_sql::ast::Expr::FunctionCall { name, .. }
13001 if name.eq_ignore_ascii_case("coalesce") =>
13002 {
13003 Some("COALESCE")
13004 }
13005 _ => None,
13006 };
13007 if let Some(kind) = conditional
13008 && engine.expr_contains_srf(n)
13009 {
13010 reject = Some(EngineError::Unsupported(alloc::format!(
13011 "set-returning functions are not allowed in {kind}"
13012 )));
13013 return true;
13014 }
13015 if !engine.is_srf_node(n) {
13016 return false;
13017 }
13018 let slot = nodes.len();
13019 nodes.push(n.clone());
13020 *n = spg_sql::ast::Expr::Column(spg_sql::ast::ColumnName {
13021 qualifier: None,
13022 name: alloc::format!("__srf_{slot}"),
13023 });
13024 true
13025 });
13026 rewritten[i] = Some(e);
13027 }
13028 if let Some(err) = reject {
13029 return Err(err);
13030 }
13031 let base_cols = ctx.columns.len();
13032 let mut ext_cols: Vec<ColumnSchema> = ctx.columns.to_vec();
13033 for slot in 0..nodes.len() {
13034 ext_cols.push(ColumnSchema::new(
13035 alloc::format!("__srf_{slot}"),
13036 DataType::Text,
13037 true,
13038 ));
13039 }
13040 // v7.39 (round 743) — compile the rewritten items against the
13041 // EXTENDED schema. The slot columns' declared type is a per-row
13042 // patched detail the compiled column read does not consult.
13043 let compiled: Vec<Option<eval::CompiledExpr>> = {
13044 let mut ext_ctx = ctx.clone();
13045 ext_ctx.columns = &ext_cols;
13046 projection
13047 .iter()
13048 .enumerate()
13049 .map(|(i, p)| {
13050 let e = rewritten[i].as_ref().unwrap_or(&p.expr);
13051 if eval::fully_compilable(e) {
13052 Some(eval::compile_expr(e, &ext_ctx))
13053 } else {
13054 None
13055 }
13056 })
13057 .collect()
13058 };
13059 Ok(SrfPlan {
13060 nodes,
13061 rewritten,
13062 ext_cols,
13063 compiled,
13064 base_cols,
13065 })
13066}
13067
13068/// One input row expanded through a plan built once for the whole scan.
13069/// v7.39 (round 621) — expand a projection whose target list contains
13070/// set-returning items, remembering which INPUT row each output row came from.
13071///
13072/// The three materialised-source tails — `FROM unnest(…)`, `FROM
13073/// generate_series(…)`, and the one that serves VALUES / a derived table /
13074/// `ROWS FROM (…)` — are near-copies of each other, and only the first knew
13075/// about target-list SRFs. So `SELECT unnest(ARRAY[1,2]), x FROM (VALUES (3),(4))
13076/// v(x)` answered `function unnest(integer[]) does not exist` on all the
13077/// others, for a query PG answers. Sharing the expansion is the point: a
13078/// fourth copy would have been the fourth place to forget.
13079fn expand_projection_srfs(
13080 engine: &Engine,
13081 projection: &[ProjectedItem],
13082 srf_idxs: &[usize],
13083 filtered: &[Row<'static>],
13084 ctx: &EvalContext<'_>,
13085) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<usize>), EngineError> {
13086 let mut out = alloc::vec::Vec::with_capacity(filtered.len());
13087 let mut src = alloc::vec::Vec::with_capacity(filtered.len());
13088 // v7.39 (round 726) — ONE plan for the whole scan. The per-row
13089 // spelling rebuilt it for every input row: a full clone of the
13090 // rewritten projection trees and the extended schema, 50k times on
13091 // the panel's unnest cell.
13092 let mut plan = build_srf_plan(engine, projection, srf_idxs, ctx)?;
13093 // v7.39 (round 733) — shard the expansion. Each shard clones the
13094 // plan (its ext_cols slot types are per-row mutable) and builds a
13095 // MINIMAL context — EvalContext is not Sync — which is sound only
13096 // when every expression involved is pure: the whole projection and
13097 // every SRF argument must be fully_compilable, or the row loop
13098 // stays serial with the full session context.
13099 // The projection is judged in its REWRITTEN form — the SRF call
13100 // itself is never compilable, but after the lift it is a plain
13101 // `__srf_N` column reference.
13102 let all_pure = projection
13103 .iter()
13104 .enumerate()
13105 .all(|(i, p)| eval::fully_compilable(plan.rewritten[i].as_ref().unwrap_or(&p.expr)))
13106 && plan.nodes.iter().all(|n| match n {
13107 Expr::FunctionCall { args, .. } => args.iter().all(eval::fully_compilable),
13108 other => eval::fully_compilable(other),
13109 });
13110 if all_pure
13111 && filtered.len() >= crate::PARALLEL_MIN_ROWS / 5
13112 && let Some(r) = engine.parallel_runner.0.as_deref()
13113 {
13114 let n_shards = (filtered.len() / (crate::PARALLEL_MIN_ROWS / 5)).clamp(2, 8);
13115 let chunk = filtered.len().div_ceil(n_shards);
13116 type ShardOut = Result<(Vec<Row<'static>>, Vec<usize>), EngineError>;
13117 let schema_cols = ctx.columns;
13118 let alias = ctx.table_alias;
13119 let mysql = ctx.mysql_dialect;
13120 let style = ctx.render_style;
13121 let plan_ref = &plan;
13122 let results = r.run_shards(n_shards, &|si| {
13123 let lo = si * chunk;
13124 let hi = ((si + 1) * chunk).min(filtered.len());
13125 let mut sctx = eval::EvalContext::new(schema_cols, alias);
13126 sctx.mysql_dialect = mysql;
13127 sctx.render_style = style;
13128 // v7.39 (round 743) — SrfPlan is no longer Clone (it carries
13129 // compiled programs); each shard rebuilds it, which also
13130 // recompiles against the shard's own context. Build errors
13131 // were already surfaced by the outer build above.
13132 let mut local_plan = match build_srf_plan(engine, projection, srf_idxs, &sctx) {
13133 Ok(p) => p,
13134 Err(e) => return alloc::boxed::Box::new(ShardOut::Err(e)) as _,
13135 };
13136 let mut run = || -> ShardOut {
13137 let mut o: Vec<Row<'static>> = Vec::with_capacity(hi - lo);
13138 let mut sidx: Vec<usize> = Vec::with_capacity(hi - lo);
13139 for (i, row) in filtered[lo..hi].iter().enumerate() {
13140 let expanded =
13141 expand_srf_row_with(engine, &mut local_plan, projection, row, &sctx)?;
13142 sidx.extend(core::iter::repeat_n(lo + i, expanded.len()));
13143 o.extend(expanded);
13144 }
13145 Ok((o, sidx))
13146 };
13147 alloc::boxed::Box::new(run())
13148 });
13149 for boxed in results {
13150 let shard = boxed
13151 .downcast::<ShardOut>()
13152 .expect("runner echoes the closure's box");
13153 let (o, sidx) = (*shard)?;
13154 out.extend(o);
13155 src.extend(sidx);
13156 }
13157 return Ok((out, src));
13158 }
13159 for (i, row) in filtered.iter().enumerate() {
13160 let expanded = expand_srf_row_with(engine, &mut plan, projection, row, ctx)?;
13161 src.extend(core::iter::repeat_n(i, expanded.len()));
13162 out.extend(expanded);
13163 }
13164 Ok((out, src))
13165}
13166
13167/// v7.39 (round 621) — one ORDER BY key, read from wherever it lives.
13168///
13169/// A key that names a select-list item reads it out of the EXPANDED row,
13170/// because PG sorts after the expansion. A key that names a source column the
13171/// query does not project is evaluated against the input row that output row
13172/// came from. `out_col` is `srf_order_output_cols`'s verdict for this key.
13173fn srf_order_key(
13174 ob: &spg_sql::ast::OrderBy,
13175 out_col: Option<usize>,
13176 out: &Row<'static>,
13177 src: &Row<'static>,
13178 ctx: &EvalContext<'_>,
13179) -> Result<Value<'static>, EngineError> {
13180 match out_col {
13181 Some(i) => Ok(out.values.get(i).cloned().unwrap_or(Value::Null)),
13182 None => eval::eval_expr(&ob.expr, src, ctx).map_err(EngineError::Eval),
13183 }
13184}
13185
13186fn expand_srf_row_with(
13187 engine: &Engine,
13188 plan: &mut SrfPlan,
13189 projection: &[ProjectedItem],
13190 row: &Row<'static>,
13191 ctx: &EvalContext<'_>,
13192) -> Result<Vec<Row<'static>>, EngineError> {
13193 let mut lists: Vec<Vec<Value<'static>>> = Vec::with_capacity(plan.nodes.len());
13194 for n in &plan.nodes {
13195 lists.push(engine.srf_values(n, row, ctx)?);
13196 }
13197 let n_rows = lists.iter().map(Vec::len).max().unwrap_or(0);
13198 // Only the slots' element types depend on the row; the names and the
13199 // input schema around them do not.
13200 for (slot, list) in lists.iter().enumerate() {
13201 plan.ext_cols[plan.base_cols + slot].ty = list
13202 .iter()
13203 .find_map(|v| v.data_type())
13204 .unwrap_or(DataType::Text);
13205 }
13206 let mut ext_ctx = ctx.clone();
13207 ext_ctx.columns = &plan.ext_cols;
13208 let mut out = Vec::with_capacity(n_rows);
13209 // v7.39 (round 726) — the base columns are the SAME for every
13210 // expanded row; clone them once and rewrite only the SRF slots per
13211 // k. The old form cloned the whole input row per OUTPUT row — for
13212 // `unnest(ARRAY[id, g])` over d that was a 100k-fold clone of a
13213 // TEXT column the projection never reads.
13214 let base_len = row.values.len();
13215 let mut ext_vals = row.values.clone();
13216 ext_vals.resize(base_len + lists.len(), Value::Null);
13217 let mut eval_stack: alloc::vec::Vec<Value<'static>> = alloc::vec::Vec::new();
13218 for k in 0..n_rows {
13219 for (slot, list) in lists.iter().enumerate() {
13220 // Past the end of THIS srf's rows → NULL (PG pads).
13221 ext_vals[base_len + slot] = list.get(k).cloned().unwrap_or(Value::Null);
13222 }
13223 let ext_row = Row::new(core::mem::take(&mut ext_vals));
13224 let mut vals = Vec::with_capacity(projection.len());
13225 for (i, p) in projection.iter().enumerate() {
13226 // v7.39 (round 743) — compiled when possible; the
13227 // interpreter for the rest, with its exact wording.
13228 vals.push(match &plan.compiled[i] {
13229 Some(c) => eval::eval_compiled(c, &ext_row, &ext_ctx, &mut eval_stack)
13230 .map_err(EngineError::Eval)?,
13231 None => {
13232 let expr = plan.rewritten[i].as_ref().unwrap_or(&p.expr);
13233 eval::eval_expr(expr, &ext_row, &ext_ctx).map_err(EngineError::Eval)?
13234 }
13235 });
13236 }
13237 ext_vals = ext_row.values;
13238 out.push(Row::new(vals));
13239 }
13240 Ok(out)
13241}
13242
13243/// The one-shot spelling, for the callers that expand a single row.
13244/// v7.39 (round 600) — which output column each ORDER BY key names, for a
13245/// query whose target list contains a set-returning function.
13246///
13247/// The keys used to be built from the INPUT row, before the SRF expanded, so
13248/// anything that named the SRF's own output was evaluated as a scalar call:
13249/// `SELECT unnest(ARRAY[g,id]) v FROM sr ORDER BY v` answered
13250/// "function unnest(integer[]) does not exist", and so did the spellings that
13251/// repeat the call or reach it through `ORDER BY 1`. Where it did not error
13252/// it silently did nothing — `SELECT DISTINCT unnest(…) … ORDER BY 1` came
13253/// back in input order. PG sorts AFTER the expansion, so a key that names a
13254/// select-list item reads that item's value out of the expanded row.
13255///
13256/// `None` keeps the key on the input row, which is where an ORDER BY naming
13257/// a column the query does not project has to be evaluated.
13258fn srf_order_output_cols(
13259 order_by: &[spg_sql::ast::OrderBy],
13260 projection: &[ProjectedItem],
13261) -> Vec<Option<usize>> {
13262 order_by
13263 .iter()
13264 .map(|ob| {
13265 // A positive ordinal is the Nth output column, directly.
13266 // `resolve_positional_order_by` deliberately leaves an ordinal
13267 // pointing at a set-returning item alone — copying the call into
13268 // ORDER BY would have made the key "the whole set" back when keys
13269 // came from the input row. Reading the expanded row's column is
13270 // what it should have meant, and is what this does.
13271 if let Expr::Literal(spg_sql::ast::Literal::Integer(n)) = &ob.expr
13272 && *n >= 1
13273 && let Ok(idx) = usize::try_from(*n - 1)
13274 && idx < projection.len()
13275 {
13276 return Some(idx);
13277 }
13278 // An unqualified name matching exactly one output name. SQL
13279 // resolves ORDER BY against the select list first, so this wins
13280 // over an input column of the same name — which is the whole
13281 // point of `SELECT g AS id … ORDER BY id`.
13282 if let Expr::Column(c) = &ob.expr
13283 && c.qualifier.is_none()
13284 {
13285 let mut hit = None;
13286 for (i, p) in projection.iter().enumerate() {
13287 if p.output_name.eq_ignore_ascii_case(&c.name) {
13288 if hit.is_some() {
13289 hit = None;
13290 break;
13291 }
13292 hit = Some(i);
13293 }
13294 }
13295 if hit.is_some() {
13296 return hit;
13297 }
13298 }
13299 // Or the same expression as a select-list item — which is what
13300 // `ORDER BY 1` becomes once `resolve_positional_order_by` has
13301 // run, and what a repeated `ORDER BY unnest(…)` is.
13302 projection.iter().position(|p| p.expr == ob.expr)
13303 })
13304 .collect()
13305}
13306
13307fn expand_srf_row(
13308 engine: &Engine,
13309 projection: &[ProjectedItem],
13310 srf_idxs: &[usize],
13311 row: &Row<'static>,
13312 ctx: &EvalContext<'_>,
13313) -> Result<Vec<Row<'static>>, EngineError> {
13314 let mut plan = build_srf_plan(engine, projection, srf_idxs, ctx)?;
13315 expand_srf_row_with(engine, &mut plan, projection, row, ctx)
13316}
13317
13318impl Engine {
13319 /// The rows one target-list SRF yields for an input row. `None` from
13320 /// `srf_target_idxs` means the expression is not set-returning at all.
13321 fn srf_values(
13322 &self,
13323 expr: &spg_sql::ast::Expr,
13324 row: &Row<'static>,
13325 ctx: &EvalContext<'_>,
13326 ) -> Result<Vec<Value<'static>>, EngineError> {
13327 if top_level_srf_kind(expr).is_some() {
13328 return top_level_srf_output(expr, row, ctx);
13329 }
13330 // A user set-returning function. Its body runs through the real
13331 // executor, like every function body since round 63.
13332 let spg_sql::ast::Expr::FunctionCall { name, args } = expr else {
13333 return Err(EngineError::Unsupported(
13334 "expected a SELECT-list SRF call".into(),
13335 ));
13336 };
13337 let mut vals: alloc::vec::Vec<Value<'static>> = alloc::vec::Vec::new();
13338 for a in args {
13339 vals.push(eval::eval_expr(a, row, ctx).map_err(EngineError::Eval)?);
13340 }
13341 let (rows, cols) = self.setof_rows_of(name, &vals, None)?;
13342 // v7.39 (read01 round 68) — in a target list a multi-column function is
13343 // a RECORD, one composite value per row: `SELECT rows_of(2)` gives
13344 // `(2,b)`, `(3,c)`. Value::Composite has existed since round 56; this is
13345 // what it is for. A single-column function contributes its bare value.
13346 Ok(rows
13347 .into_iter()
13348 .map(|r| {
13349 if r.values.len() == 1 {
13350 r.values.into_iter().next().unwrap_or(Value::Null)
13351 } else {
13352 Value::Composite(
13353 cols.iter()
13354 .map(|c| c.name.clone())
13355 .zip(r.values)
13356 .collect::<alloc::vec::Vec<_>>(),
13357 )
13358 }
13359 })
13360 .collect())
13361 }
13362
13363 /// Is THIS node a set-returning call: one of the builtin kinds, or a user
13364 /// function declared `RETURNS SETOF` / `RETURNS TABLE`.
13365 fn is_srf_node(&self, e: &spg_sql::ast::Expr) -> bool {
13366 if is_top_level_unnest(e) {
13367 return true;
13368 }
13369 let spg_sql::ast::Expr::FunctionCall { name, .. } = e else {
13370 return false;
13371 };
13372 self.active_catalog().functions_named(name).iter().any(|f| {
13373 let r = f.returns.trim().to_ascii_uppercase();
13374 r.starts_with("SETOF") || r.starts_with("TABLE(")
13375 })
13376 }
13377
13378 /// Does an SRF appear ANYWHERE in this expression (not only as its root)?
13379 fn expr_contains_srf(&self, e: &spg_sql::ast::Expr) -> bool {
13380 let mut found = false;
13381 let mut probe = e.clone();
13382 crate::expr_analysis::rewrite_nodes_mut(&mut probe, &mut |n| {
13383 if self.is_srf_node(n) {
13384 found = true;
13385 return true;
13386 }
13387 false
13388 });
13389 found
13390 }
13391
13392 /// Which projection items CONTAIN a set-returning call. Before round 78 this
13393 /// asked whether the item WAS one, so `upper(unnest(a))` looked like an
13394 /// ordinary scalar call all the way down to the function dispatcher, which
13395 /// then reported `unnest` as an unknown function.
13396 fn srf_target_idxs(&self, projection: &[ProjectedItem]) -> alloc::vec::Vec<usize> {
13397 projection
13398 .iter()
13399 .enumerate()
13400 .filter(|(_, p)| self.expr_contains_srf(&p.expr))
13401 .map(|(i, _)| i)
13402 .collect()
13403 }
13404}
13405
13406impl Engine {
13407 /// v7.39 (read01 round 74) — see the call site. `None` when the statement has
13408 /// no `(f(args)).*` item.
13409 fn lower_record_expansion(
13410 &self,
13411 stmt: &SelectStatement,
13412 ) -> Result<Option<SelectStatement>, EngineError> {
13413 use spg_sql::ast::{Expr, SelectItem};
13414 let is_marker = |it: &SelectItem| {
13415 matches!(it, SelectItem::Expr { expr: Expr::FunctionCall { name, .. }, .. }
13416 if name == "__record_expand")
13417 };
13418 if !stmt.items.iter().any(is_marker) {
13419 return Ok(None);
13420 }
13421 let mut out = stmt.clone();
13422 let mut items: alloc::vec::Vec<SelectItem> = alloc::vec::Vec::new();
13423 let mut lateral_refs: alloc::vec::Vec<TableRef> = alloc::vec::Vec::new();
13424 for (n, item) in stmt.items.iter().enumerate() {
13425 if !is_marker(item) {
13426 items.push(item.clone());
13427 continue;
13428 }
13429 let SelectItem::Expr {
13430 expr: Expr::FunctionCall { args, .. },
13431 ..
13432 } = item
13433 else {
13434 unreachable!("checked by is_marker");
13435 };
13436 let Some(Expr::FunctionCall {
13437 name: fname,
13438 args: fargs,
13439 }) = args.first()
13440 else {
13441 return Err(EngineError::Unsupported(
13442 "(<expr>).* expands a function's record — it needs a function call".into(),
13443 ));
13444 };
13445 let cols = self.setof_declared_columns(fname)?;
13446 let alias = alloc::format!("__rec{n}");
13447 let mut tref = bare_table_ref_named(&alias);
13448 tref.table_fn_call = Some(alloc::boxed::Box::new((
13449 fname.to_ascii_lowercase(),
13450 fargs.clone(),
13451 )));
13452 tref.alias = Some(alias.clone());
13453 lateral_refs.push(tref);
13454 for c in cols {
13455 items.push(SelectItem::Expr {
13456 expr: Expr::Column(spg_sql::ast::ColumnName {
13457 qualifier: Some(alias.clone()),
13458 name: c,
13459 }),
13460 alias: None,
13461 });
13462 }
13463 }
13464 out.items = items;
13465 // The function joins the FROM. With no FROM it BECOMES the FROM; with one
13466 // it is a cross join, which is what `SELECT …, (f(t.c)).* FROM t` means
13467 // (the arguments may reference the outer row — the round-69 correlation).
13468 for tref in lateral_refs {
13469 match &mut out.from {
13470 None => {
13471 out.from = Some(spg_sql::ast::FromClause {
13472 primary: tref,
13473 joins: alloc::vec::Vec::new(),
13474 });
13475 }
13476 Some(from) => from.joins.push(spg_sql::ast::FromJoin {
13477 kind: spg_sql::ast::JoinKind::Cross,
13478 table: tref,
13479 on: None,
13480 using_cols: None,
13481 natural: false,
13482 }),
13483 }
13484 }
13485 Ok(Some(out))
13486 }
13487
13488 /// The column NAMES a set-returning function declares: `RETURNS TABLE(id int,
13489 /// v text)` names them; a `SETOF <scalar>` is one column named after the
13490 /// function.
13491 fn setof_declared_columns(
13492 &self,
13493 name: &str,
13494 ) -> Result<alloc::vec::Vec<alloc::string::String>, EngineError> {
13495 let cat = self.active_catalog();
13496 let overloads = cat.functions_named(name);
13497 let def = overloads.first().ok_or_else(|| {
13498 EngineError::Unsupported(alloc::format!("function {name} does not exist"))
13499 })?;
13500 let declared = def.returns.trim();
13501 let upper = declared.to_ascii_uppercase();
13502 if upper.starts_with("TABLE(") {
13503 let raw = &declared["TABLE(".len()..declared.len() - 1];
13504 return Ok(raw
13505 .split(',')
13506 .map(|d| d.split_whitespace().next().unwrap_or("col").to_string())
13507 .collect());
13508 }
13509 Ok(alloc::vec![name.to_string()])
13510 }
13511}
13512
13513/// A bare `TableRef` with a name — the FROM item a lowered record expansion adds.
13514/// v7.39 (round 205, JSON_TABLE) — the static output schema of a
13515/// COLUMNS list (data-independent), NESTED children inlined in
13516/// declaration order (PG's flattened output shape).
13517/// v7.39 (round 205) — pub(crate) shim so join.rs infers a wrapped
13518/// correlated JSON_TABLE's static schema without evaluating its doc.
13519pub(crate) fn json_table_schema_pub(
13520 cols: &[spg_sql::ast::JsonTableColumn],
13521) -> alloc::vec::Vec<ColumnSchema> {
13522 json_table_schema(cols)
13523}
13524
13525fn json_table_schema(cols: &[spg_sql::ast::JsonTableColumn]) -> alloc::vec::Vec<ColumnSchema> {
13526 use spg_sql::ast::JsonTableColumn as C;
13527 let mut out = alloc::vec::Vec::new();
13528 for c in cols {
13529 match c {
13530 C::Ordinality { name } => {
13531 out.push(ColumnSchema::new(name.clone(), DataType::BigInt, false));
13532 }
13533 C::Regular {
13534 name, ty, exists, ..
13535 } => {
13536 let dt = if *exists {
13537 DataType::Bool
13538 } else {
13539 crate::conversions::column_type_to_data_type(*ty)
13540 };
13541 out.push(ColumnSchema::new(name.clone(), dt, true));
13542 }
13543 C::Nested { columns, .. } => out.extend(json_table_schema(columns)),
13544 }
13545 }
13546 out
13547}
13548
13549/// v7.39 (round 205) — coerce a DEFAULT / literal value to a
13550/// JSON_TABLE column's declared type (the DEFAULT expr may be a
13551/// string literal like `'none'` that must land as the column type).
13552fn coerce_json_table_default(
13553 v: Value<'static>,
13554 ty: spg_sql::ast::ColumnTypeName,
13555 name: &str,
13556) -> Result<Value<'static>, EngineError> {
13557 if v.is_null() {
13558 return Ok(Value::Null);
13559 }
13560 let dt = crate::conversions::column_type_to_data_type(ty);
13561 crate::conversions::coerce_value(v, dt, name, 0)
13562}
13563
13564/// v7.39 (round 205) — a runtime Value → JsonValue for PASSING vars.
13565fn value_to_json_value(v: &Value<'_>) -> crate::json::JsonValue {
13566 use crate::json::JsonValue as J;
13567 match v {
13568 Value::Null => J::Null,
13569 Value::Bool(b) => J::Bool(*b),
13570 Value::SmallInt(n) => J::Number(f64::from(*n)),
13571 Value::Int(n) => J::Number(f64::from(*n)),
13572 Value::BigInt(n) => J::Number(*n as f64),
13573 Value::Float(x) => J::Number(*x),
13574 Value::Json(s) => crate::json::parse_doc(s).unwrap_or(J::Null),
13575 other => J::String(crate::eval::value_to_text(other)),
13576 }
13577}
13578
13579fn bare_table_ref_named(name: &str) -> TableRef {
13580 TableRef {
13581 name: name.to_string(),
13582 alias: None,
13583 only: false,
13584 as_of_segment: None,
13585 unnest_expr: None,
13586 unnest_column_aliases: alloc::vec::Vec::new(),
13587 with_ordinality: false,
13588 generate_series_args: None,
13589 lateral_subquery: None,
13590 jsonb_each_text_arg: None,
13591 table_fn_call: None,
13592 rows_from: None,
13593 json_table: None,
13594 scalar_fn_item: false,
13595 }
13596}
13597
13598impl Engine {
13599 /// v7.39 (read01 round 74) — run a `ROWS FROM (…)` list. Each entry yields its
13600 /// own rows; they zip in lockstep and a short one pads with NULL. `__array`
13601 /// entries are the array-able SRFs, already lowered by the parser into their
13602 /// scalar array form.
13603 fn rows_from_rows(
13604 &self,
13605 primary: &TableRef,
13606 ) -> Result<(alloc::vec::Vec<Row<'static>>, alloc::vec::Vec<ColumnSchema>), EngineError> {
13607 let entries = primary
13608 .rows_from
13609 .as_ref()
13610 .expect("caller guards rows_from.is_some()");
13611 let empty: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
13612 let ctx = self.ev_ctx(&empty, None);
13613 let dummy = Row::new(alloc::vec::Vec::new());
13614 let mut lists: alloc::vec::Vec<alloc::vec::Vec<Value<'static>>> = alloc::vec::Vec::new();
13615 let mut cols: alloc::vec::Vec<ColumnSchema> = alloc::vec::Vec::new();
13616 for (name, args) in entries {
13617 let (vals, colname) = if name == "__array" {
13618 // The parser lowered this one to `<array expr>`; its rows are the
13619 // array's elements.
13620 let arr = eval::eval_expr(&args[0], &dummy, &ctx).map_err(EngineError::Eval)?;
13621 (
13622 array_value_to_elements(&arr)?,
13623 alloc::string::String::from("unnest"),
13624 )
13625 } else {
13626 let call = spg_sql::ast::Expr::FunctionCall {
13627 name: name.clone(),
13628 args: args.clone(),
13629 };
13630 (self.srf_values(&call, &dummy, &ctx)?, name.clone())
13631 };
13632 let ty = vals
13633 .first()
13634 .and_then(spg_storage::Value::data_type)
13635 .unwrap_or(DataType::Text);
13636 cols.push(ColumnSchema::new(colname, ty, true));
13637 lists.push(vals);
13638 }
13639 let n = lists.iter().map(alloc::vec::Vec::len).max().unwrap_or(0);
13640 let mut rows: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::with_capacity(n);
13641 for k in 0..n {
13642 let mut vals: alloc::vec::Vec<Value<'static>> =
13643 alloc::vec::Vec::with_capacity(lists.len() + 1);
13644 for l in &lists {
13645 vals.push(l.get(k).cloned().unwrap_or(Value::Null));
13646 }
13647 rows.push(Row::new(vals));
13648 }
13649 if primary.with_ordinality {
13650 cols.push(ColumnSchema::new(
13651 "ordinality".to_string(),
13652 DataType::BigInt,
13653 false,
13654 ));
13655 rows = rows
13656 .into_iter()
13657 .enumerate()
13658 .map(|(i, r)| {
13659 let mut v = r.values;
13660 v.push(Value::BigInt(i as i64 + 1));
13661 Row::new(v)
13662 })
13663 .collect();
13664 }
13665 Ok((rows, cols))
13666 }
13667}
13668
13669/// v7.39 (round 232) — PG names the offending set operation in its
13670/// arity / type-mismatch messages ("each UNION query must have the same
13671/// number of columns"). `UNION ALL` is still spelled UNION there.
13672fn set_op_name(kind: UnionKind) -> &'static str {
13673 match kind {
13674 UnionKind::All | UnionKind::Distinct => "UNION",
13675 UnionKind::Intersect | UnionKind::IntersectAll => "INTERSECT",
13676 UnionKind::Except | UnionKind::ExceptAll => "EXCEPT",
13677 }
13678}
13679
13680/// v7.39 (round 233) — which output columns of a branch are PG's `unknown`
13681/// type: a bare string or NULL literal that no context has typed yet. SPG
13682/// has no `Unknown` DataType (both describe as TEXT), so the witness has to
13683/// be the syntax. A wildcard or a non-literal expression is never unknown.
13684/// 7.38.1 S5.1 — is this branch item a reg* cast? Its result column
13685/// LABELS as text (the wire render) but the value is an oid-carrying
13686/// dual, so a UNION with a numeric column must not be refused on the
13687/// label (pg_dump: `SELECT classid … UNION ALL SELECT
13688/// 'pg_opfamily'::regclass …`).
13689fn branch_regcast_mask(stmt: &SelectStatement) -> Vec<bool> {
13690 fn is_regcast(e: &Expr) -> bool {
13691 matches!(
13692 e,
13693 Expr::Cast {
13694 target: spg_sql::ast::CastTarget::RegType | spg_sql::ast::CastTarget::RegClass,
13695 ..
13696 }
13697 )
13698 }
13699 stmt.items
13700 .iter()
13701 .map(|item| match item {
13702 SelectItem::Expr { expr, .. } => is_regcast(expr),
13703 _ => false,
13704 })
13705 .collect()
13706}
13707
13708fn branch_unknown_mask(stmt: &SelectStatement) -> Vec<bool> {
13709 stmt.items
13710 .iter()
13711 .map(|item| match item {
13712 SelectItem::Expr { expr, .. } => matches!(
13713 expr,
13714 Expr::Literal(spg_sql::ast::Literal::String(_))
13715 | Expr::Literal(spg_sql::ast::Literal::Null)
13716 ),
13717 _ => false,
13718 })
13719 .collect()
13720}
13721
13722/// v7.39 (round 233) — retype one branch column's cells, reporting the
13723/// conversion failure the way PG does rather than leaving the column
13724/// half-converted. Used when the other branch typed an untyped literal.
13725fn coerce_branch_column(
13726 rows: &mut [Row<'static>],
13727 col_idx: usize,
13728 target: DataType,
13729 col_name: &str,
13730) -> Result<(), EngineError> {
13731 for row in rows.iter_mut() {
13732 let Some(slot) = row.values.get_mut(col_idx) else {
13733 continue;
13734 };
13735 if matches!(slot, Value::Null) {
13736 continue;
13737 }
13738 *slot = crate::conversions::coerce_value(slot.clone(), target, col_name, col_idx)?;
13739 }
13740 Ok(())
13741}
13742
13743/// v7.39 (round 727) — PG-style pull-up of a SIMPLE derived table:
13744/// `SELECT … FROM (SELECT <bare columns> FROM t [WHERE …]) q …`
13745/// rewrites to `SELECT …' FROM t [WHERE inner AND outer'] …` with every
13746/// reference to q's output columns substituted by the underlying column.
13747///
13748/// Admission is deliberately narrow — anything that changes cardinality,
13749/// order, or scope stays on the materialising path:
13750/// * outer: no CTEs / unions / DISTINCT [ON] / windows, single derived
13751/// FROM with no ordinality or positional column aliases, and no
13752/// subquery anywhere its expressions (an inner scope could reference
13753/// q too — descending is a later knife);
13754/// * inner: one stored table, bare-column projection only, no
13755/// CTE/union/DISTINCT/GROUP/HAVING/ORDER/LIMIT/OFFSET/windows/locking;
13756/// * every outer column reference must resolve inside q's output list —
13757/// a name that does not is an ERROR today, and flattening would
13758/// silently legalise it against the base table.
13759fn try_flatten_derived(stmt: &SelectStatement, primary: &TableRef) -> Option<SelectStatement> {
13760 use spg_sql::ast::SelectItem;
13761 let inner = primary.lateral_subquery.as_deref()?;
13762 // Outer shape.
13763 if !stmt.ctes.is_empty()
13764 || !stmt.unions.is_empty()
13765 || stmt.distinct
13766 || !stmt.distinct_on.is_empty()
13767 || !stmt.window_check_exprs.is_empty()
13768 || stmt.locking.is_some()
13769 || primary.with_ordinality
13770 || !primary.unnest_column_aliases.is_empty()
13771 {
13772 return None;
13773 }
13774 // Inner shape.
13775 if !inner.ctes.is_empty()
13776 || !inner.unions.is_empty()
13777 || inner.distinct
13778 || !inner.distinct_on.is_empty()
13779 || inner.group_by.is_some()
13780 || inner.group_by_all
13781 || inner.having.is_some()
13782 || !inner.order_by.is_empty()
13783 || inner.limit.is_some()
13784 || inner.offset.is_some()
13785 || !inner.window_check_exprs.is_empty()
13786 || inner.locking.is_some()
13787 {
13788 return None;
13789 }
13790 let ifrom = inner.from.as_ref()?;
13791 let it = &ifrom.primary;
13792 if !ifrom.joins.is_empty()
13793 || it.name.is_empty()
13794 || it.lateral_subquery.is_some()
13795 || it.unnest_expr.is_some()
13796 || it.generate_series_args.is_some()
13797 || it.as_of_segment.is_some()
13798 || it.jsonb_each_text_arg.is_some()
13799 || it.table_fn_call.is_some()
13800 || it.rows_from.is_some()
13801 || it.json_table.is_some()
13802 || it.with_ordinality
13803 || !it.unnest_column_aliases.is_empty()
13804 {
13805 return None;
13806 }
13807 if inner.where_.as_ref().is_some_and(crate::expr_has_subquery) {
13808 return None;
13809 }
13810 // The output map: q's visible name -> the underlying column.
13811 let inner_alias = it.alias.clone().unwrap_or_else(|| it.name.clone());
13812 let mut map: alloc::collections::BTreeMap<String, spg_sql::ast::ColumnName> =
13813 alloc::collections::BTreeMap::new();
13814 for item in &inner.items {
13815 let SelectItem::Expr { expr, alias } = item else {
13816 return None;
13817 };
13818 let Expr::Column(c) = expr else {
13819 return None;
13820 };
13821 if let Some(q) = c.qualifier.as_deref()
13822 && !q.eq_ignore_ascii_case(&inner_alias)
13823 {
13824 return None;
13825 }
13826 let out_name = alias.clone().unwrap_or_else(|| c.name.clone());
13827 // A duplicated output name would make substitution ambiguous.
13828 if map
13829 .insert(out_name.to_ascii_lowercase(), c.clone())
13830 .is_some()
13831 {
13832 return None;
13833 }
13834 }
13835 if map.is_empty() {
13836 return None;
13837 }
13838 let derived_alias = primary
13839 .alias
13840 .clone()
13841 .unwrap_or_else(|| primary.name.clone())
13842 .to_ascii_lowercase();
13843 // Substitute in a clone; bail (None) on the first reference the map
13844 // cannot answer.
13845 let mut out = stmt.clone();
13846 let ok = core::cell::Cell::new(true);
13847 let mut subst = |e: &mut Expr| -> bool {
13848 match e {
13849 Expr::Column(c) => {
13850 match c.qualifier.as_deref() {
13851 Some(q) if q.eq_ignore_ascii_case(&derived_alias) => {}
13852 None => {}
13853 Some(_) => {
13854 ok.set(false);
13855 return true;
13856 }
13857 }
13858 match map.get(&c.name.to_ascii_lowercase()) {
13859 Some(target) => *c = target.clone(),
13860 None => ok.set(false),
13861 }
13862 true
13863 }
13864 // Any subquery could reference q from its own scope;
13865 // descending is a later knife — bail for now.
13866 Expr::ScalarSubquery(_)
13867 | Expr::Exists { .. }
13868 | Expr::InSubquery { .. }
13869 | Expr::RowInSubquery { .. }
13870 | Expr::RowCmpSubquery { .. } => {
13871 ok.set(false);
13872 true
13873 }
13874 _ => false,
13875 }
13876 };
13877 for item in &mut out.items {
13878 match item {
13879 SelectItem::Expr { expr, .. } => {
13880 crate::expr_analysis::rewrite_nodes_mut(expr, &mut subst);
13881 }
13882 // `SELECT * FROM (…) q` means q's columns, in q's order.
13883 SelectItem::Wildcard | SelectItem::QualifiedWildcard(_) => return None,
13884 }
13885 }
13886 if let Some(w) = &mut out.where_ {
13887 crate::expr_analysis::rewrite_nodes_mut(w, &mut subst);
13888 }
13889 if let Some(gs) = &mut out.group_by {
13890 for g in gs {
13891 crate::expr_analysis::rewrite_nodes_mut(g, &mut subst);
13892 }
13893 }
13894 if let Some(h) = &mut out.having {
13895 crate::expr_analysis::rewrite_nodes_mut(h, &mut subst);
13896 }
13897 for o in &mut out.order_by {
13898 crate::expr_analysis::rewrite_nodes_mut(&mut o.expr, &mut subst);
13899 }
13900 for d in &mut out.distinct_on {
13901 crate::expr_analysis::rewrite_nodes_mut(d, &mut subst);
13902 }
13903 if !ok.get() {
13904 return None;
13905 }
13906 // FROM becomes the stored table; the filters conjoin.
13907 out.from = Some(spg_sql::ast::FromClause {
13908 primary: it.clone(),
13909 joins: Vec::new(),
13910 });
13911 out.where_ = match (inner.where_.clone(), out.where_.take()) {
13912 (Some(a), Some(b)) => Some(Expr::Binary {
13913 lhs: alloc::boxed::Box::new(a),
13914 op: spg_sql::ast::BinOp::And,
13915 rhs: alloc::boxed::Box::new(b),
13916 }),
13917 (Some(a), None) => Some(a),
13918 (None, b) => b,
13919 };
13920 Some(out)
13921}
13922
13923/// v7.39 (round 742) — rewrite `SELECT count(*) FROM (SELECT <plain>
13924/// FROM t [WHERE p] ORDER BY … OFFSET k [no LIMIT]) q` into
13925/// `SELECT greatest(count(*) - k, 0) FROM t [WHERE p]`. Sound because
13926/// ORDER BY is count-invariant and OFFSET k drops exactly min(k, n)
13927/// rows. Admission mirrors the flatten's conservatism; a LIMIT, a
13928/// DISTINCT, an SRF, or an unprovable inner shape stays put.
13929fn try_count_over_offset(stmt: &SelectStatement, primary: &TableRef) -> Option<SelectStatement> {
13930 use spg_sql::ast::{Expr as E, LimitExpr, SelectItem};
13931 let inner = primary.lateral_subquery.as_deref()?;
13932 // Outer: exactly `SELECT count(*)`, nothing else.
13933 if !stmt.ctes.is_empty()
13934 || !stmt.unions.is_empty()
13935 || stmt.distinct
13936 || !stmt.distinct_on.is_empty()
13937 || stmt.where_.is_some()
13938 || stmt.group_by.is_some()
13939 || stmt.having.is_some()
13940 || !stmt.order_by.is_empty()
13941 || stmt.limit.is_some()
13942 || stmt.offset.is_some()
13943 || stmt.items.len() != 1
13944 {
13945 return None;
13946 }
13947 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
13948 return None;
13949 };
13950 let E::FunctionCall { name, args } = expr else {
13951 return None;
13952 };
13953 if !name.eq_ignore_ascii_case("count_star") || !args.is_empty() {
13954 return None;
13955 }
13956 // Inner: flatten-shaped plus ORDER BY and a literal OFFSET, no LIMIT.
13957 let Some(LimitExpr::Literal(k)) = &inner.offset else {
13958 return None;
13959 };
13960 let k = i64::from(*k);
13961 if inner.limit.is_some() || inner.order_by.is_empty() {
13962 return None;
13963 }
13964 let mut counted = inner.clone();
13965 counted.order_by = Vec::new();
13966 counted.offset = None;
13967 // The stripped inner must now be a provable simple shape (its
13968 // items become irrelevant — count(*) reads none of them — but an
13969 // SRF item would change the row count, so the flatten predicate's
13970 // scrutiny still applies).
13971 let base = matview_flatten_probe(&counted)?;
13972 let mut out = stmt.clone();
13973 out.items = alloc::vec![SelectItem::Expr {
13974 expr: E::FunctionCall {
13975 name: String::from("greatest"),
13976 args: alloc::vec![
13977 E::Binary {
13978 lhs: alloc::boxed::Box::new(E::FunctionCall {
13979 name: String::from("count_star"),
13980 args: alloc::vec![],
13981 }),
13982 op: spg_sql::ast::BinOp::Sub,
13983 rhs: alloc::boxed::Box::new(E::Literal(spg_sql::ast::Literal::Integer(k))),
13984 },
13985 E::Literal(spg_sql::ast::Literal::Integer(0)),
13986 ],
13987 },
13988 alias: Some(String::from("count")),
13989 }];
13990 out.from = Some(spg_sql::ast::FromClause {
13991 primary: base,
13992 joins: Vec::new(),
13993 });
13994 out.where_ = counted.where_.clone();
13995 Some(out)
13996}
13997
13998/// The inner-shape probe `try_count_over_offset` shares with the
13999/// flatten: single stored table, no modifiers, no subqueries, no SRF
14000/// items. Returns the base TableRef.
14001fn matview_flatten_probe(inner: &SelectStatement) -> Option<TableRef> {
14002 use spg_sql::ast::SelectItem;
14003 if !inner.ctes.is_empty()
14004 || !inner.unions.is_empty()
14005 || inner.distinct
14006 || !inner.distinct_on.is_empty()
14007 || inner.group_by.is_some()
14008 || inner.group_by_all
14009 || inner.having.is_some()
14010 || !inner.order_by.is_empty()
14011 || inner.limit.is_some()
14012 || inner.offset.is_some()
14013 || !inner.window_check_exprs.is_empty()
14014 || inner.locking.is_some()
14015 {
14016 return None;
14017 }
14018 let ifrom = inner.from.as_ref()?;
14019 let it = &ifrom.primary;
14020 if !ifrom.joins.is_empty()
14021 || it.name.is_empty()
14022 || it.lateral_subquery.is_some()
14023 || it.unnest_expr.is_some()
14024 || it.generate_series_args.is_some()
14025 || it.as_of_segment.is_some()
14026 || it.jsonb_each_text_arg.is_some()
14027 || it.table_fn_call.is_some()
14028 || it.rows_from.is_some()
14029 || it.json_table.is_some()
14030 || it.with_ordinality
14031 {
14032 return None;
14033 }
14034 for item in &inner.items {
14035 match item {
14036 SelectItem::Expr { expr, .. } => {
14037 if crate::expr_has_subquery(expr) || expr_contains_builtin_srf(expr) {
14038 return None;
14039 }
14040 }
14041 SelectItem::Wildcard => {}
14042 SelectItem::QualifiedWildcard(_) => return None,
14043 }
14044 }
14045 if inner.where_.as_ref().is_some_and(crate::expr_has_subquery) {
14046 return None;
14047 }
14048 Some(it.clone())
14049}
14050
14051/// v7.39 (round 743) — rewrite `SELECT count(*) FROM (SELECT
14052/// unnest(ARRAY[e1..ek]) [AS v] FROM t [WHERE p]) q` into
14053/// `SELECT count(*) * k FROM t [WHERE p]`. Sound because a
14054/// constant-LENGTH array literal unnests to exactly k rows per input
14055/// row (NULL elements are rows too). One SRF item only, elements
14056/// subquery-free, and the stripped inner must pass the same probe the
14057/// count-over-offset rewrite uses.
14058fn try_count_over_const_unnest(
14059 stmt: &SelectStatement,
14060 primary: &TableRef,
14061) -> Option<SelectStatement> {
14062 use spg_sql::ast::{Expr as E, SelectItem};
14063 let inner = primary.lateral_subquery.as_deref()?;
14064 if !stmt.ctes.is_empty()
14065 || !stmt.unions.is_empty()
14066 || stmt.distinct
14067 || !stmt.distinct_on.is_empty()
14068 || stmt.where_.is_some()
14069 || stmt.group_by.is_some()
14070 || stmt.having.is_some()
14071 || !stmt.order_by.is_empty()
14072 || stmt.limit.is_some()
14073 || stmt.offset.is_some()
14074 || stmt.items.len() != 1
14075 {
14076 return None;
14077 }
14078 let SelectItem::Expr { expr, .. } = &stmt.items[0] else {
14079 return None;
14080 };
14081 let E::FunctionCall { name, args } = expr else {
14082 return None;
14083 };
14084 if !name.eq_ignore_ascii_case("count_star") || !args.is_empty() {
14085 return None;
14086 }
14087 // Inner: exactly one item, and it is unnest(ARRAY[...]).
14088 if inner.items.len() != 1
14089 || !inner.order_by.is_empty()
14090 || inner.limit.is_some()
14091 || inner.offset.is_some()
14092 {
14093 return None;
14094 }
14095 let SelectItem::Expr { expr: item, .. } = &inner.items[0] else {
14096 return None;
14097 };
14098 let E::FunctionCall {
14099 name: fname,
14100 args: fargs,
14101 } = item
14102 else {
14103 return None;
14104 };
14105 if !fname.eq_ignore_ascii_case("unnest") || fargs.len() != 1 {
14106 return None;
14107 }
14108 let E::Array(elems) = &fargs[0] else {
14109 return None;
14110 };
14111 if elems.is_empty() || elems.iter().any(crate::expr_has_subquery) {
14112 return None;
14113 }
14114 let k = elems.len() as i64;
14115 // The stripped inner (the SRF item replaced by a plain constant)
14116 // must be the provable simple shape.
14117 let mut counted = inner.clone();
14118 counted.items = alloc::vec![SelectItem::Expr {
14119 expr: E::Literal(spg_sql::ast::Literal::Integer(1)),
14120 alias: None,
14121 }];
14122 let base = matview_flatten_probe(&counted)?;
14123 let mut out = stmt.clone();
14124 out.items = alloc::vec![SelectItem::Expr {
14125 expr: E::Binary {
14126 lhs: alloc::boxed::Box::new(E::FunctionCall {
14127 name: String::from("count_star"),
14128 args: alloc::vec![],
14129 }),
14130 op: spg_sql::ast::BinOp::Mul,
14131 rhs: alloc::boxed::Box::new(E::Literal(spg_sql::ast::Literal::Integer(k))),
14132 },
14133 alias: Some(String::from("count")),
14134 }];
14135 out.from = Some(spg_sql::ast::FromClause {
14136 primary: base,
14137 joins: Vec::new(),
14138 });
14139 out.where_ = counted.where_.clone();
14140 Some(out)
14141}