rudb_bind/statement.rs
1//! From an `Ast` to a `Bound`, which is a statement rather than a query.
2//!
3//! A `SELECT` binds to a [`Plan`] and nothing else, and that is why [`bind`](crate::bind) can hand
4//! one back. `CREATE TABLE`, `DROP TABLE` and `INSERT` are not plans and are deliberately not being
5//! made into plans. A `Node::CreateTable` would be a node with no columns, no rows, no cost and no
6//! reason to be pushed past anything, which is to say a node the optimizer has to be told to leave
7//! alone and the executor has to special case at the root. `spec/09-optimizer.md` section 9.1 says
8//! every node in a plan produces rows, and a DDL statement does not, so it goes beside the plan and
9//! not inside it.
10//!
11//! What each variant carries is the statement with every name and type already resolved, so the
12//! thing that runs it does catalog calls and nothing else. An `INSERT` in particular arrives with
13//! a plan whose output is exactly the target's columns in the target's order and the target's
14//! types, with the casts and the nulls for unmentioned columns already in it, so appending is a
15//! loop over chunks.
16
17use rudb_catalog::{Catalog, Entry, QualifiedName, duplicate_check, same_name};
18use rudb_common::bounds::End;
19use rudb_common::{
20 Bound as ColumnBound, Clustering, Error, Field, LogicalType, Result, Session, Stat, Value,
21 Width,
22};
23use rudb_parse::ast::{self, Ast};
24use rudb_parse::{NONE, deparse, parse_ast};
25use rudb_plan::{Expr, ExprRef, Node, Plan, SortKey};
26
27use crate::binder::Binder;
28use crate::parameters::Parameters;
29
30/// One statement, bound.
31///
32/// Not `#[non_exhaustive]`. A new variant here is a new kind of statement, and the compiler
33/// pointing at every place that has to decide what to do with it is the whole value of the enum.
34#[derive(Debug)]
35pub enum Bound {
36 /// A query, which is the only one of these that produces rows.
37 Query(Plan),
38 /// `CREATE TABLE`.
39 CreateTable(CreateTable),
40 /// `CREATE VIEW`.
41 CreateView(CreateView),
42 /// `DROP TABLE` or `DROP VIEW`.
43 DropTable(DropTable),
44 /// `INSERT INTO`.
45 Insert(Insert),
46 /// `SET name = value`, or `RESET name`, which is the same thing with no value.
47 Setting(Setting),
48 /// Flushes a persistent database snapshot.
49 Checkpoint,
50 /// `EXPLAIN` over a query, holding the plan of the query rather than the query.
51 ///
52 /// The same `Plan` a [`Bound::Query`] would have carried, bound the same way and by the same
53 /// code. What makes it an explain is that the layer above optimizes it and prints it instead
54 /// of running it, which is the point: a plan that was built differently because somebody asked
55 /// to see it is not the plan that runs.
56 ///
57 /// With `analyze` set the layer above runs it as well and prints what happened on it. Still the
58 /// same plan, for the same reason.
59 ///
60 /// With `statistics` set it prints what the planner knew as well, which is the use and the class
61 /// behind every number in the plan. That one changes nothing about the plan or the run either.
62 Explain { plan: Plan, analyze: bool, statistics: bool },
63}
64
65/// A bound `SET` or `RESET`.
66///
67/// The value is a [`Value`] rather than an expression, because every setting there is takes a
68/// string or a number and nothing that runs one wants a plan. What a setting does with the value it
69/// gets is the setting's own business and is decided a layer up, since the binder has no idea what
70/// settings exist.
71///
72/// The narrow part of that is that the value has to already be a constant. `SET threads = 2 + 2` is
73/// four in DuckDB and is refused here, because folding it needs the expression rewriter and the
74/// rewriter is two layers above the binder. Nothing writes arithmetic in a `SET` and the refusal
75/// says what it is, so this waits for a reason to move.
76#[derive(Debug)]
77pub struct Setting {
78 /// The setting name, as written.
79 pub name: String,
80 /// The scope word, if one was written.
81 pub scope: ast::Scope,
82 /// The value, or `None` for a `RESET`.
83 pub value: Option<Value>,
84 /// Whether the statement was written as a bare `PRAGMA name`, which carries its value in it.
85 pub pragma: bool,
86}
87
88/// A bound `CREATE TABLE`.
89#[derive(Debug)]
90pub struct CreateTable {
91 /// The full name the table gets.
92 pub name: QualifiedName,
93 /// The columns, in order, with the types already resolved. For a `CREATE TABLE AS` these are
94 /// the query's output types under whatever names the statement or the query gave them.
95 pub columns: Vec<Field>,
96 /// The query to fill it from, for a `CREATE TABLE AS`.
97 pub source: Option<Plan>,
98 /// Whether an existing table of that name is left alone rather than being an error.
99 pub if_not_exists: bool,
100 /// Whether an existing table of that name is dropped first.
101 pub or_replace: bool,
102}
103
104/// A bound `CREATE VIEW`.
105///
106/// The body is the text that was written rather than the plan it bound to. It was bound once on the
107/// way through here, which is what refuses a view over a table that is not there, and the plan that
108/// came out of that is then thrown away, because a view follows the tables underneath it and a plan
109/// cannot. See [`rudb_catalog::View`].
110#[derive(Debug)]
111pub struct CreateView {
112 /// The full name the view gets.
113 pub name: QualifiedName,
114 /// The body, as written.
115 pub sql: String,
116 /// The whole statement written back out, which is what `duckdb_views()` reports as `sql`.
117 ///
118 /// Written here because this is the last place the tree is in reach. See
119 /// [`rudb_catalog::View::statement`] for what the column is and why it is not the text.
120 pub statement: String,
121 /// The column names the statement gave, which rename a prefix of what the body produces.
122 pub aliases: Vec<String>,
123 /// Whether an existing entry of that name is left alone rather than being an error.
124 pub if_not_exists: bool,
125 /// Whether an existing entry of that name is dropped first.
126 pub or_replace: bool,
127 /// The columns binding the body produced, after the alias list was applied.
128 ///
129 /// Worked out here because this is where the body is bound, and carried to the catalog because
130 /// that is where `duckdb_columns()` and `duckdb_views()` read it from. See the doc on
131 /// `rudb_catalog::View` for why the catalog keeps a list it will have to refresh later.
132 pub columns: Vec<Field>,
133}
134
135/// A bound `DROP TABLE` or `DROP VIEW`.
136#[derive(Debug)]
137pub struct DropTable {
138 /// The tables or views to drop, already resolved. With `IF EXISTS` a name that does not resolve
139 /// is not in here at all, which is what makes running this a sequence of drops that cannot
140 /// fail for being missing. Dropping one of these as the wrong type still can, because `DROP
141 /// TABLE IF EXISTS v` where `v` is a view is an error in DuckDB and was measured to be one.
142 pub names: Vec<QualifiedName>,
143 /// Which of the two the statement said it was dropping.
144 pub kind: Entry,
145}
146
147/// A bound `INSERT`.
148#[derive(Debug)]
149pub struct Insert {
150 /// The table to append to.
151 pub name: QualifiedName,
152 /// The rows to append. The output is the table's columns, in the table's order, with the
153 /// table's types, so nothing between here and the append has a decision left to make.
154 pub source: Plan,
155}
156
157/// Binds one parsed statement against a catalog.
158///
159/// # Errors
160///
161/// If the script does not hold exactly one statement, if a name does not resolve, if a type does
162/// not work out, or if the statement uses something that is not bound yet.
163pub fn bind_statement(ast: &Ast, catalog: &Catalog) -> Result<Bound> {
164 bind_statement_with(ast, catalog, &Parameters::new(), &Session::new())
165}
166
167/// Binds one parsed statement against a catalog, with values for its parameters and its settings.
168///
169/// This is the prepared statement path. The statement is parsed once and bound once per set of
170/// values, so a parameter is a constant by the time the plan exists and everything after the binder
171/// sees an ordinary query. That is why there is no parameter in `rudb_plan::Expr`.
172///
173/// # Errors
174///
175/// Everything [`bind_statement`] reports, plus an error for a parameter that was given no value.
176pub fn bind_statement_with(
177 ast: &Ast,
178 catalog: &Catalog,
179 parameters: &Parameters,
180 session: &Session,
181) -> Result<Bound> {
182 let statement = match ast.statements.as_slice() {
183 [statement] => *statement,
184 [] => return Err(Error::binder("no statement to bind")),
185 _ => return Err(Error::not_implemented("a script of more than one statement")),
186 };
187 match statement {
188 ast::Statement::Query(query) => {
189 let mut binder = Binder::with(catalog, parameters, session);
190 let (root, _) = binder.bind_query(ast, query)?;
191 Ok(Bound::Query(finish(binder, root)?))
192 }
193 ast::Statement::CreateTable(index) => {
194 create_table(ast, catalog, parameters, session, index)
195 }
196 ast::Statement::CreateView(index) => create_view(ast, catalog, parameters, session, index),
197 ast::Statement::DropTable(index) => drop_table(ast, catalog, index),
198 ast::Statement::Insert(index) => insert(ast, catalog, parameters, session, index),
199 ast::Statement::Set(index) | ast::Statement::Reset(index) => {
200 setting(ast, catalog, parameters, session, index)
201 }
202 ast::Statement::Checkpoint => Ok(Bound::Checkpoint),
203 ast::Statement::Explain { query, analyze, statistics } => {
204 let mut binder = Binder::with(catalog, parameters, session);
205 let (root, _) = binder.bind_query(ast, query)?;
206 Ok(Bound::Explain { plan: finish(binder, root)?, analyze, statistics })
207 }
208 }
209}
210
211/// Parses and binds one statement, which is the whole front end in one call.
212///
213/// # Errors
214///
215/// Anything the parser or the binder reports.
216pub fn bind_statement_sql(sql: &str, catalog: &Catalog) -> Result<Bound> {
217 let ast = parse_ast(sql)?;
218 bind_statement(&ast, catalog)
219}
220
221/// Roots a binder's plan and checks it.
222fn finish(binder: Binder<'_>, root: rudb_plan::NodeRef) -> Result<Plan> {
223 let mut plan = binder.into_plan();
224 plan.set_root(root);
225 plan.validate()?;
226 Ok(plan)
227}
228
229fn create_table(
230 ast: &Ast,
231 catalog: &Catalog,
232 parameters: &Parameters,
233 session: &Session,
234 index: ast::CreateTableRef,
235) -> Result<Bound> {
236 let written = ast.create_table(index);
237 let parts: Vec<&str> = ast.name(written.name).collect();
238 let name = if written.temporary {
239 catalog.resolve_for_create_temporary(&parts)?
240 } else {
241 catalog.resolve_for_create(&parts)?
242 };
243 let defs = ast.column_defs(written.columns);
244 let (columns, source) = if written.query == NONE {
245 let mut columns = Vec::with_capacity(defs.len());
246 for def in defs {
247 let text = ast.string(def.ty);
248 if text.is_empty() {
249 return Err(Error::binder(format!(
250 "Column \"{}\" was declared without a type",
251 ast.string(def.name)
252 )));
253 }
254 let ty = LogicalType::parse(text)?;
255 let column = ast.string(def.name);
256 columns.push(if def.not_null {
257 Field::required(column, ty)
258 } else {
259 Field::new(column, ty)
260 });
261 }
262 (columns, None)
263 } else {
264 let mut binder = Binder::with(catalog, parameters, session);
265 let (root, scope) = binder.bind_query(ast, written.query)?;
266 if defs.len() > scope.len() {
267 // DuckDB's sentence, typo and all. A column list shorter than the query is fine and
268 // renames a prefix, so only this direction is an error.
269 return Err(Error::binder("Target table has more colum names than query result."));
270 }
271 let mut columns = Vec::with_capacity(scope.len());
272 for (at, column) in scope.columns.iter().enumerate() {
273 let named = match defs.get(at) {
274 Some(def) => ast.string(def.name).to_string(),
275 None => column.name.clone(),
276 };
277 columns.push(Field::new(named, column.ty.clone()));
278 }
279 if defs.is_empty() {
280 deduplicate(&mut columns);
281 }
282 (columns, Some(finish(binder, root)?))
283 };
284 duplicate_check(&columns)?;
285 Ok(Bound::CreateTable(CreateTable {
286 name,
287 columns,
288 source,
289 if_not_exists: written.if_not_exists,
290 or_replace: written.or_replace,
291 }))
292}
293
294/// Renames the columns a query repeated, which is what makes `CREATE TABLE t AS SELECT 1 AS a, 2 AS
295/// a` a table rather than an error.
296///
297/// A query is allowed to produce two columns of one name and `SELECT 1 AS a, 2 AS a` prints two
298/// columns called `a`, so a statement that turns a query into a table has to decide what to do with
299/// that, and DuckDB renames rather than refusing. The suffix is `_1`, then `_2`, counting up until
300/// the name is free, so a query that already has an `a_1` in it pushes the renamed column to `a_2`
301/// rather than colliding with it.
302///
303/// This only runs when the statement wrote no column list. With a list, even a short one, duckdb
304/// v1.4.1 takes the names as they come and a repeat is an error, so `CREATE TABLE t (z) AS SELECT 1
305/// AS a, 2 AS a` is a table of `z` and `a` and adding a third `a` to that query is a refusal.
306fn deduplicate(columns: &mut [Field]) {
307 for at in 0..columns.len() {
308 let taken = |name: &str, upto: usize, columns: &[Field]| {
309 columns[..upto].iter().any(|held| same_name(&held.name, name))
310 };
311 if !taken(&columns[at].name, at, columns) {
312 continue;
313 }
314 let mut suffix = 1;
315 let mut candidate = format!("{}_{suffix}", columns[at].name);
316 while taken(&candidate, at, columns) {
317 suffix += 1;
318 candidate = format!("{}_{suffix}", columns[at].name);
319 }
320 columns[at].name = candidate;
321 }
322}
323
324/// Binds a `CREATE VIEW`, which means binding the body and then throwing the plan away.
325///
326/// Throwing it away is the point. The body is bound here so that a view over a table that is not
327/// there is refused now rather than at the first select, and so that the column list can be checked
328/// against what the body actually produces. What the catalog keeps is the text, because a view
329/// follows the tables underneath it and a plan is a photograph of the day it was built.
330fn create_view(
331 ast: &Ast,
332 catalog: &Catalog,
333 parameters: &Parameters,
334 session: &Session,
335 index: ast::CreateViewRef,
336) -> Result<Bound> {
337 let written = ast.create_view(index);
338 let parts: Vec<&str> = ast.name(written.name).collect();
339 let name = if written.temporary {
340 catalog.resolve_for_create_temporary(&parts)?
341 } else {
342 catalog.resolve_for_create(&parts)?
343 };
344 let aliases: Vec<String> = ast.name(written.columns).map(str::to_string).collect();
345
346 let mut binder = Binder::with(catalog, parameters, session);
347 let (_, mut scope) = binder.bind_query(ast, written.query)?;
348 if aliases.len() > scope.len() {
349 return Err(Error::binder("More VIEW aliases than columns in query result"));
350 }
351 if !aliases.is_empty() {
352 let written: Vec<&str> = aliases.iter().map(String::as_str).collect();
353 scope.rename(&written, "unnamed_subquery")?;
354 }
355
356 Ok(Bound::CreateView(CreateView {
357 name,
358 sql: ast.string(written.sql).to_string(),
359 statement: deparse::create_view(ast, index),
360 aliases,
361 if_not_exists: written.if_not_exists,
362 or_replace: written.or_replace,
363 columns: scope.fields(),
364 }))
365}
366
367fn drop_table(ast: &Ast, catalog: &Catalog, index: ast::DropTableRef) -> Result<Bound> {
368 let written = ast.drop_table(index);
369 let kind = if written.view { Entry::View } else { Entry::Table };
370 let mut names = Vec::new();
371 for &name in ast.name_list(written.names) {
372 let parts: Vec<&str> = ast.name(name).collect();
373 // The statement said which of the two it meant, so a name that is not there is a missing
374 // one of those and not a missing table.
375 match catalog.resolve_as(&parts, kind) {
376 Ok(resolved) => names.push(resolved),
377 Err(error) if written.if_exists => drop(error),
378 Err(error) => return Err(error),
379 }
380 }
381 Ok(Bound::DropTable(DropTable { names, kind }))
382}
383
384/// Binds a `SET` or a `RESET`, which is resolving its value and nothing else.
385///
386/// The name is not checked here. The binder knows what tables exist and has no idea what settings
387/// exist, since a setting is a knob on the engine rather than an entry in a catalog, and a version
388/// of this that held the list would be the binder holding a copy of something it cannot enforce.
389fn setting(
390 ast: &Ast,
391 catalog: &Catalog,
392 parameters: &Parameters,
393 session: &Session,
394 index: ast::SettingRef,
395) -> Result<Bound> {
396 let written = ast.setting(index);
397 let name = ast.string(written.name).to_string();
398 let value = if written.value == NONE {
399 None
400 } else {
401 let mut binder = Binder::with(catalog, parameters, session);
402 let bound = binder.bind_setting_value(ast, written.value)?;
403 let Expr::Constant(value) = *binder.plan().expr(bound) else {
404 return Err(Error::not_implemented(format!(
405 "a value for {name} that is not a constant"
406 )));
407 };
408 Some(binder.plan().value(value).clone())
409 };
410 Ok(Bound::Setting(Setting { name, scope: written.scope, value, pragma: written.pragma }))
411}
412
413/// Sorts an insert's rows into the order the target table declared.
414///
415/// Returns the input unchanged when the statement supplies none of the declared columns, because
416/// every one of them is then a constant null and sorting on a constant is a sort that buys nothing
417/// and costs a pass. A statement that supplies some of them sorts on those: the declaration is
418/// about the order the rows are written in, and the columns that are there still order them.
419///
420/// The leading key carries the width. `date_trunc('month', d)` and `d` sort the same rows into the
421/// same fragments for any predicate a month wide or wider, and the difference is what happens
422/// inside a month: bucketed, the second key orders the whole month, which is the key locality the
423/// joins want and the reason the width is part of the declaration at all.
424fn clustered(
425 binder: &mut Binder<'_>,
426 input: rudb_plan::NodeRef,
427 scope: &crate::scope::Scope,
428 clustering: &Clustering,
429 targets: &[usize],
430 fields: &[Field],
431) -> Result<rudb_plan::NodeRef> {
432 let mut keys: Vec<SortKey> = Vec::with_capacity(clustering.columns().len());
433 for (at, &column) in clustering.columns().iter().enumerate() {
434 let Some(from) = targets.iter().position(|&target| target == column as usize) else {
435 continue;
436 };
437 let source = &scope.columns[from];
438 let expr = binder.plan_mut().add_expr(Expr::Column(source.binding), source.ty.clone());
439 // Cast to the column's own type before bucketing, since the source of a load is a file
440 // whose date column can arrive as a timestamp and `date_trunc` gives back the type it was
441 // handed. Sorting on a different type than the column stores would still be an order, but
442 // it would not be the order the declaration names.
443 let expr = binder.checked_cast_to(expr, &fields[column as usize].ty, false)?;
444 let expr =
445 if at == 0 { bucketed(binder, expr, clustering.width(), fields, column) } else { expr };
446 keys.push(SortKey { expr, descending: false, nulls_first: false });
447 }
448 if keys.is_empty() {
449 return Ok(input);
450 }
451 let keys = binder.plan_mut().add_sort_keys(&keys);
452 Ok(binder.plan_mut().add_node(Node::Sort { input, keys }))
453}
454
455/// The declaration with an automatic width turned into the bucket the incoming rows ask for.
456///
457/// A declaration that named no width says the bucket should come from how many rows a partition
458/// would hold, and this is the only place that number is in reach. The rows are the source's, not
459/// the target's: a load into an empty table has a target with nothing to count, and the whole case
460/// the rule exists for is the first load of a big table. So the count and the range come off the
461/// source's own zones, which is the Parquet footer for a file and the directory for a table, and
462/// both are already on the plan because the estimator wanted them.
463///
464/// Everything about this is best effort and that is by design. The three widths hold the same rows
465/// and answer the same queries, so guessing wrong costs some pruning or some key locality and
466/// cannot cost an answer. A source that is a join, a group by or a values list has no zones to read
467/// and gets [`Width::DEFAULT`], which is what the fixed default was before the rule existed.
468fn fitted(
469 binder: &Binder<'_>,
470 scope: &crate::scope::Scope,
471 clustering: &Clustering,
472 targets: &[usize],
473) -> Clustering {
474 if clustering.width() != Width::Auto {
475 return clustering.clone();
476 }
477 let Some(from) = targets.iter().position(|&target| target == clustering.partition() as usize)
478 else {
479 return clustering.fitted(0, 0);
480 };
481 let source = &scope.columns[from];
482 let Some(zones) = binder.plan().sole_zones() else {
483 return clustering.fitted(0, 0);
484 };
485 // By name, and off whichever store the plan reads rather than off the one this column is bound
486 // to. The binding points at the projection over the scan, since a load is a projection into the
487 // target's types, and following a binding back through a projection is the optimizer's job. A
488 // load reads one table or one file, so the store with bounds on it is the store the name is in.
489 let Some(at) = zones.column(&source.name) else {
490 return clustering.fitted(0, 0);
491 };
492 let rows = zones.surviving(&[]).unwrap_or(0);
493 let days = span(&zones.extreme(at, End::Low), &zones.extreme(at, End::High)).unwrap_or(0);
494 clustering.fitted(rows, days)
495}
496
497/// How many days a column covers, from the smallest and largest values in it.
498///
499/// `None` wherever the two do not make a span, which is a column that is entirely null, a store
500/// that could not fold its parts into one answer, and a pair of bounds that are not the same shape.
501/// All of them mean the same thing here, which is that there is nothing to divide the row count by.
502fn span(low: &Stat<ColumnBound>, high: &Stat<ColumnBound>) -> Option<u64> {
503 let (Stat::Known { value: low, .. }, Stat::Known { value: high, .. }) = (low, high) else {
504 return None;
505 };
506 let days = match (low, high) {
507 // A date is a day count already, which is the common case and the only exact one.
508 (ColumnBound::Int(low), ColumnBound::Int(high)) => high.checked_sub(*low)?,
509 // A timestamp is a count of seconds at whichever unit the column keeps, so the span is that
510 // difference divided by a day's worth of them. A scale wide enough to overflow the divisor
511 // is a column no calendar covers and falls out as no span at all.
512 (
513 ColumnBound::Scaled { unscaled: low, scale: at },
514 ColumnBound::Scaled { unscaled: high, scale: to },
515 ) if at == to => {
516 let day = 86_400_i128.checked_mul(10_i128.checked_pow(u32::from(*at))?)?;
517 high.checked_sub(*low)? / day
518 }
519 _ => return None,
520 };
521 u64::try_from(days).ok()
522}
523
524/// Wraps a sort key in the calendar bucket its declaration asked for.
525fn bucketed(
526 binder: &mut Binder<'_>,
527 expr: ExprRef,
528 width: Width,
529 fields: &[Field],
530 column: u32,
531) -> ExprRef {
532 if width == Width::Exact {
533 return expr;
534 }
535 let unit = binder.plan_mut().add_value(Value::Varchar(width.to_string().to_lowercase()));
536 let unit = binder.plan_mut().add_expr(Expr::Constant(unit), LogicalType::Varchar);
537 let args = binder.plan_mut().add_expr_list(&[unit, expr]);
538 let name = binder.plan_mut().intern("date_trunc");
539 let ty = fields[column as usize].ty.clone();
540 binder.plan_mut().add_expr(Expr::Function { name, args }, ty)
541}
542
543fn insert(
544 ast: &Ast,
545 catalog: &Catalog,
546 parameters: &Parameters,
547 session: &Session,
548 index: ast::InsertRef,
549) -> Result<Bound> {
550 let written = ast.insert(index);
551 let parts: Vec<&str> = ast.name(written.name).collect();
552 let name = catalog.resolve(&parts)?;
553 if catalog.entry(&name)? == Entry::View {
554 // The binary's sentence, article and all. A view has no rows of its own to append to, and
555 // an updatable view is a rule about rewriting the insert that neither database has.
556 return Err(Error::catalog(format!("{} is not an table", name.table)));
557 }
558 let target = catalog.table(&name)?;
559 let fields: Vec<Field> = target.columns().to_vec();
560 let clustering = target.clustering().cloned();
561
562 // Which table column each source column lands in. Without a column list that is the first n
563 // columns in order, and with one it is whatever the list says, which is also the check that
564 // the list names columns the table has and names none of them twice.
565 let targets: Vec<usize> = if written.columns.is_empty() {
566 (0..fields.len()).collect()
567 } else {
568 let mut targets = Vec::new();
569 for column in ast.name(written.columns) {
570 let at = fields.iter().position(|field| same_name(&field.name, column)).ok_or_else(
571 || {
572 Error::binder(format!(
573 "Table \"{}\" does not have a column named \"{column}\"",
574 name.table
575 ))
576 },
577 )?;
578 if targets.contains(&at) {
579 return Err(Error::binder(format!(
580 "Column \"{column}\" is named twice in the same INSERT"
581 )));
582 }
583 targets.push(at);
584 }
585 targets
586 };
587
588 let mut binder = Binder::with(catalog, parameters, session);
589 let (root, scope) = binder.bind_query(ast, written.source)?;
590 if scope.len() != targets.len() {
591 return Err(Error::binder(format!(
592 "Table \"{}\" has {} columns but {} values were supplied",
593 name.table,
594 targets.len(),
595 scope.len()
596 )));
597 }
598
599 // A table that declared what order its rows go in gets the sort here, under the projection
600 // rather than over it, because a projection does not reorder rows and the bindings the sort
601 // keys need are the ones the query just produced. This is the whole of the loader honouring
602 // the declaration: the rows arrive at the writer in order and the per fragment ranges, which
603 // are built from whatever order arrives, come out narrow instead of each covering the table.
604 let root = match &clustering {
605 None => root,
606 Some(clustering) => {
607 // The width is settled here and not on the table. A declaration that left the bucket to
608 // the data is a standing instruction, so it stays on the table as one and every load
609 // answers it with the rows that load is carrying. What the sort needs is an answer, and
610 // that is what this is.
611 let fitted = fitted(&binder, &scope, clustering, &targets);
612 clustered(&mut binder, root, &scope, &fitted, &targets, &fields)?
613 }
614 };
615
616 // The projection that makes the source look exactly like the table. Every column the statement
617 // did not name becomes a null of the column's own type, so the append never has to know that a
618 // column list was written at all.
619 let mut exprs: Vec<ExprRef> = Vec::with_capacity(fields.len());
620 let mut names = Vec::with_capacity(fields.len());
621 for (at, field) in fields.iter().enumerate() {
622 let expr = match targets.iter().position(|&target| target == at) {
623 Some(from) => {
624 let column = &scope.columns[from];
625 let expr =
626 binder.plan_mut().add_expr(Expr::Column(column.binding), column.ty.clone());
627 binder.checked_cast_to(expr, &field.ty, false)?
628 }
629 None => {
630 // A typed null rather than `add_constant`, which would give it the null type and
631 // make the column's type depend on whether a row happened to be inserted into it.
632 let value = binder.plan_mut().add_value(Value::Null);
633 binder.plan_mut().add_expr(Expr::Constant(value), field.ty.clone())
634 }
635 };
636 exprs.push(expr);
637 let interned = binder.plan_mut().intern(&field.name);
638 names.push(interned);
639 }
640 let exprs = binder.plan_mut().add_expr_list(&exprs);
641 let names = binder.plan_mut().add_name_list(&names);
642 let index = binder.fresh_index();
643 let root = binder.plan_mut().add_node(Node::Project { input: root, index, exprs, names });
644 Ok(Bound::Insert(Insert { name, source: finish(binder, root)? }))
645}