Skip to main content

rudb_bind/
statement.rs

1//! From an `Ast` to a `Bound`, which is a statement rather than a query.
2//!
3//! A `SELECT` binds to a [`Plan`] and nothing else, and that is why [`bind`](crate::bind) can hand
4//! one back. `CREATE TABLE`, `DROP TABLE` and `INSERT` are not plans and are deliberately not being
5//! made into plans. A `Node::CreateTable` would be a node with no columns, no rows, no cost and no
6//! reason to be pushed past anything, which is to say a node the optimizer has to be told to leave
7//! alone and the executor has to special case at the root. `spec/09-optimizer.md` section 9.1 says
8//! every node in a plan produces rows, and a DDL statement does not, so it goes beside the plan and
9//! not inside it.
10//!
11//! What each variant carries is the statement with every name and type already resolved, so the
12//! thing that runs it does catalog calls and nothing else. An `INSERT` in particular arrives with
13//! a plan whose output is exactly the target's columns in the target's order and the target's
14//! types, with the casts and the nulls for unmentioned columns already in it, so appending is a
15//! loop over chunks.
16
17use rudb_catalog::{Catalog, Entry, QualifiedName, duplicate_check, same_name};
18use rudb_common::bounds::End;
19use rudb_common::{
20    Bound as ColumnBound, Clustering, Error, Field, LogicalType, Result, Session, Stat, Value,
21    Width,
22};
23use rudb_parse::ast::{self, Ast};
24use rudb_parse::{NONE, deparse, parse_ast};
25use rudb_plan::{Expr, ExprRef, Node, Plan, SortKey};
26
27use crate::binder::Binder;
28use crate::parameters::Parameters;
29
30/// One statement, bound.
31///
32/// Not `#[non_exhaustive]`. A new variant here is a new kind of statement, and the compiler
33/// pointing at every place that has to decide what to do with it is the whole value of the enum.
34#[derive(Debug)]
35pub enum Bound {
36    /// A query, which is the only one of these that produces rows.
37    Query(Plan),
38    /// `CREATE TABLE`.
39    CreateTable(CreateTable),
40    /// `CREATE VIEW`.
41    CreateView(CreateView),
42    /// `DROP TABLE` or `DROP VIEW`.
43    DropTable(DropTable),
44    /// `INSERT INTO`.
45    Insert(Insert),
46    /// `SET name = value`, or `RESET name`, which is the same thing with no value.
47    Setting(Setting),
48    /// Flushes a persistent database snapshot.
49    Checkpoint,
50    /// `EXPLAIN` over a query, holding the plan of the query rather than the query.
51    ///
52    /// The same `Plan` a [`Bound::Query`] would have carried, bound the same way and by the same
53    /// code. What makes it an explain is that the layer above optimizes it and prints it instead
54    /// of running it, which is the point: a plan that was built differently because somebody asked
55    /// to see it is not the plan that runs.
56    ///
57    /// With `analyze` set the layer above runs it as well and prints what happened on it. Still the
58    /// same plan, for the same reason.
59    ///
60    /// With `statistics` set it prints what the planner knew as well, which is the use and the class
61    /// behind every number in the plan. That one changes nothing about the plan or the run either.
62    Explain { plan: Plan, analyze: bool, statistics: bool },
63}
64
65/// A bound `SET` or `RESET`.
66///
67/// The value is a [`Value`] rather than an expression, because every setting there is takes a
68/// string or a number and nothing that runs one wants a plan. What a setting does with the value it
69/// gets is the setting's own business and is decided a layer up, since the binder has no idea what
70/// settings exist.
71///
72/// The narrow part of that is that the value has to already be a constant. `SET threads = 2 + 2` is
73/// four in DuckDB and is refused here, because folding it needs the expression rewriter and the
74/// rewriter is two layers above the binder. Nothing writes arithmetic in a `SET` and the refusal
75/// says what it is, so this waits for a reason to move.
76#[derive(Debug)]
77pub struct Setting {
78    /// The setting name, as written.
79    pub name: String,
80    /// The scope word, if one was written.
81    pub scope: ast::Scope,
82    /// The value, or `None` for a `RESET`.
83    pub value: Option<Value>,
84    /// Whether the statement was written as a bare `PRAGMA name`, which carries its value in it.
85    pub pragma: bool,
86}
87
88/// A bound `CREATE TABLE`.
89#[derive(Debug)]
90pub struct CreateTable {
91    /// The full name the table gets.
92    pub name: QualifiedName,
93    /// The columns, in order, with the types already resolved. For a `CREATE TABLE AS` these are
94    /// the query's output types under whatever names the statement or the query gave them.
95    pub columns: Vec<Field>,
96    /// The query to fill it from, for a `CREATE TABLE AS`.
97    pub source: Option<Plan>,
98    /// Whether an existing table of that name is left alone rather than being an error.
99    pub if_not_exists: bool,
100    /// Whether an existing table of that name is dropped first.
101    pub or_replace: bool,
102}
103
104/// A bound `CREATE VIEW`.
105///
106/// The body is the text that was written rather than the plan it bound to. It was bound once on the
107/// way through here, which is what refuses a view over a table that is not there, and the plan that
108/// came out of that is then thrown away, because a view follows the tables underneath it and a plan
109/// cannot. See [`rudb_catalog::View`].
110#[derive(Debug)]
111pub struct CreateView {
112    /// The full name the view gets.
113    pub name: QualifiedName,
114    /// The body, as written.
115    pub sql: String,
116    /// The whole statement written back out, which is what `duckdb_views()` reports as `sql`.
117    ///
118    /// Written here because this is the last place the tree is in reach. See
119    /// [`rudb_catalog::View::statement`] for what the column is and why it is not the text.
120    pub statement: String,
121    /// The column names the statement gave, which rename a prefix of what the body produces.
122    pub aliases: Vec<String>,
123    /// Whether an existing entry of that name is left alone rather than being an error.
124    pub if_not_exists: bool,
125    /// Whether an existing entry of that name is dropped first.
126    pub or_replace: bool,
127    /// The columns binding the body produced, after the alias list was applied.
128    ///
129    /// Worked out here because this is where the body is bound, and carried to the catalog because
130    /// that is where `duckdb_columns()` and `duckdb_views()` read it from. See the doc on
131    /// `rudb_catalog::View` for why the catalog keeps a list it will have to refresh later.
132    pub columns: Vec<Field>,
133}
134
135/// A bound `DROP TABLE` or `DROP VIEW`.
136#[derive(Debug)]
137pub struct DropTable {
138    /// The tables or views to drop, already resolved. With `IF EXISTS` a name that does not resolve
139    /// is not in here at all, which is what makes running this a sequence of drops that cannot
140    /// fail for being missing. Dropping one of these as the wrong type still can, because `DROP
141    /// TABLE IF EXISTS v` where `v` is a view is an error in DuckDB and was measured to be one.
142    pub names: Vec<QualifiedName>,
143    /// Which of the two the statement said it was dropping.
144    pub kind: Entry,
145}
146
147/// A bound `INSERT`.
148#[derive(Debug)]
149pub struct Insert {
150    /// The table to append to.
151    pub name: QualifiedName,
152    /// The rows to append. The output is the table's columns, in the table's order, with the
153    /// table's types, so nothing between here and the append has a decision left to make.
154    pub source: Plan,
155}
156
157/// Binds one parsed statement against a catalog.
158///
159/// # Errors
160///
161/// If the script does not hold exactly one statement, if a name does not resolve, if a type does
162/// not work out, or if the statement uses something that is not bound yet.
163pub fn bind_statement(ast: &Ast, catalog: &Catalog) -> Result<Bound> {
164    bind_statement_with(ast, catalog, &Parameters::new(), &Session::new())
165}
166
167/// Binds one parsed statement against a catalog, with values for its parameters and its settings.
168///
169/// This is the prepared statement path. The statement is parsed once and bound once per set of
170/// values, so a parameter is a constant by the time the plan exists and everything after the binder
171/// sees an ordinary query. That is why there is no parameter in `rudb_plan::Expr`.
172///
173/// # Errors
174///
175/// Everything [`bind_statement`] reports, plus an error for a parameter that was given no value.
176pub fn bind_statement_with(
177    ast: &Ast,
178    catalog: &Catalog,
179    parameters: &Parameters,
180    session: &Session,
181) -> Result<Bound> {
182    let statement = match ast.statements.as_slice() {
183        [statement] => *statement,
184        [] => return Err(Error::binder("no statement to bind")),
185        _ => return Err(Error::not_implemented("a script of more than one statement")),
186    };
187    match statement {
188        ast::Statement::Query(query) => {
189            let mut binder = Binder::with(catalog, parameters, session);
190            let (root, _) = binder.bind_query(ast, query)?;
191            Ok(Bound::Query(finish(binder, root)?))
192        }
193        ast::Statement::CreateTable(index) => {
194            create_table(ast, catalog, parameters, session, index)
195        }
196        ast::Statement::CreateView(index) => create_view(ast, catalog, parameters, session, index),
197        ast::Statement::DropTable(index) => drop_table(ast, catalog, index),
198        ast::Statement::Insert(index) => insert(ast, catalog, parameters, session, index),
199        ast::Statement::Set(index) | ast::Statement::Reset(index) => {
200            setting(ast, catalog, parameters, session, index)
201        }
202        ast::Statement::Checkpoint => Ok(Bound::Checkpoint),
203        ast::Statement::Explain { query, analyze, statistics } => {
204            let mut binder = Binder::with(catalog, parameters, session);
205            let (root, _) = binder.bind_query(ast, query)?;
206            Ok(Bound::Explain { plan: finish(binder, root)?, analyze, statistics })
207        }
208    }
209}
210
211/// Parses and binds one statement, which is the whole front end in one call.
212///
213/// # Errors
214///
215/// Anything the parser or the binder reports.
216pub fn bind_statement_sql(sql: &str, catalog: &Catalog) -> Result<Bound> {
217    let ast = parse_ast(sql)?;
218    bind_statement(&ast, catalog)
219}
220
221/// Roots a binder's plan and checks it.
222fn finish(binder: Binder<'_>, root: rudb_plan::NodeRef) -> Result<Plan> {
223    let mut plan = binder.into_plan();
224    plan.set_root(root);
225    plan.validate()?;
226    Ok(plan)
227}
228
229fn create_table(
230    ast: &Ast,
231    catalog: &Catalog,
232    parameters: &Parameters,
233    session: &Session,
234    index: ast::CreateTableRef,
235) -> Result<Bound> {
236    let written = ast.create_table(index);
237    let parts: Vec<&str> = ast.name(written.name).collect();
238    let name = if written.temporary {
239        catalog.resolve_for_create_temporary(&parts)?
240    } else {
241        catalog.resolve_for_create(&parts)?
242    };
243    let defs = ast.column_defs(written.columns);
244    let (columns, source) = if written.query == NONE {
245        let mut columns = Vec::with_capacity(defs.len());
246        for def in defs {
247            let text = ast.string(def.ty);
248            if text.is_empty() {
249                return Err(Error::binder(format!(
250                    "Column \"{}\" was declared without a type",
251                    ast.string(def.name)
252                )));
253            }
254            let ty = LogicalType::parse(text)?;
255            let column = ast.string(def.name);
256            columns.push(if def.not_null {
257                Field::required(column, ty)
258            } else {
259                Field::new(column, ty)
260            });
261        }
262        (columns, None)
263    } else {
264        let mut binder = Binder::with(catalog, parameters, session);
265        let (root, scope) = binder.bind_query(ast, written.query)?;
266        if defs.len() > scope.len() {
267            // DuckDB's sentence, typo and all. A column list shorter than the query is fine and
268            // renames a prefix, so only this direction is an error.
269            return Err(Error::binder("Target table has more colum names than query result."));
270        }
271        let mut columns = Vec::with_capacity(scope.len());
272        for (at, column) in scope.columns.iter().enumerate() {
273            let named = match defs.get(at) {
274                Some(def) => ast.string(def.name).to_string(),
275                None => column.name.clone(),
276            };
277            columns.push(Field::new(named, column.ty.clone()));
278        }
279        if defs.is_empty() {
280            deduplicate(&mut columns);
281        }
282        (columns, Some(finish(binder, root)?))
283    };
284    duplicate_check(&columns)?;
285    Ok(Bound::CreateTable(CreateTable {
286        name,
287        columns,
288        source,
289        if_not_exists: written.if_not_exists,
290        or_replace: written.or_replace,
291    }))
292}
293
294/// Renames the columns a query repeated, which is what makes `CREATE TABLE t AS SELECT 1 AS a, 2 AS
295/// a` a table rather than an error.
296///
297/// A query is allowed to produce two columns of one name and `SELECT 1 AS a, 2 AS a` prints two
298/// columns called `a`, so a statement that turns a query into a table has to decide what to do with
299/// that, and DuckDB renames rather than refusing. The suffix is `_1`, then `_2`, counting up until
300/// the name is free, so a query that already has an `a_1` in it pushes the renamed column to `a_2`
301/// rather than colliding with it.
302///
303/// This only runs when the statement wrote no column list. With a list, even a short one, duckdb
304/// v1.4.1 takes the names as they come and a repeat is an error, so `CREATE TABLE t (z) AS SELECT 1
305/// AS a, 2 AS a` is a table of `z` and `a` and adding a third `a` to that query is a refusal.
306fn deduplicate(columns: &mut [Field]) {
307    for at in 0..columns.len() {
308        let taken = |name: &str, upto: usize, columns: &[Field]| {
309            columns[..upto].iter().any(|held| same_name(&held.name, name))
310        };
311        if !taken(&columns[at].name, at, columns) {
312            continue;
313        }
314        let mut suffix = 1;
315        let mut candidate = format!("{}_{suffix}", columns[at].name);
316        while taken(&candidate, at, columns) {
317            suffix += 1;
318            candidate = format!("{}_{suffix}", columns[at].name);
319        }
320        columns[at].name = candidate;
321    }
322}
323
324/// Binds a `CREATE VIEW`, which means binding the body and then throwing the plan away.
325///
326/// Throwing it away is the point. The body is bound here so that a view over a table that is not
327/// there is refused now rather than at the first select, and so that the column list can be checked
328/// against what the body actually produces. What the catalog keeps is the text, because a view
329/// follows the tables underneath it and a plan is a photograph of the day it was built.
330fn create_view(
331    ast: &Ast,
332    catalog: &Catalog,
333    parameters: &Parameters,
334    session: &Session,
335    index: ast::CreateViewRef,
336) -> Result<Bound> {
337    let written = ast.create_view(index);
338    let parts: Vec<&str> = ast.name(written.name).collect();
339    let name = if written.temporary {
340        catalog.resolve_for_create_temporary(&parts)?
341    } else {
342        catalog.resolve_for_create(&parts)?
343    };
344    let aliases: Vec<String> = ast.name(written.columns).map(str::to_string).collect();
345
346    let mut binder = Binder::with(catalog, parameters, session);
347    let (_, mut scope) = binder.bind_query(ast, written.query)?;
348    if aliases.len() > scope.len() {
349        return Err(Error::binder("More VIEW aliases than columns in query result"));
350    }
351    if !aliases.is_empty() {
352        let written: Vec<&str> = aliases.iter().map(String::as_str).collect();
353        scope.rename(&written, "unnamed_subquery")?;
354    }
355
356    Ok(Bound::CreateView(CreateView {
357        name,
358        sql: ast.string(written.sql).to_string(),
359        statement: deparse::create_view(ast, index),
360        aliases,
361        if_not_exists: written.if_not_exists,
362        or_replace: written.or_replace,
363        columns: scope.fields(),
364    }))
365}
366
367fn drop_table(ast: &Ast, catalog: &Catalog, index: ast::DropTableRef) -> Result<Bound> {
368    let written = ast.drop_table(index);
369    let kind = if written.view { Entry::View } else { Entry::Table };
370    let mut names = Vec::new();
371    for &name in ast.name_list(written.names) {
372        let parts: Vec<&str> = ast.name(name).collect();
373        // The statement said which of the two it meant, so a name that is not there is a missing
374        // one of those and not a missing table.
375        match catalog.resolve_as(&parts, kind) {
376            Ok(resolved) => names.push(resolved),
377            Err(error) if written.if_exists => drop(error),
378            Err(error) => return Err(error),
379        }
380    }
381    Ok(Bound::DropTable(DropTable { names, kind }))
382}
383
384/// Binds a `SET` or a `RESET`, which is resolving its value and nothing else.
385///
386/// The name is not checked here. The binder knows what tables exist and has no idea what settings
387/// exist, since a setting is a knob on the engine rather than an entry in a catalog, and a version
388/// of this that held the list would be the binder holding a copy of something it cannot enforce.
389fn setting(
390    ast: &Ast,
391    catalog: &Catalog,
392    parameters: &Parameters,
393    session: &Session,
394    index: ast::SettingRef,
395) -> Result<Bound> {
396    let written = ast.setting(index);
397    let name = ast.string(written.name).to_string();
398    let value = if written.value == NONE {
399        None
400    } else {
401        let mut binder = Binder::with(catalog, parameters, session);
402        let bound = binder.bind_setting_value(ast, written.value)?;
403        let Expr::Constant(value) = *binder.plan().expr(bound) else {
404            return Err(Error::not_implemented(format!(
405                "a value for {name} that is not a constant"
406            )));
407        };
408        Some(binder.plan().value(value).clone())
409    };
410    Ok(Bound::Setting(Setting { name, scope: written.scope, value, pragma: written.pragma }))
411}
412
413/// Sorts an insert's rows into the order the target table declared.
414///
415/// Returns the input unchanged when the statement supplies none of the declared columns, because
416/// every one of them is then a constant null and sorting on a constant is a sort that buys nothing
417/// and costs a pass. A statement that supplies some of them sorts on those: the declaration is
418/// about the order the rows are written in, and the columns that are there still order them.
419///
420/// The leading key carries the width. `date_trunc('month', d)` and `d` sort the same rows into the
421/// same fragments for any predicate a month wide or wider, and the difference is what happens
422/// inside a month: bucketed, the second key orders the whole month, which is the key locality the
423/// joins want and the reason the width is part of the declaration at all.
424fn clustered(
425    binder: &mut Binder<'_>,
426    input: rudb_plan::NodeRef,
427    scope: &crate::scope::Scope,
428    clustering: &Clustering,
429    targets: &[usize],
430    fields: &[Field],
431) -> Result<rudb_plan::NodeRef> {
432    let mut keys: Vec<SortKey> = Vec::with_capacity(clustering.columns().len());
433    for (at, &column) in clustering.columns().iter().enumerate() {
434        let Some(from) = targets.iter().position(|&target| target == column as usize) else {
435            continue;
436        };
437        let source = &scope.columns[from];
438        let expr = binder.plan_mut().add_expr(Expr::Column(source.binding), source.ty.clone());
439        // Cast to the column's own type before bucketing, since the source of a load is a file
440        // whose date column can arrive as a timestamp and `date_trunc` gives back the type it was
441        // handed. Sorting on a different type than the column stores would still be an order, but
442        // it would not be the order the declaration names.
443        let expr = binder.checked_cast_to(expr, &fields[column as usize].ty, false)?;
444        let expr =
445            if at == 0 { bucketed(binder, expr, clustering.width(), fields, column) } else { expr };
446        keys.push(SortKey { expr, descending: false, nulls_first: false });
447    }
448    if keys.is_empty() {
449        return Ok(input);
450    }
451    let keys = binder.plan_mut().add_sort_keys(&keys);
452    Ok(binder.plan_mut().add_node(Node::Sort { input, keys }))
453}
454
455/// The declaration with an automatic width turned into the bucket the incoming rows ask for.
456///
457/// A declaration that named no width says the bucket should come from how many rows a partition
458/// would hold, and this is the only place that number is in reach. The rows are the source's, not
459/// the target's: a load into an empty table has a target with nothing to count, and the whole case
460/// the rule exists for is the first load of a big table. So the count and the range come off the
461/// source's own zones, which is the Parquet footer for a file and the directory for a table, and
462/// both are already on the plan because the estimator wanted them.
463///
464/// Everything about this is best effort and that is by design. The three widths hold the same rows
465/// and answer the same queries, so guessing wrong costs some pruning or some key locality and
466/// cannot cost an answer. A source that is a join, a group by or a values list has no zones to read
467/// and gets [`Width::DEFAULT`], which is what the fixed default was before the rule existed.
468fn fitted(
469    binder: &Binder<'_>,
470    scope: &crate::scope::Scope,
471    clustering: &Clustering,
472    targets: &[usize],
473) -> Clustering {
474    if clustering.width() != Width::Auto {
475        return clustering.clone();
476    }
477    let Some(from) = targets.iter().position(|&target| target == clustering.partition() as usize)
478    else {
479        return clustering.fitted(0, 0);
480    };
481    let source = &scope.columns[from];
482    let Some(zones) = binder.plan().sole_zones() else {
483        return clustering.fitted(0, 0);
484    };
485    // By name, and off whichever store the plan reads rather than off the one this column is bound
486    // to. The binding points at the projection over the scan, since a load is a projection into the
487    // target's types, and following a binding back through a projection is the optimizer's job. A
488    // load reads one table or one file, so the store with bounds on it is the store the name is in.
489    let Some(at) = zones.column(&source.name) else {
490        return clustering.fitted(0, 0);
491    };
492    let rows = zones.surviving(&[]).unwrap_or(0);
493    let days = span(&zones.extreme(at, End::Low), &zones.extreme(at, End::High)).unwrap_or(0);
494    clustering.fitted(rows, days)
495}
496
497/// How many days a column covers, from the smallest and largest values in it.
498///
499/// `None` wherever the two do not make a span, which is a column that is entirely null, a store
500/// that could not fold its parts into one answer, and a pair of bounds that are not the same shape.
501/// All of them mean the same thing here, which is that there is nothing to divide the row count by.
502fn span(low: &Stat<ColumnBound>, high: &Stat<ColumnBound>) -> Option<u64> {
503    let (Stat::Known { value: low, .. }, Stat::Known { value: high, .. }) = (low, high) else {
504        return None;
505    };
506    let days = match (low, high) {
507        // A date is a day count already, which is the common case and the only exact one.
508        (ColumnBound::Int(low), ColumnBound::Int(high)) => high.checked_sub(*low)?,
509        // A timestamp is a count of seconds at whichever unit the column keeps, so the span is that
510        // difference divided by a day's worth of them. A scale wide enough to overflow the divisor
511        // is a column no calendar covers and falls out as no span at all.
512        (
513            ColumnBound::Scaled { unscaled: low, scale: at },
514            ColumnBound::Scaled { unscaled: high, scale: to },
515        ) if at == to => {
516            let day = 86_400_i128.checked_mul(10_i128.checked_pow(u32::from(*at))?)?;
517            high.checked_sub(*low)? / day
518        }
519        _ => return None,
520    };
521    u64::try_from(days).ok()
522}
523
524/// Wraps a sort key in the calendar bucket its declaration asked for.
525fn bucketed(
526    binder: &mut Binder<'_>,
527    expr: ExprRef,
528    width: Width,
529    fields: &[Field],
530    column: u32,
531) -> ExprRef {
532    if width == Width::Exact {
533        return expr;
534    }
535    let unit = binder.plan_mut().add_value(Value::Varchar(width.to_string().to_lowercase()));
536    let unit = binder.plan_mut().add_expr(Expr::Constant(unit), LogicalType::Varchar);
537    let args = binder.plan_mut().add_expr_list(&[unit, expr]);
538    let name = binder.plan_mut().intern("date_trunc");
539    let ty = fields[column as usize].ty.clone();
540    binder.plan_mut().add_expr(Expr::Function { name, args }, ty)
541}
542
543fn insert(
544    ast: &Ast,
545    catalog: &Catalog,
546    parameters: &Parameters,
547    session: &Session,
548    index: ast::InsertRef,
549) -> Result<Bound> {
550    let written = ast.insert(index);
551    let parts: Vec<&str> = ast.name(written.name).collect();
552    let name = catalog.resolve(&parts)?;
553    if catalog.entry(&name)? == Entry::View {
554        // The binary's sentence, article and all. A view has no rows of its own to append to, and
555        // an updatable view is a rule about rewriting the insert that neither database has.
556        return Err(Error::catalog(format!("{} is not an table", name.table)));
557    }
558    let target = catalog.table(&name)?;
559    let fields: Vec<Field> = target.columns().to_vec();
560    let clustering = target.clustering().cloned();
561
562    // Which table column each source column lands in. Without a column list that is the first n
563    // columns in order, and with one it is whatever the list says, which is also the check that
564    // the list names columns the table has and names none of them twice.
565    let targets: Vec<usize> = if written.columns.is_empty() {
566        (0..fields.len()).collect()
567    } else {
568        let mut targets = Vec::new();
569        for column in ast.name(written.columns) {
570            let at = fields.iter().position(|field| same_name(&field.name, column)).ok_or_else(
571                || {
572                    Error::binder(format!(
573                        "Table \"{}\" does not have a column named \"{column}\"",
574                        name.table
575                    ))
576                },
577            )?;
578            if targets.contains(&at) {
579                return Err(Error::binder(format!(
580                    "Column \"{column}\" is named twice in the same INSERT"
581                )));
582            }
583            targets.push(at);
584        }
585        targets
586    };
587
588    let mut binder = Binder::with(catalog, parameters, session);
589    let (root, scope) = binder.bind_query(ast, written.source)?;
590    if scope.len() != targets.len() {
591        return Err(Error::binder(format!(
592            "Table \"{}\" has {} columns but {} values were supplied",
593            name.table,
594            targets.len(),
595            scope.len()
596        )));
597    }
598
599    // A table that declared what order its rows go in gets the sort here, under the projection
600    // rather than over it, because a projection does not reorder rows and the bindings the sort
601    // keys need are the ones the query just produced. This is the whole of the loader honouring
602    // the declaration: the rows arrive at the writer in order and the per fragment ranges, which
603    // are built from whatever order arrives, come out narrow instead of each covering the table.
604    let root = match &clustering {
605        None => root,
606        Some(clustering) => {
607            // The width is settled here and not on the table. A declaration that left the bucket to
608            // the data is a standing instruction, so it stays on the table as one and every load
609            // answers it with the rows that load is carrying. What the sort needs is an answer, and
610            // that is what this is.
611            let fitted = fitted(&binder, &scope, clustering, &targets);
612            clustered(&mut binder, root, &scope, &fitted, &targets, &fields)?
613        }
614    };
615
616    // The projection that makes the source look exactly like the table. Every column the statement
617    // did not name becomes a null of the column's own type, so the append never has to know that a
618    // column list was written at all.
619    let mut exprs: Vec<ExprRef> = Vec::with_capacity(fields.len());
620    let mut names = Vec::with_capacity(fields.len());
621    for (at, field) in fields.iter().enumerate() {
622        let expr = match targets.iter().position(|&target| target == at) {
623            Some(from) => {
624                let column = &scope.columns[from];
625                let expr =
626                    binder.plan_mut().add_expr(Expr::Column(column.binding), column.ty.clone());
627                binder.checked_cast_to(expr, &field.ty, false)?
628            }
629            None => {
630                // A typed null rather than `add_constant`, which would give it the null type and
631                // make the column's type depend on whether a row happened to be inserted into it.
632                let value = binder.plan_mut().add_value(Value::Null);
633                binder.plan_mut().add_expr(Expr::Constant(value), field.ty.clone())
634            }
635        };
636        exprs.push(expr);
637        let interned = binder.plan_mut().intern(&field.name);
638        names.push(interned);
639    }
640    let exprs = binder.plan_mut().add_expr_list(&exprs);
641    let names = binder.plan_mut().add_name_list(&names);
642    let index = binder.fresh_index();
643    let root = binder.plan_mut().add_node(Node::Project { input: root, index, exprs, names });
644    Ok(Bound::Insert(Insert { name, source: finish(binder, root)? }))
645}