Skip to main content

inillucent_sql/
function.rs

1//! The built-in function registry: names, arities, and identities.
2//!
3//! Invariant: a function is recognised here or it does not exist. The binder
4//! resolves a name to one of these identities and refuses everything else with
5//! "no such function", so an unknown name fails at prepare time rather than
6//! part-way through a scan, and the VM never dispatches on a string.
7//!
8//! Arity is checked here too, because SQLite reports "wrong number of arguments
9//! to function abs()" from prepare rather than from execution.
10
11/// The names that exist in inillucent but need a component this build has not
12/// got.
13///
14/// **`embed` is the whole list, and it is here rather than in the registry
15/// because the registry is where it is absent** (task-1979, section 8.1, gap
16/// 12). `inillucent-search` registers `embed` only when the `embed` feature is
17/// compiled in, so on a build without it the name reaches the binder's
18/// "no such function" path and answered exit 1 - which says the caller
19/// misspelled something. The statement is spelled correctly and this build has
20/// not got the function, which is exactly what exit 3 means.
21///
22/// A build that *does* have `embed` never reaches here, because the registry
23/// resolves the name before the refusal is built. A machine that has the
24/// function and not the model is a third thing again and keeps its own status:
25/// `inillucent-search`'s `no_model` answers `invalid_state` and names
26/// `inillucent setup-embeddings`, because the component is installable and
27/// exit 3 would say the opposite.
28const NEEDS_A_COMPONENT: &[(&[u8], &str)] = &[
29    (
30        b"embed",
31        "embed(TEXT): this build has no embedding support compiled in",
32    ),
33    (
34        b"embed_tokens",
35        "embed_tokens(TEXT): this build has no embedding support compiled in",
36    ),
37    (
38        b"rerank",
39        "rerank(TEXT, TEXT): this build has no embedding support compiled in",
40    ),
41];
42
43/// Returns what a name needs, when the name is one this build left out.
44///
45/// @param name - the folded function name that did not resolve
46pub fn needs_a_component(name: &[u8]) -> Option<&'static str> {
47    NEEDS_A_COMPONENT
48        .iter()
49        .find(|(known, _)| *known == name)
50        .map(|(_, said)| *said)
51}
52
53/// A scalar built-in.
54#[derive(Clone, Copy, Debug, PartialEq, Eq)]
55pub enum ScalarFunc {
56    /// `abs(x)`
57    Abs,
58    /// `char(...)`
59    Char,
60    /// `coalesce(...)`
61    Coalesce,
62    /// The `coalesce` SQLite builds for a column that a `FULL` or `RIGHT JOIN`
63    /// with `USING` merges. It computes what `coalesce` does and has the
64    /// affinity of its first argument, which a written `coalesce` does not.
65    UsingCoalesce,
66    /// `concat(...)`
67    Concat,
68    /// `concat_ws(sep, ...)`
69    ConcatWs,
70    /// `glob(pattern, text)`
71    Glob,
72    /// `hex(x)`
73    Hex,
74    /// `ifnull(a, b)`
75    IfNull,
76    /// `iif(a, b, c)`
77    Iif,
78    /// `instr(haystack, needle)`
79    Instr,
80    /// `length(x)`
81    Length,
82    /// `like(pattern, text[, escape])`
83    Like,
84    /// `likelihood(x, y)`, `likely(x)` and `unlikely(x)`, which are no-ops.
85    Likelihood,
86    /// `lower(x)`
87    Lower,
88    /// `ltrim(x[, chars])`
89    LTrim,
90    /// `max(a, b, ...)`, the scalar form.
91    Max,
92    /// `min(a, b, ...)`, the scalar form.
93    Min,
94    /// `nullif(a, b)`
95    NullIf,
96    /// `quote(x)`
97    Quote,
98    /// `replace(text, from, to)`
99    Replace,
100    /// `round(x[, digits])`
101    Round,
102    /// `rtrim(x[, chars])`
103    RTrim,
104    /// `sign(x)`
105    Sign,
106    /// `substr(x, start[, length])`
107    Substr,
108    /// `trim(x[, chars])`
109    Trim,
110    /// `typeof(x)`
111    TypeOf,
112    /// `unhex(x[, chars])`
113    Unhex,
114    /// `unicode(x)`
115    Unicode,
116    /// `upper(x)`
117    Upper,
118    /// `zeroblob(n)`
119    ZeroBlob,
120    /// `printf(format, ...)` and `format(format, ...)`
121    Printf,
122    /// `octet_length(x)`
123    OctetLength,
124    /// `random()`
125    Random,
126    /// `randomblob(n)`
127    RandomBlob,
128    /// `changes()`
129    Changes,
130    /// `total_changes()`
131    TotalChanges,
132    /// `last_insert_rowid()`
133    LastInsertRowid,
134    /// `sqlite_source_id()`
135    SourceId,
136    /// `fts5_source_id()`
137    Fts5SourceId,
138    /// `sqlite_version()`
139    Version,
140    /// `vector_distance_cos(a, b)`, the cosine distance between two vectors.
141    ///
142    /// **Not a SQLite function, and the first one this engine adds.** pgvector
143    /// spells it `a <=> b`; the whole point of Phase 2's Part 7 is that a
144    /// vector is a value a `SELECT` can order by, and an operator that is sugar
145    /// for a function needs the function to exist first. A vector is a blob of
146    /// little-endian `f32`, which is what `inillucent_search` already stores and
147    /// what `vector_distance_l2` and `vector_dot` read too.
148    VectorDistanceCos,
149    /// `vector_distance_l2(a, b)`, the Euclidean distance between two vectors.
150    VectorDistanceL2,
151    /// `vector_dot(a, b)`, the dot product of two vectors.
152    ///
153    /// Negated relative to pgvector's `<#>`, which answers the *negative* inner
154    /// product so that a smaller number is a better match. This answers the dot
155    /// product itself, because a function named `dot` that returned its negative
156    /// would be a trap; the ordering sugar negates where it needs to.
157    VectorDot,
158    /// `l1_distance(a, b)`, the taxicab distance, spelled `a <+> b`.
159    VectorDistanceL1,
160    /// `hamming_distance(a, b)`, how many components differ.
161    ///
162    /// pgvector defines it over its `bit` type and spells it `a <~> b`. Here a
163    /// bit vector is the blob `binary_quantize` produces, and the distance is
164    /// the population count of the two blobs' exclusive-or - which is the same
165    /// number, computed the same way, over the representation this engine has.
166    VectorDistanceHamming,
167    /// `jaccard_distance(a, b)`, one minus the overlap, spelled `a <%> b`.
168    VectorDistanceJaccard,
169    /// `vector_dims(a)`, how many components a vector has.
170    VectorDims,
171    /// `vector_norm(a)`, its Euclidean length.
172    VectorNorm,
173    /// `l2_normalize(a)`, the same direction with length one.
174    VectorNormalize,
175    /// `binary_quantize(a)`, one bit per component: set when it is positive.
176    VectorQuantize,
177    /// `subvector(a, start, count)`, a slice, counted from one.
178    VectorSlice,
179    /// `vector_add(a, b)`, component by component.
180    ///
181    /// **A function rather than `+`, and that is a compatibility choice rather
182    /// than a shortcut.** pgvector can overload `+` because a `vector` is a
183    /// distinct type in PostgreSQL; here a vector is a blob, and SQLite says
184    /// that a blob in arithmetic is zero. Overloading the operator for every
185    /// blob would change the answer to `x'00' + x'00'` from `0` to a blob,
186    /// which is a difference every application that adds two blobs would see.
187    ///
188    /// **The operators were given back, on the one condition that keeps
189    /// both answers.** `a + b` binds to this function when a side reads a
190    /// column *declared* `VECTOR(n)` - which is the same thing PostgreSQL is
191    /// using, a declared type - and stays SQLite's arithmetic otherwise. So
192    /// `x'00' + x'00'` is still `0` and `v + v` over a vector column is a
193    /// vector.
194    VectorAdd,
195    /// `vector_sub(a, b)`, component by component.
196    VectorSubtract,
197    /// `vector_mul(a, b)`, component by component.
198    VectorMultiply,
199    /// `vector_concat(a, b)`, one vector after the other.
200    VectorConcat,
201    /// `geopoly_area(P)`, the signed area a polygon encloses.
202    ///
203    /// **The `geopoly` surface is thirteen functions and one aggregate**, and
204    /// they are listed here individually rather than folded into one
205    /// `Geopoly(kind)` variant because arity checking reads this enum: they
206    /// take one, two, three, four, seven and any number of arguments, and a
207    /// single variant could not say so.
208    GeopolyArea,
209    /// `geopoly_blob(P)`, the stored form of a polygon.
210    GeopolyBlob,
211    /// `geopoly_json(P)`, the GeoJSON form.
212    GeopolyJson,
213    /// `geopoly_svg(P, ...)`, an SVG `<polyline>` with the extra arguments
214    /// written into the tag.
215    GeopolySvg,
216    /// `geopoly_within(P1, P2)`, whether the second is inside the first.
217    GeopolyWithin,
218    /// `geopoly_contains_point(P, X, Y)`, where a point sits.
219    GeopolyContainsPoint,
220    /// `geopoly_overlap(P1, P2)`, how two polygons meet.
221    GeopolyOverlap,
222    /// `geopoly_debug(X)`, which answers nothing.
223    ///
224    /// It switches on the reference's own tracing, which only exists in a build
225    /// made with `GEOPOLY_ENABLE_DEBUG`; in every other build it reads its
226    /// argument and returns nothing at all. That is what this does, and it is
227    /// registered because a name the reference resolves and this engine does
228    /// not is a difference an application can see.
229    GeopolyDebug,
230    /// `geopoly_bbox(P)`, the bounding box as a four-sided polygon.
231    GeopolyBbox,
232    /// `geopoly_xform(P, A, B, C, D, E, F)`, an affine transform.
233    GeopolyXform,
234    /// `geopoly_regular(X, Y, R, N)`, a regular polygon.
235    GeopolyRegular,
236    /// `geopoly_ccw(P)`, the same ring wound counter-clockwise.
237    GeopolyCcw,
238    /// `unknown(...)`, which answers NULL to anything.
239    ///
240    /// SQLite registers it, lists it in `function_list`, and returns NULL from
241    /// it whatever it is given. It is here because a name the reference resolves
242    /// and this engine does not is a difference an application can see.
243    Unknown,
244    /// `subtype(x)`, the tag a function attached to its answer.
245    Subtype,
246    /// `unistr(x)`, which expands `\uXXXX` and `\UXXXXXXXX` escapes.
247    Unistr,
248    /// `unistr_quote(x)`, `quote()` with the control characters escaped.
249    UnistrQuote,
250    /// `sqlite_compileoption_used(name)`
251    CompileOptionUsed,
252    /// `sqlite_compileoption_get(n)`
253    CompileOptionGet,
254    /// `sqlite_log(code, message)`, which writes to the log and answers NULL.
255    Log,
256    /// `load_extension(path[, entry])`
257    LoadExtension,
258    /// `regexp(pattern, subject)`, which is what `X REGEXP Y` calls.
259    Regexp,
260    /// `sqlar_compress(X)`, a blob compressed if that makes it smaller.
261    ///
262    /// **The archive format's own rule, and it is why this is not just a
263    /// compressor.** A row of a `.sqlar` table holds either a zlib stream or
264    /// the raw bytes, and which one is decided by whichever is shorter; the
265    /// stored `sz` column is what tells the two apart on the way back. So a
266    /// value that does not compress is stored as it stands, and a value that is
267    /// not a blob at all is returned unchanged, type and all.
268    SqlarCompress,
269    /// `sqlar_uncompress(Z, SZ)`, the inverse.
270    ///
271    /// `SZ` is the size the row claims the content is. When it equals the
272    /// blob's own length the blob *is* the content and is returned unchanged,
273    /// which is how the format says "this one was stored raw".
274    SqlarUncompress,
275    /// `sqlite_offset(X)`, where in the file the row holding X is.
276    ///
277    /// **The page, not the record, and that is the whole of the difference.**
278    /// SQLite reports the byte offset of the *record* a value would be read
279    /// from, because a row there is one contiguous run of bytes. A leaf here is
280    /// PAX: each column is its own run, so one row occupies several places on
281    /// its page and there is no single offset for it. What is reported is the
282    /// offset of the page, which is where the value is genuinely read from.
283    ///
284    /// Folded to its answer by the physical pass, like `rtreecheck`, because it
285    /// is a question about a *tree* rather than about a value.
286    Offset,
287    /// `rtreedepth(X)`, the depth stored at the front of an R-Tree node.
288    RTreeDepth,
289    /// `rtreenode(D, X)`, an R-Tree node rendered as a readable list.
290    RTreeNode,
291    /// `rtreecheck(T)`, an integrity check over one R-Tree table.
292    ///
293    /// **Answered where the table is reachable, which is not here.** A scalar
294    /// is handed values and nothing else; this one is about a *table*, so the
295    /// physical pass folds it to its answer while it still has the catalog,
296    /// and what reaches the evaluator is already the text. Running once per
297    /// preparation rather than once per row is also what it means: the
298    /// argument is a table name, so the answer cannot vary down a column.
299    RTreeCheck,
300}
301
302/// An aggregate built-in.
303#[derive(Clone, Copy, Debug, PartialEq, Eq)]
304pub enum AggregateFunc {
305    /// `count(x)` and `count(*)`
306    Count,
307    /// `sum(x)`
308    Sum,
309    /// `total(x)`
310    Total,
311    /// `avg(x)`
312    Avg,
313    /// `min(x)`
314    Min,
315    /// `max(x)`
316    Max,
317    /// `group_concat(x[, sep])` and `string_agg(x, sep)`
318    GroupConcat,
319    /// `json_group_array(x)`
320    JsonGroupArray,
321    /// `jsonb_group_array(x)`
322    JsonbGroupArray,
323    /// `json_group_object(label, x)`
324    JsonGroupObject,
325    /// `jsonb_group_object(label, x)`
326    JsonbGroupObject,
327    /// `median(x)`, which is `percentile_cont(x, 0.5)` under a shorter name.
328    Median,
329    /// `geopoly_group_bbox(P)`, the box that holds every polygon in the group.
330    GeopolyGroupBbox,
331    /// `sum(v)` and `total(v)` over a vector column, component by component.
332    ///
333    /// Not a name a caller writes: the binder picks it when `sum`'s argument
334    /// reads a vector, because that is where the argument's type is known.
335    VectorSum,
336    /// `avg(v)` over a vector column, component by component.
337    VectorAvg,
338    /// `percentile(x, p)`, where `p` runs 0 to 100.
339    Percentile,
340    /// `percentile_cont(x, f)`, where `f` runs 0 to 1 and the answer is
341    /// interpolated between the two rows it falls between.
342    PercentileCont,
343    /// `percentile_disc(x, f)`, which answers one of the rows rather than a
344    /// value between two of them.
345    PercentileDisc,
346    /// An aggregate an application registered, named beside the call.
347    ///
348    /// The name is not in here because this enum is `Copy` and travels through
349    /// the program's operands; it rides in `AggregateCall` instead.
350    External,
351}
352
353/// What a registered function promises about itself.
354///
355/// It lives here, below `inillucent-ext`, because two different layers have to
356/// read the same promise: `inillucent_ext::registry::Registry` records it when
357/// an application registers a function, and the binder enforces it when a
358/// schema names one. `inillucent-ext` re-exports this type, so a registrant
359/// writes `inillucent_ext::registry::FunctionFlags` exactly as before.
360#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
361pub struct FunctionFlags {
362    /// The function may only be called from top-level SQL, never from a
363    /// schema: not from a `DEFAULT`, a `CHECK`, a generated column, an index
364    /// expression, a partial-index predicate, a view or a trigger.
365    ///
366    /// [`FunctionFlags::external`] sets this, because the safe assumption about
367    /// code somebody else wrote is that it does something. **It is not what the
368    /// `Default` derive gives**, which is every flag false: a registrant who
369    /// writes `..FunctionFlags::default()` gets a function a schema may name.
370    /// That is the hole `embed` was registered through (task-1969, 7.4), and
371    /// `inillucent_ext::registry::UserFunction::external` is the constructor to
372    /// reach for instead.
373    pub direct_only: bool,
374    /// The function does nothing an ordinary expression could not: no side
375    /// effects, no file access, no dependence on anything but its arguments.
376    pub innocuous: bool,
377    /// The function returns the same answer for the same arguments within one
378    /// statement, so the planner may call it once.
379    pub deterministic: bool,
380}
381
382impl FunctionFlags {
383    /// Returns the flags a built-in carries: safe for a schema to call.
384    pub fn builtin() -> FunctionFlags {
385        FunctionFlags {
386            direct_only: false,
387            innocuous: true,
388            deterministic: true,
389        }
390    }
391
392    /// Returns the flags anything registered from outside carries by default.
393    pub fn external() -> FunctionFlags {
394        FunctionFlags {
395            direct_only: true,
396            innocuous: false,
397            deterministic: false,
398        }
399    }
400}
401
402/// Which context a name is being resolved from.
403#[derive(Clone, Copy, Debug, PartialEq, Eq)]
404pub enum CallSite {
405    /// The statement an application submitted.
406    Statement,
407    /// A `DEFAULT`, `CHECK`, generated column, index expression, partial-index
408    /// predicate, view or trigger stored in the schema.
409    Schema,
410}
411
412/// Returns why a schema may not call this function, or nothing when it may.
413///
414/// **One rule, read by two layers (task-1972).** `Registry::authorize_function`
415/// wraps the answer in a `DbError` for an application that asks the registry
416/// directly, and the binder wraps it in a `ParseError` for the statement it is
417/// compiling. Writing the rule twice is how the two would eventually disagree,
418/// and the half nobody exercised would be the permissive one.
419///
420/// The rule reads the same way SQLite's does: a direct-only function is never
421/// callable from a schema; anything else is callable from a schema only when
422/// the connection trusts the schema or the function is innocuous.
423///
424/// @param flags - what the function promises about itself
425/// @param site - where the call was written
426/// @param trusted_schema - whether the connection trusts the schema it read
427pub fn schema_refusal(
428    flags: FunctionFlags,
429    site: CallSite,
430    trusted_schema: bool,
431) -> Option<&'static str> {
432    if site == CallSite::Statement {
433        return None;
434    }
435    if flags.direct_only {
436        return Some("may only be used from top-level SQL");
437    }
438    if trusted_schema || flags.innocuous {
439        return None;
440    }
441    Some("is not allowed in a schema")
442}
443
444/// A function an application registered, as the binder needs to see it.
445///
446/// Only what resolution needs: a name, how many arguments it takes, whether it
447/// reduces a group, and what it promises about itself. What it *does* is the
448/// machine's business.
449///
450/// **The flags are here because the binder is where the promise is kept
451/// (task-1972).** `Registry::authorize_function` had no caller, so
452/// `direct_only`, `innocuous` and `PRAGMA trusted_schema` were a policy with a
453/// passing unit test and no effect on the engine: a `CHECK`, an index
454/// expression or a generated column could name any registered function whatever
455/// its flags. `inillucent-sql` sits below `inillucent-ext` and cannot reach the
456/// registry, so what the registry knows travels down here with the name.
457#[derive(Clone, Debug, PartialEq, Eq)]
458pub struct ExternalFunction {
459    /// The folded name.
460    pub name: Vec<u8>,
461    /// How many arguments it takes, or -1 for any number.
462    pub arity: i32,
463    /// Whether it reduces a group rather than a row.
464    pub aggregate: bool,
465    /// What it promises about itself, which decides whether a schema may name
466    /// it.
467    pub flags: FunctionFlags,
468}
469
470impl ExternalFunction {
471    /// Returns whether this registration answers a call with this many
472    /// arguments.
473    pub fn accepts(&self, argc: usize) -> bool {
474        self.arity < 0 || self.arity as usize == argc
475    }
476}
477
478/// Returns the registration that answers a call, preferring an exact arity.
479///
480/// SQLite resolves the same way: a function registered for exactly this many
481/// arguments wins over one registered for any number, so an application can
482/// define both a fast two-argument form and a general one.
483pub fn lookup_external<'a>(
484    functions: &'a [ExternalFunction],
485    name: &[u8],
486    argc: usize,
487) -> Option<&'a ExternalFunction> {
488    let folded = name.to_ascii_lowercase();
489    functions
490        .iter()
491        .find(|function| function.name == folded && function.arity as usize == argc)
492        .or_else(|| {
493            functions
494                .iter()
495                .find(|function| function.name == folded && function.arity < 0)
496        })
497}
498
499/// A date or time built-in.
500#[derive(Clone, Copy, Debug, PartialEq, Eq)]
501pub enum TimeFunc {
502    /// `date(...)`
503    Date,
504    /// `time(...)`
505    Time,
506    /// `datetime(...)`
507    DateTime,
508    /// `julianday(...)`
509    JulianDay,
510    /// `unixepoch(...)`
511    UnixEpoch,
512    /// `strftime(format, ...)`
513    StrfTime,
514    /// `timediff(a, b)`
515    TimeDiff,
516}
517
518/// Returns the date or time function a folded name spells.
519pub fn lookup_time(folded: &[u8]) -> Option<TimeFunc> {
520    let func = match folded {
521        b"date" => TimeFunc::Date,
522        b"time" => TimeFunc::Time,
523        b"datetime" => TimeFunc::DateTime,
524        b"julianday" => TimeFunc::JulianDay,
525        b"unixepoch" => TimeFunc::UnixEpoch,
526        b"strftime" => TimeFunc::StrfTime,
527        b"timediff" => TimeFunc::TimeDiff,
528        _ => return None,
529    };
530    Some(func)
531}
532
533/// A math built-in.
534///
535/// They are their own enum rather than more `ScalarFunc` variants because they
536/// are a compile-time option in SQLite (`SQLITE_ENABLE_MATH_FUNCTIONS`) and
537/// share one rule the others do not: an argument outside the domain is NULL
538/// rather than an error or a NaN.
539#[derive(Clone, Copy, Debug, PartialEq, Eq)]
540pub enum MathFunc {
541    /// `acos(x)`
542    Acos,
543    /// `acosh(x)`
544    Acosh,
545    /// `asin(x)`
546    Asin,
547    /// `asinh(x)`
548    Asinh,
549    /// `atan(x)`
550    Atan,
551    /// `atan2(y, x)`
552    Atan2,
553    /// `atanh(x)`
554    Atanh,
555    /// `ceil(x)` and `ceiling(x)`
556    Ceil,
557    /// `cos(x)`
558    Cos,
559    /// `cosh(x)`
560    Cosh,
561    /// `degrees(x)`
562    Degrees,
563    /// `exp(x)`
564    Exp,
565    /// `floor(x)`
566    Floor,
567    /// `ln(x)`
568    Ln,
569    /// `log(x)` base 10, or `log(b, x)` base b.
570    Log,
571    /// `log10(x)`
572    Log10,
573    /// `log2(x)`
574    Log2,
575    /// `mod(x, y)`
576    Mod,
577    /// `pi()`
578    Pi,
579    /// `pow(x, y)` and `power(x, y)`
580    Pow,
581    /// `radians(x)`
582    Radians,
583    /// `sin(x)`
584    Sin,
585    /// `sinh(x)`
586    Sinh,
587    /// `sqrt(x)`
588    Sqrt,
589    /// `tan(x)`
590    Tan,
591    /// `tanh(x)`
592    Tanh,
593    /// `trunc(x)`
594    Trunc,
595}
596
597impl MathFunc {
598    /// Returns how many arguments the function takes, as `(least, most)`.
599    pub fn arity(self) -> (usize, usize) {
600        match self {
601            MathFunc::Pi => (0, 0),
602            MathFunc::Atan2 | MathFunc::Mod | MathFunc::Pow => (2, 2),
603            MathFunc::Log => (1, 2),
604            _ => (1, 1),
605        }
606    }
607}
608
609/// Returns the math function a folded name spells.
610pub fn lookup_math(folded: &[u8]) -> Option<MathFunc> {
611    let func = match folded {
612        b"acos" => MathFunc::Acos,
613        b"acosh" => MathFunc::Acosh,
614        b"asin" => MathFunc::Asin,
615        b"asinh" => MathFunc::Asinh,
616        b"atan" => MathFunc::Atan,
617        b"atan2" => MathFunc::Atan2,
618        b"atanh" => MathFunc::Atanh,
619        b"ceil" | b"ceiling" => MathFunc::Ceil,
620        b"cos" => MathFunc::Cos,
621        b"cosh" => MathFunc::Cosh,
622        b"degrees" => MathFunc::Degrees,
623        b"exp" => MathFunc::Exp,
624        b"floor" => MathFunc::Floor,
625        b"ln" => MathFunc::Ln,
626        b"log" => MathFunc::Log,
627        b"log10" => MathFunc::Log10,
628        b"log2" => MathFunc::Log2,
629        b"mod" => MathFunc::Mod,
630        b"pi" => MathFunc::Pi,
631        b"pow" | b"power" => MathFunc::Pow,
632        b"radians" => MathFunc::Radians,
633        b"sin" => MathFunc::Sin,
634        b"sinh" => MathFunc::Sinh,
635        b"sqrt" => MathFunc::Sqrt,
636        b"tan" => MathFunc::Tan,
637        b"tanh" => MathFunc::Tanh,
638        b"trunc" => MathFunc::Trunc,
639        _ => return None,
640    };
641    Some(func)
642}
643
644/// A window function that is not an aggregate.
645///
646/// The aggregates are the same functions in a different frame, so they are not
647/// listed again here: `sum(x) OVER (...)` is `AggregateFunc::Sum` with a frame,
648/// and giving it a second spelling would mean two implementations of `sum`.
649#[derive(Clone, Copy, Debug, PartialEq, Eq)]
650pub enum WindowFunc {
651    /// `row_number()`
652    RowNumber,
653    /// `rank()`
654    Rank,
655    /// `dense_rank()`
656    DenseRank,
657    /// `percent_rank()`
658    PercentRank,
659    /// `cume_dist()`
660    CumeDist,
661    /// `ntile(n)`
662    Ntile,
663    /// `lag(x[, offset[, default]])`
664    Lag,
665    /// `lead(x[, offset[, default]])`
666    Lead,
667    /// `first_value(x)`
668    FirstValue,
669    /// `last_value(x)`
670    LastValue,
671    /// `nth_value(x, n)`
672    NthValue,
673}
674
675impl WindowFunc {
676    /// Returns how many arguments the function takes, as `(least, most)`.
677    pub fn arity(self) -> (usize, usize) {
678        match self {
679            WindowFunc::RowNumber
680            | WindowFunc::Rank
681            | WindowFunc::DenseRank
682            | WindowFunc::PercentRank
683            | WindowFunc::CumeDist => (0, 0),
684            WindowFunc::Ntile | WindowFunc::FirstValue | WindowFunc::LastValue => (1, 1),
685            WindowFunc::NthValue => (2, 2),
686            WindowFunc::Lag | WindowFunc::Lead => (1, 3),
687        }
688    }
689}
690
691/// Returns the window function a folded name spells.
692pub fn lookup_window(folded: &[u8]) -> Option<WindowFunc> {
693    let func = match folded {
694        b"row_number" => WindowFunc::RowNumber,
695        b"rank" => WindowFunc::Rank,
696        b"dense_rank" => WindowFunc::DenseRank,
697        b"percent_rank" => WindowFunc::PercentRank,
698        b"cume_dist" => WindowFunc::CumeDist,
699        b"ntile" => WindowFunc::Ntile,
700        b"lag" => WindowFunc::Lag,
701        b"lead" => WindowFunc::Lead,
702        b"first_value" => WindowFunc::FirstValue,
703        b"last_value" => WindowFunc::LastValue,
704        b"nth_value" => WindowFunc::NthValue,
705        _ => return None,
706    };
707    Some(func)
708}
709
710/// A JSON built-in.
711///
712/// They are their own enum for the same reason the math functions are: they
713/// share a rule none of the others has. Every one of them can fail - a document
714/// that will not parse is an error and not a NULL - and every one of them cares
715/// whether its arguments are already JSON, which is a property of the value
716/// rather than of the expression. Folding them into `ScalarFunc` would push
717/// both facts onto eighty functions that have neither.
718///
719/// The `b` spellings return the binary format rather than text. They are
720/// separate identities rather than a flag because `json_extract` and
721/// `jsonb_extract` differ in more than their output: the text form answers a
722/// SQL value for a leaf and the binary form answers a document.
723#[derive(Clone, Copy, Debug, PartialEq, Eq)]
724pub enum JsonFunc {
725    /// `json(X)`
726    Json,
727    /// `jsonb(X)`
728    Jsonb,
729    /// `json_array(...)`
730    Array,
731    /// `jsonb_array(...)`
732    ArrayB,
733    /// `json_array_length(X[, P])`
734    ArrayLength,
735    /// `json_error_position(X)`
736    ErrorPosition,
737    /// `json_extract(X, P, ...)`
738    Extract,
739    /// `jsonb_extract(X, P, ...)`
740    ExtractB,
741    /// The `->` operator.
742    Arrow,
743    /// The `->>` operator.
744    ArrowShift,
745    /// `json_insert(X, P, V, ...)`
746    Insert,
747    /// `jsonb_insert(X, P, V, ...)`
748    InsertB,
749    /// `json_object(...)`
750    Object,
751    /// `jsonb_object(...)`
752    ObjectB,
753    /// `json_patch(T, P)`
754    Patch,
755    /// `jsonb_patch(T, P)`
756    PatchB,
757    /// `json_pretty(X[, indent])`
758    Pretty,
759    /// `json_remove(X, P, ...)`
760    Remove,
761    /// `jsonb_remove(X, P, ...)`
762    RemoveB,
763    /// `json_replace(X, P, V, ...)`
764    Replace,
765    /// `jsonb_replace(X, P, V, ...)`
766    ReplaceB,
767    /// `json_set(X, P, V, ...)`
768    Set,
769    /// `jsonb_set(X, P, V, ...)`
770    SetB,
771    /// `json_type(X[, P])`
772    Type,
773    /// `json_valid(X[, flags])`
774    Valid,
775    /// `json_quote(X)`
776    Quote,
777    /// `json_array_insert(X, P, V, ...)`
778    ArrayInsert,
779    /// `jsonb_array_insert(X, P, V, ...)`
780    ArrayInsertB,
781    /// The `value` of a `json_each` or `json_tree` row, marked as JSON when
782    /// the row's `type` is an array or an object.
783    ///
784    /// Not a name anybody can call: the binder puts it around such a column
785    /// where a JSON function reads it. SQLite gives that value the JSON
786    /// subtype for a container, so `json_group_array(value)` over the
787    /// members of `[{"a":1}]` rebuilds `[{"a":1}]`. Read as plain text here,
788    /// every nested object came back as a quoted string.
789    WalkValue,
790}
791
792impl JsonFunc {
793    /// Returns how many arguments the function takes, as `(least, most)`.
794    ///
795    /// `usize::MAX` as the upper bound means "any number", which the editing
796    /// functions further restrict to an odd count in
797    /// [`JsonFunc::arity_ok`] - a rule a pair of bounds cannot express.
798    pub fn arity(self) -> (usize, usize) {
799        match self {
800            JsonFunc::Json | JsonFunc::Jsonb | JsonFunc::ErrorPosition | JsonFunc::Quote => (1, 1),
801            JsonFunc::WalkValue => (2, 2),
802            JsonFunc::Array | JsonFunc::ArrayB | JsonFunc::Object | JsonFunc::ObjectB => {
803                (0, usize::MAX)
804            }
805            JsonFunc::ArrayLength | JsonFunc::Type | JsonFunc::Valid | JsonFunc::Pretty => (1, 2),
806            JsonFunc::Patch | JsonFunc::PatchB | JsonFunc::Arrow | JsonFunc::ArrowShift => (2, 2),
807            // **SQLite registers these with any argument count and checks it when the
808            // function runs.** `json_extract()` and `json_extract(X)` answer NULL, an
809            // editing function with an even count fails with `json_set() needs an odd
810            // number of arguments`, and `json_object('a')` fails with `json_object()
811            // requires an even number of arguments`. Refusing the count while binding
812            // answered a parse error with `wrong number of arguments to function`
813            // where SQLite answers a runtime error with its own sentence.
814            JsonFunc::Extract
815            | JsonFunc::ExtractB
816            | JsonFunc::Remove
817            | JsonFunc::RemoveB
818            | JsonFunc::Insert
819            | JsonFunc::InsertB
820            | JsonFunc::Replace
821            | JsonFunc::ReplaceB
822            | JsonFunc::Set
823            | JsonFunc::SetB
824            | JsonFunc::ArrayInsert
825            | JsonFunc::ArrayInsertB => (0, usize::MAX),
826        }
827    }
828
829    /// Returns whether an argument count is legal for this function.
830    ///
831    /// The counts that depend on parity, `json_object` and the editing functions,
832    /// are checked when the function runs, because that is where SQLite checks them
833    /// and the error it gives is a runtime one.
834    pub fn arity_ok(self, count: usize) -> bool {
835        let (least, most) = self.arity();
836        count >= least && count <= most
837    }
838
839    /// Returns whether the function answers the binary format.
840    pub fn is_binary(self) -> bool {
841        matches!(
842            self,
843            JsonFunc::Jsonb
844                | JsonFunc::ArrayB
845                | JsonFunc::ExtractB
846                | JsonFunc::InsertB
847                | JsonFunc::ObjectB
848                | JsonFunc::PatchB
849                | JsonFunc::RemoveB
850                | JsonFunc::ReplaceB
851                | JsonFunc::SetB
852        )
853    }
854
855    /// Returns whether this function's first argument names a document to be
856    /// read, rather than a value to be embedded or quoted.
857    ///
858    /// The distinction an executor's document-cache optimisation needs: it
859    /// may only substitute a pre-parsed JSONB blob for the first argument
860    /// when that argument *is* the document a call reads, such as `X` in
861    /// `json_extract(X, P)`. `json_array`, `json_object` and `json_quote`
862    /// take that same position as a **value** - one that merely happens to
863    /// look like JSON is still meant to be embedded or quoted as a string,
864    /// per the subtype rule this module's own doc comment states. Handing
865    /// them a blob instead answered "JSON cannot hold BLOB values" for a
866    /// perfectly ordinary unmarked string, which is what
867    /// `json_array('[1]')` did before this existed. `Valid` reads its
868    /// argument as a document too, but is excluded by its caller for the
869    /// unrelated reason that substituting a re-encoded blob changes what its
870    /// flags answer about the original text.
871    pub fn first_argument_is_a_document(self) -> bool {
872        !matches!(
873            self,
874            JsonFunc::Array
875                | JsonFunc::ArrayB
876                | JsonFunc::Object
877                | JsonFunc::ObjectB
878                | JsonFunc::Quote
879                | JsonFunc::WalkValue
880        )
881    }
882}
883
884/// Returns the JSON function a folded name spells.
885pub fn lookup_json(folded: &[u8]) -> Option<JsonFunc> {
886    let func = match folded {
887        b"json" => JsonFunc::Json,
888        b"jsonb" => JsonFunc::Jsonb,
889        b"json_array" => JsonFunc::Array,
890        b"jsonb_array" => JsonFunc::ArrayB,
891        b"json_array_length" => JsonFunc::ArrayLength,
892        b"json_error_position" => JsonFunc::ErrorPosition,
893        b"json_extract" => JsonFunc::Extract,
894        // **The operators are function names too.** SQLite registers `->` and
895        // `->>` as ordinary two-argument functions, so `"->"(a, b)` binds and
896        // `pragma_function_list` reports them. The parser lowered the operators
897        // here already; only the spellings were missing, which made this engine
898        // report two fewer functions than it has and refuse a call SQLite
899        // answers.
900        b"->" => JsonFunc::Arrow,
901        b"->>" => JsonFunc::ArrowShift,
902        b"jsonb_extract" => JsonFunc::ExtractB,
903        b"json_array_insert" => JsonFunc::ArrayInsert,
904        b"jsonb_array_insert" => JsonFunc::ArrayInsertB,
905        b"json_insert" => JsonFunc::Insert,
906        b"jsonb_insert" => JsonFunc::InsertB,
907        b"json_object" => JsonFunc::Object,
908        b"jsonb_object" => JsonFunc::ObjectB,
909        b"json_patch" => JsonFunc::Patch,
910        b"jsonb_patch" => JsonFunc::PatchB,
911        b"json_pretty" => JsonFunc::Pretty,
912        b"json_remove" => JsonFunc::Remove,
913        b"jsonb_remove" => JsonFunc::RemoveB,
914        b"json_replace" => JsonFunc::Replace,
915        b"jsonb_replace" => JsonFunc::ReplaceB,
916        b"json_set" => JsonFunc::Set,
917        b"jsonb_set" => JsonFunc::SetB,
918        b"json_type" => JsonFunc::Type,
919        b"json_valid" => JsonFunc::Valid,
920        b"json_quote" => JsonFunc::Quote,
921        _ => return None,
922    };
923    Some(func)
924}
925
926/// Returns the scalar function a folded name spells.
927pub fn lookup_scalar(folded: &[u8]) -> Option<ScalarFunc> {
928    let func = match folded {
929        b"abs" => ScalarFunc::Abs,
930        b"char" => ScalarFunc::Char,
931        b"coalesce" => ScalarFunc::Coalesce,
932        b"concat" => ScalarFunc::Concat,
933        b"concat_ws" => ScalarFunc::ConcatWs,
934        b"glob" => ScalarFunc::Glob,
935        b"hex" => ScalarFunc::Hex,
936        b"ifnull" => ScalarFunc::IfNull,
937        b"iif" | b"if" => ScalarFunc::Iif,
938        b"instr" => ScalarFunc::Instr,
939        b"length" => ScalarFunc::Length,
940        b"like" => ScalarFunc::Like,
941        b"likelihood" | b"likely" | b"unlikely" => ScalarFunc::Likelihood,
942        b"lower" => ScalarFunc::Lower,
943        b"ltrim" => ScalarFunc::LTrim,
944        b"max" => ScalarFunc::Max,
945        b"min" => ScalarFunc::Min,
946        b"nullif" => ScalarFunc::NullIf,
947        b"quote" => ScalarFunc::Quote,
948        b"replace" => ScalarFunc::Replace,
949        b"round" => ScalarFunc::Round,
950        b"rtrim" => ScalarFunc::RTrim,
951        b"sign" => ScalarFunc::Sign,
952        b"substr" | b"substring" => ScalarFunc::Substr,
953        b"printf" | b"format" => ScalarFunc::Printf,
954        b"octet_length" => ScalarFunc::OctetLength,
955        b"random" => ScalarFunc::Random,
956        b"randomblob" => ScalarFunc::RandomBlob,
957        b"changes" => ScalarFunc::Changes,
958        b"total_changes" => ScalarFunc::TotalChanges,
959        b"last_insert_rowid" => ScalarFunc::LastInsertRowid,
960        b"sqlite_source_id" => ScalarFunc::SourceId,
961        b"fts5_source_id" => ScalarFunc::Fts5SourceId,
962        b"trim" => ScalarFunc::Trim,
963        b"typeof" => ScalarFunc::TypeOf,
964        b"unhex" => ScalarFunc::Unhex,
965        b"unicode" => ScalarFunc::Unicode,
966        b"upper" => ScalarFunc::Upper,
967        b"zeroblob" => ScalarFunc::ZeroBlob,
968        b"sqlite_version" => ScalarFunc::Version,
969        b"vector_distance_cos" | b"cosine_distance" => ScalarFunc::VectorDistanceCos,
970        b"vector_distance_l2" | b"l2_distance" => ScalarFunc::VectorDistanceL2,
971        b"vector_dot" | b"inner_product" => ScalarFunc::VectorDot,
972        // **Both spellings of each distance.** `l1_distance` is pgvector's name
973        // and `vector_distance_l1` is this engine's own, and the family reads
974        // as a family only if every member answers to both - `cos` and `l2`
975        // already did, and `l1` answered to one of the two.
976        b"l1_distance" | b"vector_distance_l1" => ScalarFunc::VectorDistanceL1,
977        b"hamming_distance" | b"vector_distance_hamming" => ScalarFunc::VectorDistanceHamming,
978        b"jaccard_distance" | b"vector_distance_jaccard" => ScalarFunc::VectorDistanceJaccard,
979        b"vector_dims" => ScalarFunc::VectorDims,
980        b"vector_norm" => ScalarFunc::VectorNorm,
981        b"l2_normalize" => ScalarFunc::VectorNormalize,
982        b"binary_quantize" => ScalarFunc::VectorQuantize,
983        b"subvector" => ScalarFunc::VectorSlice,
984        b"vector_add" => ScalarFunc::VectorAdd,
985        b"vector_sub" => ScalarFunc::VectorSubtract,
986        b"vector_mul" => ScalarFunc::VectorMultiply,
987        b"vector_concat" => ScalarFunc::VectorConcat,
988        b"geopoly_area" => ScalarFunc::GeopolyArea,
989        b"geopoly_blob" => ScalarFunc::GeopolyBlob,
990        b"geopoly_json" => ScalarFunc::GeopolyJson,
991        b"geopoly_svg" => ScalarFunc::GeopolySvg,
992        b"geopoly_within" => ScalarFunc::GeopolyWithin,
993        b"geopoly_contains_point" => ScalarFunc::GeopolyContainsPoint,
994        b"geopoly_overlap" => ScalarFunc::GeopolyOverlap,
995        b"geopoly_debug" => ScalarFunc::GeopolyDebug,
996        b"geopoly_bbox" => ScalarFunc::GeopolyBbox,
997        b"geopoly_xform" => ScalarFunc::GeopolyXform,
998        b"geopoly_regular" => ScalarFunc::GeopolyRegular,
999        b"geopoly_ccw" => ScalarFunc::GeopolyCcw,
1000        b"unknown" => ScalarFunc::Unknown,
1001        b"subtype" => ScalarFunc::Subtype,
1002        b"unistr" => ScalarFunc::Unistr,
1003        b"unistr_quote" => ScalarFunc::UnistrQuote,
1004        b"sqlite_compileoption_used" => ScalarFunc::CompileOptionUsed,
1005        b"sqlite_compileoption_get" => ScalarFunc::CompileOptionGet,
1006        b"sqlite_log" => ScalarFunc::Log,
1007        b"load_extension" => ScalarFunc::LoadExtension,
1008        b"regexp" => ScalarFunc::Regexp,
1009        b"sqlite_offset" => ScalarFunc::Offset,
1010        b"sqlar_compress" => ScalarFunc::SqlarCompress,
1011        b"sqlar_uncompress" => ScalarFunc::SqlarUncompress,
1012        b"rtreedepth" => ScalarFunc::RTreeDepth,
1013        b"rtreenode" => ScalarFunc::RTreeNode,
1014        b"rtreecheck" => ScalarFunc::RTreeCheck,
1015        _ => return None,
1016    };
1017    Some(func)
1018}
1019
1020/// Returns the aggregate a folded name spells.
1021///
1022/// `min` and `max` are both: one argument makes them aggregates and two or more
1023/// make them scalars, which is why the binder asks about the argument count
1024/// before it decides.
1025pub fn lookup_aggregate(folded: &[u8]) -> Option<AggregateFunc> {
1026    let func = match folded {
1027        b"count" => AggregateFunc::Count,
1028        b"sum" => AggregateFunc::Sum,
1029        b"total" => AggregateFunc::Total,
1030        b"avg" => AggregateFunc::Avg,
1031        b"group_concat" | b"string_agg" => AggregateFunc::GroupConcat,
1032        b"json_group_array" => AggregateFunc::JsonGroupArray,
1033        b"jsonb_group_array" => AggregateFunc::JsonbGroupArray,
1034        b"json_group_object" => AggregateFunc::JsonGroupObject,
1035        b"geopoly_group_bbox" => AggregateFunc::GeopolyGroupBbox,
1036        b"median" => AggregateFunc::Median,
1037        b"percentile" => AggregateFunc::Percentile,
1038        b"percentile_cont" => AggregateFunc::PercentileCont,
1039        b"percentile_disc" => AggregateFunc::PercentileDisc,
1040        b"jsonb_group_object" => AggregateFunc::JsonbGroupObject,
1041        _ => return None,
1042    };
1043    Some(func)
1044}
1045
1046/// Returns whether an argument count is legal for a scalar function.
1047pub fn scalar_arity_ok(func: ScalarFunc, count: usize) -> bool {
1048    match func {
1049        ScalarFunc::Abs
1050        | ScalarFunc::Hex
1051        | ScalarFunc::Length
1052        | ScalarFunc::Lower
1053        | ScalarFunc::Quote
1054        | ScalarFunc::Sign
1055        | ScalarFunc::TypeOf
1056        | ScalarFunc::Unicode
1057        | ScalarFunc::Upper
1058        | ScalarFunc::ZeroBlob => count == 1,
1059        ScalarFunc::IfNull | ScalarFunc::NullIf | ScalarFunc::Glob => count == 2,
1060        ScalarFunc::VectorDistanceCos
1061        | ScalarFunc::VectorDistanceL2
1062        | ScalarFunc::VectorDot
1063        | ScalarFunc::VectorDistanceL1
1064        | ScalarFunc::VectorDistanceHamming
1065        | ScalarFunc::VectorDistanceJaccard
1066        | ScalarFunc::VectorAdd
1067        | ScalarFunc::VectorSubtract
1068        | ScalarFunc::VectorMultiply
1069        | ScalarFunc::VectorConcat => count == 2,
1070        ScalarFunc::VectorDims
1071        | ScalarFunc::VectorNorm
1072        | ScalarFunc::VectorNormalize
1073        | ScalarFunc::VectorQuantize => count == 1,
1074        ScalarFunc::VectorSlice => count == 3,
1075        ScalarFunc::RTreeDepth | ScalarFunc::Offset | ScalarFunc::SqlarCompress => count == 1,
1076        ScalarFunc::SqlarUncompress => count == 2,
1077        ScalarFunc::RTreeNode => count == 2,
1078        // One argument is the table and two is a schema and a table, which is
1079        // the same pair `rtreecheck` takes in the reference.
1080        ScalarFunc::RTreeCheck => count == 1 || count == 2,
1081        ScalarFunc::GeopolyArea
1082        | ScalarFunc::GeopolyBlob
1083        | ScalarFunc::GeopolyJson
1084        | ScalarFunc::GeopolyDebug
1085        | ScalarFunc::GeopolyBbox
1086        | ScalarFunc::GeopolyCcw => count == 1,
1087        ScalarFunc::GeopolyWithin | ScalarFunc::GeopolyOverlap => count == 2,
1088        ScalarFunc::GeopolyContainsPoint => count == 3,
1089        ScalarFunc::GeopolyRegular => count == 4,
1090        ScalarFunc::GeopolyXform => count == 7,
1091        ScalarFunc::GeopolySvg => count >= 1,
1092        ScalarFunc::Replace => count == 3,
1093        // `iif` is `CASE` written as a call: pairs of a test and a value, with
1094        // an optional final answer. Two arguments is the shortest legal form
1095        // and there is no upper bound, which is why it is not `count == 3`.
1096        ScalarFunc::Iif => count >= 2,
1097        ScalarFunc::Unknown => true,
1098        ScalarFunc::Subtype
1099        | ScalarFunc::Unistr
1100        | ScalarFunc::UnistrQuote
1101        | ScalarFunc::CompileOptionUsed
1102        | ScalarFunc::CompileOptionGet => count == 1,
1103        ScalarFunc::Log | ScalarFunc::Regexp => count == 2,
1104        ScalarFunc::LoadExtension => count == 1 || count == 2,
1105        ScalarFunc::Instr => count == 2,
1106        ScalarFunc::Like => count == 2 || count == 3,
1107        ScalarFunc::Likelihood => count == 1 || count == 2,
1108        ScalarFunc::LTrim | ScalarFunc::RTrim | ScalarFunc::Trim | ScalarFunc::Unhex => {
1109            count == 1 || count == 2
1110        }
1111        ScalarFunc::Round => count == 1 || count == 2,
1112        ScalarFunc::Substr => count == 2 || count == 3,
1113        ScalarFunc::Coalesce | ScalarFunc::UsingCoalesce | ScalarFunc::Max | ScalarFunc::Min => {
1114            count >= 2
1115        }
1116        // `char()` with no arguments is the empty string in SQLite, not a
1117        // parse error (task-1979, F16). `concat()` keeps its floor of one,
1118        // which is the reference's own rule for that name.
1119        ScalarFunc::Char => true,
1120        ScalarFunc::Concat => count >= 1,
1121        ScalarFunc::ConcatWs => count >= 2,
1122        ScalarFunc::Version => count == 0,
1123        ScalarFunc::Printf => count >= 1,
1124        ScalarFunc::OctetLength | ScalarFunc::RandomBlob => count == 1,
1125        ScalarFunc::Random
1126        | ScalarFunc::Changes
1127        | ScalarFunc::TotalChanges
1128        | ScalarFunc::LastInsertRowid
1129        | ScalarFunc::SourceId
1130        | ScalarFunc::Fts5SourceId => count == 0,
1131    }
1132}
1133
1134/// Returns whether an argument count is legal for an aggregate.
1135pub fn aggregate_arity_ok(func: AggregateFunc, count: usize, star: bool) -> bool {
1136    match func {
1137        AggregateFunc::Count => star || count == 1,
1138        AggregateFunc::Sum | AggregateFunc::Total | AggregateFunc::Avg => !star && count == 1,
1139        AggregateFunc::Min | AggregateFunc::Max => !star && count == 1,
1140        AggregateFunc::GroupConcat => !star && (count == 1 || count == 2),
1141        AggregateFunc::JsonGroupArray | AggregateFunc::JsonbGroupArray => !star && count == 1,
1142        AggregateFunc::JsonGroupObject | AggregateFunc::JsonbGroupObject => !star && count == 2,
1143        AggregateFunc::Median
1144        | AggregateFunc::GeopolyGroupBbox
1145        | AggregateFunc::VectorSum
1146        | AggregateFunc::VectorAvg => !star && count == 1,
1147        AggregateFunc::Percentile
1148        | AggregateFunc::PercentileCont
1149        | AggregateFunc::PercentileDisc => !star && count == 2,
1150        // An application's aggregate declared its own arity, and the binder
1151        // checked it against the registration before getting here.
1152        AggregateFunc::External => !star,
1153    }
1154}
1155
1156/// Returns whether a folded name may be an aggregate at this argument count.
1157///
1158/// `min(x)` is the aggregate and `min(x, y)` is the scalar; asking the question
1159/// this way keeps the rule in one place instead of in both lookups.
1160pub fn is_aggregate_call(folded: &[u8], count: usize, star: bool) -> bool {
1161    if folded == b"min" || folded == b"max" {
1162        return !star && count == 1;
1163    }
1164    lookup_aggregate(folded).is_some()
1165}
1166
1167/// Reports whether a built-in function's answer depends on more than its
1168/// arguments, which is what SQLite will not allow in a generated column or an
1169/// index.
1170///
1171/// @param folded - the function's folded name
1172pub fn is_volatile(folded: &[u8]) -> bool {
1173    // SQLite registers the version and compile option functions without the
1174    // deterministic flag, so an index or a generated column may not call them.
1175    const NOT_DETERMINISTIC: [&[u8]; 4] = [
1176        b"sqlite_version",
1177        b"sqlite_source_id",
1178        b"sqlite_compileoption_used",
1179        b"sqlite_compileoption_get",
1180    ];
1181    NOT_DETERMINISTIC
1182        .iter()
1183        .any(|name| name.eq_ignore_ascii_case(folded))
1184        || VOLATILE
1185            .iter()
1186            .any(|(name, _)| name.as_bytes().eq_ignore_ascii_case(folded))
1187}
1188
1189/// Returns the aggregate a `min`/`max` call resolves to at one argument.
1190pub fn minmax_aggregate(folded: &[u8]) -> Option<AggregateFunc> {
1191    match folded {
1192        b"min" => Some(AggregateFunc::Min),
1193        b"max" => Some(AggregateFunc::Max),
1194        _ => None,
1195    }
1196}
1197
1198#[cfg(test)]
1199mod tests {
1200    use super::*;
1201
1202    /// The three functions that run a model answer `unsupported` in a build with no embedding
1203    /// support, and a name nobody has is still just a name nobody has.
1204    ///
1205    /// The engine reaches this list when a call does not resolve, so a build without the feature
1206    /// tells the caller the statement is fine and the build lacks the feature, with exit code 3,
1207    /// instead of "no such function".
1208    #[test]
1209    fn the_model_functions_need_a_component() {
1210        for name in [&b"embed"[..], b"embed_tokens", b"rerank"] {
1211            let said = needs_a_component(name).expect("the function needs a component");
1212            assert!(said.contains("no embedding support compiled in"), "{said}");
1213        }
1214        assert!(
1215            needs_a_component(b"rerank").is_some_and(|said| said.starts_with("rerank(TEXT, TEXT)"))
1216        );
1217        assert_eq!(needs_a_component(b"nope"), None);
1218    }
1219
1220    /// Names are matched folded, and an unknown name is not a function.
1221    #[test]
1222    fn lookup_matches_folded_names() {
1223        assert_eq!(lookup_scalar(b"abs"), Some(ScalarFunc::Abs));
1224        assert_eq!(lookup_scalar(b"substring"), Some(ScalarFunc::Substr));
1225        assert_eq!(lookup_scalar(b"nope"), None);
1226        assert_eq!(lookup_aggregate(b"count"), Some(AggregateFunc::Count));
1227        assert_eq!(
1228            lookup_aggregate(b"string_agg"),
1229            Some(AggregateFunc::GroupConcat)
1230        );
1231    }
1232
1233    /// `min` and `max` change identity with their argument count, which is the
1234    /// one place SQLite overloads a name across the scalar/aggregate boundary.
1235    #[test]
1236    fn min_and_max_are_aggregates_only_at_one_argument() {
1237        assert!(is_aggregate_call(b"min", 1, false));
1238        assert!(!is_aggregate_call(b"min", 2, false));
1239        assert!(!is_aggregate_call(b"min", 0, true));
1240        assert_eq!(minmax_aggregate(b"max"), Some(AggregateFunc::Max));
1241    }
1242
1243    /// Arity is checked at bind time, so a wrong count is a prepare failure.
1244    #[test]
1245    fn arity_is_checked_per_function() {
1246        assert!(scalar_arity_ok(ScalarFunc::Abs, 1));
1247        assert!(!scalar_arity_ok(ScalarFunc::Abs, 2));
1248        assert!(scalar_arity_ok(ScalarFunc::Substr, 2));
1249        assert!(scalar_arity_ok(ScalarFunc::Substr, 3));
1250        assert!(!scalar_arity_ok(ScalarFunc::Substr, 4));
1251        assert!(scalar_arity_ok(ScalarFunc::Coalesce, 5));
1252        assert!(!scalar_arity_ok(ScalarFunc::Coalesce, 1));
1253        assert!(aggregate_arity_ok(AggregateFunc::Count, 0, true));
1254        assert!(!aggregate_arity_ok(AggregateFunc::Sum, 0, true));
1255    }
1256}
1257
1258/// One row of `PRAGMA function_list`.
1259#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1260pub struct FunctionEntry {
1261    /// The name as it is written.
1262    pub name: &'static str,
1263    /// `s` for a scalar, `w` for an aggregate or window function that can run
1264    /// over a window, `a` for an aggregate that cannot.
1265    pub kind: &'static str,
1266    /// How many arguments, or -1 for any number.
1267    pub arity: i64,
1268    /// The flag word the C surface reports.
1269    ///
1270    /// 2048 is `SQLITE_INNOCUOUS` and 524288 is `SQLITE_DETERMINISTIC`, which
1271    /// is what a built-in carries: it does nothing an expression could not, and
1272    /// it answers the same thing twice.
1273    pub flags: i64,
1274}
1275
1276/// The bit `function_list` sets for a function a schema may safely call.
1277///
1278/// Named rather than written twice because `inillucent-engine`'s
1279/// `function_list` reports the connection's registered functions beside these
1280/// built-ins, and it has to describe them in the same column with the same
1281/// meaning. A registered function that promised `innocuous` and was reported
1282/// with a bit nothing else uses would be a register that under-describes, which
1283/// is the defect this whole list was extended to fix.
1284pub const INNOCUOUS_FLAG: i64 = 2048;
1285
1286/// The bit `function_list` sets for a function that answers the same twice.
1287pub const DETERMINISTIC_FLAG: i64 = 524288;
1288
1289/// The flags every built-in carries: innocuous and deterministic.
1290const BUILTIN_FLAGS: i64 = INNOCUOUS_FLAG | DETERMINISTIC_FLAG;
1291
1292/// The flags a built-in that is not deterministic carries.
1293const VOLATILE_FLAGS: i64 = INNOCUOUS_FLAG;
1294
1295/// Returns every built-in this build has, in the order `function_list` reports.
1296///
1297/// The list is written out rather than derived from the lookup tables because
1298/// the arity is per *overload*: `substr` is here twice, at two and at three
1299/// arguments, which is what SQLite reports and what an application checking
1300/// whether a call will bind needs to see.
1301///
1302/// **It must name everything the binder will resolve, and a completeness check
1303/// found that it did not.** The register answered 161 names where SQLite answers 218, and
1304/// the functionality behind most of the difference was present and
1305/// byte-identical - `current_date`, `regexp`, `unistr`, `median`, `bm25`,
1306/// `matchinfo` and the rest all answered when called. A caller that
1307/// introspects the register to decide what it may use was told less than the
1308/// truth, with no error, which is the one *silent* difference this project has
1309/// had. The additions below were each verified against the engine before being
1310/// listed: a name here that the binder refuses would be the same defect
1311/// pointing the other way.
1312pub fn every_function() -> Vec<FunctionEntry> {
1313    let mut out = Vec::new();
1314    let mut scalar = |name: &'static str, arity: i64| {
1315        out.push(FunctionEntry {
1316            name,
1317            kind: "s",
1318            arity,
1319            flags: BUILTIN_FLAGS,
1320        });
1321    };
1322    for (name, arity) in SCALARS {
1323        scalar(name, *arity);
1324    }
1325    for (name, arity) in VOLATILE {
1326        out.push(FunctionEntry {
1327            name,
1328            kind: "s",
1329            arity: *arity,
1330            flags: VOLATILE_FLAGS,
1331        });
1332    }
1333    // **Every built-in aggregate is reported as `w`.** SQLite's `type` column
1334    // says `w` for an aggregate that can also run over a window, which is every
1335    // one of its own (`max`, `sum`, `count`, `group_concat`, ...); `a` is for an
1336    // aggregate that cannot, and none of these is one.
1337    for (name, arity) in AGGREGATES {
1338        out.push(FunctionEntry {
1339            name,
1340            kind: "w",
1341            arity: *arity,
1342            flags: BUILTIN_FLAGS,
1343        });
1344    }
1345    for (name, arity) in WINDOWS {
1346        out.push(FunctionEntry {
1347            name,
1348            kind: "w",
1349            arity: *arity,
1350            flags: BUILTIN_FLAGS,
1351        });
1352    }
1353    out.sort_by(|left, right| left.name.cmp(right.name).then(left.arity.cmp(&right.arity)));
1354    out
1355}
1356
1357/// The deterministic scalars, with one row per overload.
1358///
1359/// **`narg` is SQLite's own encoding, not "how many arguments".** A negative
1360/// number means variadic *and carries a minimum*: `coalesce` reads -4 and
1361/// `concat` -3 in the reference's register, not -1. A
1362/// register-completeness check compares this column because it is the one an
1363/// application reads to decide whether a call will bind, and it found seven
1364/// entries here that disagreed with the reference while answering identically.
1365const SCALARS: &[(&str, i64)] = &[
1366    ("abs", 1),
1367    ("acos", 1),
1368    ("acosh", 1),
1369    ("asin", 1),
1370    ("asinh", 1),
1371    ("atan", 1),
1372    ("atan2", 2),
1373    ("atanh", 1),
1374    ("ceil", 1),
1375    ("ceiling", 1),
1376    ("char", -1),
1377    ("coalesce", -4),
1378    ("concat", -3),
1379    ("concat_ws", -4),
1380    ("cos", 1),
1381    ("cosh", 1),
1382    ("date", -1),
1383    ("datetime", -1),
1384    ("degrees", 1),
1385    ("exp", 1),
1386    ("floor", 1),
1387    ("format", -1),
1388    ("glob", 2),
1389    ("hex", 1),
1390    ("ifnull", 2),
1391    ("iif", -4),
1392    ("instr", 2),
1393    ("json", 1),
1394    ("json_array", -1),
1395    ("json_array_length", 1),
1396    ("json_array_length", 2),
1397    ("json_error_position", 1),
1398    ("json_extract", -1),
1399    ("json_insert", -1),
1400    ("json_object", -1),
1401    ("json_patch", 2),
1402    ("json_pretty", 1),
1403    ("json_pretty", 2),
1404    ("json_quote", 1),
1405    ("json_remove", -1),
1406    ("json_replace", -1),
1407    ("json_set", -1),
1408    ("json_type", 1),
1409    ("json_type", 2),
1410    ("json_valid", 1),
1411    ("json_valid", 2),
1412    ("jsonb", 1),
1413    ("jsonb_array", -1),
1414    ("jsonb_extract", -1),
1415    ("jsonb_insert", -1),
1416    ("jsonb_object", -1),
1417    ("jsonb_patch", 2),
1418    ("jsonb_remove", -1),
1419    ("jsonb_replace", -1),
1420    ("jsonb_set", -1),
1421    ("julianday", -1),
1422    ("length", 1),
1423    ("like", 2),
1424    ("like", 3),
1425    ("likelihood", 2),
1426    ("likely", 1),
1427    ("ln", 1),
1428    ("log", 1),
1429    ("log", 2),
1430    ("log10", 1),
1431    ("log2", 1),
1432    ("lower", 1),
1433    ("ltrim", 1),
1434    ("ltrim", 2),
1435    ("max", -3),
1436    ("min", -3),
1437    ("mod", 2),
1438    ("nullif", 2),
1439    ("octet_length", 1),
1440    ("pi", 0),
1441    ("pow", 2),
1442    ("power", 2),
1443    ("printf", -1),
1444    ("quote", 1),
1445    ("radians", 1),
1446    ("replace", 3),
1447    ("round", 1),
1448    ("round", 2),
1449    ("rtrim", 1),
1450    ("rtrim", 2),
1451    ("sign", 1),
1452    ("sin", 1),
1453    ("sinh", 1),
1454    ("fts5_source_id", 0),
1455    ("optimize", 1),
1456    ("sqlite_source_id", 0),
1457    ("sqlite_version", 0),
1458    ("sqrt", 1),
1459    ("strftime", -1),
1460    ("substr", 2),
1461    ("substr", 3),
1462    ("substring", 2),
1463    ("substring", 3),
1464    ("tan", 1),
1465    ("tanh", 1),
1466    ("time", -1),
1467    ("timediff", 2),
1468    ("trim", 1),
1469    ("trim", 2),
1470    ("trunc", 1),
1471    ("typeof", 1),
1472    ("unhex", 1),
1473    ("unhex", 2),
1474    ("unicode", 1),
1475    ("unixepoch", -1),
1476    ("unlikely", 1),
1477    ("upper", 1),
1478    ("binary_quantize", 1),
1479    ("rtreecheck", -1),
1480    ("sqlar_compress", 1),
1481    ("sqlar_uncompress", 2),
1482    ("sqlite_offset", 1),
1483    ("rtreedepth", 1),
1484    ("rtreenode", 2),
1485    ("geopoly_area", 1),
1486    ("geopoly_bbox", 1),
1487    ("geopoly_blob", 1),
1488    ("geopoly_ccw", 1),
1489    ("geopoly_contains_point", 3),
1490    ("geopoly_debug", 1),
1491    ("geopoly_group_bbox", 1),
1492    ("geopoly_json", 1),
1493    ("geopoly_overlap", 2),
1494    ("geopoly_regular", 4),
1495    ("geopoly_svg", -1),
1496    ("geopoly_within", 2),
1497    ("geopoly_xform", 7),
1498    ("cosine_distance", 2),
1499    ("hamming_distance", 2),
1500    ("inner_product", 2),
1501    ("jaccard_distance", 2),
1502    ("l1_distance", 2),
1503    ("l2_distance", 2),
1504    ("l2_normalize", 1),
1505    ("subvector", 3),
1506    ("vector_add", 2),
1507    ("vector_concat", 2),
1508    ("vector_dims", 1),
1509    ("vector_distance_cos", 2),
1510    ("vector_distance_l2", 2),
1511    ("vector_dot", 2),
1512    ("vector_mul", 2),
1513    ("vector_norm", 1),
1514    ("vector_sub", 2),
1515    ("zeroblob", 1),
1516    // Present and answering, and missing from this list until now.
1517    // Each was checked against the shell before it was added.
1518    ("->", 2),
1519    ("->>", 2),
1520    ("bm25", -1),
1521    ("highlight", -1),
1522    ("if", -4),
1523    ("json_array_insert", -1),
1524    ("jsonb_array_insert", -1),
1525    ("match", 2),
1526    ("matchinfo", 1),
1527    ("matchinfo", 2),
1528    ("offsets", 1),
1529    ("regexp", 2),
1530    ("snippet", -1),
1531    ("sqlite_compileoption_get", 1),
1532    ("sqlite_compileoption_used", 1),
1533    ("subtype", 1),
1534    ("unistr", 1),
1535    ("unistr_quote", 1),
1536    ("unknown", -1),
1537];
1538
1539/// The scalars whose answer depends on something other than their arguments.
1540const VOLATILE: &[(&str, i64)] = &[
1541    ("changes", 0),
1542    // The three date keywords are functions in SQLite's register and answer
1543    // like functions here; they read the clock, so they are not deterministic.
1544    ("current_date", 0),
1545    ("current_time", 0),
1546    ("current_timestamp", 0),
1547    ("last_insert_rowid", 0),
1548    ("load_extension", 1),
1549    ("load_extension", 2),
1550    ("random", 0),
1551    ("randomblob", 1),
1552    ("sqlite_log", 2),
1553    ("total_changes", 0),
1554];
1555
1556/// The aggregates, with one row per overload.
1557const AGGREGATES: &[(&str, i64)] = &[
1558    ("avg", 1),
1559    ("count", 0),
1560    ("count", 1),
1561    ("group_concat", 1),
1562    ("group_concat", 2),
1563    ("json_group_array", 1),
1564    ("json_group_object", 2),
1565    ("jsonb_group_array", 1),
1566    ("jsonb_group_object", 2),
1567    ("max", 1),
1568    ("min", 1),
1569    ("string_agg", 2),
1570    ("sum", 1),
1571    ("total", 1),
1572];
1573
1574/// The window functions that are not aggregates.
1575const WINDOWS: &[(&str, i64)] = &[
1576    ("cume_dist", 0),
1577    ("dense_rank", 0),
1578    ("first_value", 1),
1579    ("lag", 1),
1580    ("lag", 2),
1581    ("lag", 3),
1582    ("last_value", 1),
1583    ("lead", 1),
1584    ("lead", 2),
1585    ("lead", 3),
1586    ("nth_value", 2),
1587    ("ntile", 1),
1588    ("percent_rank", 0),
1589    ("rank", 0),
1590    ("row_number", 0),
1591    // The percentile family, which SQLite reports as window functions and which
1592    // this engine answers as both aggregates and window functions.
1593    ("median", 1),
1594    ("percentile", 2),
1595    ("percentile_cont", 2),
1596    ("percentile_disc", 2),
1597];