Skip to main content

inillucent_sql/
function.rs

1//! The built-in function registry: names, arities, and identities.
2//!
3//! Invariant: a function is recognised here or it does not exist. The binder
4//! resolves a name to one of these identities and refuses everything else with
5//! "no such function", so an unknown name fails at prepare time rather than
6//! part-way through a scan, and the VM never dispatches on a string.
7//!
8//! Arity is checked here too, because SQLite reports "wrong number of arguments
9//! to function abs()" from prepare rather than from execution.
10
11/// The names that exist in inillucent but need a component this build has not
12/// got.
13///
14/// **`embed` is the whole list, and it is here rather than in the registry
15/// because the registry is where it is absent** (task-1979, section 8.1, gap
16/// 12). `inillucent-search` registers `embed` only when the `embed` feature is
17/// compiled in, so on a build without it the name reaches the binder's
18/// "no such function" path and answered exit 1 - which says the caller
19/// misspelled something. The statement is spelled correctly and this build has
20/// not got the function, which is exactly what exit 3 means.
21///
22/// A build that *does* have `embed` never reaches here, because the registry
23/// resolves the name before the refusal is built. A machine that has the
24/// function and not the model is a third thing again and keeps its own status:
25/// `inillucent-search`'s `no_model` answers `invalid_state` and names
26/// `inillucent setup-embeddings`, because the component is installable and
27/// exit 3 would say the opposite.
28const NEEDS_A_COMPONENT: &[(&[u8], &str)] = &[
29    (
30        b"embed",
31        "embed(TEXT): this build has no embedding support compiled in",
32    ),
33    (
34        b"embed_tokens",
35        "embed_tokens(TEXT): this build has no embedding support compiled in",
36    ),
37    (
38        b"rerank",
39        "rerank(TEXT, TEXT): this build has no embedding support compiled in",
40    ),
41];
42
43/// Returns what a name needs, when the name is one this build left out.
44///
45/// @param name - the folded function name that did not resolve
46pub fn needs_a_component(name: &[u8]) -> Option<&'static str> {
47    NEEDS_A_COMPONENT
48        .iter()
49        .find(|(known, _)| *known == name)
50        .map(|(_, said)| *said)
51}
52
53/// A scalar built-in.
54#[derive(Clone, Copy, Debug, PartialEq, Eq)]
55pub enum ScalarFunc {
56    /// `abs(x)`
57    Abs,
58    /// `char(...)`
59    Char,
60    /// `coalesce(...)`
61    Coalesce,
62    /// `concat(...)`
63    Concat,
64    /// `concat_ws(sep, ...)`
65    ConcatWs,
66    /// `glob(pattern, text)`
67    Glob,
68    /// `hex(x)`
69    Hex,
70    /// `ifnull(a, b)`
71    IfNull,
72    /// `iif(a, b, c)`
73    Iif,
74    /// `instr(haystack, needle)`
75    Instr,
76    /// `length(x)`
77    Length,
78    /// `like(pattern, text[, escape])`
79    Like,
80    /// `likelihood(x, y)`, `likely(x)` and `unlikely(x)`, which are no-ops.
81    Likelihood,
82    /// `lower(x)`
83    Lower,
84    /// `ltrim(x[, chars])`
85    LTrim,
86    /// `max(a, b, ...)`, the scalar form.
87    Max,
88    /// `min(a, b, ...)`, the scalar form.
89    Min,
90    /// `nullif(a, b)`
91    NullIf,
92    /// `quote(x)`
93    Quote,
94    /// `replace(text, from, to)`
95    Replace,
96    /// `round(x[, digits])`
97    Round,
98    /// `rtrim(x[, chars])`
99    RTrim,
100    /// `sign(x)`
101    Sign,
102    /// `substr(x, start[, length])`
103    Substr,
104    /// `trim(x[, chars])`
105    Trim,
106    /// `typeof(x)`
107    TypeOf,
108    /// `unhex(x[, chars])`
109    Unhex,
110    /// `unicode(x)`
111    Unicode,
112    /// `upper(x)`
113    Upper,
114    /// `zeroblob(n)`
115    ZeroBlob,
116    /// `printf(format, ...)` and `format(format, ...)`
117    Printf,
118    /// `octet_length(x)`
119    OctetLength,
120    /// `random()`
121    Random,
122    /// `randomblob(n)`
123    RandomBlob,
124    /// `changes()`
125    Changes,
126    /// `total_changes()`
127    TotalChanges,
128    /// `last_insert_rowid()`
129    LastInsertRowid,
130    /// `sqlite_source_id()`
131    SourceId,
132    /// `fts5_source_id()`
133    Fts5SourceId,
134    /// `sqlite_version()`
135    Version,
136    /// `vector_distance_cos(a, b)`, the cosine distance between two vectors.
137    ///
138    /// **Not a SQLite function, and the first one this engine adds.** pgvector
139    /// spells it `a <=> b`; the whole point of Phase 2's Part 7 is that a
140    /// vector is a value a `SELECT` can order by, and an operator that is sugar
141    /// for a function needs the function to exist first. A vector is a blob of
142    /// little-endian `f32`, which is what `inillucent_search` already stores and
143    /// what `vector_distance_l2` and `vector_dot` read too.
144    VectorDistanceCos,
145    /// `vector_distance_l2(a, b)`, the Euclidean distance between two vectors.
146    VectorDistanceL2,
147    /// `vector_dot(a, b)`, the dot product of two vectors.
148    ///
149    /// Negated relative to pgvector's `<#>`, which answers the *negative* inner
150    /// product so that a smaller number is a better match. This answers the dot
151    /// product itself, because a function named `dot` that returned its negative
152    /// would be a trap; the ordering sugar negates where it needs to.
153    VectorDot,
154    /// `l1_distance(a, b)`, the taxicab distance, spelled `a <+> b`.
155    VectorDistanceL1,
156    /// `hamming_distance(a, b)`, how many components differ.
157    ///
158    /// pgvector defines it over its `bit` type and spells it `a <~> b`. Here a
159    /// bit vector is the blob `binary_quantize` produces, and the distance is
160    /// the population count of the two blobs' exclusive-or - which is the same
161    /// number, computed the same way, over the representation this engine has.
162    VectorDistanceHamming,
163    /// `jaccard_distance(a, b)`, one minus the overlap, spelled `a <%> b`.
164    VectorDistanceJaccard,
165    /// `vector_dims(a)`, how many components a vector has.
166    VectorDims,
167    /// `vector_norm(a)`, its Euclidean length.
168    VectorNorm,
169    /// `l2_normalize(a)`, the same direction with length one.
170    VectorNormalize,
171    /// `binary_quantize(a)`, one bit per component: set when it is positive.
172    VectorQuantize,
173    /// `subvector(a, start, count)`, a slice, counted from one.
174    VectorSlice,
175    /// `vector_add(a, b)`, component by component.
176    ///
177    /// **A function rather than `+`, and that is a compatibility choice rather
178    /// than a shortcut.** pgvector can overload `+` because a `vector` is a
179    /// distinct type in PostgreSQL; here a vector is a blob, and SQLite says
180    /// that a blob in arithmetic is zero. Overloading the operator for every
181    /// blob would change the answer to `x'00' + x'00'` from `0` to a blob,
182    /// which is a difference every application that adds two blobs would see.
183    ///
184    /// **The operators were given back, on the one condition that keeps
185    /// both answers.** `a + b` binds to this function when a side reads a
186    /// column *declared* `VECTOR(n)` - which is the same thing PostgreSQL is
187    /// using, a declared type - and stays SQLite's arithmetic otherwise. So
188    /// `x'00' + x'00'` is still `0` and `v + v` over a vector column is a
189    /// vector.
190    VectorAdd,
191    /// `vector_sub(a, b)`, component by component.
192    VectorSubtract,
193    /// `vector_mul(a, b)`, component by component.
194    VectorMultiply,
195    /// `vector_concat(a, b)`, one vector after the other.
196    VectorConcat,
197    /// `geopoly_area(P)`, the signed area a polygon encloses.
198    ///
199    /// **The `geopoly` surface is thirteen functions and one aggregate**, and
200    /// they are listed here individually rather than folded into one
201    /// `Geopoly(kind)` variant because arity checking reads this enum: they
202    /// take one, two, three, four, seven and any number of arguments, and a
203    /// single variant could not say so.
204    GeopolyArea,
205    /// `geopoly_blob(P)`, the stored form of a polygon.
206    GeopolyBlob,
207    /// `geopoly_json(P)`, the GeoJSON form.
208    GeopolyJson,
209    /// `geopoly_svg(P, ...)`, an SVG `<polyline>` with the extra arguments
210    /// written into the tag.
211    GeopolySvg,
212    /// `geopoly_within(P1, P2)`, whether the second is inside the first.
213    GeopolyWithin,
214    /// `geopoly_contains_point(P, X, Y)`, where a point sits.
215    GeopolyContainsPoint,
216    /// `geopoly_overlap(P1, P2)`, how two polygons meet.
217    GeopolyOverlap,
218    /// `geopoly_debug(X)`, which answers nothing.
219    ///
220    /// It switches on the reference's own tracing, which only exists in a build
221    /// made with `GEOPOLY_ENABLE_DEBUG`; in every other build it reads its
222    /// argument and returns nothing at all. That is what this does, and it is
223    /// registered because a name the reference resolves and this engine does
224    /// not is a difference an application can see.
225    GeopolyDebug,
226    /// `geopoly_bbox(P)`, the bounding box as a four-sided polygon.
227    GeopolyBbox,
228    /// `geopoly_xform(P, A, B, C, D, E, F)`, an affine transform.
229    GeopolyXform,
230    /// `geopoly_regular(X, Y, R, N)`, a regular polygon.
231    GeopolyRegular,
232    /// `geopoly_ccw(P)`, the same ring wound counter-clockwise.
233    GeopolyCcw,
234    /// `unknown(...)`, which answers NULL to anything.
235    ///
236    /// SQLite registers it, lists it in `function_list`, and returns NULL from
237    /// it whatever it is given. It is here because a name the reference resolves
238    /// and this engine does not is a difference an application can see.
239    Unknown,
240    /// `subtype(x)`, the tag a function attached to its answer.
241    Subtype,
242    /// `unistr(x)`, which expands `\uXXXX` and `\UXXXXXXXX` escapes.
243    Unistr,
244    /// `unistr_quote(x)`, `quote()` with the control characters escaped.
245    UnistrQuote,
246    /// `sqlite_compileoption_used(name)`
247    CompileOptionUsed,
248    /// `sqlite_compileoption_get(n)`
249    CompileOptionGet,
250    /// `sqlite_log(code, message)`, which writes to the log and answers NULL.
251    Log,
252    /// `load_extension(path[, entry])`
253    LoadExtension,
254    /// `regexp(pattern, subject)`, which is what `X REGEXP Y` calls.
255    Regexp,
256    /// `sqlar_compress(X)`, a blob compressed if that makes it smaller.
257    ///
258    /// **The archive format's own rule, and it is why this is not just a
259    /// compressor.** A row of a `.sqlar` table holds either a zlib stream or
260    /// the raw bytes, and which one is decided by whichever is shorter; the
261    /// stored `sz` column is what tells the two apart on the way back. So a
262    /// value that does not compress is stored as it stands, and a value that is
263    /// not a blob at all is returned unchanged, type and all.
264    SqlarCompress,
265    /// `sqlar_uncompress(Z, SZ)`, the inverse.
266    ///
267    /// `SZ` is the size the row claims the content is. When it equals the
268    /// blob's own length the blob *is* the content and is returned unchanged,
269    /// which is how the format says "this one was stored raw".
270    SqlarUncompress,
271    /// `sqlite_offset(X)`, where in the file the row holding X is.
272    ///
273    /// **The page, not the record, and that is the whole of the difference.**
274    /// SQLite reports the byte offset of the *record* a value would be read
275    /// from, because a row there is one contiguous run of bytes. A leaf here is
276    /// PAX: each column is its own run, so one row occupies several places on
277    /// its page and there is no single offset for it. What is reported is the
278    /// offset of the page, which is where the value is genuinely read from.
279    ///
280    /// Folded to its answer by the physical pass, like `rtreecheck`, because it
281    /// is a question about a *tree* rather than about a value.
282    Offset,
283    /// `rtreedepth(X)`, the depth stored at the front of an R-Tree node.
284    RTreeDepth,
285    /// `rtreenode(D, X)`, an R-Tree node rendered as a readable list.
286    RTreeNode,
287    /// `rtreecheck(T)`, an integrity check over one R-Tree table.
288    ///
289    /// **Answered where the table is reachable, which is not here.** A scalar
290    /// is handed values and nothing else; this one is about a *table*, so the
291    /// physical pass folds it to its answer while it still has the catalog,
292    /// and what reaches the evaluator is already the text. Running once per
293    /// preparation rather than once per row is also what it means: the
294    /// argument is a table name, so the answer cannot vary down a column.
295    RTreeCheck,
296}
297
298/// An aggregate built-in.
299#[derive(Clone, Copy, Debug, PartialEq, Eq)]
300pub enum AggregateFunc {
301    /// `count(x)` and `count(*)`
302    Count,
303    /// `sum(x)`
304    Sum,
305    /// `total(x)`
306    Total,
307    /// `avg(x)`
308    Avg,
309    /// `min(x)`
310    Min,
311    /// `max(x)`
312    Max,
313    /// `group_concat(x[, sep])` and `string_agg(x, sep)`
314    GroupConcat,
315    /// `json_group_array(x)`
316    JsonGroupArray,
317    /// `jsonb_group_array(x)`
318    JsonbGroupArray,
319    /// `json_group_object(label, x)`
320    JsonGroupObject,
321    /// `jsonb_group_object(label, x)`
322    JsonbGroupObject,
323    /// `median(x)`, which is `percentile_cont(x, 0.5)` under a shorter name.
324    Median,
325    /// `geopoly_group_bbox(P)`, the box that holds every polygon in the group.
326    GeopolyGroupBbox,
327    /// `sum(v)` and `total(v)` over a vector column, component by component.
328    ///
329    /// Not a name a caller writes: the binder picks it when `sum`'s argument
330    /// reads a vector, because that is where the argument's type is known.
331    VectorSum,
332    /// `avg(v)` over a vector column, component by component.
333    VectorAvg,
334    /// `percentile(x, p)`, where `p` runs 0 to 100.
335    Percentile,
336    /// `percentile_cont(x, f)`, where `f` runs 0 to 1 and the answer is
337    /// interpolated between the two rows it falls between.
338    PercentileCont,
339    /// `percentile_disc(x, f)`, which answers one of the rows rather than a
340    /// value between two of them.
341    PercentileDisc,
342    /// An aggregate an application registered, named beside the call.
343    ///
344    /// The name is not in here because this enum is `Copy` and travels through
345    /// the program's operands; it rides in `AggregateCall` instead.
346    External,
347}
348
349/// What a registered function promises about itself.
350///
351/// It lives here, below `inillucent-ext`, because two different layers have to
352/// read the same promise: `inillucent_ext::registry::Registry` records it when
353/// an application registers a function, and the binder enforces it when a
354/// schema names one. `inillucent-ext` re-exports this type, so a registrant
355/// writes `inillucent_ext::registry::FunctionFlags` exactly as before.
356#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
357pub struct FunctionFlags {
358    /// The function may only be called from top-level SQL, never from a
359    /// schema: not from a `DEFAULT`, a `CHECK`, a generated column, an index
360    /// expression, a partial-index predicate, a view or a trigger.
361    ///
362    /// [`FunctionFlags::external`] sets this, because the safe assumption about
363    /// code somebody else wrote is that it does something. **It is not what the
364    /// `Default` derive gives**, which is every flag false: a registrant who
365    /// writes `..FunctionFlags::default()` gets a function a schema may name.
366    /// That is the hole `embed` was registered through (task-1969, 7.4), and
367    /// `inillucent_ext::registry::UserFunction::external` is the constructor to
368    /// reach for instead.
369    pub direct_only: bool,
370    /// The function does nothing an ordinary expression could not: no side
371    /// effects, no file access, no dependence on anything but its arguments.
372    pub innocuous: bool,
373    /// The function returns the same answer for the same arguments within one
374    /// statement, so the planner may call it once.
375    pub deterministic: bool,
376}
377
378impl FunctionFlags {
379    /// Returns the flags a built-in carries: safe for a schema to call.
380    pub fn builtin() -> FunctionFlags {
381        FunctionFlags {
382            direct_only: false,
383            innocuous: true,
384            deterministic: true,
385        }
386    }
387
388    /// Returns the flags anything registered from outside carries by default.
389    pub fn external() -> FunctionFlags {
390        FunctionFlags {
391            direct_only: true,
392            innocuous: false,
393            deterministic: false,
394        }
395    }
396}
397
398/// Which context a name is being resolved from.
399#[derive(Clone, Copy, Debug, PartialEq, Eq)]
400pub enum CallSite {
401    /// The statement an application submitted.
402    Statement,
403    /// A `DEFAULT`, `CHECK`, generated column, index expression, partial-index
404    /// predicate, view or trigger stored in the schema.
405    Schema,
406}
407
408/// Returns why a schema may not call this function, or nothing when it may.
409///
410/// **One rule, read by two layers (task-1972).** `Registry::authorize_function`
411/// wraps the answer in a `DbError` for an application that asks the registry
412/// directly, and the binder wraps it in a `ParseError` for the statement it is
413/// compiling. Writing the rule twice is how the two would eventually disagree,
414/// and the half nobody exercised would be the permissive one.
415///
416/// The rule reads the same way SQLite's does: a direct-only function is never
417/// callable from a schema; anything else is callable from a schema only when
418/// the connection trusts the schema or the function is innocuous.
419///
420/// @param flags - what the function promises about itself
421/// @param site - where the call was written
422/// @param trusted_schema - whether the connection trusts the schema it read
423pub fn schema_refusal(
424    flags: FunctionFlags,
425    site: CallSite,
426    trusted_schema: bool,
427) -> Option<&'static str> {
428    if site == CallSite::Statement {
429        return None;
430    }
431    if flags.direct_only {
432        return Some("may only be used from top-level SQL");
433    }
434    if trusted_schema || flags.innocuous {
435        return None;
436    }
437    Some("is not allowed in a schema")
438}
439
440/// A function an application registered, as the binder needs to see it.
441///
442/// Only what resolution needs: a name, how many arguments it takes, whether it
443/// reduces a group, and what it promises about itself. What it *does* is the
444/// machine's business.
445///
446/// **The flags are here because the binder is where the promise is kept
447/// (task-1972).** `Registry::authorize_function` had no caller, so
448/// `direct_only`, `innocuous` and `PRAGMA trusted_schema` were a policy with a
449/// passing unit test and no effect on the engine: a `CHECK`, an index
450/// expression or a generated column could name any registered function whatever
451/// its flags. `inillucent-sql` sits below `inillucent-ext` and cannot reach the
452/// registry, so what the registry knows travels down here with the name.
453#[derive(Clone, Debug, PartialEq, Eq)]
454pub struct ExternalFunction {
455    /// The folded name.
456    pub name: Vec<u8>,
457    /// How many arguments it takes, or -1 for any number.
458    pub arity: i32,
459    /// Whether it reduces a group rather than a row.
460    pub aggregate: bool,
461    /// What it promises about itself, which decides whether a schema may name
462    /// it.
463    pub flags: FunctionFlags,
464}
465
466impl ExternalFunction {
467    /// Returns whether this registration answers a call with this many
468    /// arguments.
469    pub fn accepts(&self, argc: usize) -> bool {
470        self.arity < 0 || self.arity as usize == argc
471    }
472}
473
474/// Returns the registration that answers a call, preferring an exact arity.
475///
476/// SQLite resolves the same way: a function registered for exactly this many
477/// arguments wins over one registered for any number, so an application can
478/// define both a fast two-argument form and a general one.
479pub fn lookup_external<'a>(
480    functions: &'a [ExternalFunction],
481    name: &[u8],
482    argc: usize,
483) -> Option<&'a ExternalFunction> {
484    let folded = name.to_ascii_lowercase();
485    functions
486        .iter()
487        .find(|function| function.name == folded && function.arity as usize == argc)
488        .or_else(|| {
489            functions
490                .iter()
491                .find(|function| function.name == folded && function.arity < 0)
492        })
493}
494
495/// A date or time built-in.
496#[derive(Clone, Copy, Debug, PartialEq, Eq)]
497pub enum TimeFunc {
498    /// `date(...)`
499    Date,
500    /// `time(...)`
501    Time,
502    /// `datetime(...)`
503    DateTime,
504    /// `julianday(...)`
505    JulianDay,
506    /// `unixepoch(...)`
507    UnixEpoch,
508    /// `strftime(format, ...)`
509    StrfTime,
510    /// `timediff(a, b)`
511    TimeDiff,
512}
513
514/// Returns the date or time function a folded name spells.
515pub fn lookup_time(folded: &[u8]) -> Option<TimeFunc> {
516    let func = match folded {
517        b"date" => TimeFunc::Date,
518        b"time" => TimeFunc::Time,
519        b"datetime" => TimeFunc::DateTime,
520        b"julianday" => TimeFunc::JulianDay,
521        b"unixepoch" => TimeFunc::UnixEpoch,
522        b"strftime" => TimeFunc::StrfTime,
523        b"timediff" => TimeFunc::TimeDiff,
524        _ => return None,
525    };
526    Some(func)
527}
528
529/// A math built-in.
530///
531/// They are their own enum rather than more `ScalarFunc` variants because they
532/// are a compile-time option in SQLite (`SQLITE_ENABLE_MATH_FUNCTIONS`) and
533/// share one rule the others do not: an argument outside the domain is NULL
534/// rather than an error or a NaN.
535#[derive(Clone, Copy, Debug, PartialEq, Eq)]
536pub enum MathFunc {
537    /// `acos(x)`
538    Acos,
539    /// `acosh(x)`
540    Acosh,
541    /// `asin(x)`
542    Asin,
543    /// `asinh(x)`
544    Asinh,
545    /// `atan(x)`
546    Atan,
547    /// `atan2(y, x)`
548    Atan2,
549    /// `atanh(x)`
550    Atanh,
551    /// `ceil(x)` and `ceiling(x)`
552    Ceil,
553    /// `cos(x)`
554    Cos,
555    /// `cosh(x)`
556    Cosh,
557    /// `degrees(x)`
558    Degrees,
559    /// `exp(x)`
560    Exp,
561    /// `floor(x)`
562    Floor,
563    /// `ln(x)`
564    Ln,
565    /// `log(x)` base 10, or `log(b, x)` base b.
566    Log,
567    /// `log10(x)`
568    Log10,
569    /// `log2(x)`
570    Log2,
571    /// `mod(x, y)`
572    Mod,
573    /// `pi()`
574    Pi,
575    /// `pow(x, y)` and `power(x, y)`
576    Pow,
577    /// `radians(x)`
578    Radians,
579    /// `sin(x)`
580    Sin,
581    /// `sinh(x)`
582    Sinh,
583    /// `sqrt(x)`
584    Sqrt,
585    /// `tan(x)`
586    Tan,
587    /// `tanh(x)`
588    Tanh,
589    /// `trunc(x)`
590    Trunc,
591}
592
593impl MathFunc {
594    /// Returns how many arguments the function takes, as `(least, most)`.
595    pub fn arity(self) -> (usize, usize) {
596        match self {
597            MathFunc::Pi => (0, 0),
598            MathFunc::Atan2 | MathFunc::Mod | MathFunc::Pow => (2, 2),
599            MathFunc::Log => (1, 2),
600            _ => (1, 1),
601        }
602    }
603}
604
605/// Returns the math function a folded name spells.
606pub fn lookup_math(folded: &[u8]) -> Option<MathFunc> {
607    let func = match folded {
608        b"acos" => MathFunc::Acos,
609        b"acosh" => MathFunc::Acosh,
610        b"asin" => MathFunc::Asin,
611        b"asinh" => MathFunc::Asinh,
612        b"atan" => MathFunc::Atan,
613        b"atan2" => MathFunc::Atan2,
614        b"atanh" => MathFunc::Atanh,
615        b"ceil" | b"ceiling" => MathFunc::Ceil,
616        b"cos" => MathFunc::Cos,
617        b"cosh" => MathFunc::Cosh,
618        b"degrees" => MathFunc::Degrees,
619        b"exp" => MathFunc::Exp,
620        b"floor" => MathFunc::Floor,
621        b"ln" => MathFunc::Ln,
622        b"log" => MathFunc::Log,
623        b"log10" => MathFunc::Log10,
624        b"log2" => MathFunc::Log2,
625        b"mod" => MathFunc::Mod,
626        b"pi" => MathFunc::Pi,
627        b"pow" | b"power" => MathFunc::Pow,
628        b"radians" => MathFunc::Radians,
629        b"sin" => MathFunc::Sin,
630        b"sinh" => MathFunc::Sinh,
631        b"sqrt" => MathFunc::Sqrt,
632        b"tan" => MathFunc::Tan,
633        b"tanh" => MathFunc::Tanh,
634        b"trunc" => MathFunc::Trunc,
635        _ => return None,
636    };
637    Some(func)
638}
639
640/// A window function that is not an aggregate.
641///
642/// The aggregates are the same functions in a different frame, so they are not
643/// listed again here: `sum(x) OVER (...)` is `AggregateFunc::Sum` with a frame,
644/// and giving it a second spelling would mean two implementations of `sum`.
645#[derive(Clone, Copy, Debug, PartialEq, Eq)]
646pub enum WindowFunc {
647    /// `row_number()`
648    RowNumber,
649    /// `rank()`
650    Rank,
651    /// `dense_rank()`
652    DenseRank,
653    /// `percent_rank()`
654    PercentRank,
655    /// `cume_dist()`
656    CumeDist,
657    /// `ntile(n)`
658    Ntile,
659    /// `lag(x[, offset[, default]])`
660    Lag,
661    /// `lead(x[, offset[, default]])`
662    Lead,
663    /// `first_value(x)`
664    FirstValue,
665    /// `last_value(x)`
666    LastValue,
667    /// `nth_value(x, n)`
668    NthValue,
669}
670
671impl WindowFunc {
672    /// Returns how many arguments the function takes, as `(least, most)`.
673    pub fn arity(self) -> (usize, usize) {
674        match self {
675            WindowFunc::RowNumber
676            | WindowFunc::Rank
677            | WindowFunc::DenseRank
678            | WindowFunc::PercentRank
679            | WindowFunc::CumeDist => (0, 0),
680            WindowFunc::Ntile | WindowFunc::FirstValue | WindowFunc::LastValue => (1, 1),
681            WindowFunc::NthValue => (2, 2),
682            WindowFunc::Lag | WindowFunc::Lead => (1, 3),
683        }
684    }
685}
686
687/// Returns the window function a folded name spells.
688pub fn lookup_window(folded: &[u8]) -> Option<WindowFunc> {
689    let func = match folded {
690        b"row_number" => WindowFunc::RowNumber,
691        b"rank" => WindowFunc::Rank,
692        b"dense_rank" => WindowFunc::DenseRank,
693        b"percent_rank" => WindowFunc::PercentRank,
694        b"cume_dist" => WindowFunc::CumeDist,
695        b"ntile" => WindowFunc::Ntile,
696        b"lag" => WindowFunc::Lag,
697        b"lead" => WindowFunc::Lead,
698        b"first_value" => WindowFunc::FirstValue,
699        b"last_value" => WindowFunc::LastValue,
700        b"nth_value" => WindowFunc::NthValue,
701        _ => return None,
702    };
703    Some(func)
704}
705
706/// A JSON built-in.
707///
708/// They are their own enum for the same reason the math functions are: they
709/// share a rule none of the others has. Every one of them can fail - a document
710/// that will not parse is an error and not a NULL - and every one of them cares
711/// whether its arguments are already JSON, which is a property of the value
712/// rather than of the expression. Folding them into `ScalarFunc` would push
713/// both facts onto eighty functions that have neither.
714///
715/// The `b` spellings return the binary format rather than text. They are
716/// separate identities rather than a flag because `json_extract` and
717/// `jsonb_extract` differ in more than their output: the text form answers a
718/// SQL value for a leaf and the binary form answers a document.
719#[derive(Clone, Copy, Debug, PartialEq, Eq)]
720pub enum JsonFunc {
721    /// `json(X)`
722    Json,
723    /// `jsonb(X)`
724    Jsonb,
725    /// `json_array(...)`
726    Array,
727    /// `jsonb_array(...)`
728    ArrayB,
729    /// `json_array_length(X[, P])`
730    ArrayLength,
731    /// `json_error_position(X)`
732    ErrorPosition,
733    /// `json_extract(X, P, ...)`
734    Extract,
735    /// `jsonb_extract(X, P, ...)`
736    ExtractB,
737    /// The `->` operator.
738    Arrow,
739    /// The `->>` operator.
740    ArrowShift,
741    /// `json_insert(X, P, V, ...)`
742    Insert,
743    /// `jsonb_insert(X, P, V, ...)`
744    InsertB,
745    /// `json_object(...)`
746    Object,
747    /// `jsonb_object(...)`
748    ObjectB,
749    /// `json_patch(T, P)`
750    Patch,
751    /// `jsonb_patch(T, P)`
752    PatchB,
753    /// `json_pretty(X[, indent])`
754    Pretty,
755    /// `json_remove(X, P, ...)`
756    Remove,
757    /// `jsonb_remove(X, P, ...)`
758    RemoveB,
759    /// `json_replace(X, P, V, ...)`
760    Replace,
761    /// `jsonb_replace(X, P, V, ...)`
762    ReplaceB,
763    /// `json_set(X, P, V, ...)`
764    Set,
765    /// `jsonb_set(X, P, V, ...)`
766    SetB,
767    /// `json_type(X[, P])`
768    Type,
769    /// `json_valid(X[, flags])`
770    Valid,
771    /// `json_quote(X)`
772    Quote,
773    /// `json_array_insert(X, P, V, ...)`
774    ArrayInsert,
775    /// `jsonb_array_insert(X, P, V, ...)`
776    ArrayInsertB,
777}
778
779impl JsonFunc {
780    /// Returns how many arguments the function takes, as `(least, most)`.
781    ///
782    /// `usize::MAX` as the upper bound means "any number", which the editing
783    /// functions further restrict to an odd count in
784    /// [`JsonFunc::arity_ok`] - a rule a pair of bounds cannot express.
785    pub fn arity(self) -> (usize, usize) {
786        match self {
787            JsonFunc::Json | JsonFunc::Jsonb | JsonFunc::ErrorPosition | JsonFunc::Quote => (1, 1),
788            JsonFunc::Array | JsonFunc::ArrayB | JsonFunc::Object | JsonFunc::ObjectB => {
789                (0, usize::MAX)
790            }
791            JsonFunc::ArrayLength | JsonFunc::Type | JsonFunc::Valid | JsonFunc::Pretty => (1, 2),
792            JsonFunc::Patch | JsonFunc::PatchB | JsonFunc::Arrow | JsonFunc::ArrowShift => (2, 2),
793            // **SQLite registers these with any argument count and checks it when the
794            // function runs.** `json_extract()` and `json_extract(X)` answer NULL, an
795            // editing function with an even count fails with `json_set() needs an odd
796            // number of arguments`, and `json_object('a')` fails with `json_object()
797            // requires an even number of arguments`. Refusing the count while binding
798            // answered a parse error with `wrong number of arguments to function`
799            // where SQLite answers a runtime error with its own sentence.
800            JsonFunc::Extract
801            | JsonFunc::ExtractB
802            | JsonFunc::Remove
803            | JsonFunc::RemoveB
804            | JsonFunc::Insert
805            | JsonFunc::InsertB
806            | JsonFunc::Replace
807            | JsonFunc::ReplaceB
808            | JsonFunc::Set
809            | JsonFunc::SetB
810            | JsonFunc::ArrayInsert
811            | JsonFunc::ArrayInsertB => (0, usize::MAX),
812        }
813    }
814
815    /// Returns whether an argument count is legal for this function.
816    ///
817    /// The counts that depend on parity, `json_object` and the editing functions,
818    /// are checked when the function runs, because that is where SQLite checks them
819    /// and the error it gives is a runtime one.
820    pub fn arity_ok(self, count: usize) -> bool {
821        let (least, most) = self.arity();
822        count >= least && count <= most
823    }
824
825    /// Returns whether the function answers the binary format.
826    pub fn is_binary(self) -> bool {
827        matches!(
828            self,
829            JsonFunc::Jsonb
830                | JsonFunc::ArrayB
831                | JsonFunc::ExtractB
832                | JsonFunc::InsertB
833                | JsonFunc::ObjectB
834                | JsonFunc::PatchB
835                | JsonFunc::RemoveB
836                | JsonFunc::ReplaceB
837                | JsonFunc::SetB
838        )
839    }
840
841    /// Returns whether this function's first argument names a document to be
842    /// read, rather than a value to be embedded or quoted.
843    ///
844    /// The distinction an executor's document-cache optimisation needs: it
845    /// may only substitute a pre-parsed JSONB blob for the first argument
846    /// when that argument *is* the document a call reads, such as `X` in
847    /// `json_extract(X, P)`. `json_array`, `json_object` and `json_quote`
848    /// take that same position as a **value** - one that merely happens to
849    /// look like JSON is still meant to be embedded or quoted as a string,
850    /// per the subtype rule this module's own doc comment states. Handing
851    /// them a blob instead answered "JSON cannot hold BLOB values" for a
852    /// perfectly ordinary unmarked string, which is what
853    /// `json_array('[1]')` did before this existed. `Valid` reads its
854    /// argument as a document too, but is excluded by its caller for the
855    /// unrelated reason that substituting a re-encoded blob changes what its
856    /// flags answer about the original text.
857    pub fn first_argument_is_a_document(self) -> bool {
858        !matches!(
859            self,
860            JsonFunc::Array
861                | JsonFunc::ArrayB
862                | JsonFunc::Object
863                | JsonFunc::ObjectB
864                | JsonFunc::Quote
865        )
866    }
867}
868
869/// Returns the JSON function a folded name spells.
870pub fn lookup_json(folded: &[u8]) -> Option<JsonFunc> {
871    let func = match folded {
872        b"json" => JsonFunc::Json,
873        b"jsonb" => JsonFunc::Jsonb,
874        b"json_array" => JsonFunc::Array,
875        b"jsonb_array" => JsonFunc::ArrayB,
876        b"json_array_length" => JsonFunc::ArrayLength,
877        b"json_error_position" => JsonFunc::ErrorPosition,
878        b"json_extract" => JsonFunc::Extract,
879        // **The operators are function names too.** SQLite registers `->` and
880        // `->>` as ordinary two-argument functions, so `"->"(a, b)` binds and
881        // `pragma_function_list` reports them. The parser lowered the operators
882        // here already; only the spellings were missing, which made this engine
883        // report two fewer functions than it has and refuse a call SQLite
884        // answers.
885        b"->" => JsonFunc::Arrow,
886        b"->>" => JsonFunc::ArrowShift,
887        b"jsonb_extract" => JsonFunc::ExtractB,
888        b"json_array_insert" => JsonFunc::ArrayInsert,
889        b"jsonb_array_insert" => JsonFunc::ArrayInsertB,
890        b"json_insert" => JsonFunc::Insert,
891        b"jsonb_insert" => JsonFunc::InsertB,
892        b"json_object" => JsonFunc::Object,
893        b"jsonb_object" => JsonFunc::ObjectB,
894        b"json_patch" => JsonFunc::Patch,
895        b"jsonb_patch" => JsonFunc::PatchB,
896        b"json_pretty" => JsonFunc::Pretty,
897        b"json_remove" => JsonFunc::Remove,
898        b"jsonb_remove" => JsonFunc::RemoveB,
899        b"json_replace" => JsonFunc::Replace,
900        b"jsonb_replace" => JsonFunc::ReplaceB,
901        b"json_set" => JsonFunc::Set,
902        b"jsonb_set" => JsonFunc::SetB,
903        b"json_type" => JsonFunc::Type,
904        b"json_valid" => JsonFunc::Valid,
905        b"json_quote" => JsonFunc::Quote,
906        _ => return None,
907    };
908    Some(func)
909}
910
911/// Returns the scalar function a folded name spells.
912pub fn lookup_scalar(folded: &[u8]) -> Option<ScalarFunc> {
913    let func = match folded {
914        b"abs" => ScalarFunc::Abs,
915        b"char" => ScalarFunc::Char,
916        b"coalesce" => ScalarFunc::Coalesce,
917        b"concat" => ScalarFunc::Concat,
918        b"concat_ws" => ScalarFunc::ConcatWs,
919        b"glob" => ScalarFunc::Glob,
920        b"hex" => ScalarFunc::Hex,
921        b"ifnull" => ScalarFunc::IfNull,
922        b"iif" | b"if" => ScalarFunc::Iif,
923        b"instr" => ScalarFunc::Instr,
924        b"length" => ScalarFunc::Length,
925        b"like" => ScalarFunc::Like,
926        b"likelihood" | b"likely" | b"unlikely" => ScalarFunc::Likelihood,
927        b"lower" => ScalarFunc::Lower,
928        b"ltrim" => ScalarFunc::LTrim,
929        b"max" => ScalarFunc::Max,
930        b"min" => ScalarFunc::Min,
931        b"nullif" => ScalarFunc::NullIf,
932        b"quote" => ScalarFunc::Quote,
933        b"replace" => ScalarFunc::Replace,
934        b"round" => ScalarFunc::Round,
935        b"rtrim" => ScalarFunc::RTrim,
936        b"sign" => ScalarFunc::Sign,
937        b"substr" | b"substring" => ScalarFunc::Substr,
938        b"printf" | b"format" => ScalarFunc::Printf,
939        b"octet_length" => ScalarFunc::OctetLength,
940        b"random" => ScalarFunc::Random,
941        b"randomblob" => ScalarFunc::RandomBlob,
942        b"changes" => ScalarFunc::Changes,
943        b"total_changes" => ScalarFunc::TotalChanges,
944        b"last_insert_rowid" => ScalarFunc::LastInsertRowid,
945        b"sqlite_source_id" => ScalarFunc::SourceId,
946        b"fts5_source_id" => ScalarFunc::Fts5SourceId,
947        b"trim" => ScalarFunc::Trim,
948        b"typeof" => ScalarFunc::TypeOf,
949        b"unhex" => ScalarFunc::Unhex,
950        b"unicode" => ScalarFunc::Unicode,
951        b"upper" => ScalarFunc::Upper,
952        b"zeroblob" => ScalarFunc::ZeroBlob,
953        b"sqlite_version" => ScalarFunc::Version,
954        b"vector_distance_cos" | b"cosine_distance" => ScalarFunc::VectorDistanceCos,
955        b"vector_distance_l2" | b"l2_distance" => ScalarFunc::VectorDistanceL2,
956        b"vector_dot" | b"inner_product" => ScalarFunc::VectorDot,
957        // **Both spellings of each distance.** `l1_distance` is pgvector's name
958        // and `vector_distance_l1` is this engine's own, and the family reads
959        // as a family only if every member answers to both - `cos` and `l2`
960        // already did, and `l1` answered to one of the two.
961        b"l1_distance" | b"vector_distance_l1" => ScalarFunc::VectorDistanceL1,
962        b"hamming_distance" | b"vector_distance_hamming" => ScalarFunc::VectorDistanceHamming,
963        b"jaccard_distance" | b"vector_distance_jaccard" => ScalarFunc::VectorDistanceJaccard,
964        b"vector_dims" => ScalarFunc::VectorDims,
965        b"vector_norm" => ScalarFunc::VectorNorm,
966        b"l2_normalize" => ScalarFunc::VectorNormalize,
967        b"binary_quantize" => ScalarFunc::VectorQuantize,
968        b"subvector" => ScalarFunc::VectorSlice,
969        b"vector_add" => ScalarFunc::VectorAdd,
970        b"vector_sub" => ScalarFunc::VectorSubtract,
971        b"vector_mul" => ScalarFunc::VectorMultiply,
972        b"vector_concat" => ScalarFunc::VectorConcat,
973        b"geopoly_area" => ScalarFunc::GeopolyArea,
974        b"geopoly_blob" => ScalarFunc::GeopolyBlob,
975        b"geopoly_json" => ScalarFunc::GeopolyJson,
976        b"geopoly_svg" => ScalarFunc::GeopolySvg,
977        b"geopoly_within" => ScalarFunc::GeopolyWithin,
978        b"geopoly_contains_point" => ScalarFunc::GeopolyContainsPoint,
979        b"geopoly_overlap" => ScalarFunc::GeopolyOverlap,
980        b"geopoly_debug" => ScalarFunc::GeopolyDebug,
981        b"geopoly_bbox" => ScalarFunc::GeopolyBbox,
982        b"geopoly_xform" => ScalarFunc::GeopolyXform,
983        b"geopoly_regular" => ScalarFunc::GeopolyRegular,
984        b"geopoly_ccw" => ScalarFunc::GeopolyCcw,
985        b"unknown" => ScalarFunc::Unknown,
986        b"subtype" => ScalarFunc::Subtype,
987        b"unistr" => ScalarFunc::Unistr,
988        b"unistr_quote" => ScalarFunc::UnistrQuote,
989        b"sqlite_compileoption_used" => ScalarFunc::CompileOptionUsed,
990        b"sqlite_compileoption_get" => ScalarFunc::CompileOptionGet,
991        b"sqlite_log" => ScalarFunc::Log,
992        b"load_extension" => ScalarFunc::LoadExtension,
993        b"regexp" => ScalarFunc::Regexp,
994        b"sqlite_offset" => ScalarFunc::Offset,
995        b"sqlar_compress" => ScalarFunc::SqlarCompress,
996        b"sqlar_uncompress" => ScalarFunc::SqlarUncompress,
997        b"rtreedepth" => ScalarFunc::RTreeDepth,
998        b"rtreenode" => ScalarFunc::RTreeNode,
999        b"rtreecheck" => ScalarFunc::RTreeCheck,
1000        _ => return None,
1001    };
1002    Some(func)
1003}
1004
1005/// Returns the aggregate a folded name spells.
1006///
1007/// `min` and `max` are both: one argument makes them aggregates and two or more
1008/// make them scalars, which is why the binder asks about the argument count
1009/// before it decides.
1010pub fn lookup_aggregate(folded: &[u8]) -> Option<AggregateFunc> {
1011    let func = match folded {
1012        b"count" => AggregateFunc::Count,
1013        b"sum" => AggregateFunc::Sum,
1014        b"total" => AggregateFunc::Total,
1015        b"avg" => AggregateFunc::Avg,
1016        b"group_concat" | b"string_agg" => AggregateFunc::GroupConcat,
1017        b"json_group_array" => AggregateFunc::JsonGroupArray,
1018        b"jsonb_group_array" => AggregateFunc::JsonbGroupArray,
1019        b"json_group_object" => AggregateFunc::JsonGroupObject,
1020        b"geopoly_group_bbox" => AggregateFunc::GeopolyGroupBbox,
1021        b"median" => AggregateFunc::Median,
1022        b"percentile" => AggregateFunc::Percentile,
1023        b"percentile_cont" => AggregateFunc::PercentileCont,
1024        b"percentile_disc" => AggregateFunc::PercentileDisc,
1025        b"jsonb_group_object" => AggregateFunc::JsonbGroupObject,
1026        _ => return None,
1027    };
1028    Some(func)
1029}
1030
1031/// Returns whether an argument count is legal for a scalar function.
1032pub fn scalar_arity_ok(func: ScalarFunc, count: usize) -> bool {
1033    match func {
1034        ScalarFunc::Abs
1035        | ScalarFunc::Hex
1036        | ScalarFunc::Length
1037        | ScalarFunc::Lower
1038        | ScalarFunc::Quote
1039        | ScalarFunc::Sign
1040        | ScalarFunc::TypeOf
1041        | ScalarFunc::Unicode
1042        | ScalarFunc::Upper
1043        | ScalarFunc::ZeroBlob => count == 1,
1044        ScalarFunc::IfNull | ScalarFunc::NullIf | ScalarFunc::Glob => count == 2,
1045        ScalarFunc::VectorDistanceCos
1046        | ScalarFunc::VectorDistanceL2
1047        | ScalarFunc::VectorDot
1048        | ScalarFunc::VectorDistanceL1
1049        | ScalarFunc::VectorDistanceHamming
1050        | ScalarFunc::VectorDistanceJaccard
1051        | ScalarFunc::VectorAdd
1052        | ScalarFunc::VectorSubtract
1053        | ScalarFunc::VectorMultiply
1054        | ScalarFunc::VectorConcat => count == 2,
1055        ScalarFunc::VectorDims
1056        | ScalarFunc::VectorNorm
1057        | ScalarFunc::VectorNormalize
1058        | ScalarFunc::VectorQuantize => count == 1,
1059        ScalarFunc::VectorSlice => count == 3,
1060        ScalarFunc::RTreeDepth | ScalarFunc::Offset | ScalarFunc::SqlarCompress => count == 1,
1061        ScalarFunc::SqlarUncompress => count == 2,
1062        ScalarFunc::RTreeNode => count == 2,
1063        // One argument is the table and two is a schema and a table, which is
1064        // the same pair `rtreecheck` takes in the reference.
1065        ScalarFunc::RTreeCheck => count == 1 || count == 2,
1066        ScalarFunc::GeopolyArea
1067        | ScalarFunc::GeopolyBlob
1068        | ScalarFunc::GeopolyJson
1069        | ScalarFunc::GeopolyDebug
1070        | ScalarFunc::GeopolyBbox
1071        | ScalarFunc::GeopolyCcw => count == 1,
1072        ScalarFunc::GeopolyWithin | ScalarFunc::GeopolyOverlap => count == 2,
1073        ScalarFunc::GeopolyContainsPoint => count == 3,
1074        ScalarFunc::GeopolyRegular => count == 4,
1075        ScalarFunc::GeopolyXform => count == 7,
1076        ScalarFunc::GeopolySvg => count >= 1,
1077        ScalarFunc::Replace => count == 3,
1078        // `iif` is `CASE` written as a call: pairs of a test and a value, with
1079        // an optional final answer. Two arguments is the shortest legal form
1080        // and there is no upper bound, which is why it is not `count == 3`.
1081        ScalarFunc::Iif => count >= 2,
1082        ScalarFunc::Unknown => true,
1083        ScalarFunc::Subtype
1084        | ScalarFunc::Unistr
1085        | ScalarFunc::UnistrQuote
1086        | ScalarFunc::CompileOptionUsed
1087        | ScalarFunc::CompileOptionGet => count == 1,
1088        ScalarFunc::Log | ScalarFunc::Regexp => count == 2,
1089        ScalarFunc::LoadExtension => count == 1 || count == 2,
1090        ScalarFunc::Instr => count == 2,
1091        ScalarFunc::Like => count == 2 || count == 3,
1092        ScalarFunc::Likelihood => count == 1 || count == 2,
1093        ScalarFunc::LTrim | ScalarFunc::RTrim | ScalarFunc::Trim | ScalarFunc::Unhex => {
1094            count == 1 || count == 2
1095        }
1096        ScalarFunc::Round => count == 1 || count == 2,
1097        ScalarFunc::Substr => count == 2 || count == 3,
1098        ScalarFunc::Coalesce | ScalarFunc::Max | ScalarFunc::Min => count >= 2,
1099        // `char()` with no arguments is the empty string in SQLite, not a
1100        // parse error (task-1979, F16). `concat()` keeps its floor of one,
1101        // which is the reference's own rule for that name.
1102        ScalarFunc::Char => true,
1103        ScalarFunc::Concat => count >= 1,
1104        ScalarFunc::ConcatWs => count >= 2,
1105        ScalarFunc::Version => count == 0,
1106        ScalarFunc::Printf => count >= 1,
1107        ScalarFunc::OctetLength | ScalarFunc::RandomBlob => count == 1,
1108        ScalarFunc::Random
1109        | ScalarFunc::Changes
1110        | ScalarFunc::TotalChanges
1111        | ScalarFunc::LastInsertRowid
1112        | ScalarFunc::SourceId
1113        | ScalarFunc::Fts5SourceId => count == 0,
1114    }
1115}
1116
1117/// Returns whether an argument count is legal for an aggregate.
1118pub fn aggregate_arity_ok(func: AggregateFunc, count: usize, star: bool) -> bool {
1119    match func {
1120        AggregateFunc::Count => star || count == 1,
1121        AggregateFunc::Sum | AggregateFunc::Total | AggregateFunc::Avg => !star && count == 1,
1122        AggregateFunc::Min | AggregateFunc::Max => !star && count == 1,
1123        AggregateFunc::GroupConcat => !star && (count == 1 || count == 2),
1124        AggregateFunc::JsonGroupArray | AggregateFunc::JsonbGroupArray => !star && count == 1,
1125        AggregateFunc::JsonGroupObject | AggregateFunc::JsonbGroupObject => !star && count == 2,
1126        AggregateFunc::Median
1127        | AggregateFunc::GeopolyGroupBbox
1128        | AggregateFunc::VectorSum
1129        | AggregateFunc::VectorAvg => !star && count == 1,
1130        AggregateFunc::Percentile
1131        | AggregateFunc::PercentileCont
1132        | AggregateFunc::PercentileDisc => !star && count == 2,
1133        // An application's aggregate declared its own arity, and the binder
1134        // checked it against the registration before getting here.
1135        AggregateFunc::External => !star,
1136    }
1137}
1138
1139/// Returns whether a folded name may be an aggregate at this argument count.
1140///
1141/// `min(x)` is the aggregate and `min(x, y)` is the scalar; asking the question
1142/// this way keeps the rule in one place instead of in both lookups.
1143pub fn is_aggregate_call(folded: &[u8], count: usize, star: bool) -> bool {
1144    if folded == b"min" || folded == b"max" {
1145        return !star && count == 1;
1146    }
1147    lookup_aggregate(folded).is_some()
1148}
1149
1150/// Reports whether a built-in function's answer depends on more than its
1151/// arguments, which is what SQLite will not allow in a generated column or an
1152/// index.
1153///
1154/// @param folded - the function's folded name
1155pub fn is_volatile(folded: &[u8]) -> bool {
1156    // SQLite registers the version and compile option functions without the
1157    // deterministic flag, so an index or a generated column may not call them.
1158    const NOT_DETERMINISTIC: [&[u8]; 4] = [
1159        b"sqlite_version",
1160        b"sqlite_source_id",
1161        b"sqlite_compileoption_used",
1162        b"sqlite_compileoption_get",
1163    ];
1164    NOT_DETERMINISTIC
1165        .iter()
1166        .any(|name| name.eq_ignore_ascii_case(folded))
1167        || VOLATILE
1168            .iter()
1169            .any(|(name, _)| name.as_bytes().eq_ignore_ascii_case(folded))
1170}
1171
1172/// Returns the aggregate a `min`/`max` call resolves to at one argument.
1173pub fn minmax_aggregate(folded: &[u8]) -> Option<AggregateFunc> {
1174    match folded {
1175        b"min" => Some(AggregateFunc::Min),
1176        b"max" => Some(AggregateFunc::Max),
1177        _ => None,
1178    }
1179}
1180
1181#[cfg(test)]
1182mod tests {
1183    use super::*;
1184
1185    /// The three functions that run a model answer `unsupported` in a build with no embedding
1186    /// support, and a name nobody has is still just a name nobody has.
1187    ///
1188    /// The engine reaches this list when a call does not resolve, so a build without the feature
1189    /// tells the caller the statement is fine and the build lacks the feature, with exit code 3,
1190    /// instead of "no such function".
1191    #[test]
1192    fn the_model_functions_need_a_component() {
1193        for name in [&b"embed"[..], b"embed_tokens", b"rerank"] {
1194            let said = needs_a_component(name).expect("the function needs a component");
1195            assert!(said.contains("no embedding support compiled in"), "{said}");
1196        }
1197        assert!(
1198            needs_a_component(b"rerank").is_some_and(|said| said.starts_with("rerank(TEXT, TEXT)"))
1199        );
1200        assert_eq!(needs_a_component(b"nope"), None);
1201    }
1202
1203    /// Names are matched folded, and an unknown name is not a function.
1204    #[test]
1205    fn lookup_matches_folded_names() {
1206        assert_eq!(lookup_scalar(b"abs"), Some(ScalarFunc::Abs));
1207        assert_eq!(lookup_scalar(b"substring"), Some(ScalarFunc::Substr));
1208        assert_eq!(lookup_scalar(b"nope"), None);
1209        assert_eq!(lookup_aggregate(b"count"), Some(AggregateFunc::Count));
1210        assert_eq!(
1211            lookup_aggregate(b"string_agg"),
1212            Some(AggregateFunc::GroupConcat)
1213        );
1214    }
1215
1216    /// `min` and `max` change identity with their argument count, which is the
1217    /// one place SQLite overloads a name across the scalar/aggregate boundary.
1218    #[test]
1219    fn min_and_max_are_aggregates_only_at_one_argument() {
1220        assert!(is_aggregate_call(b"min", 1, false));
1221        assert!(!is_aggregate_call(b"min", 2, false));
1222        assert!(!is_aggregate_call(b"min", 0, true));
1223        assert_eq!(minmax_aggregate(b"max"), Some(AggregateFunc::Max));
1224    }
1225
1226    /// Arity is checked at bind time, so a wrong count is a prepare failure.
1227    #[test]
1228    fn arity_is_checked_per_function() {
1229        assert!(scalar_arity_ok(ScalarFunc::Abs, 1));
1230        assert!(!scalar_arity_ok(ScalarFunc::Abs, 2));
1231        assert!(scalar_arity_ok(ScalarFunc::Substr, 2));
1232        assert!(scalar_arity_ok(ScalarFunc::Substr, 3));
1233        assert!(!scalar_arity_ok(ScalarFunc::Substr, 4));
1234        assert!(scalar_arity_ok(ScalarFunc::Coalesce, 5));
1235        assert!(!scalar_arity_ok(ScalarFunc::Coalesce, 1));
1236        assert!(aggregate_arity_ok(AggregateFunc::Count, 0, true));
1237        assert!(!aggregate_arity_ok(AggregateFunc::Sum, 0, true));
1238    }
1239}
1240
1241/// One row of `PRAGMA function_list`.
1242#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1243pub struct FunctionEntry {
1244    /// The name as it is written.
1245    pub name: &'static str,
1246    /// `s` for a scalar, `w` for an aggregate or window function that can run
1247    /// over a window, `a` for an aggregate that cannot.
1248    pub kind: &'static str,
1249    /// How many arguments, or -1 for any number.
1250    pub arity: i64,
1251    /// The flag word the C surface reports.
1252    ///
1253    /// 2048 is `SQLITE_INNOCUOUS` and 524288 is `SQLITE_DETERMINISTIC`, which
1254    /// is what a built-in carries: it does nothing an expression could not, and
1255    /// it answers the same thing twice.
1256    pub flags: i64,
1257}
1258
1259/// The bit `function_list` sets for a function a schema may safely call.
1260///
1261/// Named rather than written twice because `inillucent-engine`'s
1262/// `function_list` reports the connection's registered functions beside these
1263/// built-ins, and it has to describe them in the same column with the same
1264/// meaning. A registered function that promised `innocuous` and was reported
1265/// with a bit nothing else uses would be a register that under-describes, which
1266/// is the defect this whole list was extended to fix.
1267pub const INNOCUOUS_FLAG: i64 = 2048;
1268
1269/// The bit `function_list` sets for a function that answers the same twice.
1270pub const DETERMINISTIC_FLAG: i64 = 524288;
1271
1272/// The flags every built-in carries: innocuous and deterministic.
1273const BUILTIN_FLAGS: i64 = INNOCUOUS_FLAG | DETERMINISTIC_FLAG;
1274
1275/// The flags a built-in that is not deterministic carries.
1276const VOLATILE_FLAGS: i64 = INNOCUOUS_FLAG;
1277
1278/// Returns every built-in this build has, in the order `function_list` reports.
1279///
1280/// The list is written out rather than derived from the lookup tables because
1281/// the arity is per *overload*: `substr` is here twice, at two and at three
1282/// arguments, which is what SQLite reports and what an application checking
1283/// whether a call will bind needs to see.
1284///
1285/// **It must name everything the binder will resolve, and a completeness check
1286/// found that it did not.** The register answered 161 names where SQLite answers 218, and
1287/// the functionality behind most of the difference was present and
1288/// byte-identical - `current_date`, `regexp`, `unistr`, `median`, `bm25`,
1289/// `matchinfo` and the rest all answered when called. A caller that
1290/// introspects the register to decide what it may use was told less than the
1291/// truth, with no error, which is the one *silent* difference this project has
1292/// had. The additions below were each verified against the engine before being
1293/// listed: a name here that the binder refuses would be the same defect
1294/// pointing the other way.
1295pub fn every_function() -> Vec<FunctionEntry> {
1296    let mut out = Vec::new();
1297    let mut scalar = |name: &'static str, arity: i64| {
1298        out.push(FunctionEntry {
1299            name,
1300            kind: "s",
1301            arity,
1302            flags: BUILTIN_FLAGS,
1303        });
1304    };
1305    for (name, arity) in SCALARS {
1306        scalar(name, *arity);
1307    }
1308    for (name, arity) in VOLATILE {
1309        out.push(FunctionEntry {
1310            name,
1311            kind: "s",
1312            arity: *arity,
1313            flags: VOLATILE_FLAGS,
1314        });
1315    }
1316    // **Every built-in aggregate is reported as `w`.** SQLite's `type` column
1317    // says `w` for an aggregate that can also run over a window, which is every
1318    // one of its own (`max`, `sum`, `count`, `group_concat`, ...); `a` is for an
1319    // aggregate that cannot, and none of these is one.
1320    for (name, arity) in AGGREGATES {
1321        out.push(FunctionEntry {
1322            name,
1323            kind: "w",
1324            arity: *arity,
1325            flags: BUILTIN_FLAGS,
1326        });
1327    }
1328    for (name, arity) in WINDOWS {
1329        out.push(FunctionEntry {
1330            name,
1331            kind: "w",
1332            arity: *arity,
1333            flags: BUILTIN_FLAGS,
1334        });
1335    }
1336    out.sort_by(|left, right| left.name.cmp(right.name).then(left.arity.cmp(&right.arity)));
1337    out
1338}
1339
1340/// The deterministic scalars, with one row per overload.
1341///
1342/// **`narg` is SQLite's own encoding, not "how many arguments".** A negative
1343/// number means variadic *and carries a minimum*: `coalesce` reads -4 and
1344/// `concat` -3 in the reference's register, not -1. A
1345/// register-completeness check compares this column because it is the one an
1346/// application reads to decide whether a call will bind, and it found seven
1347/// entries here that disagreed with the reference while answering identically.
1348const SCALARS: &[(&str, i64)] = &[
1349    ("abs", 1),
1350    ("acos", 1),
1351    ("acosh", 1),
1352    ("asin", 1),
1353    ("asinh", 1),
1354    ("atan", 1),
1355    ("atan2", 2),
1356    ("atanh", 1),
1357    ("ceil", 1),
1358    ("ceiling", 1),
1359    ("char", -1),
1360    ("coalesce", -4),
1361    ("concat", -3),
1362    ("concat_ws", -4),
1363    ("cos", 1),
1364    ("cosh", 1),
1365    ("date", -1),
1366    ("datetime", -1),
1367    ("degrees", 1),
1368    ("exp", 1),
1369    ("floor", 1),
1370    ("format", -1),
1371    ("glob", 2),
1372    ("hex", 1),
1373    ("ifnull", 2),
1374    ("iif", -4),
1375    ("instr", 2),
1376    ("json", 1),
1377    ("json_array", -1),
1378    ("json_array_length", 1),
1379    ("json_array_length", 2),
1380    ("json_error_position", 1),
1381    ("json_extract", -1),
1382    ("json_insert", -1),
1383    ("json_object", -1),
1384    ("json_patch", 2),
1385    ("json_pretty", 1),
1386    ("json_pretty", 2),
1387    ("json_quote", 1),
1388    ("json_remove", -1),
1389    ("json_replace", -1),
1390    ("json_set", -1),
1391    ("json_type", 1),
1392    ("json_type", 2),
1393    ("json_valid", 1),
1394    ("json_valid", 2),
1395    ("jsonb", 1),
1396    ("jsonb_array", -1),
1397    ("jsonb_extract", -1),
1398    ("jsonb_insert", -1),
1399    ("jsonb_object", -1),
1400    ("jsonb_patch", 2),
1401    ("jsonb_remove", -1),
1402    ("jsonb_replace", -1),
1403    ("jsonb_set", -1),
1404    ("julianday", -1),
1405    ("length", 1),
1406    ("like", 2),
1407    ("like", 3),
1408    ("likelihood", 2),
1409    ("likely", 1),
1410    ("ln", 1),
1411    ("log", 1),
1412    ("log", 2),
1413    ("log10", 1),
1414    ("log2", 1),
1415    ("lower", 1),
1416    ("ltrim", 1),
1417    ("ltrim", 2),
1418    ("max", -3),
1419    ("min", -3),
1420    ("mod", 2),
1421    ("nullif", 2),
1422    ("octet_length", 1),
1423    ("pi", 0),
1424    ("pow", 2),
1425    ("power", 2),
1426    ("printf", -1),
1427    ("quote", 1),
1428    ("radians", 1),
1429    ("replace", 3),
1430    ("round", 1),
1431    ("round", 2),
1432    ("rtrim", 1),
1433    ("rtrim", 2),
1434    ("sign", 1),
1435    ("sin", 1),
1436    ("sinh", 1),
1437    ("fts5_source_id", 0),
1438    ("optimize", 1),
1439    ("sqlite_source_id", 0),
1440    ("sqlite_version", 0),
1441    ("sqrt", 1),
1442    ("strftime", -1),
1443    ("substr", 2),
1444    ("substr", 3),
1445    ("substring", 2),
1446    ("substring", 3),
1447    ("tan", 1),
1448    ("tanh", 1),
1449    ("time", -1),
1450    ("timediff", 2),
1451    ("trim", 1),
1452    ("trim", 2),
1453    ("trunc", 1),
1454    ("typeof", 1),
1455    ("unhex", 1),
1456    ("unhex", 2),
1457    ("unicode", 1),
1458    ("unixepoch", -1),
1459    ("unlikely", 1),
1460    ("upper", 1),
1461    ("binary_quantize", 1),
1462    ("rtreecheck", -1),
1463    ("sqlar_compress", 1),
1464    ("sqlar_uncompress", 2),
1465    ("sqlite_offset", 1),
1466    ("rtreedepth", 1),
1467    ("rtreenode", 2),
1468    ("geopoly_area", 1),
1469    ("geopoly_bbox", 1),
1470    ("geopoly_blob", 1),
1471    ("geopoly_ccw", 1),
1472    ("geopoly_contains_point", 3),
1473    ("geopoly_debug", 1),
1474    ("geopoly_group_bbox", 1),
1475    ("geopoly_json", 1),
1476    ("geopoly_overlap", 2),
1477    ("geopoly_regular", 4),
1478    ("geopoly_svg", -1),
1479    ("geopoly_within", 2),
1480    ("geopoly_xform", 7),
1481    ("cosine_distance", 2),
1482    ("hamming_distance", 2),
1483    ("inner_product", 2),
1484    ("jaccard_distance", 2),
1485    ("l1_distance", 2),
1486    ("l2_distance", 2),
1487    ("l2_normalize", 1),
1488    ("subvector", 3),
1489    ("vector_add", 2),
1490    ("vector_concat", 2),
1491    ("vector_dims", 1),
1492    ("vector_distance_cos", 2),
1493    ("vector_distance_l2", 2),
1494    ("vector_dot", 2),
1495    ("vector_mul", 2),
1496    ("vector_norm", 1),
1497    ("vector_sub", 2),
1498    ("zeroblob", 1),
1499    // Present and answering, and missing from this list until now.
1500    // Each was checked against the shell before it was added.
1501    ("->", 2),
1502    ("->>", 2),
1503    ("bm25", -1),
1504    ("highlight", -1),
1505    ("if", -4),
1506    ("json_array_insert", -1),
1507    ("jsonb_array_insert", -1),
1508    ("match", 2),
1509    ("matchinfo", 1),
1510    ("matchinfo", 2),
1511    ("offsets", 1),
1512    ("regexp", 2),
1513    ("snippet", -1),
1514    ("sqlite_compileoption_get", 1),
1515    ("sqlite_compileoption_used", 1),
1516    ("subtype", 1),
1517    ("unistr", 1),
1518    ("unistr_quote", 1),
1519    ("unknown", -1),
1520];
1521
1522/// The scalars whose answer depends on something other than their arguments.
1523const VOLATILE: &[(&str, i64)] = &[
1524    ("changes", 0),
1525    // The three date keywords are functions in SQLite's register and answer
1526    // like functions here; they read the clock, so they are not deterministic.
1527    ("current_date", 0),
1528    ("current_time", 0),
1529    ("current_timestamp", 0),
1530    ("last_insert_rowid", 0),
1531    ("load_extension", 1),
1532    ("load_extension", 2),
1533    ("random", 0),
1534    ("randomblob", 1),
1535    ("sqlite_log", 2),
1536    ("total_changes", 0),
1537];
1538
1539/// The aggregates, with one row per overload.
1540const AGGREGATES: &[(&str, i64)] = &[
1541    ("avg", 1),
1542    ("count", 0),
1543    ("count", 1),
1544    ("group_concat", 1),
1545    ("group_concat", 2),
1546    ("json_group_array", 1),
1547    ("json_group_object", 2),
1548    ("jsonb_group_array", 1),
1549    ("jsonb_group_object", 2),
1550    ("max", 1),
1551    ("min", 1),
1552    ("string_agg", 2),
1553    ("sum", 1),
1554    ("total", 1),
1555];
1556
1557/// The window functions that are not aggregates.
1558const WINDOWS: &[(&str, i64)] = &[
1559    ("cume_dist", 0),
1560    ("dense_rank", 0),
1561    ("first_value", 1),
1562    ("lag", 1),
1563    ("lag", 2),
1564    ("lag", 3),
1565    ("last_value", 1),
1566    ("lead", 1),
1567    ("lead", 2),
1568    ("lead", 3),
1569    ("nth_value", 2),
1570    ("ntile", 1),
1571    ("percent_rank", 0),
1572    ("rank", 0),
1573    ("row_number", 0),
1574    // The percentile family, which SQLite reports as window functions and which
1575    // this engine answers as both aggregates and window functions.
1576    ("median", 1),
1577    ("percentile", 2),
1578    ("percentile_cont", 2),
1579    ("percentile_disc", 2),
1580];