inillucent_sql/function.rs
1//! The built-in function registry: names, arities, and identities.
2//!
3//! Invariant: a function is recognised here or it does not exist. The binder
4//! resolves a name to one of these identities and refuses everything else with
5//! "no such function", so an unknown name fails at prepare time rather than
6//! part-way through a scan, and the VM never dispatches on a string.
7//!
8//! Arity is checked here too, because SQLite reports "wrong number of arguments
9//! to function abs()" from prepare rather than from execution.
10
11/// The names that exist in inillucent but need a component this build has not
12/// got.
13///
14/// **`embed` is the whole list, and it is here rather than in the registry
15/// because the registry is where it is absent** (task-1979, section 8.1, gap
16/// 12). `inillucent-search` registers `embed` only when the `embed` feature is
17/// compiled in, so on a build without it the name reaches the binder's
18/// "no such function" path and answered exit 1 - which says the caller
19/// misspelled something. The statement is spelled correctly and this build has
20/// not got the function, which is exactly what exit 3 means.
21///
22/// A build that *does* have `embed` never reaches here, because the registry
23/// resolves the name before the refusal is built. A machine that has the
24/// function and not the model is a third thing again and keeps its own status:
25/// `inillucent-search`'s `no_model` answers `invalid_state` and names
26/// `inillucent setup-embeddings`, because the component is installable and
27/// exit 3 would say the opposite.
28const NEEDS_A_COMPONENT: &[(&[u8], &str)] = &[
29 (
30 b"embed",
31 "embed(TEXT): this build has no embedding support compiled in",
32 ),
33 (
34 b"embed_tokens",
35 "embed_tokens(TEXT): this build has no embedding support compiled in",
36 ),
37 (
38 b"rerank",
39 "rerank(TEXT, TEXT): this build has no embedding support compiled in",
40 ),
41];
42
43/// Returns what a name needs, when the name is one this build left out.
44///
45/// @param name - the folded function name that did not resolve
46pub fn needs_a_component(name: &[u8]) -> Option<&'static str> {
47 NEEDS_A_COMPONENT
48 .iter()
49 .find(|(known, _)| *known == name)
50 .map(|(_, said)| *said)
51}
52
53/// A scalar built-in.
54#[derive(Clone, Copy, Debug, PartialEq, Eq)]
55pub enum ScalarFunc {
56 /// `abs(x)`
57 Abs,
58 /// `char(...)`
59 Char,
60 /// `coalesce(...)`
61 Coalesce,
62 /// `concat(...)`
63 Concat,
64 /// `concat_ws(sep, ...)`
65 ConcatWs,
66 /// `glob(pattern, text)`
67 Glob,
68 /// `hex(x)`
69 Hex,
70 /// `ifnull(a, b)`
71 IfNull,
72 /// `iif(a, b, c)`
73 Iif,
74 /// `instr(haystack, needle)`
75 Instr,
76 /// `length(x)`
77 Length,
78 /// `like(pattern, text[, escape])`
79 Like,
80 /// `likelihood(x, y)`, `likely(x)` and `unlikely(x)`, which are no-ops.
81 Likelihood,
82 /// `lower(x)`
83 Lower,
84 /// `ltrim(x[, chars])`
85 LTrim,
86 /// `max(a, b, ...)`, the scalar form.
87 Max,
88 /// `min(a, b, ...)`, the scalar form.
89 Min,
90 /// `nullif(a, b)`
91 NullIf,
92 /// `quote(x)`
93 Quote,
94 /// `replace(text, from, to)`
95 Replace,
96 /// `round(x[, digits])`
97 Round,
98 /// `rtrim(x[, chars])`
99 RTrim,
100 /// `sign(x)`
101 Sign,
102 /// `substr(x, start[, length])`
103 Substr,
104 /// `trim(x[, chars])`
105 Trim,
106 /// `typeof(x)`
107 TypeOf,
108 /// `unhex(x[, chars])`
109 Unhex,
110 /// `unicode(x)`
111 Unicode,
112 /// `upper(x)`
113 Upper,
114 /// `zeroblob(n)`
115 ZeroBlob,
116 /// `printf(format, ...)` and `format(format, ...)`
117 Printf,
118 /// `octet_length(x)`
119 OctetLength,
120 /// `random()`
121 Random,
122 /// `randomblob(n)`
123 RandomBlob,
124 /// `changes()`
125 Changes,
126 /// `total_changes()`
127 TotalChanges,
128 /// `last_insert_rowid()`
129 LastInsertRowid,
130 /// `sqlite_source_id()`
131 SourceId,
132 /// `fts5_source_id()`
133 Fts5SourceId,
134 /// `sqlite_version()`
135 Version,
136 /// `vector_distance_cos(a, b)`, the cosine distance between two vectors.
137 ///
138 /// **Not a SQLite function, and the first one this engine adds.** pgvector
139 /// spells it `a <=> b`; the whole point of Phase 2's Part 7 is that a
140 /// vector is a value a `SELECT` can order by, and an operator that is sugar
141 /// for a function needs the function to exist first. A vector is a blob of
142 /// little-endian `f32`, which is what `inillucent_search` already stores and
143 /// what `vector_distance_l2` and `vector_dot` read too.
144 VectorDistanceCos,
145 /// `vector_distance_l2(a, b)`, the Euclidean distance between two vectors.
146 VectorDistanceL2,
147 /// `vector_dot(a, b)`, the dot product of two vectors.
148 ///
149 /// Negated relative to pgvector's `<#>`, which answers the *negative* inner
150 /// product so that a smaller number is a better match. This answers the dot
151 /// product itself, because a function named `dot` that returned its negative
152 /// would be a trap; the ordering sugar negates where it needs to.
153 VectorDot,
154 /// `l1_distance(a, b)`, the taxicab distance, spelled `a <+> b`.
155 VectorDistanceL1,
156 /// `hamming_distance(a, b)`, how many components differ.
157 ///
158 /// pgvector defines it over its `bit` type and spells it `a <~> b`. Here a
159 /// bit vector is the blob `binary_quantize` produces, and the distance is
160 /// the population count of the two blobs' exclusive-or - which is the same
161 /// number, computed the same way, over the representation this engine has.
162 VectorDistanceHamming,
163 /// `jaccard_distance(a, b)`, one minus the overlap, spelled `a <%> b`.
164 VectorDistanceJaccard,
165 /// `vector_dims(a)`, how many components a vector has.
166 VectorDims,
167 /// `vector_norm(a)`, its Euclidean length.
168 VectorNorm,
169 /// `l2_normalize(a)`, the same direction with length one.
170 VectorNormalize,
171 /// `binary_quantize(a)`, one bit per component: set when it is positive.
172 VectorQuantize,
173 /// `subvector(a, start, count)`, a slice, counted from one.
174 VectorSlice,
175 /// `vector_add(a, b)`, component by component.
176 ///
177 /// **A function rather than `+`, and that is a compatibility choice rather
178 /// than a shortcut.** pgvector can overload `+` because a `vector` is a
179 /// distinct type in PostgreSQL; here a vector is a blob, and SQLite says
180 /// that a blob in arithmetic is zero. Overloading the operator for every
181 /// blob would change the answer to `x'00' + x'00'` from `0` to a blob,
182 /// which is a difference every application that adds two blobs would see.
183 ///
184 /// **The operators were given back, on the one condition that keeps
185 /// both answers.** `a + b` binds to this function when a side reads a
186 /// column *declared* `VECTOR(n)` - which is the same thing PostgreSQL is
187 /// using, a declared type - and stays SQLite's arithmetic otherwise. So
188 /// `x'00' + x'00'` is still `0` and `v + v` over a vector column is a
189 /// vector.
190 VectorAdd,
191 /// `vector_sub(a, b)`, component by component.
192 VectorSubtract,
193 /// `vector_mul(a, b)`, component by component.
194 VectorMultiply,
195 /// `vector_concat(a, b)`, one vector after the other.
196 VectorConcat,
197 /// `geopoly_area(P)`, the signed area a polygon encloses.
198 ///
199 /// **The `geopoly` surface is thirteen functions and one aggregate**, and
200 /// they are listed here individually rather than folded into one
201 /// `Geopoly(kind)` variant because arity checking reads this enum: they
202 /// take one, two, three, four, seven and any number of arguments, and a
203 /// single variant could not say so.
204 GeopolyArea,
205 /// `geopoly_blob(P)`, the stored form of a polygon.
206 GeopolyBlob,
207 /// `geopoly_json(P)`, the GeoJSON form.
208 GeopolyJson,
209 /// `geopoly_svg(P, ...)`, an SVG `<polyline>` with the extra arguments
210 /// written into the tag.
211 GeopolySvg,
212 /// `geopoly_within(P1, P2)`, whether the second is inside the first.
213 GeopolyWithin,
214 /// `geopoly_contains_point(P, X, Y)`, where a point sits.
215 GeopolyContainsPoint,
216 /// `geopoly_overlap(P1, P2)`, how two polygons meet.
217 GeopolyOverlap,
218 /// `geopoly_debug(X)`, which answers nothing.
219 ///
220 /// It switches on the reference's own tracing, which only exists in a build
221 /// made with `GEOPOLY_ENABLE_DEBUG`; in every other build it reads its
222 /// argument and returns nothing at all. That is what this does, and it is
223 /// registered because a name the reference resolves and this engine does
224 /// not is a difference an application can see.
225 GeopolyDebug,
226 /// `geopoly_bbox(P)`, the bounding box as a four-sided polygon.
227 GeopolyBbox,
228 /// `geopoly_xform(P, A, B, C, D, E, F)`, an affine transform.
229 GeopolyXform,
230 /// `geopoly_regular(X, Y, R, N)`, a regular polygon.
231 GeopolyRegular,
232 /// `geopoly_ccw(P)`, the same ring wound counter-clockwise.
233 GeopolyCcw,
234 /// `unknown(...)`, which answers NULL to anything.
235 ///
236 /// SQLite registers it, lists it in `function_list`, and returns NULL from
237 /// it whatever it is given. It is here because a name the reference resolves
238 /// and this engine does not is a difference an application can see.
239 Unknown,
240 /// `subtype(x)`, the tag a function attached to its answer.
241 Subtype,
242 /// `unistr(x)`, which expands `\uXXXX` and `\UXXXXXXXX` escapes.
243 Unistr,
244 /// `unistr_quote(x)`, `quote()` with the control characters escaped.
245 UnistrQuote,
246 /// `sqlite_compileoption_used(name)`
247 CompileOptionUsed,
248 /// `sqlite_compileoption_get(n)`
249 CompileOptionGet,
250 /// `sqlite_log(code, message)`, which writes to the log and answers NULL.
251 Log,
252 /// `load_extension(path[, entry])`
253 LoadExtension,
254 /// `regexp(pattern, subject)`, which is what `X REGEXP Y` calls.
255 Regexp,
256 /// `sqlar_compress(X)`, a blob compressed if that makes it smaller.
257 ///
258 /// **The archive format's own rule, and it is why this is not just a
259 /// compressor.** A row of a `.sqlar` table holds either a zlib stream or
260 /// the raw bytes, and which one is decided by whichever is shorter; the
261 /// stored `sz` column is what tells the two apart on the way back. So a
262 /// value that does not compress is stored as it stands, and a value that is
263 /// not a blob at all is returned unchanged, type and all.
264 SqlarCompress,
265 /// `sqlar_uncompress(Z, SZ)`, the inverse.
266 ///
267 /// `SZ` is the size the row claims the content is. When it equals the
268 /// blob's own length the blob *is* the content and is returned unchanged,
269 /// which is how the format says "this one was stored raw".
270 SqlarUncompress,
271 /// `sqlite_offset(X)`, where in the file the row holding X is.
272 ///
273 /// **The page, not the record, and that is the whole of the difference.**
274 /// SQLite reports the byte offset of the *record* a value would be read
275 /// from, because a row there is one contiguous run of bytes. A leaf here is
276 /// PAX: each column is its own run, so one row occupies several places on
277 /// its page and there is no single offset for it. What is reported is the
278 /// offset of the page, which is where the value is genuinely read from.
279 ///
280 /// Folded to its answer by the physical pass, like `rtreecheck`, because it
281 /// is a question about a *tree* rather than about a value.
282 Offset,
283 /// `rtreedepth(X)`, the depth stored at the front of an R-Tree node.
284 RTreeDepth,
285 /// `rtreenode(D, X)`, an R-Tree node rendered as a readable list.
286 RTreeNode,
287 /// `rtreecheck(T)`, an integrity check over one R-Tree table.
288 ///
289 /// **Answered where the table is reachable, which is not here.** A scalar
290 /// is handed values and nothing else; this one is about a *table*, so the
291 /// physical pass folds it to its answer while it still has the catalog,
292 /// and what reaches the evaluator is already the text. Running once per
293 /// preparation rather than once per row is also what it means: the
294 /// argument is a table name, so the answer cannot vary down a column.
295 RTreeCheck,
296}
297
298/// An aggregate built-in.
299#[derive(Clone, Copy, Debug, PartialEq, Eq)]
300pub enum AggregateFunc {
301 /// `count(x)` and `count(*)`
302 Count,
303 /// `sum(x)`
304 Sum,
305 /// `total(x)`
306 Total,
307 /// `avg(x)`
308 Avg,
309 /// `min(x)`
310 Min,
311 /// `max(x)`
312 Max,
313 /// `group_concat(x[, sep])` and `string_agg(x, sep)`
314 GroupConcat,
315 /// `json_group_array(x)`
316 JsonGroupArray,
317 /// `jsonb_group_array(x)`
318 JsonbGroupArray,
319 /// `json_group_object(label, x)`
320 JsonGroupObject,
321 /// `jsonb_group_object(label, x)`
322 JsonbGroupObject,
323 /// `median(x)`, which is `percentile_cont(x, 0.5)` under a shorter name.
324 Median,
325 /// `geopoly_group_bbox(P)`, the box that holds every polygon in the group.
326 GeopolyGroupBbox,
327 /// `sum(v)` and `total(v)` over a vector column, component by component.
328 ///
329 /// Not a name a caller writes: the binder picks it when `sum`'s argument
330 /// reads a vector, because that is where the argument's type is known.
331 VectorSum,
332 /// `avg(v)` over a vector column, component by component.
333 VectorAvg,
334 /// `percentile(x, p)`, where `p` runs 0 to 100.
335 Percentile,
336 /// `percentile_cont(x, f)`, where `f` runs 0 to 1 and the answer is
337 /// interpolated between the two rows it falls between.
338 PercentileCont,
339 /// `percentile_disc(x, f)`, which answers one of the rows rather than a
340 /// value between two of them.
341 PercentileDisc,
342 /// An aggregate an application registered, named beside the call.
343 ///
344 /// The name is not in here because this enum is `Copy` and travels through
345 /// the program's operands; it rides in `AggregateCall` instead.
346 External,
347}
348
349/// What a registered function promises about itself.
350///
351/// It lives here, below `inillucent-ext`, because two different layers have to
352/// read the same promise: `inillucent_ext::registry::Registry` records it when
353/// an application registers a function, and the binder enforces it when a
354/// schema names one. `inillucent-ext` re-exports this type, so a registrant
355/// writes `inillucent_ext::registry::FunctionFlags` exactly as before.
356#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
357pub struct FunctionFlags {
358 /// The function may only be called from top-level SQL, never from a
359 /// schema: not from a `DEFAULT`, a `CHECK`, a generated column, an index
360 /// expression, a partial-index predicate, a view or a trigger.
361 ///
362 /// [`FunctionFlags::external`] sets this, because the safe assumption about
363 /// code somebody else wrote is that it does something. **It is not what the
364 /// `Default` derive gives**, which is every flag false: a registrant who
365 /// writes `..FunctionFlags::default()` gets a function a schema may name.
366 /// That is the hole `embed` was registered through (task-1969, 7.4), and
367 /// `inillucent_ext::registry::UserFunction::external` is the constructor to
368 /// reach for instead.
369 pub direct_only: bool,
370 /// The function does nothing an ordinary expression could not: no side
371 /// effects, no file access, no dependence on anything but its arguments.
372 pub innocuous: bool,
373 /// The function returns the same answer for the same arguments within one
374 /// statement, so the planner may call it once.
375 pub deterministic: bool,
376}
377
378impl FunctionFlags {
379 /// Returns the flags a built-in carries: safe for a schema to call.
380 pub fn builtin() -> FunctionFlags {
381 FunctionFlags {
382 direct_only: false,
383 innocuous: true,
384 deterministic: true,
385 }
386 }
387
388 /// Returns the flags anything registered from outside carries by default.
389 pub fn external() -> FunctionFlags {
390 FunctionFlags {
391 direct_only: true,
392 innocuous: false,
393 deterministic: false,
394 }
395 }
396}
397
398/// Which context a name is being resolved from.
399#[derive(Clone, Copy, Debug, PartialEq, Eq)]
400pub enum CallSite {
401 /// The statement an application submitted.
402 Statement,
403 /// A `DEFAULT`, `CHECK`, generated column, index expression, partial-index
404 /// predicate, view or trigger stored in the schema.
405 Schema,
406}
407
408/// Returns why a schema may not call this function, or nothing when it may.
409///
410/// **One rule, read by two layers (task-1972).** `Registry::authorize_function`
411/// wraps the answer in a `DbError` for an application that asks the registry
412/// directly, and the binder wraps it in a `ParseError` for the statement it is
413/// compiling. Writing the rule twice is how the two would eventually disagree,
414/// and the half nobody exercised would be the permissive one.
415///
416/// The rule reads the same way SQLite's does: a direct-only function is never
417/// callable from a schema; anything else is callable from a schema only when
418/// the connection trusts the schema or the function is innocuous.
419///
420/// @param flags - what the function promises about itself
421/// @param site - where the call was written
422/// @param trusted_schema - whether the connection trusts the schema it read
423pub fn schema_refusal(
424 flags: FunctionFlags,
425 site: CallSite,
426 trusted_schema: bool,
427) -> Option<&'static str> {
428 if site == CallSite::Statement {
429 return None;
430 }
431 if flags.direct_only {
432 return Some("may only be used from top-level SQL");
433 }
434 if trusted_schema || flags.innocuous {
435 return None;
436 }
437 Some("is not allowed in a schema")
438}
439
440/// A function an application registered, as the binder needs to see it.
441///
442/// Only what resolution needs: a name, how many arguments it takes, whether it
443/// reduces a group, and what it promises about itself. What it *does* is the
444/// machine's business.
445///
446/// **The flags are here because the binder is where the promise is kept
447/// (task-1972).** `Registry::authorize_function` had no caller, so
448/// `direct_only`, `innocuous` and `PRAGMA trusted_schema` were a policy with a
449/// passing unit test and no effect on the engine: a `CHECK`, an index
450/// expression or a generated column could name any registered function whatever
451/// its flags. `inillucent-sql` sits below `inillucent-ext` and cannot reach the
452/// registry, so what the registry knows travels down here with the name.
453#[derive(Clone, Debug, PartialEq, Eq)]
454pub struct ExternalFunction {
455 /// The folded name.
456 pub name: Vec<u8>,
457 /// How many arguments it takes, or -1 for any number.
458 pub arity: i32,
459 /// Whether it reduces a group rather than a row.
460 pub aggregate: bool,
461 /// What it promises about itself, which decides whether a schema may name
462 /// it.
463 pub flags: FunctionFlags,
464}
465
466impl ExternalFunction {
467 /// Returns whether this registration answers a call with this many
468 /// arguments.
469 pub fn accepts(&self, argc: usize) -> bool {
470 self.arity < 0 || self.arity as usize == argc
471 }
472}
473
474/// Returns the registration that answers a call, preferring an exact arity.
475///
476/// SQLite resolves the same way: a function registered for exactly this many
477/// arguments wins over one registered for any number, so an application can
478/// define both a fast two-argument form and a general one.
479pub fn lookup_external<'a>(
480 functions: &'a [ExternalFunction],
481 name: &[u8],
482 argc: usize,
483) -> Option<&'a ExternalFunction> {
484 let folded = name.to_ascii_lowercase();
485 functions
486 .iter()
487 .find(|function| function.name == folded && function.arity as usize == argc)
488 .or_else(|| {
489 functions
490 .iter()
491 .find(|function| function.name == folded && function.arity < 0)
492 })
493}
494
495/// A date or time built-in.
496#[derive(Clone, Copy, Debug, PartialEq, Eq)]
497pub enum TimeFunc {
498 /// `date(...)`
499 Date,
500 /// `time(...)`
501 Time,
502 /// `datetime(...)`
503 DateTime,
504 /// `julianday(...)`
505 JulianDay,
506 /// `unixepoch(...)`
507 UnixEpoch,
508 /// `strftime(format, ...)`
509 StrfTime,
510 /// `timediff(a, b)`
511 TimeDiff,
512}
513
514/// Returns the date or time function a folded name spells.
515pub fn lookup_time(folded: &[u8]) -> Option<TimeFunc> {
516 let func = match folded {
517 b"date" => TimeFunc::Date,
518 b"time" => TimeFunc::Time,
519 b"datetime" => TimeFunc::DateTime,
520 b"julianday" => TimeFunc::JulianDay,
521 b"unixepoch" => TimeFunc::UnixEpoch,
522 b"strftime" => TimeFunc::StrfTime,
523 b"timediff" => TimeFunc::TimeDiff,
524 _ => return None,
525 };
526 Some(func)
527}
528
529/// A math built-in.
530///
531/// They are their own enum rather than more `ScalarFunc` variants because they
532/// are a compile-time option in SQLite (`SQLITE_ENABLE_MATH_FUNCTIONS`) and
533/// share one rule the others do not: an argument outside the domain is NULL
534/// rather than an error or a NaN.
535#[derive(Clone, Copy, Debug, PartialEq, Eq)]
536pub enum MathFunc {
537 /// `acos(x)`
538 Acos,
539 /// `acosh(x)`
540 Acosh,
541 /// `asin(x)`
542 Asin,
543 /// `asinh(x)`
544 Asinh,
545 /// `atan(x)`
546 Atan,
547 /// `atan2(y, x)`
548 Atan2,
549 /// `atanh(x)`
550 Atanh,
551 /// `ceil(x)` and `ceiling(x)`
552 Ceil,
553 /// `cos(x)`
554 Cos,
555 /// `cosh(x)`
556 Cosh,
557 /// `degrees(x)`
558 Degrees,
559 /// `exp(x)`
560 Exp,
561 /// `floor(x)`
562 Floor,
563 /// `ln(x)`
564 Ln,
565 /// `log(x)` base 10, or `log(b, x)` base b.
566 Log,
567 /// `log10(x)`
568 Log10,
569 /// `log2(x)`
570 Log2,
571 /// `mod(x, y)`
572 Mod,
573 /// `pi()`
574 Pi,
575 /// `pow(x, y)` and `power(x, y)`
576 Pow,
577 /// `radians(x)`
578 Radians,
579 /// `sin(x)`
580 Sin,
581 /// `sinh(x)`
582 Sinh,
583 /// `sqrt(x)`
584 Sqrt,
585 /// `tan(x)`
586 Tan,
587 /// `tanh(x)`
588 Tanh,
589 /// `trunc(x)`
590 Trunc,
591}
592
593impl MathFunc {
594 /// Returns how many arguments the function takes, as `(least, most)`.
595 pub fn arity(self) -> (usize, usize) {
596 match self {
597 MathFunc::Pi => (0, 0),
598 MathFunc::Atan2 | MathFunc::Mod | MathFunc::Pow => (2, 2),
599 MathFunc::Log => (1, 2),
600 _ => (1, 1),
601 }
602 }
603}
604
605/// Returns the math function a folded name spells.
606pub fn lookup_math(folded: &[u8]) -> Option<MathFunc> {
607 let func = match folded {
608 b"acos" => MathFunc::Acos,
609 b"acosh" => MathFunc::Acosh,
610 b"asin" => MathFunc::Asin,
611 b"asinh" => MathFunc::Asinh,
612 b"atan" => MathFunc::Atan,
613 b"atan2" => MathFunc::Atan2,
614 b"atanh" => MathFunc::Atanh,
615 b"ceil" | b"ceiling" => MathFunc::Ceil,
616 b"cos" => MathFunc::Cos,
617 b"cosh" => MathFunc::Cosh,
618 b"degrees" => MathFunc::Degrees,
619 b"exp" => MathFunc::Exp,
620 b"floor" => MathFunc::Floor,
621 b"ln" => MathFunc::Ln,
622 b"log" => MathFunc::Log,
623 b"log10" => MathFunc::Log10,
624 b"log2" => MathFunc::Log2,
625 b"mod" => MathFunc::Mod,
626 b"pi" => MathFunc::Pi,
627 b"pow" | b"power" => MathFunc::Pow,
628 b"radians" => MathFunc::Radians,
629 b"sin" => MathFunc::Sin,
630 b"sinh" => MathFunc::Sinh,
631 b"sqrt" => MathFunc::Sqrt,
632 b"tan" => MathFunc::Tan,
633 b"tanh" => MathFunc::Tanh,
634 b"trunc" => MathFunc::Trunc,
635 _ => return None,
636 };
637 Some(func)
638}
639
640/// A window function that is not an aggregate.
641///
642/// The aggregates are the same functions in a different frame, so they are not
643/// listed again here: `sum(x) OVER (...)` is `AggregateFunc::Sum` with a frame,
644/// and giving it a second spelling would mean two implementations of `sum`.
645#[derive(Clone, Copy, Debug, PartialEq, Eq)]
646pub enum WindowFunc {
647 /// `row_number()`
648 RowNumber,
649 /// `rank()`
650 Rank,
651 /// `dense_rank()`
652 DenseRank,
653 /// `percent_rank()`
654 PercentRank,
655 /// `cume_dist()`
656 CumeDist,
657 /// `ntile(n)`
658 Ntile,
659 /// `lag(x[, offset[, default]])`
660 Lag,
661 /// `lead(x[, offset[, default]])`
662 Lead,
663 /// `first_value(x)`
664 FirstValue,
665 /// `last_value(x)`
666 LastValue,
667 /// `nth_value(x, n)`
668 NthValue,
669}
670
671impl WindowFunc {
672 /// Returns how many arguments the function takes, as `(least, most)`.
673 pub fn arity(self) -> (usize, usize) {
674 match self {
675 WindowFunc::RowNumber
676 | WindowFunc::Rank
677 | WindowFunc::DenseRank
678 | WindowFunc::PercentRank
679 | WindowFunc::CumeDist => (0, 0),
680 WindowFunc::Ntile | WindowFunc::FirstValue | WindowFunc::LastValue => (1, 1),
681 WindowFunc::NthValue => (2, 2),
682 WindowFunc::Lag | WindowFunc::Lead => (1, 3),
683 }
684 }
685}
686
687/// Returns the window function a folded name spells.
688pub fn lookup_window(folded: &[u8]) -> Option<WindowFunc> {
689 let func = match folded {
690 b"row_number" => WindowFunc::RowNumber,
691 b"rank" => WindowFunc::Rank,
692 b"dense_rank" => WindowFunc::DenseRank,
693 b"percent_rank" => WindowFunc::PercentRank,
694 b"cume_dist" => WindowFunc::CumeDist,
695 b"ntile" => WindowFunc::Ntile,
696 b"lag" => WindowFunc::Lag,
697 b"lead" => WindowFunc::Lead,
698 b"first_value" => WindowFunc::FirstValue,
699 b"last_value" => WindowFunc::LastValue,
700 b"nth_value" => WindowFunc::NthValue,
701 _ => return None,
702 };
703 Some(func)
704}
705
706/// A JSON built-in.
707///
708/// They are their own enum for the same reason the math functions are: they
709/// share a rule none of the others has. Every one of them can fail - a document
710/// that will not parse is an error and not a NULL - and every one of them cares
711/// whether its arguments are already JSON, which is a property of the value
712/// rather than of the expression. Folding them into `ScalarFunc` would push
713/// both facts onto eighty functions that have neither.
714///
715/// The `b` spellings return the binary format rather than text. They are
716/// separate identities rather than a flag because `json_extract` and
717/// `jsonb_extract` differ in more than their output: the text form answers a
718/// SQL value for a leaf and the binary form answers a document.
719#[derive(Clone, Copy, Debug, PartialEq, Eq)]
720pub enum JsonFunc {
721 /// `json(X)`
722 Json,
723 /// `jsonb(X)`
724 Jsonb,
725 /// `json_array(...)`
726 Array,
727 /// `jsonb_array(...)`
728 ArrayB,
729 /// `json_array_length(X[, P])`
730 ArrayLength,
731 /// `json_error_position(X)`
732 ErrorPosition,
733 /// `json_extract(X, P, ...)`
734 Extract,
735 /// `jsonb_extract(X, P, ...)`
736 ExtractB,
737 /// The `->` operator.
738 Arrow,
739 /// The `->>` operator.
740 ArrowShift,
741 /// `json_insert(X, P, V, ...)`
742 Insert,
743 /// `jsonb_insert(X, P, V, ...)`
744 InsertB,
745 /// `json_object(...)`
746 Object,
747 /// `jsonb_object(...)`
748 ObjectB,
749 /// `json_patch(T, P)`
750 Patch,
751 /// `jsonb_patch(T, P)`
752 PatchB,
753 /// `json_pretty(X[, indent])`
754 Pretty,
755 /// `json_remove(X, P, ...)`
756 Remove,
757 /// `jsonb_remove(X, P, ...)`
758 RemoveB,
759 /// `json_replace(X, P, V, ...)`
760 Replace,
761 /// `jsonb_replace(X, P, V, ...)`
762 ReplaceB,
763 /// `json_set(X, P, V, ...)`
764 Set,
765 /// `jsonb_set(X, P, V, ...)`
766 SetB,
767 /// `json_type(X[, P])`
768 Type,
769 /// `json_valid(X[, flags])`
770 Valid,
771 /// `json_quote(X)`
772 Quote,
773 /// `json_array_insert(X, P, V, ...)`
774 ArrayInsert,
775 /// `jsonb_array_insert(X, P, V, ...)`
776 ArrayInsertB,
777}
778
779impl JsonFunc {
780 /// Returns how many arguments the function takes, as `(least, most)`.
781 ///
782 /// `usize::MAX` as the upper bound means "any number", which the editing
783 /// functions further restrict to an odd count in
784 /// [`JsonFunc::arity_ok`] - a rule a pair of bounds cannot express.
785 pub fn arity(self) -> (usize, usize) {
786 match self {
787 JsonFunc::Json | JsonFunc::Jsonb | JsonFunc::ErrorPosition | JsonFunc::Quote => (1, 1),
788 JsonFunc::Array | JsonFunc::ArrayB | JsonFunc::Object | JsonFunc::ObjectB => {
789 (0, usize::MAX)
790 }
791 JsonFunc::ArrayLength | JsonFunc::Type | JsonFunc::Valid | JsonFunc::Pretty => (1, 2),
792 JsonFunc::Patch | JsonFunc::PatchB | JsonFunc::Arrow | JsonFunc::ArrowShift => (2, 2),
793 JsonFunc::Extract | JsonFunc::ExtractB | JsonFunc::Remove | JsonFunc::RemoveB => {
794 (2, usize::MAX)
795 }
796 JsonFunc::Insert
797 | JsonFunc::InsertB
798 | JsonFunc::Replace
799 | JsonFunc::ReplaceB
800 | JsonFunc::Set
801 | JsonFunc::SetB
802 | JsonFunc::ArrayInsert
803 | JsonFunc::ArrayInsertB => (3, usize::MAX),
804 }
805 }
806
807 /// Returns whether an argument count is legal for this function.
808 pub fn arity_ok(self, count: usize) -> bool {
809 let (least, most) = self.arity();
810 if count < least || count > most {
811 return false;
812 }
813 match self {
814 // A path and a value go together, so the count past the document
815 // has to be even and the whole count therefore odd.
816 JsonFunc::Insert
817 | JsonFunc::InsertB
818 | JsonFunc::Replace
819 | JsonFunc::ReplaceB
820 | JsonFunc::Set
821 | JsonFunc::SetB
822 | JsonFunc::ArrayInsert
823 | JsonFunc::ArrayInsertB => count % 2 == 1,
824 JsonFunc::Object | JsonFunc::ObjectB => count.is_multiple_of(2),
825 _ => true,
826 }
827 }
828
829 /// Returns whether the function answers the binary format.
830 pub fn is_binary(self) -> bool {
831 matches!(
832 self,
833 JsonFunc::Jsonb
834 | JsonFunc::ArrayB
835 | JsonFunc::ExtractB
836 | JsonFunc::InsertB
837 | JsonFunc::ObjectB
838 | JsonFunc::PatchB
839 | JsonFunc::RemoveB
840 | JsonFunc::ReplaceB
841 | JsonFunc::SetB
842 )
843 }
844
845 /// Returns whether this function's first argument names a document to be
846 /// read, rather than a value to be embedded or quoted.
847 ///
848 /// The distinction an executor's document-cache optimisation needs: it
849 /// may only substitute a pre-parsed JSONB blob for the first argument
850 /// when that argument *is* the document a call reads, such as `X` in
851 /// `json_extract(X, P)`. `json_array`, `json_object` and `json_quote`
852 /// take that same position as a **value** - one that merely happens to
853 /// look like JSON is still meant to be embedded or quoted as a string,
854 /// per the subtype rule this module's own doc comment states. Handing
855 /// them a blob instead answered "JSON cannot hold BLOB values" for a
856 /// perfectly ordinary unmarked string, which is what
857 /// `json_array('[1]')` did before this existed. `Valid` reads its
858 /// argument as a document too, but is excluded by its caller for the
859 /// unrelated reason that substituting a re-encoded blob changes what its
860 /// flags answer about the original text.
861 pub fn first_argument_is_a_document(self) -> bool {
862 !matches!(
863 self,
864 JsonFunc::Array
865 | JsonFunc::ArrayB
866 | JsonFunc::Object
867 | JsonFunc::ObjectB
868 | JsonFunc::Quote
869 )
870 }
871}
872
873/// Returns the JSON function a folded name spells.
874pub fn lookup_json(folded: &[u8]) -> Option<JsonFunc> {
875 let func = match folded {
876 b"json" => JsonFunc::Json,
877 b"jsonb" => JsonFunc::Jsonb,
878 b"json_array" => JsonFunc::Array,
879 b"jsonb_array" => JsonFunc::ArrayB,
880 b"json_array_length" => JsonFunc::ArrayLength,
881 b"json_error_position" => JsonFunc::ErrorPosition,
882 b"json_extract" => JsonFunc::Extract,
883 // **The operators are function names too.** SQLite registers `->` and
884 // `->>` as ordinary two-argument functions, so `"->"(a, b)` binds and
885 // `pragma_function_list` reports them. The parser lowered the operators
886 // here already; only the spellings were missing, which made this engine
887 // report two fewer functions than it has and refuse a call SQLite
888 // answers.
889 b"->" => JsonFunc::Arrow,
890 b"->>" => JsonFunc::ArrowShift,
891 b"jsonb_extract" => JsonFunc::ExtractB,
892 b"json_array_insert" => JsonFunc::ArrayInsert,
893 b"jsonb_array_insert" => JsonFunc::ArrayInsertB,
894 b"json_insert" => JsonFunc::Insert,
895 b"jsonb_insert" => JsonFunc::InsertB,
896 b"json_object" => JsonFunc::Object,
897 b"jsonb_object" => JsonFunc::ObjectB,
898 b"json_patch" => JsonFunc::Patch,
899 b"jsonb_patch" => JsonFunc::PatchB,
900 b"json_pretty" => JsonFunc::Pretty,
901 b"json_remove" => JsonFunc::Remove,
902 b"jsonb_remove" => JsonFunc::RemoveB,
903 b"json_replace" => JsonFunc::Replace,
904 b"jsonb_replace" => JsonFunc::ReplaceB,
905 b"json_set" => JsonFunc::Set,
906 b"jsonb_set" => JsonFunc::SetB,
907 b"json_type" => JsonFunc::Type,
908 b"json_valid" => JsonFunc::Valid,
909 b"json_quote" => JsonFunc::Quote,
910 _ => return None,
911 };
912 Some(func)
913}
914
915/// Returns the scalar function a folded name spells.
916pub fn lookup_scalar(folded: &[u8]) -> Option<ScalarFunc> {
917 let func = match folded {
918 b"abs" => ScalarFunc::Abs,
919 b"char" => ScalarFunc::Char,
920 b"coalesce" => ScalarFunc::Coalesce,
921 b"concat" => ScalarFunc::Concat,
922 b"concat_ws" => ScalarFunc::ConcatWs,
923 b"glob" => ScalarFunc::Glob,
924 b"hex" => ScalarFunc::Hex,
925 b"ifnull" => ScalarFunc::IfNull,
926 b"iif" | b"if" => ScalarFunc::Iif,
927 b"instr" => ScalarFunc::Instr,
928 b"length" => ScalarFunc::Length,
929 b"like" => ScalarFunc::Like,
930 b"likelihood" | b"likely" | b"unlikely" => ScalarFunc::Likelihood,
931 b"lower" => ScalarFunc::Lower,
932 b"ltrim" => ScalarFunc::LTrim,
933 b"max" => ScalarFunc::Max,
934 b"min" => ScalarFunc::Min,
935 b"nullif" => ScalarFunc::NullIf,
936 b"quote" => ScalarFunc::Quote,
937 b"replace" => ScalarFunc::Replace,
938 b"round" => ScalarFunc::Round,
939 b"rtrim" => ScalarFunc::RTrim,
940 b"sign" => ScalarFunc::Sign,
941 b"substr" | b"substring" => ScalarFunc::Substr,
942 b"printf" | b"format" => ScalarFunc::Printf,
943 b"octet_length" => ScalarFunc::OctetLength,
944 b"random" => ScalarFunc::Random,
945 b"randomblob" => ScalarFunc::RandomBlob,
946 b"changes" => ScalarFunc::Changes,
947 b"total_changes" => ScalarFunc::TotalChanges,
948 b"last_insert_rowid" => ScalarFunc::LastInsertRowid,
949 b"sqlite_source_id" => ScalarFunc::SourceId,
950 b"fts5_source_id" => ScalarFunc::Fts5SourceId,
951 b"trim" => ScalarFunc::Trim,
952 b"typeof" => ScalarFunc::TypeOf,
953 b"unhex" => ScalarFunc::Unhex,
954 b"unicode" => ScalarFunc::Unicode,
955 b"upper" => ScalarFunc::Upper,
956 b"zeroblob" => ScalarFunc::ZeroBlob,
957 b"sqlite_version" => ScalarFunc::Version,
958 b"vector_distance_cos" | b"cosine_distance" => ScalarFunc::VectorDistanceCos,
959 b"vector_distance_l2" | b"l2_distance" => ScalarFunc::VectorDistanceL2,
960 b"vector_dot" | b"inner_product" => ScalarFunc::VectorDot,
961 // **Both spellings of each distance.** `l1_distance` is pgvector's name
962 // and `vector_distance_l1` is this engine's own, and the family reads
963 // as a family only if every member answers to both - `cos` and `l2`
964 // already did, and `l1` answered to one of the two.
965 b"l1_distance" | b"vector_distance_l1" => ScalarFunc::VectorDistanceL1,
966 b"hamming_distance" | b"vector_distance_hamming" => ScalarFunc::VectorDistanceHamming,
967 b"jaccard_distance" | b"vector_distance_jaccard" => ScalarFunc::VectorDistanceJaccard,
968 b"vector_dims" => ScalarFunc::VectorDims,
969 b"vector_norm" => ScalarFunc::VectorNorm,
970 b"l2_normalize" => ScalarFunc::VectorNormalize,
971 b"binary_quantize" => ScalarFunc::VectorQuantize,
972 b"subvector" => ScalarFunc::VectorSlice,
973 b"vector_add" => ScalarFunc::VectorAdd,
974 b"vector_sub" => ScalarFunc::VectorSubtract,
975 b"vector_mul" => ScalarFunc::VectorMultiply,
976 b"vector_concat" => ScalarFunc::VectorConcat,
977 b"geopoly_area" => ScalarFunc::GeopolyArea,
978 b"geopoly_blob" => ScalarFunc::GeopolyBlob,
979 b"geopoly_json" => ScalarFunc::GeopolyJson,
980 b"geopoly_svg" => ScalarFunc::GeopolySvg,
981 b"geopoly_within" => ScalarFunc::GeopolyWithin,
982 b"geopoly_contains_point" => ScalarFunc::GeopolyContainsPoint,
983 b"geopoly_overlap" => ScalarFunc::GeopolyOverlap,
984 b"geopoly_debug" => ScalarFunc::GeopolyDebug,
985 b"geopoly_bbox" => ScalarFunc::GeopolyBbox,
986 b"geopoly_xform" => ScalarFunc::GeopolyXform,
987 b"geopoly_regular" => ScalarFunc::GeopolyRegular,
988 b"geopoly_ccw" => ScalarFunc::GeopolyCcw,
989 b"unknown" => ScalarFunc::Unknown,
990 b"subtype" => ScalarFunc::Subtype,
991 b"unistr" => ScalarFunc::Unistr,
992 b"unistr_quote" => ScalarFunc::UnistrQuote,
993 b"sqlite_compileoption_used" => ScalarFunc::CompileOptionUsed,
994 b"sqlite_compileoption_get" => ScalarFunc::CompileOptionGet,
995 b"sqlite_log" => ScalarFunc::Log,
996 b"load_extension" => ScalarFunc::LoadExtension,
997 b"regexp" => ScalarFunc::Regexp,
998 b"sqlite_offset" => ScalarFunc::Offset,
999 b"sqlar_compress" => ScalarFunc::SqlarCompress,
1000 b"sqlar_uncompress" => ScalarFunc::SqlarUncompress,
1001 b"rtreedepth" => ScalarFunc::RTreeDepth,
1002 b"rtreenode" => ScalarFunc::RTreeNode,
1003 b"rtreecheck" => ScalarFunc::RTreeCheck,
1004 _ => return None,
1005 };
1006 Some(func)
1007}
1008
1009/// Returns the aggregate a folded name spells.
1010///
1011/// `min` and `max` are both: one argument makes them aggregates and two or more
1012/// make them scalars, which is why the binder asks about the argument count
1013/// before it decides.
1014pub fn lookup_aggregate(folded: &[u8]) -> Option<AggregateFunc> {
1015 let func = match folded {
1016 b"count" => AggregateFunc::Count,
1017 b"sum" => AggregateFunc::Sum,
1018 b"total" => AggregateFunc::Total,
1019 b"avg" => AggregateFunc::Avg,
1020 b"group_concat" | b"string_agg" => AggregateFunc::GroupConcat,
1021 b"json_group_array" => AggregateFunc::JsonGroupArray,
1022 b"jsonb_group_array" => AggregateFunc::JsonbGroupArray,
1023 b"json_group_object" => AggregateFunc::JsonGroupObject,
1024 b"geopoly_group_bbox" => AggregateFunc::GeopolyGroupBbox,
1025 b"median" => AggregateFunc::Median,
1026 b"percentile" => AggregateFunc::Percentile,
1027 b"percentile_cont" => AggregateFunc::PercentileCont,
1028 b"percentile_disc" => AggregateFunc::PercentileDisc,
1029 b"jsonb_group_object" => AggregateFunc::JsonbGroupObject,
1030 _ => return None,
1031 };
1032 Some(func)
1033}
1034
1035/// Returns whether an argument count is legal for a scalar function.
1036pub fn scalar_arity_ok(func: ScalarFunc, count: usize) -> bool {
1037 match func {
1038 ScalarFunc::Abs
1039 | ScalarFunc::Hex
1040 | ScalarFunc::Length
1041 | ScalarFunc::Lower
1042 | ScalarFunc::Quote
1043 | ScalarFunc::Sign
1044 | ScalarFunc::TypeOf
1045 | ScalarFunc::Unicode
1046 | ScalarFunc::Upper
1047 | ScalarFunc::ZeroBlob => count == 1,
1048 ScalarFunc::IfNull | ScalarFunc::NullIf | ScalarFunc::Glob => count == 2,
1049 ScalarFunc::VectorDistanceCos
1050 | ScalarFunc::VectorDistanceL2
1051 | ScalarFunc::VectorDot
1052 | ScalarFunc::VectorDistanceL1
1053 | ScalarFunc::VectorDistanceHamming
1054 | ScalarFunc::VectorDistanceJaccard
1055 | ScalarFunc::VectorAdd
1056 | ScalarFunc::VectorSubtract
1057 | ScalarFunc::VectorMultiply
1058 | ScalarFunc::VectorConcat => count == 2,
1059 ScalarFunc::VectorDims
1060 | ScalarFunc::VectorNorm
1061 | ScalarFunc::VectorNormalize
1062 | ScalarFunc::VectorQuantize => count == 1,
1063 ScalarFunc::VectorSlice => count == 3,
1064 ScalarFunc::RTreeDepth | ScalarFunc::Offset | ScalarFunc::SqlarCompress => count == 1,
1065 ScalarFunc::SqlarUncompress => count == 2,
1066 ScalarFunc::RTreeNode => count == 2,
1067 // One argument is the table and two is a schema and a table, which is
1068 // the same pair `rtreecheck` takes in the reference.
1069 ScalarFunc::RTreeCheck => count == 1 || count == 2,
1070 ScalarFunc::GeopolyArea
1071 | ScalarFunc::GeopolyBlob
1072 | ScalarFunc::GeopolyJson
1073 | ScalarFunc::GeopolyDebug
1074 | ScalarFunc::GeopolyBbox
1075 | ScalarFunc::GeopolyCcw => count == 1,
1076 ScalarFunc::GeopolyWithin | ScalarFunc::GeopolyOverlap => count == 2,
1077 ScalarFunc::GeopolyContainsPoint => count == 3,
1078 ScalarFunc::GeopolyRegular => count == 4,
1079 ScalarFunc::GeopolyXform => count == 7,
1080 ScalarFunc::GeopolySvg => count >= 1,
1081 ScalarFunc::Replace => count == 3,
1082 // `iif` is `CASE` written as a call: pairs of a test and a value, with
1083 // an optional final answer. Two arguments is the shortest legal form
1084 // and there is no upper bound, which is why it is not `count == 3`.
1085 ScalarFunc::Iif => count >= 2,
1086 ScalarFunc::Unknown => true,
1087 ScalarFunc::Subtype
1088 | ScalarFunc::Unistr
1089 | ScalarFunc::UnistrQuote
1090 | ScalarFunc::CompileOptionUsed
1091 | ScalarFunc::CompileOptionGet => count == 1,
1092 ScalarFunc::Log | ScalarFunc::Regexp => count == 2,
1093 ScalarFunc::LoadExtension => count == 1 || count == 2,
1094 ScalarFunc::Instr => count == 2,
1095 ScalarFunc::Like => count == 2 || count == 3,
1096 ScalarFunc::Likelihood => count == 1 || count == 2,
1097 ScalarFunc::LTrim | ScalarFunc::RTrim | ScalarFunc::Trim | ScalarFunc::Unhex => {
1098 count == 1 || count == 2
1099 }
1100 ScalarFunc::Round => count == 1 || count == 2,
1101 ScalarFunc::Substr => count == 2 || count == 3,
1102 ScalarFunc::Coalesce | ScalarFunc::Max | ScalarFunc::Min => count >= 2,
1103 // `char()` with no arguments is the empty string in SQLite, not a
1104 // parse error (task-1979, F16). `concat()` keeps its floor of one,
1105 // which is the reference's own rule for that name.
1106 ScalarFunc::Char => true,
1107 ScalarFunc::Concat => count >= 1,
1108 ScalarFunc::ConcatWs => count >= 2,
1109 ScalarFunc::Version => count == 0,
1110 ScalarFunc::Printf => count >= 1,
1111 ScalarFunc::OctetLength | ScalarFunc::RandomBlob => count == 1,
1112 ScalarFunc::Random
1113 | ScalarFunc::Changes
1114 | ScalarFunc::TotalChanges
1115 | ScalarFunc::LastInsertRowid
1116 | ScalarFunc::SourceId
1117 | ScalarFunc::Fts5SourceId => count == 0,
1118 }
1119}
1120
1121/// Returns whether an argument count is legal for an aggregate.
1122pub fn aggregate_arity_ok(func: AggregateFunc, count: usize, star: bool) -> bool {
1123 match func {
1124 AggregateFunc::Count => star || count == 1,
1125 AggregateFunc::Sum | AggregateFunc::Total | AggregateFunc::Avg => !star && count == 1,
1126 AggregateFunc::Min | AggregateFunc::Max => !star && count == 1,
1127 AggregateFunc::GroupConcat => !star && (count == 1 || count == 2),
1128 AggregateFunc::JsonGroupArray | AggregateFunc::JsonbGroupArray => !star && count == 1,
1129 AggregateFunc::JsonGroupObject | AggregateFunc::JsonbGroupObject => !star && count == 2,
1130 AggregateFunc::Median
1131 | AggregateFunc::GeopolyGroupBbox
1132 | AggregateFunc::VectorSum
1133 | AggregateFunc::VectorAvg => !star && count == 1,
1134 AggregateFunc::Percentile
1135 | AggregateFunc::PercentileCont
1136 | AggregateFunc::PercentileDisc => !star && count == 2,
1137 // An application's aggregate declared its own arity, and the binder
1138 // checked it against the registration before getting here.
1139 AggregateFunc::External => !star,
1140 }
1141}
1142
1143/// Returns whether a folded name may be an aggregate at this argument count.
1144///
1145/// `min(x)` is the aggregate and `min(x, y)` is the scalar; asking the question
1146/// this way keeps the rule in one place instead of in both lookups.
1147pub fn is_aggregate_call(folded: &[u8], count: usize, star: bool) -> bool {
1148 if folded == b"min" || folded == b"max" {
1149 return !star && count == 1;
1150 }
1151 lookup_aggregate(folded).is_some()
1152}
1153
1154/// Returns the aggregate a `min`/`max` call resolves to at one argument.
1155pub fn minmax_aggregate(folded: &[u8]) -> Option<AggregateFunc> {
1156 match folded {
1157 b"min" => Some(AggregateFunc::Min),
1158 b"max" => Some(AggregateFunc::Max),
1159 _ => None,
1160 }
1161}
1162
1163#[cfg(test)]
1164mod tests {
1165 use super::*;
1166
1167 /// The three functions that run a model answer `unsupported` in a build with no embedding
1168 /// support, and a name nobody has is still just a name nobody has.
1169 ///
1170 /// The engine reaches this list when a call does not resolve, so a build without the feature
1171 /// tells the caller the statement is fine and the build lacks the feature, with exit code 3,
1172 /// instead of "no such function".
1173 #[test]
1174 fn the_model_functions_need_a_component() {
1175 for name in [&b"embed"[..], b"embed_tokens", b"rerank"] {
1176 let said = needs_a_component(name).expect("the function needs a component");
1177 assert!(said.contains("no embedding support compiled in"), "{said}");
1178 }
1179 assert!(
1180 needs_a_component(b"rerank").is_some_and(|said| said.starts_with("rerank(TEXT, TEXT)"))
1181 );
1182 assert_eq!(needs_a_component(b"nope"), None);
1183 }
1184
1185 /// Names are matched folded, and an unknown name is not a function.
1186 #[test]
1187 fn lookup_matches_folded_names() {
1188 assert_eq!(lookup_scalar(b"abs"), Some(ScalarFunc::Abs));
1189 assert_eq!(lookup_scalar(b"substring"), Some(ScalarFunc::Substr));
1190 assert_eq!(lookup_scalar(b"nope"), None);
1191 assert_eq!(lookup_aggregate(b"count"), Some(AggregateFunc::Count));
1192 assert_eq!(
1193 lookup_aggregate(b"string_agg"),
1194 Some(AggregateFunc::GroupConcat)
1195 );
1196 }
1197
1198 /// `min` and `max` change identity with their argument count, which is the
1199 /// one place SQLite overloads a name across the scalar/aggregate boundary.
1200 #[test]
1201 fn min_and_max_are_aggregates_only_at_one_argument() {
1202 assert!(is_aggregate_call(b"min", 1, false));
1203 assert!(!is_aggregate_call(b"min", 2, false));
1204 assert!(!is_aggregate_call(b"min", 0, true));
1205 assert_eq!(minmax_aggregate(b"max"), Some(AggregateFunc::Max));
1206 }
1207
1208 /// Arity is checked at bind time, so a wrong count is a prepare failure.
1209 #[test]
1210 fn arity_is_checked_per_function() {
1211 assert!(scalar_arity_ok(ScalarFunc::Abs, 1));
1212 assert!(!scalar_arity_ok(ScalarFunc::Abs, 2));
1213 assert!(scalar_arity_ok(ScalarFunc::Substr, 2));
1214 assert!(scalar_arity_ok(ScalarFunc::Substr, 3));
1215 assert!(!scalar_arity_ok(ScalarFunc::Substr, 4));
1216 assert!(scalar_arity_ok(ScalarFunc::Coalesce, 5));
1217 assert!(!scalar_arity_ok(ScalarFunc::Coalesce, 1));
1218 assert!(aggregate_arity_ok(AggregateFunc::Count, 0, true));
1219 assert!(!aggregate_arity_ok(AggregateFunc::Sum, 0, true));
1220 }
1221}
1222
1223/// One row of `PRAGMA function_list`.
1224#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1225pub struct FunctionEntry {
1226 /// The name as it is written.
1227 pub name: &'static str,
1228 /// `s` for a scalar, `w` for a window function, `a` for an aggregate.
1229 pub kind: &'static str,
1230 /// How many arguments, or -1 for any number.
1231 pub arity: i64,
1232 /// The flag word the C surface reports.
1233 ///
1234 /// 2048 is `SQLITE_INNOCUOUS` and 524288 is `SQLITE_DETERMINISTIC`, which
1235 /// is what a built-in carries: it does nothing an expression could not, and
1236 /// it answers the same thing twice.
1237 pub flags: i64,
1238}
1239
1240/// The bit `function_list` sets for a function a schema may safely call.
1241///
1242/// Named rather than written twice because `inillucent-engine`'s
1243/// `function_list` reports the connection's registered functions beside these
1244/// built-ins, and it has to describe them in the same column with the same
1245/// meaning. A registered function that promised `innocuous` and was reported
1246/// with a bit nothing else uses would be a register that under-describes, which
1247/// is the defect this whole list was extended to fix.
1248pub const INNOCUOUS_FLAG: i64 = 2048;
1249
1250/// The bit `function_list` sets for a function that answers the same twice.
1251pub const DETERMINISTIC_FLAG: i64 = 524288;
1252
1253/// The flags every built-in carries: innocuous and deterministic.
1254const BUILTIN_FLAGS: i64 = INNOCUOUS_FLAG | DETERMINISTIC_FLAG;
1255
1256/// The flags a built-in that is not deterministic carries.
1257const VOLATILE_FLAGS: i64 = INNOCUOUS_FLAG;
1258
1259/// Returns every built-in this build has, in the order `function_list` reports.
1260///
1261/// The list is written out rather than derived from the lookup tables because
1262/// the arity is per *overload*: `substr` is here twice, at two and at three
1263/// arguments, which is what SQLite reports and what an application checking
1264/// whether a call will bind needs to see.
1265///
1266/// **It must name everything the binder will resolve, and a completeness check
1267/// found that it did not.** The register answered 161 names where SQLite answers 218, and
1268/// the functionality behind most of the difference was present and
1269/// byte-identical - `current_date`, `regexp`, `unistr`, `median`, `bm25`,
1270/// `matchinfo` and the rest all answered when called. A caller that
1271/// introspects the register to decide what it may use was told less than the
1272/// truth, with no error, which is the one *silent* difference this project has
1273/// had. The additions below were each verified against the engine before being
1274/// listed: a name here that the binder refuses would be the same defect
1275/// pointing the other way.
1276pub fn every_function() -> Vec<FunctionEntry> {
1277 let mut out = Vec::new();
1278 let mut scalar = |name: &'static str, arity: i64| {
1279 out.push(FunctionEntry {
1280 name,
1281 kind: "s",
1282 arity,
1283 flags: BUILTIN_FLAGS,
1284 });
1285 };
1286 for (name, arity) in SCALARS {
1287 scalar(name, *arity);
1288 }
1289 for (name, arity) in VOLATILE {
1290 out.push(FunctionEntry {
1291 name,
1292 kind: "s",
1293 arity: *arity,
1294 flags: VOLATILE_FLAGS,
1295 });
1296 }
1297 for (name, arity) in AGGREGATES {
1298 out.push(FunctionEntry {
1299 name,
1300 kind: "a",
1301 arity: *arity,
1302 flags: BUILTIN_FLAGS,
1303 });
1304 }
1305 for (name, arity) in WINDOWS {
1306 out.push(FunctionEntry {
1307 name,
1308 kind: "w",
1309 arity: *arity,
1310 flags: BUILTIN_FLAGS,
1311 });
1312 }
1313 out.sort_by(|left, right| left.name.cmp(right.name).then(left.arity.cmp(&right.arity)));
1314 out
1315}
1316
1317/// The deterministic scalars, with one row per overload.
1318///
1319/// **`narg` is SQLite's own encoding, not "how many arguments".** A negative
1320/// number means variadic *and carries a minimum*: `coalesce` reads -4 and
1321/// `concat` -3 in the reference's register, not -1. A
1322/// register-completeness check compares this column because it is the one an
1323/// application reads to decide whether a call will bind, and it found seven
1324/// entries here that disagreed with the reference while answering identically.
1325const SCALARS: &[(&str, i64)] = &[
1326 ("abs", 1),
1327 ("acos", 1),
1328 ("acosh", 1),
1329 ("asin", 1),
1330 ("asinh", 1),
1331 ("atan", 1),
1332 ("atan2", 2),
1333 ("atanh", 1),
1334 ("ceil", 1),
1335 ("ceiling", 1),
1336 ("char", -1),
1337 ("coalesce", -4),
1338 ("concat", -3),
1339 ("concat_ws", -4),
1340 ("cos", 1),
1341 ("cosh", 1),
1342 ("date", -1),
1343 ("datetime", -1),
1344 ("degrees", 1),
1345 ("exp", 1),
1346 ("floor", 1),
1347 ("format", -1),
1348 ("glob", 2),
1349 ("hex", 1),
1350 ("ifnull", 2),
1351 ("iif", -4),
1352 ("instr", 2),
1353 ("json", 1),
1354 ("json_array", -1),
1355 ("json_array_length", 1),
1356 ("json_array_length", 2),
1357 ("json_error_position", 1),
1358 ("json_extract", -1),
1359 ("json_insert", -1),
1360 ("json_object", -1),
1361 ("json_patch", 2),
1362 ("json_pretty", 1),
1363 ("json_pretty", 2),
1364 ("json_quote", 1),
1365 ("json_remove", -1),
1366 ("json_replace", -1),
1367 ("json_set", -1),
1368 ("json_type", 1),
1369 ("json_type", 2),
1370 ("json_valid", 1),
1371 ("json_valid", 2),
1372 ("jsonb", 1),
1373 ("jsonb_array", -1),
1374 ("jsonb_extract", -1),
1375 ("jsonb_insert", -1),
1376 ("jsonb_object", -1),
1377 ("jsonb_patch", 2),
1378 ("jsonb_remove", -1),
1379 ("jsonb_replace", -1),
1380 ("jsonb_set", -1),
1381 ("julianday", -1),
1382 ("length", 1),
1383 ("like", 2),
1384 ("like", 3),
1385 ("likelihood", 2),
1386 ("likely", 1),
1387 ("ln", 1),
1388 ("log", 1),
1389 ("log", 2),
1390 ("log10", 1),
1391 ("log2", 1),
1392 ("lower", 1),
1393 ("ltrim", 1),
1394 ("ltrim", 2),
1395 ("max", -3),
1396 ("min", -3),
1397 ("mod", 2),
1398 ("nullif", 2),
1399 ("octet_length", 1),
1400 ("pi", 0),
1401 ("pow", 2),
1402 ("power", 2),
1403 ("printf", -1),
1404 ("quote", 1),
1405 ("radians", 1),
1406 ("replace", 3),
1407 ("round", 1),
1408 ("round", 2),
1409 ("rtrim", 1),
1410 ("rtrim", 2),
1411 ("sign", 1),
1412 ("sin", 1),
1413 ("sinh", 1),
1414 ("fts5_source_id", 0),
1415 ("optimize", 1),
1416 ("sqlite_source_id", 0),
1417 ("sqlite_version", 0),
1418 ("sqrt", 1),
1419 ("strftime", -1),
1420 ("substr", 2),
1421 ("substr", 3),
1422 ("substring", 2),
1423 ("substring", 3),
1424 ("tan", 1),
1425 ("tanh", 1),
1426 ("time", -1),
1427 ("timediff", 2),
1428 ("trim", 1),
1429 ("trim", 2),
1430 ("trunc", 1),
1431 ("typeof", 1),
1432 ("unhex", 1),
1433 ("unhex", 2),
1434 ("unicode", 1),
1435 ("unixepoch", -1),
1436 ("unlikely", 1),
1437 ("upper", 1),
1438 ("binary_quantize", 1),
1439 ("rtreecheck", -1),
1440 ("sqlar_compress", 1),
1441 ("sqlar_uncompress", 2),
1442 ("sqlite_offset", 1),
1443 ("rtreedepth", 1),
1444 ("rtreenode", 2),
1445 ("geopoly_area", 1),
1446 ("geopoly_bbox", 1),
1447 ("geopoly_blob", 1),
1448 ("geopoly_ccw", 1),
1449 ("geopoly_contains_point", 3),
1450 ("geopoly_debug", 1),
1451 ("geopoly_group_bbox", 1),
1452 ("geopoly_json", 1),
1453 ("geopoly_overlap", 2),
1454 ("geopoly_regular", 4),
1455 ("geopoly_svg", -1),
1456 ("geopoly_within", 2),
1457 ("geopoly_xform", 7),
1458 ("cosine_distance", 2),
1459 ("hamming_distance", 2),
1460 ("inner_product", 2),
1461 ("jaccard_distance", 2),
1462 ("l1_distance", 2),
1463 ("l2_distance", 2),
1464 ("l2_normalize", 1),
1465 ("subvector", 3),
1466 ("vector_add", 2),
1467 ("vector_concat", 2),
1468 ("vector_dims", 1),
1469 ("vector_distance_cos", 2),
1470 ("vector_distance_l2", 2),
1471 ("vector_dot", 2),
1472 ("vector_mul", 2),
1473 ("vector_norm", 1),
1474 ("vector_sub", 2),
1475 ("zeroblob", 1),
1476 // Present and answering, and missing from this list until now.
1477 // Each was checked against the shell before it was added.
1478 ("->", 2),
1479 ("->>", 2),
1480 ("bm25", -1),
1481 ("highlight", -1),
1482 ("if", -4),
1483 ("json_array_insert", -1),
1484 ("jsonb_array_insert", -1),
1485 ("match", 2),
1486 ("matchinfo", 1),
1487 ("matchinfo", 2),
1488 ("offsets", 1),
1489 ("regexp", 2),
1490 ("snippet", -1),
1491 ("sqlite_compileoption_get", 1),
1492 ("sqlite_compileoption_used", 1),
1493 ("subtype", 1),
1494 ("unistr", 1),
1495 ("unistr_quote", 1),
1496 ("unknown", -1),
1497];
1498
1499/// The scalars whose answer depends on something other than their arguments.
1500const VOLATILE: &[(&str, i64)] = &[
1501 ("changes", 0),
1502 // The three date keywords are functions in SQLite's register and answer
1503 // like functions here; they read the clock, so they are not deterministic.
1504 ("current_date", 0),
1505 ("current_time", 0),
1506 ("current_timestamp", 0),
1507 ("last_insert_rowid", 0),
1508 ("load_extension", 1),
1509 ("load_extension", 2),
1510 ("random", 0),
1511 ("randomblob", 1),
1512 ("sqlite_log", 2),
1513 ("total_changes", 0),
1514];
1515
1516/// The aggregates, with one row per overload.
1517const AGGREGATES: &[(&str, i64)] = &[
1518 ("avg", 1),
1519 ("count", 0),
1520 ("count", 1),
1521 ("group_concat", 1),
1522 ("group_concat", 2),
1523 ("json_group_array", 1),
1524 ("json_group_object", 2),
1525 ("jsonb_group_array", 1),
1526 ("jsonb_group_object", 2),
1527 ("max", 1),
1528 ("min", 1),
1529 ("string_agg", 2),
1530 ("sum", 1),
1531 ("total", 1),
1532];
1533
1534/// The window functions that are not aggregates.
1535const WINDOWS: &[(&str, i64)] = &[
1536 ("cume_dist", 0),
1537 ("dense_rank", 0),
1538 ("first_value", 1),
1539 ("lag", 1),
1540 ("lag", 2),
1541 ("lag", 3),
1542 ("last_value", 1),
1543 ("lead", 1),
1544 ("lead", 2),
1545 ("lead", 3),
1546 ("nth_value", 2),
1547 ("ntile", 1),
1548 ("percent_rank", 0),
1549 ("rank", 0),
1550 ("row_number", 0),
1551 // The percentile family, which SQLite reports as window functions and which
1552 // this engine answers as both aggregates and window functions.
1553 ("median", 1),
1554 ("percentile", 2),
1555 ("percentile_cont", 2),
1556 ("percentile_disc", 2),
1557];