oxilite-core 0.9.0

The I/O-free core of oxilite: SPARQL to SQL compiler, term encoding and SQLite schema (sans-IO jobs for any SQLite backend)
Documentation
//! Database schema.
//!
// @lat: [[architecture#Storage schema]]

use crate::encoding::{Tag, INT_OFFSET, PAYLOAD_BITS};
use crate::sql::{Request, Statement};

/// Current schema version stored in `oxilite_meta`.
/// Version 2 keeps the schema registry as RDF in `<oxilite:schema>` (no `schema_graphs`
/// table) and scopes `tbox_closure`; opening a version 1 store migrates it (`ops::open_job`).
pub const SCHEMA_VERSION: &str = "2";

/// The TBox closure cache, one closure per scope (see `reason::closure_statements`).
pub const TBOX_TABLE: &str = "CREATE TABLE IF NOT EXISTS tbox_closure (\
    kind INTEGER NOT NULL, scope INTEGER NOT NULL, sub INTEGER NOT NULL, sup INTEGER NOT NULL, \
    PRIMARY KEY (kind, scope, sup, sub)) WITHOUT ROWID, STRICT";
pub const TBOX_INDEX: &str =
    "CREATE INDEX IF NOT EXISTS tbox_closure_sub ON tbox_closure(kind, scope, sub, sup)";

/// Options chosen when a store is created.
#[derive(Debug, Clone, PartialEq, Eq)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
#[cfg_attr(feature = "serde", serde(default, rename_all = "camelCase"))]
pub struct StoreOptions {
    /// Create the optional `quads_gspo` index (fast `GRAPH <g> { ?s ?p ?o }`, `CLEAR GRAPH`).
    pub graph_index: bool,
    /// Create the full-text index over string literals (FTS5, see [`crate::text`]).
    pub text_index: bool,
    /// How much history the store keeps (see [`crate::version`]); `Off` by default. Applied when
    /// the store is created; an existing store changes level only through an explicit level
    /// change.
    pub versioning: crate::version::Versioning,
    /// With `stamped` or `log`: index `quads.t` (fast "added since" queries, one more row
    /// written per quad).
    pub stamp_index: bool,
    /// With `log`: index the change log by predicate and object too (fast as-of queries on
    /// any pattern, two more rows written per change).
    pub as_of_index: bool,
    /// Install the system graphs in a blank store: the oxilite vocabulary in
    /// `<oxilite:vocabulary>` and the registry's own description in `<oxilite:schema>` (see
    /// `registry::system_quads`). Off by default, so a new store is empty as in Oxigraph.
    pub system_graphs: bool,
}

impl Default for StoreOptions {
    fn default() -> Self {
        Self {
            graph_index: true,
            text_index: false,
            versioning: crate::version::Versioning::Off,
            stamp_index: false,
            as_of_index: false,
            system_graphs: false,
        }
    }
}

impl StoreOptions {
    /// The level change that creates this store's versioning.
    pub fn level_change(&self) -> crate::version::LevelChange {
        crate::version::LevelChange {
            as_of_index: self.as_of_index.then_some(true),
            stamp_index: self.stamp_index.then_some(true),
            ..Default::default()
        }
    }
}

/// The schema as one SQL script (for `wrangler d1 migrations`).
pub fn schema_sql(options: &StoreOptions) -> String {
    schema_sql_with(options, &[])
}

/// The schema plus extra DDL of optional modules (e.g. `oxilite-jsonld`'s tables).
pub fn schema_sql_with(options: &StoreOptions, extra: &[Statement]) -> String {
    let mut out =
        String::from("-- oxilite schema (generated by oxilite_core::schema::schema_sql)\n");
    for s in create_schema(options).statements.iter().chain(extra) {
        out.push_str(&s.sql);
        out.push_str(";\n");
    }
    out
}

/// DDL creating the oxilite schema, versioning included: a script for a new database (a D1
/// migration). Not idempotent when versioning is on (`ALTER TABLE`); stores opened in place use
/// [`base_schema`] and apply their level through `version::change_statements`.
pub fn create_schema(options: &StoreOptions) -> Request {
    let mut r = base_schema(options);
    if options.versioning > crate::version::Versioning::Off {
        let change = crate::version::change_statements(
            &crate::version::VersionState::default(),
            options.versioning,
            &options.level_change(),
        )
        .expect("versioning from off is always possible");
        r.statements.extend(change);
    }
    if options.system_graphs {
        r.statements
            .extend(system_graph_statements(&crate::sql::Capabilities::d1()));
    }
    r
}

/// Statements writing the system graphs (see `StoreOptions::system_graphs`) and rebuilding the
/// schema caches they scope.
pub fn system_graph_statements(caps: &crate::sql::Capabilities) -> Vec<Statement> {
    let quads = crate::registry::system_quads();
    let mut s = crate::writer::EncodedQuads::new(quads.iter().map(oxrdf::Quad::as_ref))
        .insert_statements(caps);
    s.extend(crate::reason::closure_statements());
    s.extend(crate::shapes::refresh_statements());
    s
}

/// DDL statements creating (idempotently) the oxilite schema without versioning.
pub fn base_schema(options: &StoreOptions) -> Request {
    let mut s = vec![
        "CREATE TABLE IF NOT EXISTS oxilite_meta (key TEXT PRIMARY KEY, value TEXT NOT NULL) STRICT",
        // Hashed terms. `id` is the rowid alias: the fastest possible key.
        "CREATE TABLE IF NOT EXISTS terms (\
            id INTEGER PRIMARY KEY, \
            lex TEXT NOT NULL, \
            dt TEXT, \
            lang TEXT, \
            dir INTEGER, \
            num REAL, \
            nt INTEGER, \
            ts REAL) STRICT",
        "CREATE INDEX IF NOT EXISTS terms_num ON terms(num) WHERE num IS NOT NULL",
        "CREATE INDEX IF NOT EXISTS terms_ts ON terms(ts) WHERE ts IS NOT NULL",
        // Detects xxh3 collisions atomically: aborts the whole batch.
        "CREATE TRIGGER IF NOT EXISTS terms_collision BEFORE INSERT ON terms \
         WHEN EXISTS (SELECT 1 FROM terms t WHERE t.id = NEW.id AND \
            (t.lex IS NOT NEW.lex OR t.dt IS NOT NEW.dt OR t.lang IS NOT NEW.lang OR t.dir IS NOT NEW.dir)) \
         BEGIN SELECT RAISE(ABORT, 'oxilite: term hash collision'); END",
        "CREATE TABLE IF NOT EXISTS triple_terms (\
            id INTEGER PRIMARY KEY, s INTEGER NOT NULL, p INTEGER NOT NULL, o INTEGER NOT NULL, vk TEXT NOT NULL, sk TEXT NOT NULL) STRICT",
        // The quad table is its own clustered SPOG index; secondary indexes contain every
        // column, so every triple-pattern scan is index-only.
        "CREATE TABLE IF NOT EXISTS quads (\
            s INTEGER NOT NULL, p INTEGER NOT NULL, o INTEGER NOT NULL, g INTEGER NOT NULL DEFAULT 0, \
            PRIMARY KEY (s, p, o, g)) WITHOUT ROWID, STRICT",
        "CREATE INDEX IF NOT EXISTS quads_posg ON quads(p, o, s, g)",
        "CREATE INDEX IF NOT EXISTS quads_ospg ON quads(o, s, p, g)",
        "CREATE TABLE IF NOT EXISTS graphs (id INTEGER PRIMARY KEY) STRICT",
        "CREATE TABLE IF NOT EXISTS stats_pred (\
            p INTEGER PRIMARY KEY, triples INTEGER NOT NULL, distinct_s INTEGER NOT NULL, distinct_o INTEGER NOT NULL) STRICT",
        "CREATE TABLE IF NOT EXISTS stats_class (o INTEGER PRIMARY KEY, instances INTEGER NOT NULL) STRICT",
        // Frequent (predicate, object) pairs of low-cardinality predicates (planner skew).
        "CREATE TABLE IF NOT EXISTS stats_po (p INTEGER NOT NULL, o INTEGER NOT NULL, n INTEGER NOT NULL, PRIMARY KEY (p, o)) WITHOUT ROWID, STRICT",
        // Reasoning: the schema closure (see `reason::closure_statements`) and materialized
        // OWL 2 RL inferences, kept apart from asserted quads.
        // `scope`: the graph a closure applies to, or the id of `oxl:AllGraphs` (see `registry`).
        TBOX_TABLE,
        TBOX_INDEX,
        "CREATE TABLE IF NOT EXISTS quads_inf (\
            s INTEGER NOT NULL, p INTEGER NOT NULL, o INTEGER NOT NULL, g INTEGER NOT NULL DEFAULT 0, \
            PRIMARY KEY (s, p, o, g)) WITHOUT ROWID, STRICT",
        "CREATE INDEX IF NOT EXISTS quads_inf_posg ON quads_inf(p, o, s, g)",
        "CREATE INDEX IF NOT EXISTS quads_inf_ospg ON quads_inf(o, s, p, g)",
        // Which producer (OWL 2 RL, a named rule set) derived each inference, so one producer
        // can be recomputed without discarding the others' conclusions (see `reason`).
        "CREATE TABLE IF NOT EXISTS quads_inf_src (\
            src INTEGER NOT NULL, s INTEGER NOT NULL, p INTEGER NOT NULL, o INTEGER NOT NULL, g INTEGER NOT NULL DEFAULT 0, \
            PRIMARY KEY (s, p, o, g, src)) WITHOUT ROWID, STRICT",
        "CREATE TABLE IF NOT EXISTS inf_producers (id INTEGER PRIMARY KEY, name TEXT NOT NULL) STRICT",
        // The compiled SHACL property shapes of the registered shapes graphs, and the values of
        // their `sh:in` lists (see `shapes`). Both are caches, rebuilt from `quads`.
        "CREATE TABLE IF NOT EXISTS shapes_index (\
            target TEXT NOT NULL, path TEXT NOT NULL, datatype TEXT, min_count INTEGER, \
            max_count INTEGER, pattern TEXT, relationship INTEGER NOT NULL DEFAULT 0, \
            PRIMARY KEY (target, path)) WITHOUT ROWID, STRICT",
        "CREATE TABLE IF NOT EXISTS shapes_in (\
            target TEXT NOT NULL, path TEXT NOT NULL, id INTEGER NOT NULL, \
            lex TEXT, dt TEXT, lang TEXT, dir INTEGER, \
            PRIMARY KEY (target, path, id)) WITHOUT ROWID, STRICT",
        // Work table for Datalog components that need iteration (non-linear recursion).
        // Rows are term ids, padded to a fixed width so one table serves every arity; `run`
        // scopes an evaluation, so concurrent programs do not see each other and cleanup is
        // exact. Unused columns default to 0 because a WITHOUT ROWID primary key is NOT NULL.
        "CREATE TABLE IF NOT EXISTS datalog_work (\
            run INTEGER NOT NULL, rel INTEGER NOT NULL, \
            c0 INTEGER NOT NULL DEFAULT 0, c1 INTEGER NOT NULL DEFAULT 0, \
            c2 INTEGER NOT NULL DEFAULT 0, c3 INTEGER NOT NULL DEFAULT 0, \
            c4 INTEGER NOT NULL DEFAULT 0, c5 INTEGER NOT NULL DEFAULT 0, \
            PRIMARY KEY (run, rel, c0, c1, c2, c3, c4, c5)) WITHOUT ROWID, STRICT",
        // Staging table for SPARQL UPDATE (DELETE/INSERT … WHERE) inside one atomic batch.
        "CREATE TABLE IF NOT EXISTS update_buffer (\
            op INTEGER NOT NULL, s INTEGER NOT NULL, p INTEGER NOT NULL, o INTEGER NOT NULL, g INTEGER NOT NULL) STRICT",
        // Assertions inside atomic batches: inserting a non-NULL value aborts the batch with a
        // "CHECK constraint failed: <name>" error naming the violated SPARQL condition.
        "CREATE TABLE IF NOT EXISTS oxilite_guard (\
            graph_does_not_exist INTEGER CHECK (graph_does_not_exist IS NULL), \
            graph_already_exists INTEGER CHECK (graph_already_exists IS NULL), \
            computed_value_not_storable INTEGER CHECK (computed_value_not_storable IS NULL)) STRICT",
    ]
    .into_iter()
    .map(Statement::from)
    .collect::<Vec<_>>();
    if options.graph_index {
        s.push("CREATE INDEX IF NOT EXISTS quads_gspo ON quads(g, s, p, o)".into());
    }
    if options.text_index {
        s.extend(crate::text::schema_statements());
    }
    s.push(
        format!(
            "INSERT OR IGNORE INTO oxilite_meta(key, value) VALUES ('schema_version', '{SCHEMA_VERSION}'), ('graph_index', '{}'), ('int_offset', '{INT_OFFSET}'), ('payload_bits', '{PAYLOAD_BITS}'), ('integer_tag', '{}')",
            u8::from(options.graph_index),
            Tag::Integer as u8
        )
        .into(),
    );
    Request::atomic(s)
}