Skip to main content

oxilite_core/
encoding.rs

1//! Term encoding: every RDF term maps to a tagged 64-bit integer id.
2//!
3//! Layout (the sign bit is always 0, so ids are positive SQLite INTEGERs):
4//!
5//! ```text
6//!  63 | 62..59 | 58..0
7//!   0 |  tag   | payload (hash, or inline value)
8//! ```
9//!
10//! Hashed kinds store a 59-bit xxh3 hash of a canonical key and need a row in `terms`.
11//! Inline kinds (canonical `xsd:integer` in ±2^58 and canonical `xsd:boolean`) carry their
12//! value in the payload, need no `terms` row, and — because the integer payload is offset —
13//! sort by value, so range filters on inline integers are id range scans.
14//!
15// @lat: [[architecture#Term encoding]]
16
17use crate::error::{Error, Result};
18use oxrdf::vocab::{rdf, xsd};
19use oxrdf::{
20    BaseDirection, BlankNode, GraphName, GraphNameRef, Literal, LiteralRef, NamedNode,
21    NamedOrBlankNode, NamedOrBlankNodeRef, Quad, QuadRef, Term, TermRef, Triple, TripleRef,
22};
23use std::str::FromStr;
24use xxhash_rust::xxh3::xxh3_64;
25
26/// Number of bits used by the payload.
27pub const PAYLOAD_BITS: u32 = 59;
28/// Mask selecting the payload.
29pub const PAYLOAD_MASK: i64 = (1_i64 << PAYLOAD_BITS) - 1;
30/// Offset applied to inline integers so that ids sort by value.
31pub const INT_OFFSET: i64 = 1_i64 << (PAYLOAD_BITS - 1);
32/// Smallest inline integer.
33pub const INT_MIN: i64 = -INT_OFFSET;
34/// Largest inline integer.
35pub const INT_MAX: i64 = INT_OFFSET - 1;
36
37/// The id of the default graph.
38pub const DEFAULT_GRAPH_ID: i64 = 0;
39
40/// Term kinds, stored in the 4 tag bits.
41#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
42#[repr(u8)]
43pub enum Tag {
44    /// Reserved: the default graph (id 0).
45    Default = 0,
46    /// IRI (hashed).
47    Iri = 1,
48    /// Blank node (hashed).
49    BlankNode = 2,
50    /// Simple literal / `xsd:string` (hashed).
51    String = 3,
52    /// Language-tagged string (hashed).
53    LangString = 4,
54    /// Any other typed literal, including non-canonical or out-of-range numbers (hashed).
55    Typed = 5,
56    /// Canonical `xsd:integer` in ±2^58 (inline).
57    Integer = 6,
58    /// Canonical `xsd:boolean` (inline).
59    Boolean = 7,
60    /// RDF 1.2 triple term (hashed, components in `triple_terms`).
61    Triple = 9,
62    /// RDF 1.2 directional language-tagged string (hashed).
63    DirLangString = 10,
64}
65
66impl Tag {
67    pub fn from_u8(v: u8) -> Option<Self> {
68        Some(match v {
69            0 => Self::Default,
70            1 => Self::Iri,
71            2 => Self::BlankNode,
72            3 => Self::String,
73            4 => Self::LangString,
74            5 => Self::Typed,
75            6 => Self::Integer,
76            7 => Self::Boolean,
77            9 => Self::Triple,
78            10 => Self::DirLangString,
79            _ => return None,
80        })
81    }
82
83    /// The first id carrying this tag.
84    pub const fn base(self) -> i64 {
85        (self as i64) << PAYLOAD_BITS
86    }
87
88    /// Does this kind of term need a row in `terms`?
89    pub const fn is_hashed(self) -> bool {
90        matches!(
91            self,
92            Self::Iri
93                | Self::BlankNode
94                | Self::String
95                | Self::LangString
96                | Self::Typed
97                | Self::DirLangString
98        )
99    }
100
101    /// Is this kind of term a literal?
102    pub const fn is_literal(self) -> bool {
103        matches!(
104            self,
105            Self::String
106                | Self::LangString
107                | Self::Typed
108                | Self::Integer
109                | Self::Boolean
110                | Self::DirLangString
111        )
112    }
113}
114
115/// Extracts the tag of an id.
116pub fn tag_of(id: i64) -> Option<Tag> {
117    Tag::from_u8(((id >> PAYLOAD_BITS) & 0xF) as u8)
118}
119
120fn make(tag: Tag, payload: i64) -> i64 {
121    tag.base() | (payload & PAYLOAD_MASK)
122}
123
124fn hashed(tag: Tag, key: &[&[u8]]) -> i64 {
125    let mut buf = Vec::with_capacity(key.iter().map(|k| k.len() + 1).sum::<usize>() + 1);
126    buf.push(tag as u8);
127    for (i, part) in key.iter().enumerate() {
128        if i > 0 {
129            buf.push(0);
130        }
131        buf.extend_from_slice(part);
132    }
133    make(tag, (xxh3_64(&buf) as i64) & PAYLOAD_MASK)
134}
135
136/// Builds the id of an inline integer, if it fits.
137pub fn integer_id(value: i64) -> Option<i64> {
138    (INT_MIN..=INT_MAX)
139        .contains(&value)
140        .then(|| make(Tag::Integer, value + INT_OFFSET))
141}
142
143/// Builds the id of a boolean.
144pub fn boolean_id(value: bool) -> i64 {
145    make(Tag::Boolean, i64::from(value))
146}
147
148/// Numeric type ranks used for SPARQL type promotion.
149pub mod numeric_type {
150    pub const INTEGER: i64 = 1;
151    pub const DECIMAL: i64 = 2;
152    pub const FLOAT: i64 = 3;
153    pub const DOUBLE: i64 = 4;
154}
155
156/// The row stored in `terms` for a hashed term.
157#[derive(Debug, Clone, PartialEq)]
158pub struct TermRow {
159    pub id: i64,
160    /// IRI, blank node label, or literal lexical form.
161    pub lex: String,
162    /// Datatype IRI, only for [`Tag::Typed`].
163    pub dt: Option<String>,
164    /// Language tag, for [`Tag::LangString`] and [`Tag::DirLangString`].
165    pub lang: Option<String>,
166    /// Base direction (1 = ltr, 2 = rtl) for [`Tag::DirLangString`]; for date/time literals,
167    /// whether the value has a timezone (1) or not (0).
168    pub dir: Option<i64>,
169    /// Numeric value for numeric datatypes, or 0/1 for non-canonical `xsd:boolean`.
170    pub num: Option<f64>,
171    /// Numeric type rank (see [`numeric_type`]).
172    pub nt: Option<i64>,
173    /// `xsd:dateTime` / `xsd:date` as seconds since the Unix epoch (timezone-normalized).
174    pub ts: Option<f64>,
175}
176
177/// The row stored in `triple_terms` for an RDF 1.2 triple term.
178#[derive(Debug, Clone, PartialEq, Eq)]
179pub struct TripleRow {
180    pub id: i64,
181    pub s: i64,
182    pub p: i64,
183    pub o: i64,
184    /// Value key: equal for value-equal triple terms (`<<a b 1>> = <<a b 1.0>>`).
185    pub vk: String,
186    /// Sort key: orders triple terms like Oxigraph (subject, predicate, then object).
187    pub sk: String,
188}
189
190/// Everything that must be written so that an encoded term can be decoded later.
191#[derive(Debug, Default, Clone)]
192pub struct EncodedRows {
193    pub terms: Vec<TermRow>,
194    pub triples: Vec<TripleRow>,
195}
196
197/// Encodes a named node.
198pub fn named_node_id(iri: &str) -> i64 {
199    hashed(Tag::Iri, &[iri.as_bytes()])
200}
201
202/// Encodes a blank node.
203pub fn blank_node_id(label: &str) -> i64 {
204    hashed(Tag::BlankNode, &[label.as_bytes()])
205}
206
207fn integer_family(dt: &str) -> bool {
208    matches!(
209        dt,
210        "http://www.w3.org/2001/XMLSchema#integer"
211            | "http://www.w3.org/2001/XMLSchema#int"
212            | "http://www.w3.org/2001/XMLSchema#long"
213            | "http://www.w3.org/2001/XMLSchema#short"
214            | "http://www.w3.org/2001/XMLSchema#byte"
215            | "http://www.w3.org/2001/XMLSchema#nonNegativeInteger"
216            | "http://www.w3.org/2001/XMLSchema#positiveInteger"
217            | "http://www.w3.org/2001/XMLSchema#nonPositiveInteger"
218            | "http://www.w3.org/2001/XMLSchema#negativeInteger"
219            | "http://www.w3.org/2001/XMLSchema#unsignedLong"
220            | "http://www.w3.org/2001/XMLSchema#unsignedInt"
221            | "http://www.w3.org/2001/XMLSchema#unsignedShort"
222            | "http://www.w3.org/2001/XMLSchema#unsignedByte"
223    )
224}
225
226/// Returns the numeric type rank of a datatype, if numeric.
227pub fn numeric_rank(dt: &str) -> Option<i64> {
228    if integer_family(dt) {
229        Some(numeric_type::INTEGER)
230    } else {
231        match dt {
232            "http://www.w3.org/2001/XMLSchema#decimal" => Some(numeric_type::DECIMAL),
233            "http://www.w3.org/2001/XMLSchema#float" => Some(numeric_type::FLOAT),
234            "http://www.w3.org/2001/XMLSchema#double" => Some(numeric_type::DOUBLE),
235            _ => None,
236        }
237    }
238}
239
240fn parse_xsd_float(lex: &str) -> Option<f64> {
241    match lex {
242        "INF" | "+INF" => Some(f64::INFINITY),
243        "-INF" => Some(f64::NEG_INFINITY),
244        "NaN" => None, // SQLite stores NaN as NULL anyway
245        _ => {
246            let v = f64::from_str(lex.trim()).ok()?;
247            v.is_finite().then_some(v)
248        }
249    }
250}
251
252/// Days since 1970-01-01 of a proleptic Gregorian civil date (Howard Hinnant's algorithm).
253fn days_from_civil(y: i64, m: i64, d: i64) -> i64 {
254    let y = if m <= 2 { y - 1 } else { y };
255    let era = if y >= 0 { y } else { y - 399 } / 400;
256    let yoe = y - era * 400;
257    let mp = (m + 9) % 12;
258    let doy = (153 * mp + 2) / 5 + d - 1;
259    let doe = yoe * 365 + yoe / 4 - yoe / 100 + doy;
260    era * 146_097 + doe - 719_468
261}
262
263fn tz_seconds(tz: Option<oxsdatatypes::DayTimeDuration>) -> f64 {
264    tz.map_or(0.0, |tz| (tz.hours() * 3600 + tz.minutes() * 60) as f64)
265}
266
267/// Seconds since the epoch of an `xsd:dateTime` or `xsd:date` lexical form (no timezone = UTC).
268pub fn timestamp(lex: &str, dt: &str) -> Option<f64> {
269    match dt {
270        "http://www.w3.org/2001/XMLSchema#dateTime"
271        | "http://www.w3.org/2001/XMLSchema#dateTimeStamp" => {
272            let v = oxsdatatypes::DateTime::from_str(lex).ok()?;
273            let days = days_from_civil(v.year(), v.month().into(), v.day().into());
274            let secs: f64 = f64::from(oxsdatatypes::Double::from(v.second()));
275            Some(
276                days as f64 * 86_400.0
277                    + f64::from(v.hour()) * 3600.0
278                    + f64::from(v.minute()) * 60.0
279                    + secs
280                    - tz_seconds(v.timezone()),
281            )
282        }
283        "http://www.w3.org/2001/XMLSchema#date" => {
284            let v = oxsdatatypes::Date::from_str(lex).ok()?;
285            let days = days_from_civil(v.year(), v.month().into(), v.day().into());
286            Some(days as f64 * 86_400.0 - tz_seconds(v.timezone()))
287        }
288        _ => None,
289    }
290}
291
292/// For `xsd:dateTime` / `xsd:date` values: 1 if the lexical form has a timezone, else 0.
293pub fn timezone_flag(lex: &str, dt: &str) -> Option<i64> {
294    timestamp(lex, dt)?;
295    let tz = lex.ends_with('Z')
296        || (lex.len() > 6 && {
297            let b = lex.as_bytes();
298            let n = b.len();
299            (b[n - 6] == b'+' || b[n - 6] == b'-') && b[n - 3] == b':'
300        });
301    Some(i64::from(tz))
302}
303
304/// A string that is equal for value-equal terms (numbers by value, dates by instant): used to
305/// compare triple terms with SPARQL `=` in one SQL comparison.
306pub fn value_key(term: TermRef<'_>) -> String {
307    match term {
308        TermRef::NamedNode(n) => format!("i{}", named_node_id(n.as_str())),
309        TermRef::BlankNode(b) => format!("b{}", blank_node_id(b.as_str())),
310        TermRef::Literal(l) => {
311            let (id, row) = encode_literal(l);
312            match (tag_of(id), row) {
313                (Some(Tag::Integer), _) => format!("n{}", (id & PAYLOAD_MASK) - INT_OFFSET),
314                (Some(Tag::Boolean), _) => format!("B{}", id & 1),
315                (_, Some(r)) => match (r.num, r.nt, r.ts) {
316                    (Some(n), Some(_), _) if n.fract() == 0.0 && n.abs() < 9.0e15 => {
317                        format!("n{}", n as i64)
318                    }
319                    (Some(n), Some(_), _) => format!("n{n:?}"),
320                    (Some(b), None, _) if l.datatype() == xsd::BOOLEAN => format!("B{}", b as i64),
321                    (_, _, Some(ts)) => {
322                        format!("t{}|{}|{ts:?}", l.datatype().as_str(), r.dir.unwrap_or(0))
323                    }
324                    _ => format!("l{id}"),
325                },
326                _ => format!("l{id}"),
327            }
328        }
329        TermRef::Triple(t) => format!(
330            "({} {} {})",
331            value_key(t.subject.as_ref().into()),
332            value_key(t.predicate.as_ref().into()),
333            value_key(t.object.as_ref())
334        ),
335    }
336}
337
338/// A string whose byte order is the ORDER BY order of terms (blank nodes, IRIs, numbers by
339/// value, other literals by lexical form, triple terms by components).
340pub fn sort_key(term: TermRef<'_>) -> String {
341    fn number(x: f64) -> String {
342        // Order-preserving bit transform: lexicographic order of the hex = numeric order.
343        let bits = x.to_bits();
344        let key = if x.is_sign_negative() {
345            !bits
346        } else {
347            bits | (1 << 63)
348        };
349        format!("{key:016x}")
350    }
351    match term {
352        TermRef::BlankNode(b) => format!("0{}", b.as_str()),
353        TermRef::NamedNode(n) => format!("1{}", n.as_str()),
354        TermRef::Literal(l) => {
355            let (id, row) = encode_literal(l);
356            let num = match (tag_of(id), &row) {
357                (Some(Tag::Integer), _) => Some(((id & PAYLOAD_MASK) - INT_OFFSET) as f64),
358                (_, Some(r)) if r.nt.is_some() => r.num,
359                _ => None,
360            };
361            match num {
362                Some(n) => format!("2{}", number(n)),
363                None => format!("3{}\u{1}{}", l.value(), l.datatype().as_str()),
364            }
365        }
366        TermRef::Triple(t) => format!(
367            "4{}\u{2}{}\u{2}{}",
368            sort_key(t.subject.as_ref().into()),
369            sort_key(t.predicate.as_ref().into()),
370            sort_key(t.object.as_ref())
371        ),
372    }
373}
374
375/// Encodes a literal, returning its id and (for hashed literals) the row to store.
376pub fn encode_literal(literal: LiteralRef<'_>) -> (i64, Option<TermRow>) {
377    let lex = literal.value();
378    if let Some(lang) = literal.language() {
379        if let Some(dir) = literal.direction() {
380            let d = match dir {
381                BaseDirection::Ltr => 1,
382                BaseDirection::Rtl => 2,
383            };
384            let id = hashed(
385                Tag::DirLangString,
386                &[
387                    lang.as_bytes(),
388                    if d == 1 { b"ltr" } else { b"rtl" },
389                    lex.as_bytes(),
390                ],
391            );
392            return (
393                id,
394                Some(TermRow {
395                    id,
396                    lex: lex.into(),
397                    dt: None,
398                    lang: Some(lang.into()),
399                    dir: Some(d),
400                    num: None,
401                    nt: None,
402                    ts: None,
403                }),
404            );
405        }
406        let id = hashed(Tag::LangString, &[lang.as_bytes(), lex.as_bytes()]);
407        return (
408            id,
409            Some(TermRow {
410                id,
411                lex: lex.into(),
412                dt: None,
413                lang: Some(lang.into()),
414                dir: None,
415                num: None,
416                nt: None,
417                ts: None,
418            }),
419        );
420    }
421    let dt = literal.datatype();
422    if dt == xsd::STRING {
423        let id = hashed(Tag::String, &[lex.as_bytes()]);
424        return (
425            id,
426            Some(TermRow {
427                id,
428                lex: lex.into(),
429                dt: None,
430                lang: None,
431                dir: None,
432                num: None,
433                nt: None,
434                ts: None,
435            }),
436        );
437    }
438    if dt == xsd::INTEGER {
439        if let Ok(v) = i64::from_str(lex) {
440            if v.to_string() == lex {
441                if let Some(id) = integer_id(v) {
442                    return (id, None);
443                }
444            }
445        }
446    }
447    if dt == xsd::BOOLEAN {
448        match lex {
449            "true" => return (boolean_id(true), None),
450            "false" => return (boolean_id(false), None),
451            _ => {}
452        }
453    }
454    let dt_str = dt.as_str();
455    let id = hashed(Tag::Typed, &[dt_str.as_bytes(), lex.as_bytes()]);
456    let nt = numeric_rank(dt_str);
457    let mut num = nt.and_then(|_| parse_xsd_float(lex));
458    if dt == xsd::BOOLEAN {
459        num = match lex.trim() {
460            "1" | "true" => Some(1.0),
461            "0" | "false" => Some(0.0),
462            _ => None,
463        };
464    }
465    (
466        id,
467        Some(TermRow {
468            id,
469            lex: lex.into(),
470            dt: Some(dt_str.into()),
471            lang: None,
472            dir: timezone_flag(lex, dt_str),
473            num,
474            nt: if num.is_some() { nt } else { None },
475            ts: timestamp(lex, dt_str),
476        }),
477    )
478}
479
480/// Encodes a term without collecting rows (only the id).
481pub fn term_id(term: TermRef<'_>) -> i64 {
482    match term {
483        TermRef::NamedNode(n) => named_node_id(n.as_str()),
484        TermRef::BlankNode(b) => blank_node_id(b.as_str()),
485        TermRef::Literal(l) => encode_literal(l).0,
486        TermRef::Triple(t) => triple_id(t.as_ref()),
487    }
488}
489
490/// Id of a triple term (hash of its component ids).
491pub fn triple_id(t: TripleRef<'_>) -> i64 {
492    let s = subject_id(t.subject);
493    let p = named_node_id(t.predicate.as_str());
494    let o = term_id(t.object);
495    hashed(
496        Tag::Triple,
497        &[&s.to_be_bytes(), &p.to_be_bytes(), &o.to_be_bytes()],
498    )
499}
500
501pub fn subject_id(s: NamedOrBlankNodeRef<'_>) -> i64 {
502    match s {
503        NamedOrBlankNodeRef::NamedNode(n) => named_node_id(n.as_str()),
504        NamedOrBlankNodeRef::BlankNode(b) => blank_node_id(b.as_str()),
505    }
506}
507
508pub fn graph_id(g: GraphNameRef<'_>) -> i64 {
509    match g {
510        GraphNameRef::DefaultGraph => DEFAULT_GRAPH_ID,
511        GraphNameRef::NamedNode(n) => named_node_id(n.as_str()),
512        GraphNameRef::BlankNode(b) => blank_node_id(b.as_str()),
513    }
514}
515
516impl EncodedRows {
517    /// Encodes a term, recording the rows needed to decode it.
518    pub fn term(&mut self, term: TermRef<'_>) -> i64 {
519        match term {
520            TermRef::NamedNode(n) => self.iri(n.as_str()),
521            TermRef::BlankNode(b) => self.bnode(b.as_str()),
522            TermRef::Literal(l) => {
523                let (id, row) = encode_literal(l);
524                if let Some(row) = row {
525                    self.terms.push(row);
526                }
527                id
528            }
529            TermRef::Triple(t) => self.triple(t.as_ref()),
530        }
531    }
532
533    pub fn iri(&mut self, iri: &str) -> i64 {
534        let id = named_node_id(iri);
535        self.terms.push(TermRow {
536            id,
537            lex: iri.into(),
538            dt: None,
539            lang: None,
540            dir: None,
541            num: None,
542            nt: None,
543            ts: None,
544        });
545        id
546    }
547
548    pub fn bnode(&mut self, label: &str) -> i64 {
549        let id = blank_node_id(label);
550        self.terms.push(TermRow {
551            id,
552            lex: label.into(),
553            dt: None,
554            lang: None,
555            dir: None,
556            num: None,
557            nt: None,
558            ts: None,
559        });
560        id
561    }
562
563    pub fn subject(&mut self, s: NamedOrBlankNodeRef<'_>) -> i64 {
564        match s {
565            NamedOrBlankNodeRef::NamedNode(n) => self.iri(n.as_str()),
566            NamedOrBlankNodeRef::BlankNode(b) => self.bnode(b.as_str()),
567        }
568    }
569
570    pub fn graph(&mut self, g: GraphNameRef<'_>) -> i64 {
571        match g {
572            GraphNameRef::DefaultGraph => DEFAULT_GRAPH_ID,
573            GraphNameRef::NamedNode(n) => self.iri(n.as_str()),
574            GraphNameRef::BlankNode(b) => self.bnode(b.as_str()),
575        }
576    }
577
578    pub fn triple(&mut self, t: TripleRef<'_>) -> i64 {
579        let s = self.subject(t.subject);
580        let p = self.iri(t.predicate.as_str());
581        let o = self.term(t.object);
582        let id = hashed(
583            Tag::Triple,
584            &[&s.to_be_bytes(), &p.to_be_bytes(), &o.to_be_bytes()],
585        );
586        let owned = t.into_owned();
587        let vk = value_key(TermRef::Triple(&owned));
588        let sk = sort_key(TermRef::Triple(&owned));
589        self.triples.push(TripleRow {
590            id,
591            s,
592            p,
593            o,
594            vk,
595            sk,
596        });
597        id
598    }
599
600    /// Encodes a quad, returning `[s, p, o, g]`.
601    pub fn quad(&mut self, q: QuadRef<'_>) -> [i64; 4] {
602        [
603            self.subject(q.subject),
604            self.iri(q.predicate.as_str()),
605            self.term(q.object),
606            self.graph(q.graph_name),
607        ]
608    }
609
610    pub fn is_empty(&self) -> bool {
611        self.terms.is_empty() && self.triples.is_empty()
612    }
613
614    /// Sorts and removes duplicate rows (by id).
615    pub fn dedup(&mut self) {
616        self.terms.sort_by_key(|r| r.id);
617        self.terms.dedup_by_key(|r| r.id);
618        self.triples.sort_by_key(|r| r.id);
619        self.triples.dedup_by_key(|r| r.id);
620    }
621}
622
623/// Decodes an inline id (integer, boolean) without any lookup.
624pub fn decode_inline(id: i64) -> Option<Term> {
625    match tag_of(id)? {
626        Tag::Integer => Some(Literal::from((id & PAYLOAD_MASK) - INT_OFFSET).into()),
627        Tag::Boolean => Some(Literal::from((id & PAYLOAD_MASK) != 0).into()),
628        _ => None,
629    }
630}
631
632/// Decodes a hashed id from its `terms` row.
633pub fn decode_row(
634    id: i64,
635    lex: String,
636    dt: Option<String>,
637    lang: Option<String>,
638    dir: Option<i64>,
639) -> Result<Term> {
640    let tag = tag_of(id).ok_or_else(|| Error::corrupted(format!("invalid term id {id}")))?;
641    Ok(match tag {
642        Tag::Iri => NamedNode::new_unchecked(lex).into(),
643        Tag::BlankNode => BlankNode::new_unchecked(lex).into(),
644        Tag::String => Literal::new_simple_literal(lex).into(),
645        Tag::LangString => Literal::new_language_tagged_literal_unchecked(
646            lex,
647            lang.ok_or_else(|| Error::corrupted("language tag missing"))?,
648        )
649        .into(),
650        Tag::DirLangString => Literal::new_directional_language_tagged_literal_unchecked(
651            lex,
652            lang.ok_or_else(|| Error::corrupted("language tag missing"))?,
653            if dir == Some(2) {
654                BaseDirection::Rtl
655            } else {
656                BaseDirection::Ltr
657            },
658        )
659        .into(),
660        Tag::Typed => Literal::new_typed_literal(
661            lex,
662            NamedNode::new_unchecked(dt.ok_or_else(|| Error::corrupted("datatype missing"))?),
663        )
664        .into(),
665        Tag::Integer | Tag::Boolean => {
666            decode_inline(id).ok_or_else(|| Error::corrupted("bad inline term"))?
667        }
668        Tag::Triple | Tag::Default => {
669            return Err(Error::corrupted(format!("id {id} is not a plain term")))
670        }
671    })
672}
673
674/// Rebuilds a triple from its decoded components.
675pub fn make_triple(s: Term, p: Term, o: Term) -> Result<Triple> {
676    let s = match s {
677        Term::NamedNode(n) => NamedOrBlankNode::NamedNode(n),
678        Term::BlankNode(b) => NamedOrBlankNode::BlankNode(b),
679        _ => return Err(Error::corrupted("invalid triple term subject")),
680    };
681    let Term::NamedNode(p) = p else {
682        return Err(Error::corrupted("invalid triple term predicate"));
683    };
684    Ok(Triple::new(s, p, o))
685}
686
687/// Converts a decoded term into a subject.
688pub fn to_subject(t: Term) -> Result<NamedOrBlankNode> {
689    match t {
690        Term::NamedNode(n) => Ok(n.into()),
691        Term::BlankNode(b) => Ok(b.into()),
692        _ => Err(Error::corrupted("invalid subject")),
693    }
694}
695
696/// Converts a decoded term into a graph name.
697pub fn to_graph_name(id: i64, t: Option<Term>) -> Result<GraphName> {
698    if id == DEFAULT_GRAPH_ID {
699        return Ok(GraphName::DefaultGraph);
700    }
701    match t {
702        Some(Term::NamedNode(n)) => Ok(n.into()),
703        Some(Term::BlankNode(b)) => Ok(b.into()),
704        _ => Err(Error::corrupted("invalid graph name")),
705    }
706}
707
708/// Builds a quad from decoded parts.
709pub fn make_quad(s: Term, p: Term, o: Term, g: GraphName) -> Result<Quad> {
710    let Term::NamedNode(p) = p else {
711        return Err(Error::corrupted("invalid predicate"));
712    };
713    Ok(Quad::new(to_subject(s)?, p, o, g))
714}
715
716/// Well-known ids used by the planner and reasoner.
717pub fn rdf_type_id() -> i64 {
718    named_node_id(rdf::TYPE.as_str())
719}
720
721#[cfg(test)]
722mod tests {
723    use super::*;
724
725    // @lat: [[tests#Encoding#Inline integers sort by value]]
726    #[test]
727    fn inline_integers_sort_by_value() {
728        let ids: Vec<i64> = [-5_i64, -1, 0, 1, 42, 1_000_000]
729            .iter()
730            .map(|v| integer_id(*v).unwrap())
731            .collect();
732        let mut sorted = ids.clone();
733        sorted.sort();
734        assert_eq!(ids, sorted);
735        assert!(ids.iter().all(|id| *id > 0));
736        for v in [-5_i64, 0, 7, INT_MAX, INT_MIN] {
737            let id = integer_id(v).unwrap();
738            assert_eq!(decode_inline(id), Some(Literal::from(v).into()));
739        }
740        assert!(integer_id(INT_MAX + 1).is_none());
741    }
742
743    // @lat: [[tests#Encoding#Non-canonical literals are hashed]]
744    #[test]
745    fn non_canonical_literals_are_hashed() {
746        let canonical = Literal::new_typed_literal("12", xsd::INTEGER);
747        let padded = Literal::new_typed_literal("012", xsd::INTEGER);
748        let (a, row_a) = encode_literal(canonical.as_ref());
749        let (b, row_b) = encode_literal(padded.as_ref());
750        assert_eq!(tag_of(a), Some(Tag::Integer));
751        assert!(row_a.is_none());
752        assert_eq!(tag_of(b), Some(Tag::Typed));
753        let row_b = row_b.unwrap();
754        assert_eq!(row_b.num, Some(12.0));
755        assert_eq!(row_b.nt, Some(numeric_type::INTEGER));
756        assert_ne!(a, b);
757    }
758
759    // @lat: [[tests#Encoding#Simple literal equals xsd string]]
760    #[test]
761    fn simple_literal_equals_xsd_string() {
762        let a = Literal::new_simple_literal("abc");
763        let b = Literal::new_typed_literal("abc", xsd::STRING);
764        assert_eq!(encode_literal(a.as_ref()).0, encode_literal(b.as_ref()).0);
765        let c = Literal::new_language_tagged_literal("abc", "en").unwrap();
766        assert_ne!(encode_literal(a.as_ref()).0, encode_literal(c.as_ref()).0);
767    }
768
769    // @lat: [[tests#Encoding#Tags partition the id space]]
770    #[test]
771    fn tags_partition_the_id_space() {
772        let iri = named_node_id("http://example.com/a");
773        let bnode = blank_node_id("http://example.com/a");
774        assert_eq!(tag_of(iri), Some(Tag::Iri));
775        assert_eq!(tag_of(bnode), Some(Tag::BlankNode));
776        assert!(iri >= Tag::Iri.base() && iri < Tag::BlankNode.base());
777        assert_ne!(iri, bnode);
778    }
779
780    #[test]
781    fn timestamps() {
782        assert_eq!(
783            timestamp("1970-01-02T00:00:00Z", xsd::DATE_TIME.as_str()),
784            Some(86_400.0)
785        );
786        assert_eq!(
787            timestamp("1970-01-01T02:00:00+02:00", xsd::DATE_TIME.as_str()),
788            Some(0.0)
789        );
790        assert_eq!(
791            timestamp("2000-03-01", xsd::DATE.as_str()),
792            Some(951_868_800.0)
793        );
794    }
795}