Skip to main content

rete_core/
terms.rs

1//! Term identifiers and N-Triples term-token helpers.
2//!
3//! Two distinct things live in one place here because they describe the same
4//! domain — "what is a term, and how is it identified":
5//!
6//! 1. **ID aliases.** Every term in a `.rete` file is interned into the
7//!    dictionary and addressed by a `u32`. Code passes these `u32`s through
8//!    many signatures where the bare type says nothing about *which* id space a
9//!    value lives in. The aliases below ([`NodeId`], [`SubjectId`],
10//!    [`PredicateId`], [`ObjectId`]) are documentation: they are all `u32`
11//!    today (so they cost nothing and mix freely), but they let a signature
12//!    state its intent — `subject_node(sid: SubjectId) -> NodeId` reads as the
13//!    role-id → unified-node mapping it is. A later pass can promote them to
14//!    true newtypes (`struct NodeId(u32)`) without touching call sites that
15//!    already name the alias.
16//!
17//! 2. **Term-token helpers.** A [`TermToken`] is the textual form of a term as
18//!    it appears in N-Triples and in the dictionary: an IRI `<http://…>`, a
19//!    blank node `_:b0`, or a literal `"text"`, `"text"@en`, `"text"^^<dt>`.
20//!    Several modules (SPARQL evaluation, SHACL validation, doc rendering)
21//!    independently grew the same little parsers for "is this an IRI", "what's
22//!    the lexical value", "what's the datatype". They are consolidated here so
23//!    there is one definition of the term grammar to reason about.
24
25use std::borrow::Cow;
26
27/// A dictionary id in the **unified node space** — the single id space that
28/// covers every term that ever appears as a subject or an object. This is the
29/// id reachability, the community pyramid, and the graph index work in.
30pub type NodeId = u32;
31
32/// A dictionary id in the **subject** id space (terms seen in subject
33/// position). Map to a [`NodeId`] with [`Dictionary::subject_node`].
34///
35/// [`Dictionary::subject_node`]: crate::dictionary::Dictionary::subject_node
36pub type SubjectId = u32;
37
38/// A dictionary id in the **predicate** id space. Predicates have their own
39/// dense id space and are never part of the unified node space.
40pub type PredicateId = u32;
41
42/// A dictionary id in the **object** id space (terms seen in object position).
43/// Map to a [`NodeId`] with [`Dictionary::object_node`].
44///
45/// [`Dictionary::object_node`]: crate::dictionary::Dictionary::object_node
46pub type ObjectId = u32;
47
48/// The textual form of an RDF term as stored in the dictionary and emitted in
49/// N-Triples: an IRI (`<…>`), a blank node (`_:…`), or a literal (`"…"`,
50/// optionally with an `@lang` or `^^<datatype>` suffix). An alias for `str`;
51/// it names intent at API boundaries that take a term rather than arbitrary
52/// text.
53pub type TermToken = str;
54
55const XSD_STRING: &str = "http://www.w3.org/2001/XMLSchema#string";
56const RDF_LANG_STRING: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#langString";
57/// RDF 1.2 base-direction language string: `"…"@lang--dir` (dir = `ltr`/`rtl`).
58const RDF_DIR_LANG_STRING: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#dirLangString";
59
60/// Is `t` an IRI term (`<…>`)?
61#[inline]
62pub fn is_iri(t: &TermToken) -> bool {
63    // A quoted triple (`<< … >>`, RDF-star) also starts with `<` and ends with
64    // `>`, so exclude it explicitly — it is its own term kind, not an IRI.
65    t.starts_with('<') && !t.starts_with("<<") && t.ends_with('>')
66}
67
68/// Is `t` a **quoted triple** term (`<< s p o >>`, RDF-star)? These appear only
69/// in subject/object position and are stored in the dictionary as their
70/// canonical N-Triples-star surface, exactly like any other term.
71#[inline]
72pub fn is_quoted_triple(t: &TermToken) -> bool {
73    t.starts_with("<<") && t.ends_with(">>")
74}
75
76/// The content of an IRI term without its angle brackets (`<http://x>` →
77/// `http://x`), or `None` if `t` is not an IRI term.
78#[inline]
79pub fn iri_content(t: &TermToken) -> Option<&str> {
80    if is_quoted_triple(t) {
81        return None;
82    }
83    t.strip_prefix('<').and_then(|s| s.strip_suffix('>'))
84}
85
86/// Is `t` a blank-node term (`_:…`)?
87#[inline]
88pub fn is_blank(t: &TermToken) -> bool {
89    t.starts_with("_:")
90}
91
92/// Is `t` a literal term (`"…"`)?
93#[inline]
94pub fn is_literal(t: &TermToken) -> bool {
95    t.starts_with('"')
96}
97
98/// Index of the closing quote of a literal term, scanning from the opening
99/// quote and honoring `\"` escapes. `t` must start with `"`.
100fn closing_quote(t: &TermToken) -> usize {
101    let bytes = t.as_bytes();
102    let mut i = 1;
103    while i < bytes.len() {
104        match bytes[i] {
105            b'\\' => i += 2,
106            b'"' => break,
107            _ => i += 1,
108        }
109    }
110    i.min(t.len())
111}
112
113/// The **lexical value** of a literal term — the text between the quotes with
114/// N-Triples escapes resolved — or `None` for IRIs and blank nodes. The
115/// datatype and language suffix are dropped (`"42"^^<…int>` → `42`,
116/// `"hi"@en` → `hi`).
117pub fn literal_lexical(token: &TermToken) -> Option<String> {
118    if !is_literal(token) {
119        return None;
120    }
121    Some(unescape_literal(&token[1..closing_quote(token)]))
122}
123
124/// The **lexical value** of any term: a literal's unescaped body, an IRI's
125/// content, or a blank-node token unchanged. Always succeeds. Useful where a
126/// plain comparable string is wanted regardless of term kind.
127pub fn lexical(token: &TermToken) -> Cow<'_, str> {
128    if is_literal(token) {
129        Cow::Owned(unescape_literal(&token[1..closing_quote(token)]))
130    } else if let Some(iri) = iri_content(token) {
131        Cow::Borrowed(iri)
132    } else {
133        Cow::Borrowed(token)
134    }
135}
136
137/// The part of a literal term after its closing quote (`"x"^^<dt>` → `^^<dt>`,
138/// `"x"@en` → `@en`, `"x"` → ``), or `None` if `token` is not a literal.
139fn literal_suffix(token: &TermToken) -> Option<&str> {
140    if !is_literal(token) {
141        return None;
142    }
143    token.get(closing_quote(token) + 1..)
144}
145
146/// The datatype IRI **content** of a literal term (no angle brackets): the
147/// explicit `^^<dt>`, else `rdf:langString` for a language-tagged literal,
148/// else `xsd:string` for a plain one. `None` for a non-literal or a malformed
149/// suffix.
150pub fn literal_datatype(token: &TermToken) -> Option<String> {
151    let suffix = literal_suffix(token)?;
152    if let Some(dt) = suffix.strip_prefix("^^<").and_then(|s| s.strip_suffix('>')) {
153        Some(dt.to_string())
154    } else if suffix.starts_with('@') {
155        // RDF 1.2: a language string WITH a base direction (`@lang--dir`) is an
156        // `rdf:dirLangString`; a plain `@lang` is `rdf:langString`. `--` never
157        // occurs in a well-formed BCP-47 tag, so this is unambiguous.
158        if suffix.contains("--") {
159            Some(RDF_DIR_LANG_STRING.to_string())
160        } else {
161            Some(RDF_LANG_STRING.to_string())
162        }
163    } else if suffix.is_empty() {
164        Some(XSD_STRING.to_string())
165    } else {
166        None
167    }
168}
169
170/// The language tag of a literal term (`"hi"@en` → `en`), `""` when the literal
171/// is untagged, or `None` for a non-literal. For an RDF 1.2 directional string
172/// (`"x"@ar--rtl`) this is the LANGUAGE only (`ar`) — the base direction is
173/// separate (see [`lang_dir`]), matching SPARQL 1.2 `LANG`.
174pub fn lang_tag(token: &TermToken) -> Option<String> {
175    literal_suffix(token).map(|s| {
176        s.strip_prefix('@')
177            .unwrap_or("")
178            .split("--")
179            .next()
180            .unwrap_or("")
181            .to_string()
182    })
183}
184
185/// The base direction of an RDF 1.2 directional language string (`"x"@ar--rtl` →
186/// `rtl`), or `None` if the literal has no direction (or is not a literal).
187pub fn lang_dir(token: &TermToken) -> Option<String> {
188    let tag = literal_suffix(token)?.strip_prefix('@')?;
189    tag.split("--").nth(1).map(str::to_string)
190}
191
192/// Numeric value of a term: the lexical part of a literal parsed as `f64`
193/// (`"30"^^<…int>` → `30.0`) or a bare numeric token, else `None`.
194///
195/// The literal's lexical value is taken through the escape-aware
196/// [`literal_lexical`] — closing quote located correctly, body unescaped, any
197/// `^^<datatype>` suffix dropped — rather than scanning to the *first* `"`. A
198/// value containing an embedded escaped quote (`"1\"2"`) is therefore read
199/// whole and simply fails to parse, instead of being truncated at the escaped
200/// quote. A language-tagged literal (`"5"@en`) is never a numeric literal in
201/// SPARQL/RDF, so it yields `None`.
202pub fn as_number(token: &TermToken) -> Option<f64> {
203    if is_literal(token) {
204        if lang_tag(token).is_some_and(|tag| !tag.is_empty()) {
205            return None;
206        }
207        literal_lexical(token)?.parse::<f64>().ok()
208    } else {
209        token.parse::<f64>().ok()
210    }
211}
212
213/// Escape a string for use as the body of an N-Triples literal (`"…"`): the
214/// inverse of [`unescape_literal`] for the characters that must be escaped
215/// (`\`, `"`, newline, carriage return, tab). The common case (no special
216/// characters) returns the input untouched.
217pub fn escape_literal(s: &str) -> String {
218    if !s.contains(['\\', '"', '\n', '\r', '\t']) {
219        return s.to_string();
220    }
221    let mut out = String::with_capacity(s.len() + 2);
222    for c in s.chars() {
223        match c {
224            '\\' => out.push_str("\\\\"),
225            '"' => out.push_str("\\\""),
226            '\n' => out.push_str("\\n"),
227            '\r' => out.push_str("\\r"),
228            '\t' => out.push_str("\\t"),
229            _ => out.push(c),
230        }
231    }
232    out
233}
234
235/// Build a literal term token from a (raw, unescaped) lexical value, attaching
236/// an optional non-empty language tag (`@lang`) or datatype IRI content
237/// (`^^<dt>`). `lang` wins over `datatype` if both are given (a tagged literal
238/// is implicitly `rdf:langString`).
239pub fn make_literal(lexical: &str, lang: Option<&str>, datatype: Option<&str>) -> String {
240    let body = escape_literal(lexical);
241    match (lang.filter(|l| !l.is_empty()), datatype) {
242        (Some(l), _) => format!("\"{body}\"@{l}"),
243        (None, Some(dt)) => format!("\"{body}\"^^<{dt}>"),
244        (None, None) => format!("\"{body}\""),
245    }
246}
247
248/// Resolve the N-Triples escape sequences in a literal's body to actual chars
249/// (`\n`, `\t`, `\"`, `\\`, `\uXXXX`, `\UXXXXXXXX`, …). Strings without a
250/// backslash — the overwhelming majority — are returned untouched.
251pub fn unescape_literal(s: &str) -> String {
252    if !s.contains('\\') {
253        return s.to_string();
254    }
255    let mut out = String::with_capacity(s.len());
256    let mut chars = s.chars();
257    while let Some(c) = chars.next() {
258        if c != '\\' {
259            out.push(c);
260            continue;
261        }
262        let unicode = |chars: &mut std::str::Chars, n: usize, out: &mut String| {
263            let hex: String = chars.take(n).collect();
264            match u32::from_str_radix(&hex, 16).ok().and_then(char::from_u32) {
265                Some(ch) => out.push(ch),
266                None => out.push('\u{FFFD}'),
267            }
268        };
269        match chars.next() {
270            Some('t') => out.push('\t'),
271            Some('b') => out.push('\u{08}'),
272            Some('n') => out.push('\n'),
273            Some('r') => out.push('\r'),
274            Some('f') => out.push('\u{0C}'),
275            Some('"') => out.push('"'),
276            Some('\'') => out.push('\''),
277            Some('\\') => out.push('\\'),
278            Some('u') => unicode(&mut chars, 4, &mut out),
279            Some('U') => unicode(&mut chars, 8, &mut out),
280            Some(other) => {
281                out.push('\\');
282                out.push(other);
283            }
284            None => out.push('\\'),
285        }
286    }
287    out
288}
289
290/// Rewrite a term token into the **RDF 1.2 triple-term surface** for a text
291/// serializer.
292///
293/// rete stores a quoted triple in one canonical token, the RDF-star surface
294/// `<<s p o>>` — see `ingest::take_term`, which accepts both surfaces
295/// on ingest and canonicalises them to that one. Current RDF 1.2 parsers
296/// (oxttl 0.2 and anything built on it, including the `oxigraph` CLI) do not
297/// read that surface: in N-Triples/N-Quads they **reject** it outright, and in
298/// Turtle/TriG they read `<< s p o >>` as a *reifier* — one statement silently
299/// becomes two, with a blank node where the triple term was. So a dump in the
300/// stored surface is not interoperable, quietly in one format and loudly in the
301/// other. This is the translation that makes it so, at write time only: nothing
302/// about the file changes.
303///
304/// Returns:
305///
306/// * `Some(Borrowed(token))` when `token` is not a quoted triple. This is the
307///   overwhelming majority of terms and the only cost is a two-byte prefix
308///   check, so a dump with no quoted triples in it is byte-for-byte unchanged.
309/// * `Some(Owned(…))` with the token rewritten to `<<( s p o )>>`, recursively:
310///   a triple term nested in the object slot is rewritten too.
311/// * `None` when the token has **no RDF 1.2 spelling at all**. RDF 1.2 puts a
312///   triple term in *object position only* — the grammar is
313///   `tripleTerm ::= '<<(' ttSubject predicate ttObject ')>>'` with
314///   `ttSubject ::= iri | BlankNode` — so a quoted triple standing in the
315///   subject slot of another quoted triple cannot be written. (The caller is
316///   responsible for the same rule at statement level: a quoted triple in the
317///   *statement's* subject or predicate slot is equally unwritable, and the
318///   caller is the one that knows which slot a term came from.)
319///
320/// A token that is not well-formed — `<<` … `>>` that does not parse as three
321/// terms — also yields `None` rather than a mangled rewrite.
322pub fn rdf12_triple_term(token: &TermToken) -> Option<Cow<'_, TermToken>> {
323    if !is_quoted_triple(token) {
324        return Some(Cow::Borrowed(token));
325    }
326    rewrite_rdf12(token).map(Cow::Owned)
327}
328
329/// The owned half of [`rdf12_triple_term`], split out so the recursion does not
330/// re-run the `is_quoted_triple` fast path on a token it already classified.
331fn rewrite_rdf12(token: &TermToken) -> Option<String> {
332    let (s, p, o) = crate::ingest::quoted_triple_parts(token)?;
333    // `ttSubject ::= iri | BlankNode` and `predicate ::= iri`: neither slot
334    // admits a triple term, at any depth.
335    if is_quoted_triple(&s) || is_quoted_triple(&p) {
336        return None;
337    }
338    // `ttObject` does admit one, which is where nesting lives.
339    let o = if is_quoted_triple(&o) {
340        rewrite_rdf12(&o)?
341    } else {
342        o
343    };
344    Some(format!("<<( {s} {p} {o} )>>"))
345}
346
347#[cfg(test)]
348mod tests {
349    use super::*;
350
351    // --- the RDF 1.2 writer surface ----------------------------------------
352
353    #[test]
354    fn a_plain_term_is_borrowed_unchanged() {
355        // The hot path. Every term that is not a quoted triple comes back
356        // borrowed, which is what makes a dump of a quoted-triple-free file
357        // byte-for-byte what it was.
358        for t in [
359            "<http://example.org/x>",
360            "_:b0",
361            "\"lit\"@en",
362            "\"5\"^^<http://www.w3.org/2001/XMLSchema#integer>",
363            "\"a > b\"",
364        ] {
365            match rdf12_triple_term(t) {
366                Some(Cow::Borrowed(got)) => assert_eq!(got, t),
367                other => panic!("{t} should borrow unchanged, got {other:?}"),
368            }
369        }
370    }
371
372    #[test]
373    fn an_object_triple_term_gets_the_rdf12_surface() {
374        assert_eq!(
375            rdf12_triple_term("<<<http://ex/s> <http://ex/p> <http://ex/o>>>").unwrap(),
376            "<<( <http://ex/s> <http://ex/p> <http://ex/o> )>>"
377        );
378    }
379
380    #[test]
381    fn nesting_in_the_object_slot_recurses() {
382        // `ttObject` admits another triple term, so depth works — and the
383        // recursion has to rewrite the inner one too, not just the outer.
384        assert_eq!(
385            rdf12_triple_term(
386                "<<<http://ex/a> <http://ex/b> <<<http://ex/x> <http://ex/y> <http://ex/z>>>>>"
387            )
388            .unwrap(),
389            "<<( <http://ex/a> <http://ex/b> <<( <http://ex/x> <http://ex/y> <http://ex/z> )>> )>>"
390        );
391    }
392
393    #[test]
394    fn a_literal_object_survives_verbatim() {
395        // Term boundaries come from `take_term`, not from splitting on spaces,
396        // so a literal carrying spaces, a `>` and a `<<` does not derail it.
397        assert_eq!(
398            rdf12_triple_term("<<_:b1 <http://ex/p> \"a > b << c\"@en>>").unwrap(),
399            "<<( _:b1 <http://ex/p> \"a > b << c\"@en )>>"
400        );
401    }
402
403    #[test]
404    fn a_triple_term_in_a_subject_slot_has_no_rdf12_spelling() {
405        // RDF 1.2: `ttSubject ::= iri | BlankNode`. A quoted triple nested in
406        // another one's subject cannot be written, at any depth, and the honest
407        // answer is `None` rather than a token no parser accepts.
408        assert!(rdf12_triple_term(
409            "<<<<<http://ex/x> <http://ex/y> <http://ex/z>>> <http://ex/p> <http://ex/o>>>"
410        )
411        .is_none());
412        // …including one level down.
413        assert!(rdf12_triple_term(
414            "<<<http://ex/a> <http://ex/b> <<<<<http://ex/x> <http://ex/y> <http://ex/z>>> <http://ex/p> <http://ex/o>>>>>"
415        )
416        .is_none());
417    }
418
419    #[test]
420    fn a_malformed_quoted_triple_is_refused_not_mangled() {
421        assert!(rdf12_triple_term("<<<http://ex/s> <http://ex/p>>>").is_none());
422        assert!(rdf12_triple_term("<<>>").is_none());
423    }
424
425    #[test]
426    fn the_rewrite_is_what_ingest_accepts_back() {
427        // The round-trip property, at the term level: what the writer emits is
428        // what `take_term` canonicalises back to the stored token. This is why
429        // rete -> nq -> rete is safe in either surface.
430        let stored =
431            "<<<http://ex/a> <http://ex/b> <<<http://ex/x> <http://ex/y> <http://ex/z>>>>>";
432        let written = rdf12_triple_term(stored).unwrap().into_owned();
433        let (back, rest) = crate::ingest::take_term(&written).unwrap();
434        assert_eq!(back, stored);
435        assert!(rest.trim().is_empty());
436    }
437
438    #[test]
439    fn term_kinds() {
440        assert!(is_iri("<http://example.org/x>"));
441        assert!(!is_iri("\"x\""));
442        assert!(!is_iri("_:b0"));
443        assert!(is_blank("_:b0"));
444        assert!(is_literal("\"x\"@en"));
445        assert_eq!(iri_content("<http://x>"), Some("http://x"));
446        assert_eq!(iri_content("\"x\""), None);
447    }
448
449    #[test]
450    fn lexical_values() {
451        assert_eq!(literal_lexical("\"hello\""), Some("hello".to_string()));
452        assert_eq!(literal_lexical("\"42\"^^<int>"), Some("42".to_string()));
453        assert_eq!(literal_lexical("\"hi\"@en"), Some("hi".to_string()));
454        assert_eq!(literal_lexical("<http://x>"), None);
455        // any-term lexical
456        assert_eq!(lexical("\"hi\"@en"), "hi");
457        assert_eq!(lexical("<http://x>"), "http://x");
458        assert_eq!(lexical("_:b0"), "_:b0");
459    }
460
461    #[test]
462    fn datatype_and_lang() {
463        assert_eq!(literal_datatype("\"42\"^^<int>").as_deref(), Some("int"));
464        assert_eq!(
465            literal_datatype("\"hi\"@en").as_deref(),
466            Some(RDF_LANG_STRING)
467        );
468        assert_eq!(literal_datatype("\"plain\"").as_deref(), Some(XSD_STRING));
469        assert_eq!(literal_datatype("<http://x>"), None);
470        assert_eq!(lang_tag("\"hi\"@en").as_deref(), Some("en"));
471        assert_eq!(lang_tag("\"plain\"").as_deref(), Some(""));
472        assert_eq!(lang_tag("<http://x>"), None);
473    }
474
475    #[test]
476    fn numbers() {
477        assert_eq!(as_number("\"30\"^^<int>"), Some(30.0));
478        assert_eq!(as_number("3.5"), Some(3.5));
479        assert_eq!(as_number("\"nope\""), None);
480        assert_eq!(as_number("<http://x>"), None);
481    }
482
483    #[test]
484    fn as_number_escape_aware() {
485        // Plain and typed numeric literals parse as before.
486        assert_eq!(
487            as_number("\"42\"^^<http://www.w3.org/2001/XMLSchema#integer>"),
488            Some(42.0)
489        );
490        assert_eq!(
491            as_number("\"12.5\"^^<http://www.w3.org/2001/XMLSchema#decimal>"),
492            Some(12.5)
493        );
494        assert_eq!(
495            as_number("\"6.022e23\"^^<http://www.w3.org/2001/XMLSchema#double>"),
496            Some(6.022e23)
497        );
498        // Plain literal, negative, and leading-`+`.
499        assert_eq!(as_number("\"5\""), Some(5.0));
500        assert_eq!(as_number("\"-5\"^^<int>"), Some(-5.0));
501        assert_eq!(as_number("\"+7\""), Some(7.0));
502        // Non-numeric literal → None.
503        assert_eq!(as_number("\"not a number\""), None);
504        // A value with an EMBEDDED escaped quote must be read whole (`1"2`),
505        // fail to parse, and never be truncated to `1` (or panic). This is the
506        // escape-aware path: the old first-`"` scan would have stopped early.
507        assert_eq!(as_number("\"1\\\"2\""), None);
508        // IRI and blank node → None.
509        assert_eq!(as_number("<http://example.org/n>"), None);
510        assert_eq!(as_number("_:b0"), None);
511        // A language-tagged literal is never numeric (SPARQL): `"5"@en` → None.
512        assert_eq!(as_number("\"5\"@en"), None);
513    }
514
515    #[test]
516    fn escapes() {
517        assert_eq!(unescape_literal("plain"), "plain");
518        assert_eq!(unescape_literal("a\\nb"), "a\nb");
519        assert_eq!(unescape_literal("a\\\"b"), "a\"b");
520        assert_eq!(unescape_literal("\\u0041"), "A");
521        // an escaped quote inside the body is honored by the closing-quote scan
522        assert_eq!(literal_lexical("\"a\\\"b\"@en"), Some("a\"b".to_string()));
523    }
524}