Skip to main content

rete_core/
terms.rs

1//! Term identifiers and N-Triples term-token helpers.
2//!
3//! Two distinct things live in one place here because they describe the same
4//! domain — "what is a term, and how is it identified":
5//!
6//! 1. **ID aliases.** Every term in a `.rete` file is interned into the
7//!    dictionary and addressed by a `u32`. Code passes these `u32`s through
8//!    many signatures where the bare type says nothing about *which* id space a
9//!    value lives in. The aliases below ([`NodeId`], [`SubjectId`],
10//!    [`PredicateId`], [`ObjectId`]) are documentation: they are all `u32`
11//!    today (so they cost nothing and mix freely), but they let a signature
12//!    state its intent — `subject_node(sid: SubjectId) -> NodeId` reads as the
13//!    role-id → unified-node mapping it is. A later pass can promote them to
14//!    true newtypes (`struct NodeId(u32)`) without touching call sites that
15//!    already name the alias.
16//!
17//! 2. **Term-token helpers.** A [`TermToken`] is the textual form of a term as
18//!    it appears in N-Triples and in the dictionary: an IRI `<http://…>`, a
19//!    blank node `_:b0`, or a literal `"text"`, `"text"@en`, `"text"^^<dt>`.
20//!    Several modules (SPARQL evaluation, SHACL validation, doc rendering)
21//!    independently grew the same little parsers for "is this an IRI", "what's
22//!    the lexical value", "what's the datatype". They are consolidated here so
23//!    there is one definition of the term grammar to reason about.
24
25use std::borrow::Cow;
26
27/// A dictionary id in the **unified node space** — the single id space that
28/// covers every term that ever appears as a subject or an object. This is the
29/// id reachability, the community pyramid, and the graph index work in.
30pub type NodeId = u32;
31
32/// A dictionary id in the **subject** id space (terms seen in subject
33/// position). Map to a [`NodeId`] with [`Dictionary::subject_node`].
34///
35/// [`Dictionary::subject_node`]: crate::dictionary::Dictionary::subject_node
36pub type SubjectId = u32;
37
38/// A dictionary id in the **predicate** id space. Predicates have their own
39/// dense id space and are never part of the unified node space.
40pub type PredicateId = u32;
41
42/// A dictionary id in the **object** id space (terms seen in object position).
43/// Map to a [`NodeId`] with [`Dictionary::object_node`].
44///
45/// [`Dictionary::object_node`]: crate::dictionary::Dictionary::object_node
46pub type ObjectId = u32;
47
48/// The textual form of an RDF term as stored in the dictionary and emitted in
49/// N-Triples: an IRI (`<…>`), a blank node (`_:…`), or a literal (`"…"`,
50/// optionally with an `@lang` or `^^<datatype>` suffix). An alias for `str`;
51/// it names intent at API boundaries that take a term rather than arbitrary
52/// text.
53pub type TermToken = str;
54
55const XSD_STRING: &str = "http://www.w3.org/2001/XMLSchema#string";
56const RDF_LANG_STRING: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#langString";
57/// RDF 1.2 base-direction language string: `"…"@lang--dir` (dir = `ltr`/`rtl`).
58const RDF_DIR_LANG_STRING: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#dirLangString";
59
60/// Is `t` an IRI term (`<…>`)?
61#[inline]
62pub fn is_iri(t: &TermToken) -> bool {
63    // A quoted triple (`<< … >>`, RDF-star) also starts with `<` and ends with
64    // `>`, so exclude it explicitly — it is its own term kind, not an IRI.
65    t.starts_with('<') && !t.starts_with("<<") && t.ends_with('>')
66}
67
68/// Is `t` a **quoted triple** term (`<< s p o >>`, RDF-star)? These appear only
69/// in subject/object position and are stored in the dictionary as their
70/// canonical N-Triples-star surface, exactly like any other term.
71#[inline]
72pub fn is_quoted_triple(t: &TermToken) -> bool {
73    t.starts_with("<<") && t.ends_with(">>")
74}
75
76/// The content of an IRI term without its angle brackets (`<http://x>` →
77/// `http://x`), or `None` if `t` is not an IRI term.
78#[inline]
79pub fn iri_content(t: &TermToken) -> Option<&str> {
80    if is_quoted_triple(t) {
81        return None;
82    }
83    t.strip_prefix('<').and_then(|s| s.strip_suffix('>'))
84}
85
86/// Is `t` a blank-node term (`_:…`)?
87#[inline]
88pub fn is_blank(t: &TermToken) -> bool {
89    t.starts_with("_:")
90}
91
92/// Is `t` a literal term (`"…"`)?
93#[inline]
94pub fn is_literal(t: &TermToken) -> bool {
95    t.starts_with('"')
96}
97
98/// Index of the closing quote of a literal term, scanning from the opening
99/// quote and honoring `\"` escapes. `t` must start with `"`.
100fn closing_quote(t: &TermToken) -> usize {
101    let bytes = t.as_bytes();
102    let mut i = 1;
103    while i < bytes.len() {
104        match bytes[i] {
105            b'\\' => i += 2,
106            b'"' => break,
107            _ => i += 1,
108        }
109    }
110    i.min(t.len())
111}
112
113/// The **lexical value** of a literal term — the text between the quotes with
114/// N-Triples escapes resolved — or `None` for IRIs and blank nodes. The
115/// datatype and language suffix are dropped (`"42"^^<…int>` → `42`,
116/// `"hi"@en` → `hi`).
117pub fn literal_lexical(token: &TermToken) -> Option<String> {
118    if !is_literal(token) {
119        return None;
120    }
121    Some(unescape_literal(&token[1..closing_quote(token)]))
122}
123
124/// The **lexical value** of any term: a literal's unescaped body, an IRI's
125/// content, or a blank-node token unchanged. Always succeeds. Useful where a
126/// plain comparable string is wanted regardless of term kind.
127pub fn lexical(token: &TermToken) -> Cow<'_, str> {
128    if is_literal(token) {
129        Cow::Owned(unescape_literal(&token[1..closing_quote(token)]))
130    } else if let Some(iri) = iri_content(token) {
131        Cow::Borrowed(iri)
132    } else {
133        Cow::Borrowed(token)
134    }
135}
136
137/// The part of a literal term after its closing quote (`"x"^^<dt>` → `^^<dt>`,
138/// `"x"@en` → `@en`, `"x"` → ``), or `None` if `token` is not a literal.
139fn literal_suffix(token: &TermToken) -> Option<&str> {
140    if !is_literal(token) {
141        return None;
142    }
143    token.get(closing_quote(token) + 1..)
144}
145
146/// The datatype IRI **content** of a literal term (no angle brackets): the
147/// explicit `^^<dt>`, else `rdf:langString` for a language-tagged literal,
148/// else `xsd:string` for a plain one. `None` for a non-literal or a malformed
149/// suffix.
150pub fn literal_datatype(token: &TermToken) -> Option<String> {
151    let suffix = literal_suffix(token)?;
152    if let Some(dt) = suffix.strip_prefix("^^<").and_then(|s| s.strip_suffix('>')) {
153        Some(dt.to_string())
154    } else if suffix.starts_with('@') {
155        // RDF 1.2: a language string WITH a base direction (`@lang--dir`) is an
156        // `rdf:dirLangString`; a plain `@lang` is `rdf:langString`. `--` never
157        // occurs in a well-formed BCP-47 tag, so this is unambiguous.
158        if suffix.contains("--") {
159            Some(RDF_DIR_LANG_STRING.to_string())
160        } else {
161            Some(RDF_LANG_STRING.to_string())
162        }
163    } else if suffix.is_empty() {
164        Some(XSD_STRING.to_string())
165    } else {
166        None
167    }
168}
169
170/// The language tag of a literal term (`"hi"@en` → `en`), `""` when the literal
171/// is untagged, or `None` for a non-literal. For an RDF 1.2 directional string
172/// (`"x"@ar--rtl`) this is the LANGUAGE only (`ar`) — the base direction is
173/// separate (see [`lang_dir`]), matching SPARQL 1.2 `LANG`.
174pub fn lang_tag(token: &TermToken) -> Option<String> {
175    literal_suffix(token).map(|s| {
176        s.strip_prefix('@')
177            .unwrap_or("")
178            .split("--")
179            .next()
180            .unwrap_or("")
181            .to_string()
182    })
183}
184
185/// The base direction of an RDF 1.2 directional language string (`"x"@ar--rtl` →
186/// `rtl`), or `None` if the literal has no direction (or is not a literal).
187pub fn lang_dir(token: &TermToken) -> Option<String> {
188    let tag = literal_suffix(token)?.strip_prefix('@')?;
189    tag.split("--").nth(1).map(str::to_string)
190}
191
192/// Numeric value of a term: the lexical part of a literal parsed as `f64`
193/// (`"30"^^<…int>` → `30.0`) or a bare numeric token, else `None`.
194pub fn as_number(token: &TermToken) -> Option<f64> {
195    let lex = if let Some(rest) = token.strip_prefix('"') {
196        &rest[..rest.find('"')?]
197    } else {
198        token
199    };
200    lex.parse::<f64>().ok()
201}
202
203/// Escape a string for use as the body of an N-Triples literal (`"…"`): the
204/// inverse of [`unescape_literal`] for the characters that must be escaped
205/// (`\`, `"`, newline, carriage return, tab). The common case (no special
206/// characters) returns the input untouched.
207pub fn escape_literal(s: &str) -> String {
208    if !s.contains(['\\', '"', '\n', '\r', '\t']) {
209        return s.to_string();
210    }
211    let mut out = String::with_capacity(s.len() + 2);
212    for c in s.chars() {
213        match c {
214            '\\' => out.push_str("\\\\"),
215            '"' => out.push_str("\\\""),
216            '\n' => out.push_str("\\n"),
217            '\r' => out.push_str("\\r"),
218            '\t' => out.push_str("\\t"),
219            _ => out.push(c),
220        }
221    }
222    out
223}
224
225/// Build a literal term token from a (raw, unescaped) lexical value, attaching
226/// an optional non-empty language tag (`@lang`) or datatype IRI content
227/// (`^^<dt>`). `lang` wins over `datatype` if both are given (a tagged literal
228/// is implicitly `rdf:langString`).
229pub fn make_literal(lexical: &str, lang: Option<&str>, datatype: Option<&str>) -> String {
230    let body = escape_literal(lexical);
231    match (lang.filter(|l| !l.is_empty()), datatype) {
232        (Some(l), _) => format!("\"{body}\"@{l}"),
233        (None, Some(dt)) => format!("\"{body}\"^^<{dt}>"),
234        (None, None) => format!("\"{body}\""),
235    }
236}
237
238/// Resolve the N-Triples escape sequences in a literal's body to actual chars
239/// (`\n`, `\t`, `\"`, `\\`, `\uXXXX`, `\UXXXXXXXX`, …). Strings without a
240/// backslash — the overwhelming majority — are returned untouched.
241pub fn unescape_literal(s: &str) -> String {
242    if !s.contains('\\') {
243        return s.to_string();
244    }
245    let mut out = String::with_capacity(s.len());
246    let mut chars = s.chars();
247    while let Some(c) = chars.next() {
248        if c != '\\' {
249            out.push(c);
250            continue;
251        }
252        let unicode = |chars: &mut std::str::Chars, n: usize, out: &mut String| {
253            let hex: String = chars.take(n).collect();
254            match u32::from_str_radix(&hex, 16).ok().and_then(char::from_u32) {
255                Some(ch) => out.push(ch),
256                None => out.push('\u{FFFD}'),
257            }
258        };
259        match chars.next() {
260            Some('t') => out.push('\t'),
261            Some('b') => out.push('\u{08}'),
262            Some('n') => out.push('\n'),
263            Some('r') => out.push('\r'),
264            Some('f') => out.push('\u{0C}'),
265            Some('"') => out.push('"'),
266            Some('\'') => out.push('\''),
267            Some('\\') => out.push('\\'),
268            Some('u') => unicode(&mut chars, 4, &mut out),
269            Some('U') => unicode(&mut chars, 8, &mut out),
270            Some(other) => {
271                out.push('\\');
272                out.push(other);
273            }
274            None => out.push('\\'),
275        }
276    }
277    out
278}
279
280#[cfg(test)]
281mod tests {
282    use super::*;
283
284    #[test]
285    fn term_kinds() {
286        assert!(is_iri("<http://example.org/x>"));
287        assert!(!is_iri("\"x\""));
288        assert!(!is_iri("_:b0"));
289        assert!(is_blank("_:b0"));
290        assert!(is_literal("\"x\"@en"));
291        assert_eq!(iri_content("<http://x>"), Some("http://x"));
292        assert_eq!(iri_content("\"x\""), None);
293    }
294
295    #[test]
296    fn lexical_values() {
297        assert_eq!(literal_lexical("\"hello\""), Some("hello".to_string()));
298        assert_eq!(literal_lexical("\"42\"^^<int>"), Some("42".to_string()));
299        assert_eq!(literal_lexical("\"hi\"@en"), Some("hi".to_string()));
300        assert_eq!(literal_lexical("<http://x>"), None);
301        // any-term lexical
302        assert_eq!(lexical("\"hi\"@en"), "hi");
303        assert_eq!(lexical("<http://x>"), "http://x");
304        assert_eq!(lexical("_:b0"), "_:b0");
305    }
306
307    #[test]
308    fn datatype_and_lang() {
309        assert_eq!(literal_datatype("\"42\"^^<int>").as_deref(), Some("int"));
310        assert_eq!(
311            literal_datatype("\"hi\"@en").as_deref(),
312            Some(RDF_LANG_STRING)
313        );
314        assert_eq!(literal_datatype("\"plain\"").as_deref(), Some(XSD_STRING));
315        assert_eq!(literal_datatype("<http://x>"), None);
316        assert_eq!(lang_tag("\"hi\"@en").as_deref(), Some("en"));
317        assert_eq!(lang_tag("\"plain\"").as_deref(), Some(""));
318        assert_eq!(lang_tag("<http://x>"), None);
319    }
320
321    #[test]
322    fn numbers() {
323        assert_eq!(as_number("\"30\"^^<int>"), Some(30.0));
324        assert_eq!(as_number("3.5"), Some(3.5));
325        assert_eq!(as_number("\"nope\""), None);
326        assert_eq!(as_number("<http://x>"), None);
327    }
328
329    #[test]
330    fn escapes() {
331        assert_eq!(unescape_literal("plain"), "plain");
332        assert_eq!(unescape_literal("a\\nb"), "a\nb");
333        assert_eq!(unescape_literal("a\\\"b"), "a\"b");
334        assert_eq!(unescape_literal("\\u0041"), "A");
335        // an escaped quote inside the body is honored by the closing-quote scan
336        assert_eq!(literal_lexical("\"a\\\"b\"@en"), Some("a\"b".to_string()));
337    }
338}