rete_core/terms.rs
1//! Term identifiers and N-Triples term-token helpers.
2//!
3//! Two distinct things live in one place here because they describe the same
4//! domain — "what is a term, and how is it identified":
5//!
6//! 1. **ID aliases.** Every term in a `.rete` file is interned into the
7//! dictionary and addressed by a `u32`. Code passes these `u32`s through
8//! many signatures where the bare type says nothing about *which* id space a
9//! value lives in. The aliases below ([`NodeId`], [`SubjectId`],
10//! [`PredicateId`], [`ObjectId`]) are documentation: they are all `u32`
11//! today (so they cost nothing and mix freely), but they let a signature
12//! state its intent — `subject_node(sid: SubjectId) -> NodeId` reads as the
13//! role-id → unified-node mapping it is. A later pass can promote them to
14//! true newtypes (`struct NodeId(u32)`) without touching call sites that
15//! already name the alias.
16//!
17//! 2. **Term-token helpers.** A [`TermToken`] is the textual form of a term as
18//! it appears in N-Triples and in the dictionary: an IRI `<http://…>`, a
19//! blank node `_:b0`, or a literal `"text"`, `"text"@en`, `"text"^^<dt>`.
20//! Several modules (SPARQL evaluation, SHACL validation, doc rendering)
21//! independently grew the same little parsers for "is this an IRI", "what's
22//! the lexical value", "what's the datatype". They are consolidated here so
23//! there is one definition of the term grammar to reason about.
24
25use std::borrow::Cow;
26
27/// A dictionary id in the **unified node space** — the single id space that
28/// covers every term that ever appears as a subject or an object. This is the
29/// id reachability, the community pyramid, and the graph index work in.
30pub type NodeId = u32;
31
32/// A dictionary id in the **subject** id space (terms seen in subject
33/// position). Map to a [`NodeId`] with [`Dictionary::subject_node`].
34///
35/// [`Dictionary::subject_node`]: crate::dictionary::Dictionary::subject_node
36pub type SubjectId = u32;
37
38/// A dictionary id in the **predicate** id space. Predicates have their own
39/// dense id space and are never part of the unified node space.
40pub type PredicateId = u32;
41
42/// A dictionary id in the **object** id space (terms seen in object position).
43/// Map to a [`NodeId`] with [`Dictionary::object_node`].
44///
45/// [`Dictionary::object_node`]: crate::dictionary::Dictionary::object_node
46pub type ObjectId = u32;
47
48/// The textual form of an RDF term as stored in the dictionary and emitted in
49/// N-Triples: an IRI (`<…>`), a blank node (`_:…`), or a literal (`"…"`,
50/// optionally with an `@lang` or `^^<datatype>` suffix). An alias for `str`;
51/// it names intent at API boundaries that take a term rather than arbitrary
52/// text.
53pub type TermToken = str;
54
55const XSD_STRING: &str = "http://www.w3.org/2001/XMLSchema#string";
56const RDF_LANG_STRING: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#langString";
57/// RDF 1.2 base-direction language string: `"…"@lang--dir` (dir = `ltr`/`rtl`).
58const RDF_DIR_LANG_STRING: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#dirLangString";
59
60/// Is `t` an IRI term (`<…>`)?
61#[inline]
62pub fn is_iri(t: &TermToken) -> bool {
63 // A quoted triple (`<< … >>`, RDF-star) also starts with `<` and ends with
64 // `>`, so exclude it explicitly — it is its own term kind, not an IRI.
65 t.starts_with('<') && !t.starts_with("<<") && t.ends_with('>')
66}
67
68/// Is `t` a **quoted triple** term (`<< s p o >>`, RDF-star)? These appear only
69/// in subject/object position and are stored in the dictionary as their
70/// canonical N-Triples-star surface, exactly like any other term.
71#[inline]
72pub fn is_quoted_triple(t: &TermToken) -> bool {
73 t.starts_with("<<") && t.ends_with(">>")
74}
75
76/// The content of an IRI term without its angle brackets (`<http://x>` →
77/// `http://x`), or `None` if `t` is not an IRI term.
78#[inline]
79pub fn iri_content(t: &TermToken) -> Option<&str> {
80 if is_quoted_triple(t) {
81 return None;
82 }
83 t.strip_prefix('<').and_then(|s| s.strip_suffix('>'))
84}
85
86/// Is `t` a blank-node term (`_:…`)?
87#[inline]
88pub fn is_blank(t: &TermToken) -> bool {
89 t.starts_with("_:")
90}
91
92/// Is `t` a literal term (`"…"`)?
93#[inline]
94pub fn is_literal(t: &TermToken) -> bool {
95 t.starts_with('"')
96}
97
98/// Index of the closing quote of a literal term, scanning from the opening
99/// quote and honoring `\"` escapes. `t` must start with `"`.
100fn closing_quote(t: &TermToken) -> usize {
101 let bytes = t.as_bytes();
102 let mut i = 1;
103 while i < bytes.len() {
104 match bytes[i] {
105 b'\\' => i += 2,
106 b'"' => break,
107 _ => i += 1,
108 }
109 }
110 i.min(t.len())
111}
112
113/// The **lexical value** of a literal term — the text between the quotes with
114/// N-Triples escapes resolved — or `None` for IRIs and blank nodes. The
115/// datatype and language suffix are dropped (`"42"^^<…int>` → `42`,
116/// `"hi"@en` → `hi`).
117pub fn literal_lexical(token: &TermToken) -> Option<String> {
118 if !is_literal(token) {
119 return None;
120 }
121 Some(unescape_literal(&token[1..closing_quote(token)]))
122}
123
124/// The **lexical value** of any term: a literal's unescaped body, an IRI's
125/// content, or a blank-node token unchanged. Always succeeds. Useful where a
126/// plain comparable string is wanted regardless of term kind.
127pub fn lexical(token: &TermToken) -> Cow<'_, str> {
128 if is_literal(token) {
129 Cow::Owned(unescape_literal(&token[1..closing_quote(token)]))
130 } else if let Some(iri) = iri_content(token) {
131 Cow::Borrowed(iri)
132 } else {
133 Cow::Borrowed(token)
134 }
135}
136
137/// The part of a literal term after its closing quote (`"x"^^<dt>` → `^^<dt>`,
138/// `"x"@en` → `@en`, `"x"` → ``), or `None` if `token` is not a literal.
139fn literal_suffix(token: &TermToken) -> Option<&str> {
140 if !is_literal(token) {
141 return None;
142 }
143 token.get(closing_quote(token) + 1..)
144}
145
146/// The datatype IRI **content** of a literal term (no angle brackets): the
147/// explicit `^^<dt>`, else `rdf:langString` for a language-tagged literal,
148/// else `xsd:string` for a plain one. `None` for a non-literal or a malformed
149/// suffix.
150pub fn literal_datatype(token: &TermToken) -> Option<String> {
151 let suffix = literal_suffix(token)?;
152 if let Some(dt) = suffix.strip_prefix("^^<").and_then(|s| s.strip_suffix('>')) {
153 Some(dt.to_string())
154 } else if suffix.starts_with('@') {
155 // RDF 1.2: a language string WITH a base direction (`@lang--dir`) is an
156 // `rdf:dirLangString`; a plain `@lang` is `rdf:langString`. `--` never
157 // occurs in a well-formed BCP-47 tag, so this is unambiguous.
158 if suffix.contains("--") {
159 Some(RDF_DIR_LANG_STRING.to_string())
160 } else {
161 Some(RDF_LANG_STRING.to_string())
162 }
163 } else if suffix.is_empty() {
164 Some(XSD_STRING.to_string())
165 } else {
166 None
167 }
168}
169
170/// The language tag of a literal term (`"hi"@en` → `en`), `""` when the literal
171/// is untagged, or `None` for a non-literal. For an RDF 1.2 directional string
172/// (`"x"@ar--rtl`) this is the LANGUAGE only (`ar`) — the base direction is
173/// separate (see [`lang_dir`]), matching SPARQL 1.2 `LANG`.
174pub fn lang_tag(token: &TermToken) -> Option<String> {
175 literal_suffix(token).map(|s| {
176 s.strip_prefix('@')
177 .unwrap_or("")
178 .split("--")
179 .next()
180 .unwrap_or("")
181 .to_string()
182 })
183}
184
185/// The base direction of an RDF 1.2 directional language string (`"x"@ar--rtl` →
186/// `rtl`), or `None` if the literal has no direction (or is not a literal).
187pub fn lang_dir(token: &TermToken) -> Option<String> {
188 let tag = literal_suffix(token)?.strip_prefix('@')?;
189 tag.split("--").nth(1).map(str::to_string)
190}
191
192/// Numeric value of a term: the lexical part of a literal parsed as `f64`
193/// (`"30"^^<…int>` → `30.0`) or a bare numeric token, else `None`.
194pub fn as_number(token: &TermToken) -> Option<f64> {
195 let lex = if let Some(rest) = token.strip_prefix('"') {
196 &rest[..rest.find('"')?]
197 } else {
198 token
199 };
200 lex.parse::<f64>().ok()
201}
202
203/// Escape a string for use as the body of an N-Triples literal (`"…"`): the
204/// inverse of [`unescape_literal`] for the characters that must be escaped
205/// (`\`, `"`, newline, carriage return, tab). The common case (no special
206/// characters) returns the input untouched.
207pub fn escape_literal(s: &str) -> String {
208 if !s.contains(['\\', '"', '\n', '\r', '\t']) {
209 return s.to_string();
210 }
211 let mut out = String::with_capacity(s.len() + 2);
212 for c in s.chars() {
213 match c {
214 '\\' => out.push_str("\\\\"),
215 '"' => out.push_str("\\\""),
216 '\n' => out.push_str("\\n"),
217 '\r' => out.push_str("\\r"),
218 '\t' => out.push_str("\\t"),
219 _ => out.push(c),
220 }
221 }
222 out
223}
224
225/// Build a literal term token from a (raw, unescaped) lexical value, attaching
226/// an optional non-empty language tag (`@lang`) or datatype IRI content
227/// (`^^<dt>`). `lang` wins over `datatype` if both are given (a tagged literal
228/// is implicitly `rdf:langString`).
229pub fn make_literal(lexical: &str, lang: Option<&str>, datatype: Option<&str>) -> String {
230 let body = escape_literal(lexical);
231 match (lang.filter(|l| !l.is_empty()), datatype) {
232 (Some(l), _) => format!("\"{body}\"@{l}"),
233 (None, Some(dt)) => format!("\"{body}\"^^<{dt}>"),
234 (None, None) => format!("\"{body}\""),
235 }
236}
237
238/// Resolve the N-Triples escape sequences in a literal's body to actual chars
239/// (`\n`, `\t`, `\"`, `\\`, `\uXXXX`, `\UXXXXXXXX`, …). Strings without a
240/// backslash — the overwhelming majority — are returned untouched.
241pub fn unescape_literal(s: &str) -> String {
242 if !s.contains('\\') {
243 return s.to_string();
244 }
245 let mut out = String::with_capacity(s.len());
246 let mut chars = s.chars();
247 while let Some(c) = chars.next() {
248 if c != '\\' {
249 out.push(c);
250 continue;
251 }
252 let unicode = |chars: &mut std::str::Chars, n: usize, out: &mut String| {
253 let hex: String = chars.take(n).collect();
254 match u32::from_str_radix(&hex, 16).ok().and_then(char::from_u32) {
255 Some(ch) => out.push(ch),
256 None => out.push('\u{FFFD}'),
257 }
258 };
259 match chars.next() {
260 Some('t') => out.push('\t'),
261 Some('b') => out.push('\u{08}'),
262 Some('n') => out.push('\n'),
263 Some('r') => out.push('\r'),
264 Some('f') => out.push('\u{0C}'),
265 Some('"') => out.push('"'),
266 Some('\'') => out.push('\''),
267 Some('\\') => out.push('\\'),
268 Some('u') => unicode(&mut chars, 4, &mut out),
269 Some('U') => unicode(&mut chars, 8, &mut out),
270 Some(other) => {
271 out.push('\\');
272 out.push(other);
273 }
274 None => out.push('\\'),
275 }
276 }
277 out
278}
279
280#[cfg(test)]
281mod tests {
282 use super::*;
283
284 #[test]
285 fn term_kinds() {
286 assert!(is_iri("<http://example.org/x>"));
287 assert!(!is_iri("\"x\""));
288 assert!(!is_iri("_:b0"));
289 assert!(is_blank("_:b0"));
290 assert!(is_literal("\"x\"@en"));
291 assert_eq!(iri_content("<http://x>"), Some("http://x"));
292 assert_eq!(iri_content("\"x\""), None);
293 }
294
295 #[test]
296 fn lexical_values() {
297 assert_eq!(literal_lexical("\"hello\""), Some("hello".to_string()));
298 assert_eq!(literal_lexical("\"42\"^^<int>"), Some("42".to_string()));
299 assert_eq!(literal_lexical("\"hi\"@en"), Some("hi".to_string()));
300 assert_eq!(literal_lexical("<http://x>"), None);
301 // any-term lexical
302 assert_eq!(lexical("\"hi\"@en"), "hi");
303 assert_eq!(lexical("<http://x>"), "http://x");
304 assert_eq!(lexical("_:b0"), "_:b0");
305 }
306
307 #[test]
308 fn datatype_and_lang() {
309 assert_eq!(literal_datatype("\"42\"^^<int>").as_deref(), Some("int"));
310 assert_eq!(
311 literal_datatype("\"hi\"@en").as_deref(),
312 Some(RDF_LANG_STRING)
313 );
314 assert_eq!(literal_datatype("\"plain\"").as_deref(), Some(XSD_STRING));
315 assert_eq!(literal_datatype("<http://x>"), None);
316 assert_eq!(lang_tag("\"hi\"@en").as_deref(), Some("en"));
317 assert_eq!(lang_tag("\"plain\"").as_deref(), Some(""));
318 assert_eq!(lang_tag("<http://x>"), None);
319 }
320
321 #[test]
322 fn numbers() {
323 assert_eq!(as_number("\"30\"^^<int>"), Some(30.0));
324 assert_eq!(as_number("3.5"), Some(3.5));
325 assert_eq!(as_number("\"nope\""), None);
326 assert_eq!(as_number("<http://x>"), None);
327 }
328
329 #[test]
330 fn escapes() {
331 assert_eq!(unescape_literal("plain"), "plain");
332 assert_eq!(unescape_literal("a\\nb"), "a\nb");
333 assert_eq!(unescape_literal("a\\\"b"), "a\"b");
334 assert_eq!(unescape_literal("\\u0041"), "A");
335 // an escaped quote inside the body is honored by the closing-quote scan
336 assert_eq!(literal_lexical("\"a\\\"b\"@en"), Some("a\"b".to_string()));
337 }
338}