rete_core/terms.rs
1//! Term identifiers and N-Triples term-token helpers.
2//!
3//! Two distinct things live in one place here because they describe the same
4//! domain — "what is a term, and how is it identified":
5//!
6//! 1. **ID aliases.** Every term in a `.rete` file is interned into the
7//! dictionary and addressed by a `u32`. Code passes these `u32`s through
8//! many signatures where the bare type says nothing about *which* id space a
9//! value lives in. The aliases below ([`NodeId`], [`SubjectId`],
10//! [`PredicateId`], [`ObjectId`]) are documentation: they are all `u32`
11//! today (so they cost nothing and mix freely), but they let a signature
12//! state its intent — `subject_node(sid: SubjectId) -> NodeId` reads as the
13//! role-id → unified-node mapping it is. A later pass can promote them to
14//! true newtypes (`struct NodeId(u32)`) without touching call sites that
15//! already name the alias.
16//!
17//! 2. **Term-token helpers.** A [`TermToken`] is the textual form of a term as
18//! it appears in N-Triples and in the dictionary: an IRI `<http://…>`, a
19//! blank node `_:b0`, or a literal `"text"`, `"text"@en`, `"text"^^<dt>`.
20//! Several modules (SPARQL evaluation, SHACL validation, doc rendering)
21//! independently grew the same little parsers for "is this an IRI", "what's
22//! the lexical value", "what's the datatype". They are consolidated here so
23//! there is one definition of the term grammar to reason about.
24
25use std::borrow::Cow;
26
27/// A dictionary id in the **unified node space** — the single id space that
28/// covers every term that ever appears as a subject or an object. This is the
29/// id reachability, the community pyramid, and the graph index work in.
30pub type NodeId = u32;
31
32/// A dictionary id in the **subject** id space (terms seen in subject
33/// position). Map to a [`NodeId`] with [`Dictionary::subject_node`].
34///
35/// [`Dictionary::subject_node`]: crate::dictionary::Dictionary::subject_node
36pub type SubjectId = u32;
37
38/// A dictionary id in the **predicate** id space. Predicates have their own
39/// dense id space and are never part of the unified node space.
40pub type PredicateId = u32;
41
42/// A dictionary id in the **object** id space (terms seen in object position).
43/// Map to a [`NodeId`] with [`Dictionary::object_node`].
44///
45/// [`Dictionary::object_node`]: crate::dictionary::Dictionary::object_node
46pub type ObjectId = u32;
47
48/// The textual form of an RDF term as stored in the dictionary and emitted in
49/// N-Triples: an IRI (`<…>`), a blank node (`_:…`), or a literal (`"…"`,
50/// optionally with an `@lang` or `^^<datatype>` suffix). An alias for `str`;
51/// it names intent at API boundaries that take a term rather than arbitrary
52/// text.
53pub type TermToken = str;
54
55const XSD_STRING: &str = "http://www.w3.org/2001/XMLSchema#string";
56const RDF_LANG_STRING: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#langString";
57/// RDF 1.2 base-direction language string: `"…"@lang--dir` (dir = `ltr`/`rtl`).
58const RDF_DIR_LANG_STRING: &str = "http://www.w3.org/1999/02/22-rdf-syntax-ns#dirLangString";
59
60/// Is `t` an IRI term (`<…>`)?
61#[inline]
62pub fn is_iri(t: &TermToken) -> bool {
63 // A quoted triple (`<< … >>`, RDF-star) also starts with `<` and ends with
64 // `>`, so exclude it explicitly — it is its own term kind, not an IRI.
65 t.starts_with('<') && !t.starts_with("<<") && t.ends_with('>')
66}
67
68/// Is `t` a **quoted triple** term (`<< s p o >>`, RDF-star)? These appear only
69/// in subject/object position and are stored in the dictionary as their
70/// canonical N-Triples-star surface, exactly like any other term.
71#[inline]
72pub fn is_quoted_triple(t: &TermToken) -> bool {
73 t.starts_with("<<") && t.ends_with(">>")
74}
75
76/// The content of an IRI term without its angle brackets (`<http://x>` →
77/// `http://x`), or `None` if `t` is not an IRI term.
78#[inline]
79pub fn iri_content(t: &TermToken) -> Option<&str> {
80 if is_quoted_triple(t) {
81 return None;
82 }
83 t.strip_prefix('<').and_then(|s| s.strip_suffix('>'))
84}
85
86/// Is `t` a blank-node term (`_:…`)?
87#[inline]
88pub fn is_blank(t: &TermToken) -> bool {
89 t.starts_with("_:")
90}
91
92/// Is `t` a literal term (`"…"`)?
93#[inline]
94pub fn is_literal(t: &TermToken) -> bool {
95 t.starts_with('"')
96}
97
98/// Index of the closing quote of a literal term, scanning from the opening
99/// quote and honoring `\"` escapes. `t` must start with `"`.
100fn closing_quote(t: &TermToken) -> usize {
101 let bytes = t.as_bytes();
102 let mut i = 1;
103 while i < bytes.len() {
104 match bytes[i] {
105 b'\\' => i += 2,
106 b'"' => break,
107 _ => i += 1,
108 }
109 }
110 i.min(t.len())
111}
112
113/// The **lexical value** of a literal term — the text between the quotes with
114/// N-Triples escapes resolved — or `None` for IRIs and blank nodes. The
115/// datatype and language suffix are dropped (`"42"^^<…int>` → `42`,
116/// `"hi"@en` → `hi`).
117pub fn literal_lexical(token: &TermToken) -> Option<String> {
118 if !is_literal(token) {
119 return None;
120 }
121 Some(unescape_literal(&token[1..closing_quote(token)]))
122}
123
124/// The **lexical value** of any term: a literal's unescaped body, an IRI's
125/// content, or a blank-node token unchanged. Always succeeds. Useful where a
126/// plain comparable string is wanted regardless of term kind.
127pub fn lexical(token: &TermToken) -> Cow<'_, str> {
128 if is_literal(token) {
129 Cow::Owned(unescape_literal(&token[1..closing_quote(token)]))
130 } else if let Some(iri) = iri_content(token) {
131 Cow::Borrowed(iri)
132 } else {
133 Cow::Borrowed(token)
134 }
135}
136
137/// The part of a literal term after its closing quote (`"x"^^<dt>` → `^^<dt>`,
138/// `"x"@en` → `@en`, `"x"` → ``), or `None` if `token` is not a literal.
139fn literal_suffix(token: &TermToken) -> Option<&str> {
140 if !is_literal(token) {
141 return None;
142 }
143 token.get(closing_quote(token) + 1..)
144}
145
146/// The datatype IRI **content** of a literal term (no angle brackets): the
147/// explicit `^^<dt>`, else `rdf:langString` for a language-tagged literal,
148/// else `xsd:string` for a plain one. `None` for a non-literal or a malformed
149/// suffix.
150pub fn literal_datatype(token: &TermToken) -> Option<String> {
151 let suffix = literal_suffix(token)?;
152 if let Some(dt) = suffix.strip_prefix("^^<").and_then(|s| s.strip_suffix('>')) {
153 Some(dt.to_string())
154 } else if suffix.starts_with('@') {
155 // RDF 1.2: a language string WITH a base direction (`@lang--dir`) is an
156 // `rdf:dirLangString`; a plain `@lang` is `rdf:langString`. `--` never
157 // occurs in a well-formed BCP-47 tag, so this is unambiguous.
158 if suffix.contains("--") {
159 Some(RDF_DIR_LANG_STRING.to_string())
160 } else {
161 Some(RDF_LANG_STRING.to_string())
162 }
163 } else if suffix.is_empty() {
164 Some(XSD_STRING.to_string())
165 } else {
166 None
167 }
168}
169
170/// The language tag of a literal term (`"hi"@en` → `en`), `""` when the literal
171/// is untagged, or `None` for a non-literal. For an RDF 1.2 directional string
172/// (`"x"@ar--rtl`) this is the LANGUAGE only (`ar`) — the base direction is
173/// separate (see [`lang_dir`]), matching SPARQL 1.2 `LANG`.
174pub fn lang_tag(token: &TermToken) -> Option<String> {
175 literal_suffix(token).map(|s| {
176 s.strip_prefix('@')
177 .unwrap_or("")
178 .split("--")
179 .next()
180 .unwrap_or("")
181 .to_string()
182 })
183}
184
185/// The base direction of an RDF 1.2 directional language string (`"x"@ar--rtl` →
186/// `rtl`), or `None` if the literal has no direction (or is not a literal).
187pub fn lang_dir(token: &TermToken) -> Option<String> {
188 let tag = literal_suffix(token)?.strip_prefix('@')?;
189 tag.split("--").nth(1).map(str::to_string)
190}
191
192/// Numeric value of a term: the lexical part of a literal parsed as `f64`
193/// (`"30"^^<…int>` → `30.0`) or a bare numeric token, else `None`.
194///
195/// The literal's lexical value is taken through the escape-aware
196/// [`literal_lexical`] — closing quote located correctly, body unescaped, any
197/// `^^<datatype>` suffix dropped — rather than scanning to the *first* `"`. A
198/// value containing an embedded escaped quote (`"1\"2"`) is therefore read
199/// whole and simply fails to parse, instead of being truncated at the escaped
200/// quote. A language-tagged literal (`"5"@en`) is never a numeric literal in
201/// SPARQL/RDF, so it yields `None`.
202pub fn as_number(token: &TermToken) -> Option<f64> {
203 if is_literal(token) {
204 if lang_tag(token).is_some_and(|tag| !tag.is_empty()) {
205 return None;
206 }
207 literal_lexical(token)?.parse::<f64>().ok()
208 } else {
209 token.parse::<f64>().ok()
210 }
211}
212
213/// Escape a string for use as the body of an N-Triples literal (`"…"`): the
214/// inverse of [`unescape_literal`] for the characters that must be escaped
215/// (`\`, `"`, newline, carriage return, tab). The common case (no special
216/// characters) returns the input untouched.
217pub fn escape_literal(s: &str) -> String {
218 if !s.contains(['\\', '"', '\n', '\r', '\t']) {
219 return s.to_string();
220 }
221 let mut out = String::with_capacity(s.len() + 2);
222 for c in s.chars() {
223 match c {
224 '\\' => out.push_str("\\\\"),
225 '"' => out.push_str("\\\""),
226 '\n' => out.push_str("\\n"),
227 '\r' => out.push_str("\\r"),
228 '\t' => out.push_str("\\t"),
229 _ => out.push(c),
230 }
231 }
232 out
233}
234
235/// Build a literal term token from a (raw, unescaped) lexical value, attaching
236/// an optional non-empty language tag (`@lang`) or datatype IRI content
237/// (`^^<dt>`). `lang` wins over `datatype` if both are given (a tagged literal
238/// is implicitly `rdf:langString`).
239pub fn make_literal(lexical: &str, lang: Option<&str>, datatype: Option<&str>) -> String {
240 let body = escape_literal(lexical);
241 match (lang.filter(|l| !l.is_empty()), datatype) {
242 (Some(l), _) => format!("\"{body}\"@{l}"),
243 (None, Some(dt)) => format!("\"{body}\"^^<{dt}>"),
244 (None, None) => format!("\"{body}\""),
245 }
246}
247
248/// Resolve the N-Triples escape sequences in a literal's body to actual chars
249/// (`\n`, `\t`, `\"`, `\\`, `\uXXXX`, `\UXXXXXXXX`, …). Strings without a
250/// backslash — the overwhelming majority — are returned untouched.
251pub fn unescape_literal(s: &str) -> String {
252 if !s.contains('\\') {
253 return s.to_string();
254 }
255 let mut out = String::with_capacity(s.len());
256 let mut chars = s.chars();
257 while let Some(c) = chars.next() {
258 if c != '\\' {
259 out.push(c);
260 continue;
261 }
262 let unicode = |chars: &mut std::str::Chars, n: usize, out: &mut String| {
263 let hex: String = chars.take(n).collect();
264 match u32::from_str_radix(&hex, 16).ok().and_then(char::from_u32) {
265 Some(ch) => out.push(ch),
266 None => out.push('\u{FFFD}'),
267 }
268 };
269 match chars.next() {
270 Some('t') => out.push('\t'),
271 Some('b') => out.push('\u{08}'),
272 Some('n') => out.push('\n'),
273 Some('r') => out.push('\r'),
274 Some('f') => out.push('\u{0C}'),
275 Some('"') => out.push('"'),
276 Some('\'') => out.push('\''),
277 Some('\\') => out.push('\\'),
278 Some('u') => unicode(&mut chars, 4, &mut out),
279 Some('U') => unicode(&mut chars, 8, &mut out),
280 Some(other) => {
281 out.push('\\');
282 out.push(other);
283 }
284 None => out.push('\\'),
285 }
286 }
287 out
288}
289
290/// Rewrite a term token into the **RDF 1.2 triple-term surface** for a text
291/// serializer.
292///
293/// rete stores a quoted triple in one canonical token, the RDF-star surface
294/// `<<s p o>>` — see `ingest::take_term`, which accepts both surfaces
295/// on ingest and canonicalises them to that one. Current RDF 1.2 parsers
296/// (oxttl 0.2 and anything built on it, including the `oxigraph` CLI) do not
297/// read that surface: in N-Triples/N-Quads they **reject** it outright, and in
298/// Turtle/TriG they read `<< s p o >>` as a *reifier* — one statement silently
299/// becomes two, with a blank node where the triple term was. So a dump in the
300/// stored surface is not interoperable, quietly in one format and loudly in the
301/// other. This is the translation that makes it so, at write time only: nothing
302/// about the file changes.
303///
304/// Returns:
305///
306/// * `Some(Borrowed(token))` when `token` is not a quoted triple. This is the
307/// overwhelming majority of terms and the only cost is a two-byte prefix
308/// check, so a dump with no quoted triples in it is byte-for-byte unchanged.
309/// * `Some(Owned(…))` with the token rewritten to `<<( s p o )>>`, recursively:
310/// a triple term nested in the object slot is rewritten too.
311/// * `None` when the token has **no RDF 1.2 spelling at all**. RDF 1.2 puts a
312/// triple term in *object position only* — the grammar is
313/// `tripleTerm ::= '<<(' ttSubject predicate ttObject ')>>'` with
314/// `ttSubject ::= iri | BlankNode` — so a quoted triple standing in the
315/// subject slot of another quoted triple cannot be written. (The caller is
316/// responsible for the same rule at statement level: a quoted triple in the
317/// *statement's* subject or predicate slot is equally unwritable, and the
318/// caller is the one that knows which slot a term came from.)
319///
320/// A token that is not well-formed — `<<` … `>>` that does not parse as three
321/// terms — also yields `None` rather than a mangled rewrite.
322pub fn rdf12_triple_term(token: &TermToken) -> Option<Cow<'_, TermToken>> {
323 if !is_quoted_triple(token) {
324 return Some(Cow::Borrowed(token));
325 }
326 rewrite_rdf12(token).map(Cow::Owned)
327}
328
329/// The owned half of [`rdf12_triple_term`], split out so the recursion does not
330/// re-run the `is_quoted_triple` fast path on a token it already classified.
331fn rewrite_rdf12(token: &TermToken) -> Option<String> {
332 let (s, p, o) = crate::ingest::quoted_triple_parts(token)?;
333 // `ttSubject ::= iri | BlankNode` and `predicate ::= iri`: neither slot
334 // admits a triple term, at any depth.
335 if is_quoted_triple(&s) || is_quoted_triple(&p) {
336 return None;
337 }
338 // `ttObject` does admit one, which is where nesting lives.
339 let o = if is_quoted_triple(&o) {
340 rewrite_rdf12(&o)?
341 } else {
342 o
343 };
344 Some(format!("<<( {s} {p} {o} )>>"))
345}
346
347#[cfg(test)]
348mod tests {
349 use super::*;
350
351 // --- the RDF 1.2 writer surface ----------------------------------------
352
353 #[test]
354 fn a_plain_term_is_borrowed_unchanged() {
355 // The hot path. Every term that is not a quoted triple comes back
356 // borrowed, which is what makes a dump of a quoted-triple-free file
357 // byte-for-byte what it was.
358 for t in [
359 "<http://example.org/x>",
360 "_:b0",
361 "\"lit\"@en",
362 "\"5\"^^<http://www.w3.org/2001/XMLSchema#integer>",
363 "\"a > b\"",
364 ] {
365 match rdf12_triple_term(t) {
366 Some(Cow::Borrowed(got)) => assert_eq!(got, t),
367 other => panic!("{t} should borrow unchanged, got {other:?}"),
368 }
369 }
370 }
371
372 #[test]
373 fn an_object_triple_term_gets_the_rdf12_surface() {
374 assert_eq!(
375 rdf12_triple_term("<<<http://ex/s> <http://ex/p> <http://ex/o>>>").unwrap(),
376 "<<( <http://ex/s> <http://ex/p> <http://ex/o> )>>"
377 );
378 }
379
380 #[test]
381 fn nesting_in_the_object_slot_recurses() {
382 // `ttObject` admits another triple term, so depth works — and the
383 // recursion has to rewrite the inner one too, not just the outer.
384 assert_eq!(
385 rdf12_triple_term(
386 "<<<http://ex/a> <http://ex/b> <<<http://ex/x> <http://ex/y> <http://ex/z>>>>>"
387 )
388 .unwrap(),
389 "<<( <http://ex/a> <http://ex/b> <<( <http://ex/x> <http://ex/y> <http://ex/z> )>> )>>"
390 );
391 }
392
393 #[test]
394 fn a_literal_object_survives_verbatim() {
395 // Term boundaries come from `take_term`, not from splitting on spaces,
396 // so a literal carrying spaces, a `>` and a `<<` does not derail it.
397 assert_eq!(
398 rdf12_triple_term("<<_:b1 <http://ex/p> \"a > b << c\"@en>>").unwrap(),
399 "<<( _:b1 <http://ex/p> \"a > b << c\"@en )>>"
400 );
401 }
402
403 #[test]
404 fn a_triple_term_in_a_subject_slot_has_no_rdf12_spelling() {
405 // RDF 1.2: `ttSubject ::= iri | BlankNode`. A quoted triple nested in
406 // another one's subject cannot be written, at any depth, and the honest
407 // answer is `None` rather than a token no parser accepts.
408 assert!(rdf12_triple_term(
409 "<<<<<http://ex/x> <http://ex/y> <http://ex/z>>> <http://ex/p> <http://ex/o>>>"
410 )
411 .is_none());
412 // …including one level down.
413 assert!(rdf12_triple_term(
414 "<<<http://ex/a> <http://ex/b> <<<<<http://ex/x> <http://ex/y> <http://ex/z>>> <http://ex/p> <http://ex/o>>>>>"
415 )
416 .is_none());
417 }
418
419 #[test]
420 fn a_malformed_quoted_triple_is_refused_not_mangled() {
421 assert!(rdf12_triple_term("<<<http://ex/s> <http://ex/p>>>").is_none());
422 assert!(rdf12_triple_term("<<>>").is_none());
423 }
424
425 #[test]
426 fn the_rewrite_is_what_ingest_accepts_back() {
427 // The round-trip property, at the term level: what the writer emits is
428 // what `take_term` canonicalises back to the stored token. This is why
429 // rete -> nq -> rete is safe in either surface.
430 let stored =
431 "<<<http://ex/a> <http://ex/b> <<<http://ex/x> <http://ex/y> <http://ex/z>>>>>";
432 let written = rdf12_triple_term(stored).unwrap().into_owned();
433 let (back, rest) = crate::ingest::take_term(&written).unwrap();
434 assert_eq!(back, stored);
435 assert!(rest.trim().is_empty());
436 }
437
438 #[test]
439 fn term_kinds() {
440 assert!(is_iri("<http://example.org/x>"));
441 assert!(!is_iri("\"x\""));
442 assert!(!is_iri("_:b0"));
443 assert!(is_blank("_:b0"));
444 assert!(is_literal("\"x\"@en"));
445 assert_eq!(iri_content("<http://x>"), Some("http://x"));
446 assert_eq!(iri_content("\"x\""), None);
447 }
448
449 #[test]
450 fn lexical_values() {
451 assert_eq!(literal_lexical("\"hello\""), Some("hello".to_string()));
452 assert_eq!(literal_lexical("\"42\"^^<int>"), Some("42".to_string()));
453 assert_eq!(literal_lexical("\"hi\"@en"), Some("hi".to_string()));
454 assert_eq!(literal_lexical("<http://x>"), None);
455 // any-term lexical
456 assert_eq!(lexical("\"hi\"@en"), "hi");
457 assert_eq!(lexical("<http://x>"), "http://x");
458 assert_eq!(lexical("_:b0"), "_:b0");
459 }
460
461 #[test]
462 fn datatype_and_lang() {
463 assert_eq!(literal_datatype("\"42\"^^<int>").as_deref(), Some("int"));
464 assert_eq!(
465 literal_datatype("\"hi\"@en").as_deref(),
466 Some(RDF_LANG_STRING)
467 );
468 assert_eq!(literal_datatype("\"plain\"").as_deref(), Some(XSD_STRING));
469 assert_eq!(literal_datatype("<http://x>"), None);
470 assert_eq!(lang_tag("\"hi\"@en").as_deref(), Some("en"));
471 assert_eq!(lang_tag("\"plain\"").as_deref(), Some(""));
472 assert_eq!(lang_tag("<http://x>"), None);
473 }
474
475 #[test]
476 fn numbers() {
477 assert_eq!(as_number("\"30\"^^<int>"), Some(30.0));
478 assert_eq!(as_number("3.5"), Some(3.5));
479 assert_eq!(as_number("\"nope\""), None);
480 assert_eq!(as_number("<http://x>"), None);
481 }
482
483 #[test]
484 fn as_number_escape_aware() {
485 // Plain and typed numeric literals parse as before.
486 assert_eq!(
487 as_number("\"42\"^^<http://www.w3.org/2001/XMLSchema#integer>"),
488 Some(42.0)
489 );
490 assert_eq!(
491 as_number("\"12.5\"^^<http://www.w3.org/2001/XMLSchema#decimal>"),
492 Some(12.5)
493 );
494 assert_eq!(
495 as_number("\"6.022e23\"^^<http://www.w3.org/2001/XMLSchema#double>"),
496 Some(6.022e23)
497 );
498 // Plain literal, negative, and leading-`+`.
499 assert_eq!(as_number("\"5\""), Some(5.0));
500 assert_eq!(as_number("\"-5\"^^<int>"), Some(-5.0));
501 assert_eq!(as_number("\"+7\""), Some(7.0));
502 // Non-numeric literal → None.
503 assert_eq!(as_number("\"not a number\""), None);
504 // A value with an EMBEDDED escaped quote must be read whole (`1"2`),
505 // fail to parse, and never be truncated to `1` (or panic). This is the
506 // escape-aware path: the old first-`"` scan would have stopped early.
507 assert_eq!(as_number("\"1\\\"2\""), None);
508 // IRI and blank node → None.
509 assert_eq!(as_number("<http://example.org/n>"), None);
510 assert_eq!(as_number("_:b0"), None);
511 // A language-tagged literal is never numeric (SPARQL): `"5"@en` → None.
512 assert_eq!(as_number("\"5\"@en"), None);
513 }
514
515 #[test]
516 fn escapes() {
517 assert_eq!(unescape_literal("plain"), "plain");
518 assert_eq!(unescape_literal("a\\nb"), "a\nb");
519 assert_eq!(unescape_literal("a\\\"b"), "a\"b");
520 assert_eq!(unescape_literal("\\u0041"), "A");
521 // an escaped quote inside the body is honored by the closing-quote scan
522 assert_eq!(literal_lexical("\"a\\\"b\"@en"), Some("a\"b".to_string()));
523 }
524}