rete_core/ingest.rs
1//! Ingestion: parse RDF text (N-Triples / N-Quads / Turtle) and assemble a
2//! complete `.rete` file image. Shared by the CLI's `build`/`validate` commands
3//! and the wasm bindings (the playground's in-browser builder).
4//!
5//! The N-Triples/N-Quads reader is line-based and keeps terms as their exact
6//! canonical token strings (`<iri>`, `_:bnode`, `"lit"`, `"lit"^^<dt>`,
7//! `"lit"@lang`) so they double as dictionary keys and so a `query` can match
8//! by the same string. This is not a full RDF 1.1 parser — it covers the
9//! canonical N-Triples surface (and is deliberately tolerant of IRIs a strict
10//! parser would reject), enough for v0 ingestion. Turtle goes through `oxttl`.
11//!
12//! That tolerance is *measured*, not silent: pass an [`IriAudit`] to any of the
13//! `*_audited` entry points and the parse counts every IRI outside the
14//! N-Triples `IRIREF` grammar / RFC 3987 (see [`crate::iri`]) — or, with
15//! `strict`, refuses the input on the first one. `rete build` always audits.
16//!
17//! Assembly degrades with the build features: without the `compression`
18//! feature (e.g. on wasm) sections are written with the `NONE` codec — larger
19//! files, byte-compatible readers.
20
21use crate::{
22 build_pyramid_meta_algo, Dictionary, DictionaryBuilder, GraphIndexBuilder, PyramidAlgo,
23 DEFAULT_TILE_BUDGET,
24};
25
26/// A parsed triple as three canonical term tokens.
27pub type RawTriple = (String, String, String);
28
29/// A parsed quad: the triple plus an optional graph term (`None` = default graph).
30pub type RawQuad = (String, String, String, Option<String>);
31
32/// Which **surface** a `<< … >>` in Turtle/TriG input is read as — the input
33/// half of `rete export --quoted-triple-syntax`, same vocabulary, same two
34/// values.
35///
36/// The flag exists because the two standards give one piece of syntax two
37/// meanings, and no amount of looking at the bytes can tell them apart:
38///
39/// | written | under [`RdfStar`](Self::RdfStar) | under [`Rdf12`](Self::Rdf12) |
40/// |--------------------|----------------------------------|---------------------------------------|
41/// | `<< s p o >>` | a **quoted triple**, one term | a **reifier**: `_:r rdf:reifies <<( s p o )>>`, plus a blank node standing where it was |
42/// | `<<( s p o )>>` | a syntax error | a **triple term**, object position only |
43/// | `{\| … \|}` | a syntax error | an annotation on the statement before it |
44///
45/// One file, two readings, two different graphs — see the round-trip matrix in
46/// `crates/rete-cli/tests/quoted_triple_surfaces.rs`, which asserts both.
47///
48/// `<<( s p o )>>` is unambiguous: it is RDF 1.2's and nothing else's. That
49/// asymmetry is why [`RdfStar`](Self::RdfStar) is the **default**. Reading an
50/// RDF-star file as RDF 1.2 silently yields a different graph; reading an
51/// RDF 1.2 file as RDF-star is a hard parse error that names the flag. Only one
52/// of the two mistakes is survivable, so the default is the one that makes the
53/// other mistake loud.
54///
55/// **N-Triples and N-Quads ignore this entirely.** Their reader is rete's own
56/// (`take_term`), it has accepted both `<< s p o >>` and `<<( s p o )>>` since
57/// #262, and RDF 1.2 N-Triples has no reifier syntax for `<< … >>` to be — so
58/// there is no ambiguity to resolve and nothing to choose. The same goes for
59/// RDF/XML.
60///
61/// Whichever surface reads the file, what gets **stored** is the same canonical
62/// token `<<s p o>>`. rete's term model is a superset of both: a triple term is
63/// a term, and RDF 1.2 reification is ordinary RDF — a blank node, a predicate
64/// and a term — which rete already stored before this flag existed.
65#[derive(Clone, Copy, PartialEq, Eq, Debug, Default, Hash)]
66pub enum QuotedTripleSurface {
67 /// `<< s p o >>` is a **quoted triple** (RDF-star), in subject or object
68 /// position — rete's own storage token. The default, and byte-for-byte the
69 /// behaviour every rete release before this flag had.
70 #[default]
71 RdfStar,
72 /// `<<( s p o )>>` is a **triple term** and `<< s p o >>` is a **reifier**
73 /// (RDF 1.2), with `{| … |}` annotations read too.
74 Rdf12,
75}
76
77impl QuotedTripleSurface {
78 /// Parse the `--quoted-triple-syntax` value, `None` for anything else.
79 /// Shared by `rete build`, `rete validate` and `rete export` so the flag
80 /// cannot come to mean two things in two directions.
81 pub fn parse(s: &str) -> Option<Self> {
82 match s {
83 crate::card_derive::QUOTED_TRIPLE_SURFACE_RDF12 => Some(Self::Rdf12),
84 crate::card_derive::QUOTED_TRIPLE_SURFACE_RDF_STAR => Some(Self::RdfStar),
85 _ => None,
86 }
87 }
88
89 /// The flag spelling of this surface — the inverse of [`Self::parse`].
90 pub const fn as_str(self) -> &'static str {
91 match self {
92 Self::Rdf12 => crate::card_derive::QUOTED_TRIPLE_SURFACE_RDF12,
93 Self::RdfStar => crate::card_derive::QUOTED_TRIPLE_SURFACE_RDF_STAR,
94 }
95 }
96}
97
98/// The one-line hint appended to a Turtle/TriG parse error when the input looks
99/// like RDF 1.2 and was read as RDF-star — i.e. the exact mistake
100/// [`QuotedTripleSurface`]'s default makes, made loud.
101const RDF12_SURFACE_HINT: &str = "\nhint: this input contains `<<(`, the RDF 1.2 triple-term \
102 syntax — which is what `rete export` writes by default. Reading it takes \
103 `--quoted-triple-syntax rdf12`; the default, `rdf-star`, reads `<< s p o >>` as a quoted \
104 triple instead.";
105
106/// Does `text` look like it carries RDF 1.2 triple terms? A deliberately dumb
107/// substring test: `<<(` cannot begin anything else in Turtle, and this is only
108/// ever consulted to *explain a parse that already failed*, never to decide how
109/// to read a file. A `<<(` inside a string literal would make it a false
110/// positive on some other error's hint, which costs one confusing sentence and
111/// nothing else.
112fn looks_like_rdf12(text: &str) -> bool {
113 text.contains("<<(")
114}
115
116/// Why an ingest failed. Parser errors from `oxttl`/`oxrdfxml` and I/O errors
117/// are flattened to their `Display` string rather than wrapped: the variants
118/// then carry no foreign types, which keeps the error usable from the wasm
119/// bindings and stable across upstream parser upgrades.
120#[derive(Debug, thiserror::Error)]
121#[non_exhaustive]
122pub enum IngestError {
123 /// A malformed N-Triples/N-Quads line: its 1-based line number and a short
124 /// reason (`"bad subject"`, `"missing trailing '.'"`, …).
125 #[error("line {0}: {1}")]
126 Line(usize, &'static str),
127 /// The Turtle parser rejected the input.
128 #[error("turtle: {0}")]
129 Turtle(String),
130 /// The TriG parser rejected the input.
131 #[error("trig: {0}")]
132 TriG(String),
133 /// The RDF/XML parser rejected the input.
134 #[error("rdf/xml: {0}")]
135 RdfXml(String),
136 /// The requested format is not one this crate can parse.
137 #[error("unknown input format: {0} (expected nt, nq, ttl, trig, or rdf/xml)")]
138 UnknownFormat(String),
139 /// [`QuotedTripleSurface::Rdf12`] was asked for on a Turtle/TriG input, but
140 /// this build of `rete-core` has the `rdf12-turtle` feature off, so the
141 /// RDF 1.2 reader was not compiled in. Refused by name rather than read as
142 /// RDF-star, which would be the wrong graph told in silence.
143 #[error(
144 "this build cannot read the RDF 1.2 Turtle/TriG surface: rete-core was compiled without \
145 the `rdf12-turtle` feature (N-Triples/N-Quads still accept `<<( s p o )>>`)"
146 )]
147 Rdf12TurtleUnavailable,
148 /// The input could not be read (streaming builds surface the path here).
149 #[error("io: {0}")]
150 Io(String),
151 /// An IRI outside the N-Triples/N-Quads `IRIREF` grammar / RFC 3987, refused
152 /// because the caller asked for a strict [`IriAudit`] (`rete build
153 /// --strict`). Without `strict` the same finding is only counted.
154 /// `location` is `line N` for the line-based syntaxes, the format name for
155 /// the others (oxttl reports its own positions).
156 #[error("{location}: invalid IRI {token} — {reason}")]
157 InvalidIri {
158 /// Where it was seen: `line 42`, or the format name.
159 location: String,
160 /// The offending term token, verbatim.
161 token: String,
162 /// Which rule it breaks ([`crate::iri::IriDefect::reason`]).
163 reason: &'static str,
164 },
165}
166
167/// Policy **and** tally for the invalid-IRI check that runs during ingest.
168///
169/// rete's reader is deliberately tolerant of IRIs a strict parser rejects (see
170/// [`crate::iri`]), which is what lets it hold every published dataset — and
171/// also what let three of the `scholar/` exports produce N-Quads Oxigraph
172/// refuses to load. Passing an audit makes the tolerance *visible*: the build
173/// still succeeds and stores exactly what it was given, but it can now say how
174/// much of what it stored is not valid RDF. `strict` turns the same finding into
175/// a refusal, on the first offending statement.
176#[derive(Debug, Default)]
177pub struct IriAudit {
178 /// Fail the parse on the first invalid IRI instead of counting it.
179 pub strict: bool,
180 /// What the parse saw. Constant memory — see [`crate::iri::IriReport`].
181 pub report: crate::iri::IriReport,
182}
183
184impl IriAudit {
185 /// Count invalid IRIs; never fail (`rete build`'s default).
186 pub fn counting() -> Self {
187 Self::default()
188 }
189
190 /// Refuse the input at the first invalid IRI (`rete build --strict`).
191 pub fn refusing() -> Self {
192 Self {
193 strict: true,
194 report: crate::iri::IriReport::default(),
195 }
196 }
197}
198
199/// Run one statement past the audit: record any invalid IRI, and in `strict`
200/// mode turn it into an error naming the offending token. `location` is only
201/// called on the error path, so the common case allocates nothing.
202fn audit_quad<F: FnOnce() -> String>(
203 audit: Option<&mut IriAudit>,
204 s: &str,
205 p: &str,
206 o: &str,
207 g: Option<&str>,
208 location: F,
209) -> Result<(), IngestError> {
210 let Some(a) = audit else { return Ok(()) };
211 let Some(defect) = a.report.observe_quad(s, p, o, g) else {
212 return Ok(());
213 };
214 if !a.strict {
215 return Ok(());
216 }
217 let token = [Some(s), Some(p), Some(o), g]
218 .into_iter()
219 .flatten()
220 .find(|t| crate::iri::term_defect(t).is_some())
221 .unwrap_or("<?>")
222 .to_string();
223 Err(IngestError::InvalidIri {
224 location: location(),
225 token,
226 reason: defect.reason(),
227 })
228}
229
230/// Estimate the statement count of an N-Triples/N-Quads text from its newline
231/// count, so the output `Vec` can be pre-sized — a big build otherwise pays
232/// repeated `Vec` doublings, each briefly holding ~2× the (large) spine. An
233/// over-estimate by a few blank/comment lines is harmless (it's a capacity hint).
234fn estimate_statements(input: &str) -> usize {
235 bytecount_newlines(input).max(1)
236}
237
238/// Count `\n` bytes — one linear pass, far cheaper than the parse it sizes.
239fn bytecount_newlines(input: &str) -> usize {
240 input.as_bytes().iter().filter(|&&b| b == b'\n').count()
241}
242
243/// Parse one N-Quads line into a quad, or `None` for a blank/comment line.
244/// Shared by the whole-text [`parse_quads`] and the streaming [`parse_reader`].
245fn parse_nq_line(
246 raw: &str,
247 lineno: usize,
248 audit: Option<&mut IriAudit>,
249) -> Result<Option<RawQuad>, IngestError> {
250 let line = raw.trim();
251 if line.is_empty() || line.starts_with('#') {
252 return Ok(None);
253 }
254 let stripped = line
255 .strip_suffix('.')
256 .ok_or(IngestError::Line(lineno, "missing trailing '.'"))?
257 .trim_end();
258 let (s, rest) = take_term(stripped).ok_or(IngestError::Line(lineno, "bad subject"))?;
259 let (p, rest) =
260 take_term(rest.trim_start()).ok_or(IngestError::Line(lineno, "bad predicate"))?;
261 let (o, rest) = take_term(rest.trim_start()).ok_or(IngestError::Line(lineno, "bad object"))?;
262 let rest = rest.trim();
263 let graph = if rest.is_empty() {
264 None
265 } else {
266 let (g, tail) = take_term(rest).ok_or(IngestError::Line(lineno, "bad graph"))?;
267 if !tail.trim().is_empty() {
268 return Err(IngestError::Line(lineno, "trailing content after graph"));
269 }
270 Some(g)
271 };
272 audit_quad(audit, &s, &p, &o, graph.as_deref(), || {
273 format!("line {lineno}")
274 })?;
275 Ok(Some((s, p, o, graph)))
276}
277
278/// Parse one N-Triples line into a triple, or `None` for a blank/comment line.
279fn parse_nt_line(
280 raw: &str,
281 lineno: usize,
282 audit: Option<&mut IriAudit>,
283) -> Result<Option<RawTriple>, IngestError> {
284 let line = raw.trim();
285 if line.is_empty() || line.starts_with('#') {
286 return Ok(None);
287 }
288 let stripped = line
289 .strip_suffix('.')
290 .ok_or(IngestError::Line(lineno, "missing trailing '.'"))?
291 .trim_end();
292 let (s, rest) = take_term(stripped).ok_or(IngestError::Line(lineno, "bad subject"))?;
293 let (p, rest) =
294 take_term(rest.trim_start()).ok_or(IngestError::Line(lineno, "bad predicate"))?;
295 let (o, rest) = take_term(rest.trim_start()).ok_or(IngestError::Line(lineno, "bad object"))?;
296 if !rest.trim().is_empty() {
297 return Err(IngestError::Line(lineno, "trailing content after object"));
298 }
299 audit_quad(audit, &s, &p, &o, None, || format!("line {lineno}"))?;
300 Ok(Some((s, p, o)))
301}
302
303/// Parse N-Quads text: `subject predicate object [graph] .` per line.
304pub fn parse_quads(input: &str) -> Result<Vec<RawQuad>, IngestError> {
305 parse_quads_audited(input, None)
306}
307
308/// [`parse_quads`] with an optional [`IriAudit`].
309pub fn parse_quads_audited(
310 input: &str,
311 mut audit: Option<&mut IriAudit>,
312) -> Result<Vec<RawQuad>, IngestError> {
313 let mut out = Vec::with_capacity(estimate_statements(input));
314 for (i, raw) in input.lines().enumerate() {
315 if let Some(q) = parse_nq_line(raw, i + 1, audit.as_deref_mut())? {
316 out.push(q);
317 }
318 }
319 Ok(out)
320}
321
322/// Parse N-Triples text into raw term-token triples, skipping blank/comment lines.
323pub fn parse(input: &str) -> Result<Vec<RawTriple>, IngestError> {
324 parse_audited(input, None)
325}
326
327/// [`parse`] with an optional [`IriAudit`].
328pub fn parse_audited(
329 input: &str,
330 mut audit: Option<&mut IriAudit>,
331) -> Result<Vec<RawTriple>, IngestError> {
332 let mut out = Vec::with_capacity(estimate_statements(input));
333 for (i, raw) in input.lines().enumerate() {
334 if let Some(t) = parse_nt_line(raw, i + 1, audit.as_deref_mut())? {
335 out.push(t);
336 }
337 }
338 Ok(out)
339}
340
341/// **Stream-parse** N-Triples (`"nt"`) or N-Quads (`"nq"`) from a reader, one
342/// line at a time, so the whole input text is **never resident** — the big-build
343/// memory win over reading the file into a `String` first (each line String is
344/// transient, freed every iteration). `cap` pre-sizes the output `Vec` (e.g.
345/// `file_len / 64`) to avoid reallocation doublings of the large spine. Turtle is
346/// not streamable here (oxttl needs the whole input); callers use the text path
347/// for `"ttl"`.
348pub fn parse_reader<R: std::io::BufRead>(
349 reader: R,
350 format: &str,
351 cap: usize,
352) -> Result<Vec<RawQuad>, IngestError> {
353 parse_reader_audited(reader, format, cap, None)
354}
355
356/// [`parse_reader`] with an optional [`IriAudit`].
357pub fn parse_reader_audited<R: std::io::BufRead>(
358 reader: R,
359 format: &str,
360 cap: usize,
361 audit: Option<&mut IriAudit>,
362) -> Result<Vec<RawQuad>, IngestError> {
363 parse_reader_audited_surface(reader, format, cap, audit, QuotedTripleSurface::default())
364}
365
366/// [`parse_reader_audited`] reading `<< … >>` in the given
367/// [`QuotedTripleSurface`] (Turtle/TriG only — see the enum).
368pub fn parse_reader_audited_surface<R: std::io::BufRead>(
369 reader: R,
370 format: &str,
371 cap: usize,
372 audit: Option<&mut IriAudit>,
373 surface: QuotedTripleSurface,
374) -> Result<Vec<RawQuad>, IngestError> {
375 let mut out = Vec::with_capacity(cap);
376 stream_reader_audited_surface(reader, format, audit, surface, &mut |q| out.push(q))?;
377 Ok(out)
378}
379
380/// **Stream** N-Triples (`"nt"`) / N-Quads (`"nq"`) from a reader, invoking `f`
381/// once per parsed quad — like [`parse_reader`] but **without collecting** into a
382/// `Vec`. Each quad's term Strings are owned by `f` (and dropped when it returns
383/// if it doesn't retain them), so a caller that only needs to *observe* every
384/// term — e.g. the two-pass [`assemble_dataset_streaming`] building its
385/// dictionary — never materializes the whole quad multiset. The big-graph,
386/// low-RAM ingest primitive. Blank/comment lines are skipped; a parse error stops
387/// the stream and is returned.
388///
389/// Every format streams: the line-based ones a line at a time, Turtle / TriG /
390/// RDF-XML through their `oxttl`/`oxrdfxml` reader parsers, which pull from the
391/// reader incrementally. Nothing here holds the input text — so a 60 GB Turtle
392/// file is as ingestible as a 60 GB `.nt` one, which is what lets the external
393/// (memory-bounded) build accept them.
394pub fn stream_reader<R: std::io::BufRead>(
395 reader: R,
396 format: &str,
397 f: &mut dyn FnMut(RawQuad),
398) -> Result<(), IngestError> {
399 stream_reader_audited(reader, format, None, f)
400}
401
402/// [`stream_reader`] with an optional [`IriAudit`].
403///
404/// Every syntax is audited, not only the line-based ones. `oxttl`/`oxrdfxml`
405/// validate IRIs themselves and reject an invalid one before it reaches us, so
406/// in practice the Turtle/TriG/RDF-XML tallies stay at zero — but auditing them
407/// too means the count is a property of the *file*, not of which reader happened
408/// to produce it, and a future parser change cannot silently reopen the hole.
409pub fn stream_reader_audited<R: std::io::BufRead>(
410 reader: R,
411 format: &str,
412 audit: Option<&mut IriAudit>,
413 f: &mut dyn FnMut(RawQuad),
414) -> Result<(), IngestError> {
415 stream_reader_audited_surface(reader, format, audit, QuotedTripleSurface::default(), f)
416}
417
418/// [`stream_reader_audited`] reading `<< … >>` in the given
419/// [`QuotedTripleSurface`].
420///
421/// The surface reaches only the `"ttl"` and `"trig"` arms: the line-based
422/// readers are rete's own and take both spellings unconditionally, and RDF/XML
423/// has no syntax for either. Under [`QuotedTripleSurface::Rdf12`] Turtle/TriG go
424/// through `oxttl` 0.2 instead of 0.1, and the RDF 1.2 triple terms it hands
425/// back (`<<( s p o )>>`) are folded to rete's stored token on the way out, so
426/// the dictionary a file lands in does not depend on which reader read it.
427///
428/// **This entry point does not carry the RDF 1.2 hint** that the text path adds
429/// to a failed RDF-star parse: it is handed a reader, not a string, and sniffing
430/// the input would mean buffering the 60 GB file this function exists to avoid
431/// buffering. `rete build --memory-budget-mb file.ttl` therefore reports the raw
432/// parser error; every other route reports the hint.
433pub fn stream_reader_audited_surface<R: std::io::BufRead>(
434 reader: R,
435 format: &str,
436 mut audit: Option<&mut IriAudit>,
437 surface: QuotedTripleSurface,
438 f: &mut dyn FnMut(RawQuad),
439) -> Result<(), IngestError> {
440 if surface == QuotedTripleSurface::Rdf12 && matches!(format, "ttl" | "trig") {
441 return rdf12::stream(reader, format, audit, f);
442 }
443 match format {
444 "nt" | "nq" => {
445 for (i, line) in reader.lines().enumerate() {
446 let line = line.map_err(|e| IngestError::Io(e.to_string()))?;
447 if format == "nq" {
448 if let Some(q) = parse_nq_line(&line, i + 1, audit.as_deref_mut())? {
449 f(q);
450 }
451 } else if let Some((s, p, o)) = parse_nt_line(&line, i + 1, audit.as_deref_mut())? {
452 f((s, p, o, None));
453 }
454 }
455 Ok(())
456 }
457 "ttl" => turtle_star(reader, audit, f),
458 "trig" => trig_star(reader, audit, f),
459 "rdfxml" => {
460 for r in oxrdfxml::RdfXmlParser::new().for_reader(reader) {
461 let t = r.map_err(|e| IngestError::RdfXml(e.to_string()))?;
462 let q = (
463 t.subject.to_string(),
464 t.predicate.to_string(),
465 t.object.to_string(),
466 None,
467 );
468 audit_quad(audit.as_deref_mut(), &q.0, &q.1, &q.2, None, || {
469 "rdf/xml".into()
470 })?;
471 f(q);
472 }
473 Ok(())
474 }
475 other => Err(IngestError::UnknownFormat(other.to_string())),
476 }
477}
478
479/// Stream **RDF-star** Turtle: `with_quoted_triples` accepts `<< s p o >>` in
480/// subject/object position, and oxrdf 0.2's `Term::Triple` Displays as the
481/// canonical `<<…>>` token rete's own N-Triples-star tokenizer emits — so no
482/// translation is needed on the way out.
483///
484/// Its own function, and called directly from both the reader path and the text
485/// path, so that the wasm engine — which reaches only the text path and has no
486/// RDF 1.2 reader compiled in — links this and nothing else.
487fn turtle_star<R: std::io::BufRead>(
488 reader: R,
489 mut audit: Option<&mut IriAudit>,
490 f: &mut dyn FnMut(RawQuad),
491) -> Result<(), IngestError> {
492 for r in oxttl::TurtleParser::new()
493 .with_quoted_triples()
494 .for_reader(reader)
495 {
496 let t = r.map_err(|e| IngestError::Turtle(e.to_string()))?;
497 let q = (
498 t.subject.to_string(),
499 t.predicate.to_string(),
500 t.object.to_string(),
501 None,
502 );
503 audit_quad(audit.as_deref_mut(), &q.0, &q.1, &q.2, None, || {
504 "turtle".into()
505 })?;
506 f(q);
507 }
508 Ok(())
509}
510
511/// [`turtle_star`] for TriG — Turtle plus named-graph blocks (`<g> { … }`), the
512/// shape most large RDF dumps ship in.
513fn trig_star<R: std::io::BufRead>(
514 reader: R,
515 mut audit: Option<&mut IriAudit>,
516 f: &mut dyn FnMut(RawQuad),
517) -> Result<(), IngestError> {
518 for r in oxttl::TriGParser::new()
519 .with_quoted_triples()
520 .for_reader(reader)
521 {
522 let q = r.map_err(|e| IngestError::TriG(e.to_string()))?;
523 let q = (
524 q.subject.to_string(),
525 q.predicate.to_string(),
526 q.object.to_string(),
527 graph_token(&q.graph_name),
528 );
529 audit_quad(
530 audit.as_deref_mut(),
531 &q.0,
532 &q.1,
533 &q.2,
534 q.3.as_deref(),
535 || "trig".into(),
536 )?;
537 f(q);
538 }
539 Ok(())
540}
541
542/// The canonical graph token for a parsed quad — `None` for the default graph,
543/// otherwise the same `<iri>` / `_:b` spelling the N-Quads reader produces, so
544/// TriG and N-Quads inputs land identical dictionary keys. `GraphName`'s own
545/// `Display` renders the default graph as `DEFAULT`, which must never become a
546/// term, hence the explicit match.
547fn graph_token(g: &oxrdf::GraphName) -> Option<String> {
548 match g {
549 oxrdf::GraphName::DefaultGraph => None,
550 other => Some(other.to_string()),
551 }
552}
553
554/// The **RDF 1.2** Turtle/TriG reader — `oxttl` 0.2 with its `rdf-12` feature,
555/// living beside the `oxttl` 0.1 reader above rather than replacing it.
556///
557/// Everything RDF 1.2 adds over RDF 1.1 arrives through this one module:
558/// `<<( s p o )>>` triple terms, `<< s p o >>` reifiers (which expand to
559/// `_:r rdf:reifies <<( s p o )>>` plus the statement that mentioned them),
560/// `{| … |}` annotations, and `"…"@lang--dir` directional literals. None of it
561/// needs a storage change, which is the point: a reifier is a blank node, an
562/// IRI and a term, and rete has stored those since v0.
563///
564/// The one translation is at the term boundary. `oxrdf` 0.3 renders a triple
565/// term as `<<( s p o )>>`; rete's dictionary key is `<<s p o>>`. [`take_term`]
566/// already reads both — that is what made RDF 1.2 N-Quads ingestible in #262 —
567/// so the fold is a re-scan of a token oxttl has already validated.
568#[cfg(feature = "rdf12-turtle")]
569mod rdf12 {
570 use super::{audit_quad, take_term, IngestError, IriAudit, RawQuad};
571
572 /// Fold `oxrdf` 0.3's `<<( s p o )>>` rendering of a triple term into rete's
573 /// stored `<<s p o>>`, recursively (a nested triple term folds with it).
574 /// Every other token — the overwhelming majority, and all of them in a file
575 /// with no triple terms — is returned untouched, not even copied.
576 pub(super) fn fold_triple_term(t: String) -> String {
577 if !t.starts_with("<<(") {
578 return t;
579 }
580 match take_term(&t) {
581 Some((token, rest)) if rest.trim().is_empty() => token,
582 // oxttl only ever hands back a well-formed term, so this is
583 // unreachable; returning the original beats mangling it silently.
584 _ => t,
585 }
586 }
587
588 /// `super::graph_token` for `oxrdf` 0.3's `GraphName` — the same rule, a
589 /// different crate version, which is exactly why it cannot be shared.
590 fn graph_token(g: &oxrdf12::GraphName) -> Option<String> {
591 match g {
592 oxrdf12::GraphName::DefaultGraph => None,
593 other => Some(other.to_string()),
594 }
595 }
596
597 /// Stream one RDF 1.2 Turtle (`"ttl"`) or TriG (`"trig"`) input, auditing
598 /// every quad the way the RDF-star reader does.
599 pub(super) fn stream<R: std::io::BufRead>(
600 reader: R,
601 format: &str,
602 mut audit: Option<&mut IriAudit>,
603 f: &mut dyn FnMut(RawQuad),
604 ) -> Result<(), IngestError> {
605 if format == "ttl" {
606 for r in oxttl12::TurtleParser::new().for_reader(reader) {
607 let t = r.map_err(|e| IngestError::Turtle(e.to_string()))?;
608 let q = (
609 t.subject.to_string(),
610 t.predicate.to_string(),
611 fold_triple_term(t.object.to_string()),
612 None,
613 );
614 audit_quad(audit.as_deref_mut(), &q.0, &q.1, &q.2, None, || {
615 "turtle".into()
616 })?;
617 f(q);
618 }
619 } else {
620 for r in oxttl12::TriGParser::new().for_reader(reader) {
621 let q = r.map_err(|e| IngestError::TriG(e.to_string()))?;
622 let q = (
623 q.subject.to_string(),
624 q.predicate.to_string(),
625 fold_triple_term(q.object.to_string()),
626 graph_token(&q.graph_name),
627 );
628 audit_quad(
629 audit.as_deref_mut(),
630 &q.0,
631 &q.1,
632 &q.2,
633 q.3.as_deref(),
634 || "trig".into(),
635 )?;
636 f(q);
637 }
638 }
639 Ok(())
640 }
641}
642
643/// The stand-in for [`rdf12`] in a build with `rdf12-turtle` off: it refuses by
644/// name. Reading the file as RDF-star instead would answer a question nobody
645/// asked, with a different graph.
646#[cfg(not(feature = "rdf12-turtle"))]
647mod rdf12 {
648 use super::{IngestError, IriAudit, RawQuad};
649
650 pub(super) fn stream<R: std::io::BufRead>(
651 _reader: R,
652 _format: &str,
653 _audit: Option<&mut IriAudit>,
654 _f: &mut dyn FnMut(RawQuad),
655 ) -> Result<(), IngestError> {
656 Err(IngestError::Rdf12TurtleUnavailable)
657 }
658}
659
660/// Parse Turtle into canonical N-Triples-token triples via oxttl, reading
661/// `<< s p o >>` as an RDF-star quoted triple. [`parse_turtle_surface`] chooses.
662pub fn parse_turtle(text: &str) -> Result<Vec<RawTriple>, IngestError> {
663 parse_turtle_surface(text, QuotedTripleSurface::default())
664}
665
666/// [`parse_turtle`] in the given [`QuotedTripleSurface`].
667pub fn parse_turtle_surface(
668 text: &str,
669 surface: QuotedTripleSurface,
670) -> Result<Vec<RawTriple>, IngestError> {
671 Ok(parse_text_surface(text, "ttl", surface)?
672 .into_iter()
673 .map(|(s, p, o, _)| (s, p, o))
674 .collect())
675}
676
677/// Parse TriG into canonical quads via oxttl — Turtle plus named-graph blocks
678/// (`<g> { … }`), the shape most large RDF dumps ship in. Reads `<< s p o >>` as
679/// an RDF-star quoted triple; [`parse_trig_surface`] chooses.
680pub fn parse_trig(text: &str) -> Result<Vec<RawQuad>, IngestError> {
681 parse_trig_surface(text, QuotedTripleSurface::default())
682}
683
684/// [`parse_trig`] in the given [`QuotedTripleSurface`].
685pub fn parse_trig_surface(
686 text: &str,
687 surface: QuotedTripleSurface,
688) -> Result<Vec<RawQuad>, IngestError> {
689 parse_text_surface(text, "trig", surface)
690}
691
692/// The shared Turtle/TriG text path: stream the string through whichever reader
693/// `surface` selects, and — this is the part only the text path can do, because
694/// only it holds the input — say so when an RDF-star parse fails on what is
695/// plainly RDF 1.2.
696///
697/// It does not audit: the Turtle/TriG text callers audit over the finished quads
698/// (see [`parse_statements_audited_surface`]), and auditing here too would count
699/// every IRI twice.
700///
701/// The RDF-star arm calls `oxttl` 0.1 **directly** rather than routing through
702/// [`stream_reader_audited_surface`]. That is not style: this is the only ingest
703/// path the wasm engine reaches, and going through the generic reader would
704/// monomorphize it over `&[u8]` and pull the whole `match format` — every
705/// syntax's reader — into a module that previously linked one Turtle parser.
706/// Measured at +17 KB of wasm for code no browser can run, since the RDF 1.2
707/// reader is not compiled there at all. Keeping the arm direct keeps the
708/// browser artifacts what they were.
709fn parse_text_surface(
710 text: &str,
711 format: &str,
712 surface: QuotedTripleSurface,
713) -> Result<Vec<RawQuad>, IngestError> {
714 let mut out = Vec::new();
715 let r = if surface == QuotedTripleSurface::Rdf12 {
716 rdf12::stream(text.as_bytes(), format, None, &mut |q| out.push(q))
717 } else if format == "ttl" {
718 turtle_star(text.as_bytes(), None, &mut |q| out.push(q))
719 } else {
720 trig_star(text.as_bytes(), None, &mut |q| out.push(q))
721 };
722 match r {
723 Ok(()) => Ok(out),
724 // The default surface met `<<(`: that is an RDF 1.2 file being read as
725 // RDF-star, the one mistake this flag exists to prevent, and the parse
726 // error alone ("unexpected character") would send the reader hunting for
727 // a typo that is not there.
728 Err(IngestError::Turtle(m))
729 if surface == QuotedTripleSurface::RdfStar && looks_like_rdf12(text) =>
730 {
731 Err(IngestError::Turtle(format!("{m}{RDF12_SURFACE_HINT}")))
732 }
733 Err(IngestError::TriG(m))
734 if surface == QuotedTripleSurface::RdfStar && looks_like_rdf12(text) =>
735 {
736 Err(IngestError::TriG(format!("{m}{RDF12_SURFACE_HINT}")))
737 }
738 Err(e) => Err(e),
739 }
740}
741
742/// Parse RDF/XML into canonical N-Triples-token triples via oxrdfxml. This is how
743/// most OWL ontologies ship (`.rdf`/`.owl`/`.xml` with an `rdf:RDF` root) — so rete
744/// ingests them directly, no external conversion. (OWL/XML — the non-RDF functional
745/// XML serialization — is a different language; convert it with owlready2 first.)
746pub fn parse_rdfxml(text: &str) -> Result<Vec<RawTriple>, IngestError> {
747 let mut out = Vec::new();
748 for r in oxrdfxml::RdfXmlParser::new().for_reader(text.as_bytes()) {
749 let t = r.map_err(|e| IngestError::RdfXml(e.to_string()))?;
750 out.push((
751 t.subject.to_string(),
752 t.predicate.to_string(),
753 t.object.to_string(),
754 ));
755 }
756 Ok(out)
757}
758
759/// Parse one text input by format name (`"nt"`, `"nq"`, `"ttl"`, `"trig"`, or
760/// `"rdfxml"`) into quads (triples land in the default graph).
761pub fn parse_statements(text: &str, format: &str) -> Result<Vec<RawQuad>, IngestError> {
762 parse_statements_audited(text, format, None)
763}
764
765/// [`parse_statements`] with an optional [`IriAudit`]. The line-based syntaxes
766/// audit as they parse (so `strict` fails on the offending line); the others are
767/// audited over the parsed quads, which is where `oxttl` has already had its say.
768pub fn parse_statements_audited(
769 text: &str,
770 format: &str,
771 audit: Option<&mut IriAudit>,
772) -> Result<Vec<RawQuad>, IngestError> {
773 parse_statements_audited_surface(text, format, audit, QuotedTripleSurface::default())
774}
775
776/// [`parse_statements_audited`] reading `<< … >>` in the given
777/// [`QuotedTripleSurface`]. This is the entry point `rete build` and
778/// `rete validate` use for everything but a streamed line-based input, and the
779/// one that turns "RDF 1.2 file, RDF-star reader" into an error that names the
780/// flag instead of a bare syntax complaint.
781pub fn parse_statements_audited_surface(
782 text: &str,
783 format: &str,
784 mut audit: Option<&mut IriAudit>,
785 surface: QuotedTripleSurface,
786) -> Result<Vec<RawQuad>, IngestError> {
787 let quads = match format {
788 "nq" => return parse_quads_audited(text, audit),
789 "nt" => {
790 return Ok(parse_audited(text, audit)?
791 .into_iter()
792 .map(|(s, p, o)| (s, p, o, None))
793 .collect())
794 }
795 // Straight to the quad path for both: going via `parse_turtle_surface`
796 // would build a `Vec<RawQuad>`, rebuild it as a `Vec<RawTriple>` to
797 // satisfy that function's signature, and rebuild it as quads again —
798 // three copies of every term String on the biggest input rete takes.
799 "ttl" | "trig" => parse_text_surface(text, format, surface)?,
800 "rdfxml" => parse_rdfxml(text)?
801 .into_iter()
802 .map(|(s, p, o)| (s, p, o, None))
803 .collect(),
804 other => return Err(IngestError::UnknownFormat(other.to_string())),
805 };
806 for (s, p, o, g) in &quads {
807 audit_quad(audit.as_deref_mut(), s, p, o, g.as_deref(), || {
808 format.to_string()
809 })?;
810 }
811 Ok(quads)
812}
813
814/// Take one term from the front of `s`, returning `(term, remainder)`.
815pub(crate) fn take_term(s: &str) -> Option<(String, &str)> {
816 let bytes = s.as_bytes();
817 let first = *bytes.first()?;
818 match first {
819 // A quoted triple / triple term, in EITHER surface — the inner terms are
820 // themselves terms (so this recurses; nesting works). `<<` starts with `<`
821 // and an IRI scan would stop at the first inner `>`, so it must be handled
822 // before the plain-IRI case.
823 // * RDF-star: `<< subject predicate object >>`
824 // * RDF 1.2: `<<( subject predicate object )>>` (triple term)
825 // Both re-emit the SAME canonical token `<<s p o>>`, so a file written in
826 // either surface — and a query written either way — dedupe and match. This
827 // makes RDF 1.2 N-Triples interoperable with the RDF-star we already store,
828 // with no format change and no dependency swap.
829 b'<' if bytes.get(1) == Some(&b'<') => {
830 // Distinguish `<<(` (RDF 1.2) from `<<` (RDF-star) by the char after `<<`.
831 let (inner, rdf12) = match s[2..].strip_prefix('(') {
832 Some(after) => (after.trim_start(), true),
833 None => (s[2..].trim_start(), false),
834 };
835 let (subj, r) = take_term(inner)?;
836 let (pred, r) = take_term(r.trim_start())?;
837 let (obj, r) = take_term(r.trim_start())?;
838 let r = r.trim_start();
839 let rest = if rdf12 {
840 r.strip_prefix(")>>")?
841 } else {
842 r.strip_prefix(">>")?
843 };
844 // Canonical surface = oxrdf's `Triple` Display: `<<s p o>>` (tight
845 // brackets, single spaces between components) — the same token for both
846 // input surfaces, so N-Triples-star, Turtle-star, and RDF 1.2 triple
847 // terms all resolve to one dictionary entry.
848 Some((format!("<<{subj} {pred} {obj}>>"), rest))
849 }
850 b'<' => {
851 // IRI ref: up to the closing '>'.
852 let end = s.find('>')?;
853 Some((s[..=end].to_string(), &s[end + 1..]))
854 }
855 b'_' => {
856 // Blank node: up to whitespace.
857 let end = s.find(char::is_whitespace).unwrap_or(s.len());
858 Some((s[..end].to_string(), &s[end..]))
859 }
860 b'"' => {
861 // Literal: closing unescaped quote, then optional ^^<dt> or @lang.
862 let mut i = 1;
863 let b = s.as_bytes();
864 while i < b.len() {
865 match b[i] {
866 b'\\' => i += 2, // skip escaped char
867 b'"' => break,
868 _ => i += 1,
869 }
870 }
871 if i >= b.len() {
872 return None; // unterminated
873 }
874 let mut end = i + 1; // past closing quote
875 if s[end..].starts_with("^^<") {
876 let close = s[end..].find('>')? + end;
877 end = close + 1;
878 } else if s[end..].starts_with('@') {
879 // Language tag: '@' then BCP-47 subtags `[a-zA-Z0-9-]+`. Stop at
880 // the first char that can't be part of a tag — normally the
881 // whitespace before the predicate, but ALSO the `>>` that closes
882 // a quoted triple when this literal is its object (`"x"@en>>`),
883 // where there is no separating whitespace.
884 let mut j = end + 1; // past '@'
885 while j < b.len() && (b[j].is_ascii_alphanumeric() || b[j] == b'-') {
886 j += 1;
887 }
888 end = j;
889 }
890 Some((s[..end].to_string(), &s[end..]))
891 }
892 _ => None,
893 }
894}
895
896/// Split a canonical quoted-triple token `<<s p o>>` (RDF-star) into its three
897/// component term tokens, or `None` if `t` is not a quoted triple. Reuses
898/// [`take_term`] for term-boundary scanning, so nested quoting parses correctly.
899/// The inverse of the `<<…>>` construction; used by the SUBJECT/PREDICATE/OBJECT
900/// SPARQL-star builtins.
901pub(crate) fn quoted_triple_parts(t: &str) -> Option<(String, String, String)> {
902 let inner = t.strip_prefix("<<")?.strip_suffix(">>")?.trim();
903 let (s, r) = take_term(inner)?;
904 let (p, r) = take_term(r.trim_start())?;
905 let (o, r) = take_term(r.trim_start())?;
906 if !r.trim().is_empty() {
907 return None;
908 }
909 Some((s, p, o))
910}
911
912/// Counts describing an assembled file, for status lines and UIs.
913///
914/// In the [`BuildStats`] a metadata callback receives, `statements` and
915/// `default_triples` are the counts **ingested** — the input multiset, before
916/// the index deduplicates it. In the `BuildStats` an assemble function
917/// *returns*, they are the counts actually **written** (see
918/// [`FinalCounts`]), so a status line reports what the file holds. Use
919/// [`FinalCounts`] rather than these when a payload must agree with the header.
920#[derive(Debug, Clone, Copy)]
921pub struct BuildStats {
922 /// Total statements, across the default graph and every named one.
923 pub statements: usize,
924 /// Statements that landed in the default graph.
925 pub default_triples: usize,
926 /// How many named graphs the input mentioned; one index is written per graph.
927 pub named_graphs: usize,
928 /// Distinct terms in the shared dictionary.
929 pub terms: usize,
930 /// Levels in the community pyramid, or `0` when the build skipped it.
931 pub pyramid_levels: u16,
932}
933
934/// The **deduplicated** counts of the file being written — what the header
935/// records, and therefore what any embedded metadata has to report.
936///
937/// The input multiset is not the file: every permutation index sorts and
938/// deduplicates, so a harvest that pages with overlapping windows writes fewer
939/// quads than it ingested. These counts are known only once the indexes exist,
940/// which is why a payload that carries them is derived in two stages (see
941/// [`IntoMetadata`]).
942#[derive(Debug, Clone, Copy)]
943pub struct FinalCounts {
944 /// Unique triples in the default graph.
945 pub default_triples: u64,
946 /// Unique quads across the default graph and every named graph — exactly
947 /// the header's `quad_count`.
948 pub quads: u64,
949}
950
951/// What a build's metadata callback may return.
952///
953/// A `Vec<u8>` is the payload verbatim — the common case (no metadata, or a
954/// payload that carries no counts). [`DeferredMetadata`] is a payload that
955/// *does* carry counts: the callback derives everything it needs while the
956/// source quads are still resident, and returns a finalizer that the writer
957/// calls with the [`FinalCounts`], which only exist once the indexes have
958/// deduplicated the input. Without that second stage a card can only report the
959/// ingested count, which over-reports any input containing duplicates.
960pub trait IntoMetadata {
961 /// Produce the metadata section payload for a file with these counts.
962 fn into_metadata(self, counts: FinalCounts) -> Vec<u8>;
963}
964
965impl IntoMetadata for Vec<u8> {
966 fn into_metadata(self, _counts: FinalCounts) -> Vec<u8> {
967 self
968 }
969}
970
971/// A metadata payload whose final form needs the written (deduplicated) counts.
972/// See [`IntoMetadata`].
973pub struct DeferredMetadata(Box<dyn FnOnce(FinalCounts) -> Vec<u8>>);
974
975impl DeferredMetadata {
976 /// Defer `f` until the indexes are built and the true counts are known.
977 pub fn new(f: impl FnOnce(FinalCounts) -> Vec<u8> + 'static) -> Self {
978 DeferredMetadata(Box::new(f))
979 }
980
981 /// No metadata section — byte-identical to a metadata-free build. Lets one
982 /// `match` arm opt out while the other defers, without a type mismatch.
983 pub fn none() -> Self {
984 DeferredMetadata::new(|_| Vec::new())
985 }
986}
987
988impl IntoMetadata for DeferredMetadata {
989 fn into_metadata(self, counts: FinalCounts) -> Vec<u8> {
990 (self.0)(counts)
991 }
992}
993
994/// Assemble a complete `.rete` file image from parsed quads: one shared
995/// dictionary, the default-graph index, one index per named graph, and the
996/// community pyramid. `metadata` is the opaque metadata-section payload (the
997/// CLI puts a JSON Dataset Card there); pass `&[]` for none — that is
998/// byte-identical to a metadata-free build.
999pub fn assemble_dataset(quads: Vec<RawQuad>, metadata: &[u8]) -> (Vec<u8>, BuildStats) {
1000 let blob = metadata.to_vec();
1001 assemble_dataset_with(quads, move |_, _| blob)
1002}
1003
1004/// Like [`assemble_dataset`], but the metadata payload is derived from the
1005/// [`BuildStats`] and the source quads while they are still resident — for
1006/// metadata that describes the graph (the Dataset Card). Returning an empty
1007/// `Vec` is byte-identical to a metadata-free build; return a
1008/// [`DeferredMetadata`] for a payload that must also carry the file's
1009/// deduplicated counts, which exist only once the indexes are built.
1010pub fn assemble_dataset_with<M: IntoMetadata>(
1011 quads: Vec<RawQuad>,
1012 metadata: impl FnOnce(&BuildStats, &[RawQuad]) -> M,
1013) -> (Vec<u8>, BuildStats) {
1014 assemble_dataset_with_opts(quads, true, false, None, metadata)
1015}
1016
1017/// Like [`assemble_dataset_with`], but `with_pyramid = false` skips the Louvain
1018/// community pyramid entirely — no pyramid section is written (header length 0).
1019/// SPARQL / SHACL / triple / reachability queries don't use the pyramid, so a
1020/// pyramid-less file is fully queryable and markedly smaller (the pyramid is the
1021/// largest section on highly-connected graphs). Only the community / summary /
1022/// progressive paths need it.
1023pub fn assemble_dataset_with_opts<M: IntoMetadata>(
1024 quads: Vec<RawQuad>,
1025 with_pyramid: bool,
1026 with_text_index: bool,
1027 type_override: Option<&str>,
1028 metadata: impl FnOnce(&BuildStats, &[RawQuad]) -> M,
1029) -> (Vec<u8>, BuildStats) {
1030 assemble_dataset_with_opts_algo(
1031 quads,
1032 with_pyramid,
1033 with_text_index,
1034 type_override,
1035 PyramidAlgo::Louvain,
1036 metadata,
1037 )
1038}
1039
1040/// Like [`assemble_dataset_with_opts`], but selects the community [`PyramidAlgo`]
1041/// (the in-memory build path for `rete build --pyramid-algo …`).
1042#[allow(clippy::too_many_arguments)]
1043pub fn assemble_dataset_with_opts_algo<M: IntoMetadata>(
1044 quads: Vec<RawQuad>,
1045 with_pyramid: bool,
1046 with_text_index: bool,
1047 type_override: Option<&str>,
1048 algo: PyramidAlgo,
1049 metadata: impl FnOnce(&BuildStats, &[RawQuad]) -> M,
1050) -> (Vec<u8>, BuildStats) {
1051 assemble_dataset_with_perms(
1052 quads,
1053 with_pyramid,
1054 with_text_index,
1055 type_override,
1056 algo,
1057 crate::index::PermSet::ALL,
1058 metadata,
1059 )
1060}
1061
1062/// Like [`assemble_dataset_with_opts_algo`], but writes only the permutations in
1063/// `perms` (`rete build --permutations 3`). Every other byte of the file is
1064/// unchanged; [`crate::index::PermSet::ALL`] reproduces the default build
1065/// exactly, down to the header byte.
1066#[allow(clippy::too_many_arguments)]
1067pub fn assemble_dataset_with_perms<M: IntoMetadata>(
1068 quads: Vec<RawQuad>,
1069 with_pyramid: bool,
1070 with_text_index: bool,
1071 type_override: Option<&str>,
1072 algo: PyramidAlgo,
1073 perms: crate::index::PermSet,
1074 metadata: impl FnOnce(&BuildStats, &[RawQuad]) -> M,
1075) -> (Vec<u8>, BuildStats) {
1076 use std::collections::BTreeMap;
1077
1078 let mut db = DictionaryBuilder::new();
1079 for (s, p, o, _) in &quads {
1080 db.observe(s, p, o);
1081 }
1082 let dict = db.build();
1083
1084 let mut default_triples = Vec::new();
1085 let mut named: BTreeMap<String, Vec<(u32, u32, u32)>> = BTreeMap::new();
1086 for (s, p, o, g) in &quads {
1087 let t = dict.encode(s, p, o).expect("observed term");
1088 match g {
1089 None => default_triples.push(t),
1090 Some(graph) => named.entry(graph.clone()).or_default().push(t),
1091 }
1092 }
1093
1094 // Derive the metadata blob (the Dataset Card) from the raw quads NOW, while
1095 // they are resident — then DROP them before the memory-heavy pyramid + index
1096 // phases. On a big build the string quads are the largest working set (every
1097 // term an owned String, heavily duplicated) and are fully redundant with the
1098 // dictionary + id-triples once encoded, so freeing them here cuts peak RAM by
1099 // their whole size. `pyramid_levels` is not known yet (0 in the callback);
1100 // it is filled into the returned `stats` by `finish_assembly`, and no metadata
1101 // callback depends on it (the card derives from the quads + term/graph counts).
1102 let stats = BuildStats {
1103 statements: quads.len(),
1104 default_triples: default_triples.len(),
1105 named_graphs: named.len(),
1106 terms: dict.term_count() as usize,
1107 pyramid_levels: 0,
1108 };
1109 let pending = metadata(&stats, &quads);
1110 drop(quads);
1111
1112 finish_assembly(
1113 dict,
1114 default_triples,
1115 named,
1116 with_pyramid,
1117 with_text_index,
1118 type_override,
1119 algo,
1120 perms,
1121 pending,
1122 stats,
1123 )
1124}
1125
1126/// **Two-pass, low-RAM** assembly: build a `.rete` by **streaming** the input(s)
1127/// twice instead of holding every parsed quad in memory. `stream` is invoked
1128/// **twice** and MUST replay the exact same quads in the same order each time —
1129/// pass 1 observes every term into the dictionary; pass 2 encodes them to
1130/// id-triples. The raw string quads (every term an owned String, heavily
1131/// duplicated — by far the largest working set on a big graph) are **never
1132/// collected**, so peak RAM is bounded by the dictionary + id-triples + index
1133/// rather than the string-quad multiset. Output is **byte-identical** to
1134/// [`assemble_dataset_with_opts`] on the same quads (same dictionary, same
1135/// id-triples in file order, same downstream pipeline).
1136///
1137/// The metadata callback derives the Dataset Card from the built dictionary +
1138/// default-graph id-triples (resolving terms through the dictionary), since the
1139/// raw quads were never retained. `stream` propagates parse/IO errors.
1140pub fn assemble_dataset_streaming<S, M: IntoMetadata>(
1141 stream: S,
1142 with_pyramid: bool,
1143 with_text_index: bool,
1144 type_override: Option<&str>,
1145 metadata: impl FnOnce(&BuildStats, &Dictionary, &[(u32, u32, u32)]) -> M,
1146) -> Result<(Vec<u8>, BuildStats), IngestError>
1147where
1148 S: FnMut(&mut dyn FnMut(RawQuad)) -> Result<(), IngestError>,
1149{
1150 assemble_dataset_streaming_algo(
1151 stream,
1152 with_pyramid,
1153 with_text_index,
1154 type_override,
1155 PyramidAlgo::Louvain,
1156 metadata,
1157 )
1158}
1159
1160/// Like [`assemble_dataset_streaming`], but selects the community [`PyramidAlgo`]
1161/// (the streaming, low-RAM build path for `rete build --pyramid-algo …`).
1162#[allow(clippy::too_many_arguments)]
1163pub fn assemble_dataset_streaming_algo<S, M: IntoMetadata>(
1164 stream: S,
1165 with_pyramid: bool,
1166 with_text_index: bool,
1167 type_override: Option<&str>,
1168 algo: PyramidAlgo,
1169 metadata: impl FnOnce(&BuildStats, &Dictionary, &[(u32, u32, u32)]) -> M,
1170) -> Result<(Vec<u8>, BuildStats), IngestError>
1171where
1172 S: FnMut(&mut dyn FnMut(RawQuad)) -> Result<(), IngestError>,
1173{
1174 assemble_dataset_streaming_with_perms(
1175 stream,
1176 with_pyramid,
1177 with_text_index,
1178 type_override,
1179 algo,
1180 crate::index::PermSet::ALL,
1181 metadata,
1182 )
1183}
1184
1185/// Like [`assemble_dataset_streaming_algo`], but writes only the permutations in
1186/// `perms` — the two-pass low-RAM twin of [`assemble_dataset_with_perms`].
1187#[allow(clippy::too_many_arguments)]
1188pub fn assemble_dataset_streaming_with_perms<S, M: IntoMetadata>(
1189 mut stream: S,
1190 with_pyramid: bool,
1191 with_text_index: bool,
1192 type_override: Option<&str>,
1193 algo: PyramidAlgo,
1194 perms: crate::index::PermSet,
1195 metadata: impl FnOnce(&BuildStats, &Dictionary, &[(u32, u32, u32)]) -> M,
1196) -> Result<(Vec<u8>, BuildStats), IngestError>
1197where
1198 S: FnMut(&mut dyn FnMut(RawQuad)) -> Result<(), IngestError>,
1199{
1200 use std::collections::BTreeMap;
1201
1202 // Pass 1: observe every term (the dictionary dedups; the string quads are
1203 // freed line by line and never collected).
1204 let mut db = DictionaryBuilder::new();
1205 stream(&mut |(s, p, o, _g)| db.observe(&s, &p, &o))?;
1206 let dict = db.build();
1207
1208 // Pass 2: encode each quad to an id-triple, bucketing named graphs.
1209 let mut default_triples: Vec<(u32, u32, u32)> = Vec::new();
1210 let mut named: BTreeMap<String, Vec<(u32, u32, u32)>> = BTreeMap::new();
1211 stream(&mut |(s, p, o, g)| {
1212 let t = dict.encode(&s, &p, &o).expect("observed term");
1213 match g {
1214 None => default_triples.push(t),
1215 Some(graph) => named.entry(graph).or_default().push(t),
1216 }
1217 })?;
1218
1219 let statements = default_triples.len() + named.values().map(Vec::len).sum::<usize>();
1220 let stats = BuildStats {
1221 statements,
1222 default_triples: default_triples.len(),
1223 named_graphs: named.len(),
1224 terms: dict.term_count() as usize,
1225 pyramid_levels: 0,
1226 };
1227 let pending = metadata(&stats, &dict, &default_triples);
1228 Ok(finish_assembly(
1229 dict,
1230 default_triples,
1231 named,
1232 with_pyramid,
1233 with_text_index,
1234 type_override,
1235 algo,
1236 perms,
1237 pending,
1238 stats,
1239 ))
1240}
1241
1242/// The shared tail of every build path: from a finished dictionary + encoded
1243/// id-triples (default graph + named), build the community pyramid, the optional
1244/// full-text index, the permutation indexes, and serialize the file image. Takes
1245/// the id-triples **by value** so they are freed as the permutations consume them.
1246#[allow(clippy::too_many_arguments)]
1247fn finish_assembly<M: IntoMetadata>(
1248 dict: Dictionary,
1249 default_triples: Vec<(u32, u32, u32)>,
1250 named: std::collections::BTreeMap<String, Vec<(u32, u32, u32)>>,
1251 with_pyramid: bool,
1252 with_text_index: bool,
1253 type_override: Option<&str>,
1254 algo: PyramidAlgo,
1255 perms: crate::index::PermSet,
1256 pending: M,
1257 mut stats: BuildStats,
1258) -> (Vec<u8>, BuildStats) {
1259 let has_named = !named.is_empty();
1260
1261 // The pyramid and the full-text index are built FIRST, while the dictionary and
1262 // the default id-triples are both still resident — both need the two together.
1263 let (meta, levels) = if with_pyramid {
1264 build_pyramid_meta_algo(
1265 &dict,
1266 &default_triples,
1267 DEFAULT_TILE_BUDGET,
1268 type_override,
1269 algo,
1270 )
1271 } else {
1272 (Vec::new(), 0)
1273 };
1274 stats.pyramid_levels = levels;
1275
1276 let text_index = if with_text_index {
1277 crate::file::compute_text_index(&dict, &default_triples)
1278 } else {
1279 Vec::new()
1280 };
1281
1282 // From here the dictionary is only needed for its own serialized bytes. Encode
1283 // it, capture the header term count, then DROP it before building the
1284 // permutation indexes — which work purely on id-triples and never touch the
1285 // dictionary. On a large graph this frees the single biggest resident structure
1286 // (the dictionary) right when the index sort needs the headroom. The output is
1287 // byte-for-byte identical to serializing the dictionary inline.
1288 let codec = crate::file::writer_codec();
1289 let dict_container = crate::file::encode_dict_container(&dict, codec);
1290 let term_count = dict.term_count() as u64;
1291 let has_quoted_triples = dict.has_quoted_triples();
1292 drop(dict);
1293
1294 // Build each graph's six permutations one-at-a-time on a large graph (a single
1295 // permuted copy resident at a time, each sort still parallel) instead of all
1296 // six concurrently; below the threshold the faster all-permutations-parallel
1297 // build is used (its 6x transient copy is negligible there). `build_seq` is
1298 // byte-identical to `build`.
1299 let build_index = |triples: Vec<(u32, u32, u32)>| -> crate::GraphIndex {
1300 let n = triples.len();
1301 let b = GraphIndexBuilder::from_triples(triples).with_perms(perms);
1302 if n > LOWMEM_TRIPLE_THRESHOLD {
1303 b.build_seq()
1304 } else {
1305 b.build()
1306 }
1307 };
1308 let def = build_index(default_triples);
1309 let named_indexes: Vec<(String, crate::GraphIndex)> = named
1310 .into_iter()
1311 .map(|(g, ts)| (g, build_index(ts)))
1312 .collect();
1313
1314 // Only now is the DEDUPLICATED size of the file known: every index sorts and
1315 // dedups, so an input containing duplicate statements (a harvest that pages
1316 // with overlapping windows, a merge of overlapping dumps) writes fewer quads
1317 // than it ingested. `write_dataset_from_parts` stamps exactly this sum into
1318 // the header, so a metadata payload finalized with it cannot disagree with
1319 // the header — which is what an embedded Dataset Card used to do.
1320 let counts = FinalCounts {
1321 default_triples: def.triple_count() as u64,
1322 quads: def.triple_count() as u64
1323 + named_indexes
1324 .iter()
1325 .map(|(_, ix)| ix.triple_count() as u64)
1326 .sum::<u64>(),
1327 };
1328 let blob = pending.into_metadata(counts);
1329 // Report what was WRITTEN, not what was read: the returned stats drive the
1330 // CLI's "wrote …: N quads" line, which described the file, not the input.
1331 stats.statements = counts.quads as usize;
1332 stats.default_triples = counts.default_triples as usize;
1333
1334 let bytes = crate::file::write_dataset_from_parts(
1335 &dict_container,
1336 term_count,
1337 &def,
1338 &named_indexes,
1339 has_named,
1340 has_quoted_triples,
1341 &meta,
1342 levels,
1343 &blob,
1344 &text_index,
1345 codec,
1346 );
1347 (bytes, stats)
1348}
1349
1350/// Above this default-graph triple count, the index is built one permutation at a
1351/// time (low peak RAM) instead of all six in parallel. Chosen so typical builds
1352/// keep the faster parallel path while multi-hundred-million-triple graphs stay
1353/// within a bounded memory budget.
1354const LOWMEM_TRIPLE_THRESHOLD: usize = 30_000_000;
1355
1356#[cfg(test)]
1357mod tests {
1358 use super::*;
1359 use crate::Rete;
1360
1361 #[test]
1362 fn parses_iris_bnodes_literals() {
1363 let input = r#"
1364 # a comment
1365 <http://ex/Alice> <http://ex/knows> <http://ex/Bob> .
1366 <http://ex/Alice> <http://ex/age> "30"^^<http://www.w3.org/2001/XMLSchema#integer> .
1367 <http://ex/Bob> <http://ex/label> "Bob"@en .
1368 _:b0 <http://ex/p> "plain" .
1369 "#;
1370 let t = parse(input).unwrap();
1371 assert_eq!(t.len(), 4);
1372 assert_eq!(t[0].0, "<http://ex/Alice>");
1373 assert_eq!(t[0].2, "<http://ex/Bob>");
1374 assert_eq!(t[1].2, "\"30\"^^<http://www.w3.org/2001/XMLSchema#integer>");
1375 assert_eq!(t[2].2, "\"Bob\"@en");
1376 assert_eq!(t[3].0, "_:b0");
1377 assert_eq!(t[3].2, "\"plain\"");
1378 }
1379
1380 #[test]
1381 fn literal_with_spaces_and_escaped_quote() {
1382 let input = r#"<http://ex/s> <http://ex/p> "a \"quoted\" phrase here" ."#;
1383 let t = parse(input).unwrap();
1384 assert_eq!(t.len(), 1);
1385 assert_eq!(t[0].2, r#""a \"quoted\" phrase here""#);
1386 }
1387
1388 /// RDF-star: a quoted triple whose object is a **language-tagged literal**
1389 /// sits directly against the closing `>>` with no separating whitespace
1390 /// (`"name"@fr>>`). The language-tag scan must stop at `>` — a regression
1391 /// guard for the greedy scan-to-whitespace that swallowed `@fr>>` as the tag.
1392 #[test]
1393 fn quoted_triple_langtagged_object() {
1394 let s = r#"<<<http://ex/sp> <http://ex/name> "Hirondelle rustique"@fr>>"#;
1395 let (tok, rest) = take_term(s).unwrap();
1396 assert_eq!(
1397 tok,
1398 r#"<<<http://ex/sp> <http://ex/name> "Hirondelle rustique"@fr>>"#
1399 );
1400 assert_eq!(rest, "");
1401 // and it round-trips through a full annotation line
1402 let line = r#"<<<http://ex/sp> <http://ex/name> "Oreneta vulgar"@ca>> <http://purl.org/dc/terms/source> "Catalogue of Life" ."#;
1403 let t = parse(line).unwrap();
1404 assert_eq!(t.len(), 1);
1405 assert_eq!(
1406 t[0].0,
1407 r#"<<<http://ex/sp> <http://ex/name> "Oreneta vulgar"@ca>>"#
1408 );
1409 assert_eq!(t[0].2, r#""Catalogue of Life""#);
1410 // a plain lang-tagged object still terminates at whitespace
1411 assert_eq!(take_term(r#""x"@pt-BR ."#).unwrap().0, r#""x"@pt-BR"#);
1412 }
1413
1414 /// RDF 1.2 triple-term surface `<<( s p o )>>` parses to the SAME canonical
1415 /// token as the RDF-star `<< s p o >>`, so the two are interoperable (a query
1416 /// written either way matches data ingested either way).
1417 #[test]
1418 fn rdf12_triple_term_same_token_as_rdf_star() {
1419 let rdf12 = r#"<http://ex/r> <http://ex/p> <<( <http://ex/s> <http://ex/q> "o" )>> ."#;
1420 let star = r#"<http://ex/r> <http://ex/p> << <http://ex/s> <http://ex/q> "o" >> ."#;
1421 let a = parse(rdf12).unwrap();
1422 let b = parse(star).unwrap();
1423 assert_eq!(a.len(), 1);
1424 assert_eq!(a[0].2, r#"<<<http://ex/s> <http://ex/q> "o">>"#);
1425 assert_eq!(
1426 a[0].2, b[0].2,
1427 "RDF 1.2 and RDF-star must dedupe to one token"
1428 );
1429 // take_term consumes exactly the triple term (nothing left dangling).
1430 let (tok, rest) =
1431 take_term(r#"<<( <http://ex/s> <http://ex/q> <http://ex/o> )>>"#).unwrap();
1432 assert_eq!(tok, "<<<http://ex/s> <http://ex/q> <http://ex/o>>>");
1433 assert_eq!(rest, "");
1434 }
1435
1436 #[test]
1437 fn rejects_missing_dot() {
1438 assert!(parse("<a> <b> <c>").is_err());
1439 }
1440
1441 #[test]
1442 fn parses_quads_with_and_without_graph() {
1443 let input = "<http://ex/a> <http://ex/p> <http://ex/b> .\n\
1444 <http://ex/a> <http://ex/p> <http://ex/c> <http://ex/g> .";
1445 let q = parse_quads(input).unwrap();
1446 assert_eq!(q.len(), 2);
1447 assert_eq!(q[0].3, None); // default graph
1448 assert_eq!(q[1].3, Some("<http://ex/g>".to_string()));
1449 }
1450
1451 #[test]
1452 fn parse_reader_matches_text_parse() {
1453 // The streaming reader must produce exactly the same quads as parsing the
1454 // whole text — including blank/comment skipping, a named graph, and CRLF.
1455 let nt = "<http://ex/a> <http://ex/p> <http://ex/b> .\r\n\
1456 # comment\n\
1457 \n\
1458 _:b0 <http://ex/q> \"lit\"@en .\n";
1459 let via_text = parse_statements(nt, "nt").unwrap();
1460 let via_reader = parse_reader(std::io::Cursor::new(nt), "nt", 0).unwrap();
1461 assert_eq!(via_text, via_reader);
1462
1463 let nq = "<http://ex/a> <http://ex/p> <http://ex/b> .\n\
1464 <http://ex/a> <http://ex/p> <http://ex/c> <http://ex/g> .\n";
1465 assert_eq!(
1466 parse_quads(nq).unwrap(),
1467 parse_reader(std::io::Cursor::new(nq), "nq", 0).unwrap()
1468 );
1469 // Turtle and TriG stream too — that is what lets the external build take a
1470 // multi-gigabyte `.ttl.gz` without expanding it to N-Triples first. Reading
1471 // a structured syntax from a reader must agree with parsing its whole text.
1472 let ttl = "@prefix ex: <http://ex/> .\nex:A ex:knows ex:B , ex:C .\n";
1473 assert_eq!(
1474 parse_statements(ttl, "ttl").unwrap(),
1475 parse_reader(std::io::Cursor::new(ttl), "ttl", 0).unwrap()
1476 );
1477 let trig = "@prefix ex: <http://ex/> .\nex:g { ex:A ex:knows ex:B . }\n";
1478 let streamed = parse_reader(std::io::Cursor::new(trig), "trig", 0).unwrap();
1479 assert_eq!(parse_statements(trig, "trig").unwrap(), streamed);
1480 assert_eq!(streamed[0].3.as_deref(), Some("<http://ex/g>"));
1481 // An unknown format is still a hard error, not a silent guess.
1482 assert!(parse_reader(std::io::Cursor::new(nt), "jsonld", 0).is_err());
1483 }
1484
1485 #[test]
1486 fn parse_statements_dispatches_by_format() {
1487 let nt = "<http://ex/a> <http://ex/p> <http://ex/b> .";
1488 assert_eq!(parse_statements(nt, "nt").unwrap().len(), 1);
1489 let ttl = "@prefix ex: <http://ex/> .\nex:A ex:knows ex:B , ex:C .";
1490 assert_eq!(parse_statements(ttl, "ttl").unwrap().len(), 2);
1491 // N-Triples is a subset of both Turtle and TriG, so the same text parses
1492 // under all three — and lands in the default graph under each.
1493 assert_eq!(parse_statements(nt, "trig").unwrap().len(), 1);
1494 assert!(parse_statements(nt, "trig").unwrap()[0].3.is_none());
1495 assert!(parse_statements(nt, "jsonld").is_err());
1496 }
1497
1498 /// A TriG graph block carries its graph name through as the same canonical
1499 /// token N-Quads would produce — so the two syntaxes agree on dictionary keys
1500 /// and a `.trig` shard federates with a `.nq` one.
1501 #[test]
1502 fn trig_graph_names_match_nquads() {
1503 let trig = "@prefix ex: <http://ex/> .\nex:g { ex:a ex:p ex:b . }\nex:a ex:p ex:c .\n";
1504 let nq = "<http://ex/a> <http://ex/p> <http://ex/b> <http://ex/g> .\n\
1505 <http://ex/a> <http://ex/p> <http://ex/c> .\n";
1506 let mut from_trig = parse_statements(trig, "trig").unwrap();
1507 let mut from_nq = parse_statements(nq, "nq").unwrap();
1508 from_trig.sort();
1509 from_nq.sort();
1510 assert_eq!(from_trig, from_nq);
1511 }
1512
1513 /// RDF/XML (how most OWL ontologies ship) parses to the same canonical tokens,
1514 /// including the abbreviated typed-node syntax and `rdf:resource` references.
1515 #[test]
1516 fn parses_rdfxml_owl() {
1517 let xml = r#"<?xml version="1.0"?>
1518 <rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
1519 xmlns:rdfs="http://www.w3.org/2000/01/rdf-schema#"
1520 xmlns:owl="http://www.w3.org/2002/07/owl#">
1521 <owl:Class rdf:about="http://ex/Dog">
1522 <rdfs:subClassOf rdf:resource="http://ex/Animal"/>
1523 <rdfs:label>Dog</rdfs:label>
1524 </owl:Class>
1525 </rdf:RDF>"#;
1526 let triples = parse_rdfxml(xml).unwrap();
1527 // rdf:type owl:Class, rdfs:subClassOf, rdfs:label = 3 triples.
1528 assert_eq!(triples.len(), 3);
1529 assert!(triples.iter().any(|(s, p, o)| s == "<http://ex/Dog>"
1530 && p == "<http://www.w3.org/2000/01/rdf-schema#subClassOf>"
1531 && o == "<http://ex/Animal>"));
1532 // Same data through the format dispatcher (triples → default graph).
1533 assert_eq!(parse_statements(xml, "rdfxml").unwrap().len(), 3);
1534 // Malformed XML is a clear RdfXml error, not a silent empty parse.
1535 assert!(matches!(
1536 parse_statements("<rdf:RDF><not closed", "rdfxml"),
1537 Err(IngestError::RdfXml(_))
1538 ));
1539 }
1540
1541 /// Regression coverage for RUSTSEC-2026-0195: namespace declarations from
1542 /// an untrusted RDF/XML document must not make parsing panic or exhaust the
1543 /// bounded test process.
1544 #[test]
1545 fn rdfxml_namespace_fanout_is_bounded() {
1546 let declarations = (0..2_000)
1547 .map(|i| format!(r#" xmlns:p{i}="http://example.test/{i}/""#))
1548 .collect::<String>();
1549 let xml = format!(
1550 r#"<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"{declarations}/>"#
1551 );
1552 let result = parse_statements(&xml, "rdfxml");
1553 assert!(result.is_ok() || result.unwrap_err().to_string().contains("namespace"));
1554 }
1555
1556 /// Regression coverage for RUSTSEC-2026-0194: duplicate-name checking must
1557 /// complete for a large valid start tag without panicking or exhausting the
1558 /// bounded test process.
1559 #[test]
1560 fn rdfxml_attribute_fanout_completes() {
1561 let attributes = (0..2_000)
1562 .map(|i| format!(r#" p:a{i}="{i}""#))
1563 .collect::<String>();
1564 let xml = format!(
1565 r#"<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns:p="http://example.test/"><rdf:Description rdf:about="http://example.test/s"{attributes}/></rdf:RDF>"#
1566 );
1567 assert!(parse_statements(&xml, "rdfxml").is_ok());
1568 }
1569
1570 /// Text in, queryable file image out — the whole in-memory build path the
1571 /// wasm `build()` binding uses, including a named graph.
1572 #[test]
1573 fn assemble_dataset_round_trips() {
1574 let text = "<http://ex/a> <http://ex/knows> <http://ex/b> .\n\
1575 <http://ex/b> <http://ex/knows> <http://ex/c> .\n\
1576 <http://ex/a> <http://ex/age> \"30\"^^<http://www.w3.org/2001/XMLSchema#integer> .\n\
1577 <http://ex/a> <http://ex/p> <http://ex/d> <http://ex/g1> .";
1578 let quads = parse_statements(text, "nq").unwrap();
1579 let (bytes, stats) = assemble_dataset(quads, &[]);
1580 assert_eq!(stats.statements, 4);
1581 assert_eq!(stats.default_triples, 3);
1582 assert_eq!(stats.named_graphs, 1);
1583 assert!(stats.terms >= 7);
1584
1585 let rete = Rete::open(&bytes).unwrap();
1586 assert_eq!(
1587 rete.query(None, Some("<http://ex/knows>"), None).len(),
1588 2,
1589 "default-graph pattern query"
1590 );
1591 assert_eq!(rete.graph_names(), vec!["<http://ex/g1>"]);
1592 let out = crate::eval_query(
1593 &rete,
1594 "SELECT ?x WHERE { ?x <http://ex/knows> ?y . ?y <http://ex/knows> ?z }",
1595 )
1596 .unwrap();
1597 match out {
1598 crate::QueryOutput::Select(_, rows) => assert_eq!(rows.len(), 1),
1599 other => panic!("expected select result, got {other:?}"),
1600 }
1601 }
1602
1603 /// The exact minimal input the playground Build tab uses: two triples, no
1604 /// literals, no rdf:type, no named graph — the smallest graph that still
1605 /// builds a community pyramid. (The wasm `build()` panic this guards was a
1606 /// `std::time::Instant::now()` in the pyramid timing path; native has a clock
1607 /// so this passes here, while the playground harness exercises the wasm path.)
1608 #[test]
1609 fn assemble_minimal_typeless_graph() {
1610 let text = "<http://ex/A> <http://ex/knows> <http://ex/B> .\n\
1611 <http://ex/B> <http://ex/knows> <http://ex/C> .\n";
1612 let quads = parse_statements(text, "nt").unwrap();
1613 let (bytes, stats) = assemble_dataset(quads, &[]);
1614 assert_eq!(stats.default_triples, 2);
1615 let rete = Rete::open(&bytes).unwrap();
1616 assert_eq!(rete.query(None, Some("<http://ex/knows>"), None).len(), 2);
1617 }
1618}