Skip to main content

rete_core/
ingest.rs

1//! Ingestion: parse RDF text (N-Triples / N-Quads / Turtle) and assemble a
2//! complete `.rete` file image. Shared by the CLI's `build`/`validate` commands
3//! and the wasm bindings (the playground's in-browser builder).
4//!
5//! The N-Triples/N-Quads reader is line-based and keeps terms as their exact
6//! canonical token strings (`<iri>`, `_:bnode`, `"lit"`, `"lit"^^<dt>`,
7//! `"lit"@lang`) so they double as dictionary keys and so a `query` can match
8//! by the same string. This is not a full RDF 1.1 parser — it covers the
9//! canonical N-Triples surface (and is deliberately tolerant of IRIs a strict
10//! parser would reject), enough for v0 ingestion. Turtle goes through `oxttl`.
11//!
12//! That tolerance is *measured*, not silent: pass an [`IriAudit`] to any of the
13//! `*_audited` entry points and the parse counts every IRI outside the
14//! N-Triples `IRIREF` grammar / RFC 3987 (see [`crate::iri`]) — or, with
15//! `strict`, refuses the input on the first one. `rete build` always audits.
16//!
17//! Assembly degrades with the build features: without the `compression`
18//! feature (e.g. on wasm) sections are written with the `NONE` codec — larger
19//! files, byte-compatible readers.
20
21use crate::{
22    build_pyramid_meta_algo, Dictionary, DictionaryBuilder, GraphIndexBuilder, PyramidAlgo,
23    DEFAULT_TILE_BUDGET,
24};
25
26/// A parsed triple as three canonical term tokens.
27pub type RawTriple = (String, String, String);
28
29/// A parsed quad: the triple plus an optional graph term (`None` = default graph).
30pub type RawQuad = (String, String, String, Option<String>);
31
32/// Which **surface** a `<< … >>` in Turtle/TriG input is read as — the input
33/// half of `rete export --quoted-triple-syntax`, same vocabulary, same two
34/// values.
35///
36/// The flag exists because the two standards give one piece of syntax two
37/// meanings, and no amount of looking at the bytes can tell them apart:
38///
39/// | written            | under [`RdfStar`](Self::RdfStar) | under [`Rdf12`](Self::Rdf12)          |
40/// |--------------------|----------------------------------|---------------------------------------|
41/// | `<< s p o >>`      | a **quoted triple**, one term    | a **reifier**: `_:r rdf:reifies <<( s p o )>>`, plus a blank node standing where it was |
42/// | `<<( s p o )>>`    | a syntax error                   | a **triple term**, object position only |
43/// | `{\| … \|}`         | a syntax error                   | an annotation on the statement before it |
44///
45/// One file, two readings, two different graphs — see the round-trip matrix in
46/// `crates/rete-cli/tests/quoted_triple_surfaces.rs`, which asserts both.
47///
48/// `<<( s p o )>>` is unambiguous: it is RDF 1.2's and nothing else's. That
49/// asymmetry is why [`RdfStar`](Self::RdfStar) is the **default**. Reading an
50/// RDF-star file as RDF 1.2 silently yields a different graph; reading an
51/// RDF 1.2 file as RDF-star is a hard parse error that names the flag. Only one
52/// of the two mistakes is survivable, so the default is the one that makes the
53/// other mistake loud.
54///
55/// **N-Triples and N-Quads ignore this entirely.** Their reader is rete's own
56/// (`take_term`), it has accepted both `<< s p o >>` and `<<( s p o )>>` since
57/// #262, and RDF 1.2 N-Triples has no reifier syntax for `<< … >>` to be — so
58/// there is no ambiguity to resolve and nothing to choose. The same goes for
59/// RDF/XML.
60///
61/// Whichever surface reads the file, what gets **stored** is the same canonical
62/// token `<<s p o>>`. rete's term model is a superset of both: a triple term is
63/// a term, and RDF 1.2 reification is ordinary RDF — a blank node, a predicate
64/// and a term — which rete already stored before this flag existed.
65#[derive(Clone, Copy, PartialEq, Eq, Debug, Default, Hash)]
66pub enum QuotedTripleSurface {
67    /// `<< s p o >>` is a **quoted triple** (RDF-star), in subject or object
68    /// position — rete's own storage token. The default, and byte-for-byte the
69    /// behaviour every rete release before this flag had.
70    #[default]
71    RdfStar,
72    /// `<<( s p o )>>` is a **triple term** and `<< s p o >>` is a **reifier**
73    /// (RDF 1.2), with `{| … |}` annotations read too.
74    Rdf12,
75}
76
77impl QuotedTripleSurface {
78    /// Parse the `--quoted-triple-syntax` value, `None` for anything else.
79    /// Shared by `rete build`, `rete validate` and `rete export` so the flag
80    /// cannot come to mean two things in two directions.
81    pub fn parse(s: &str) -> Option<Self> {
82        match s {
83            crate::card_derive::QUOTED_TRIPLE_SURFACE_RDF12 => Some(Self::Rdf12),
84            crate::card_derive::QUOTED_TRIPLE_SURFACE_RDF_STAR => Some(Self::RdfStar),
85            _ => None,
86        }
87    }
88
89    /// The flag spelling of this surface — the inverse of [`Self::parse`].
90    pub const fn as_str(self) -> &'static str {
91        match self {
92            Self::Rdf12 => crate::card_derive::QUOTED_TRIPLE_SURFACE_RDF12,
93            Self::RdfStar => crate::card_derive::QUOTED_TRIPLE_SURFACE_RDF_STAR,
94        }
95    }
96}
97
98/// The one-line hint appended to a Turtle/TriG parse error when the input looks
99/// like RDF 1.2 and was read as RDF-star — i.e. the exact mistake
100/// [`QuotedTripleSurface`]'s default makes, made loud.
101const RDF12_SURFACE_HINT: &str = "\nhint: this input contains `<<(`, the RDF 1.2 triple-term \
102     syntax — which is what `rete export` writes by default. Reading it takes \
103     `--quoted-triple-syntax rdf12`; the default, `rdf-star`, reads `<< s p o >>` as a quoted \
104     triple instead.";
105
106/// Does `text` look like it carries RDF 1.2 triple terms? A deliberately dumb
107/// substring test: `<<(` cannot begin anything else in Turtle, and this is only
108/// ever consulted to *explain a parse that already failed*, never to decide how
109/// to read a file. A `<<(` inside a string literal would make it a false
110/// positive on some other error's hint, which costs one confusing sentence and
111/// nothing else.
112fn looks_like_rdf12(text: &str) -> bool {
113    text.contains("<<(")
114}
115
116/// Why an ingest failed. Parser errors from `oxttl`/`oxrdfxml` and I/O errors
117/// are flattened to their `Display` string rather than wrapped: the variants
118/// then carry no foreign types, which keeps the error usable from the wasm
119/// bindings and stable across upstream parser upgrades.
120#[derive(Debug, thiserror::Error)]
121#[non_exhaustive]
122pub enum IngestError {
123    /// A malformed N-Triples/N-Quads line: its 1-based line number and a short
124    /// reason (`"bad subject"`, `"missing trailing '.'"`, …).
125    #[error("line {0}: {1}")]
126    Line(usize, &'static str),
127    /// The Turtle parser rejected the input.
128    #[error("turtle: {0}")]
129    Turtle(String),
130    /// The TriG parser rejected the input.
131    #[error("trig: {0}")]
132    TriG(String),
133    /// The RDF/XML parser rejected the input.
134    #[error("rdf/xml: {0}")]
135    RdfXml(String),
136    /// The requested format is not one this crate can parse.
137    #[error("unknown input format: {0} (expected nt, nq, ttl, trig, or rdf/xml)")]
138    UnknownFormat(String),
139    /// [`QuotedTripleSurface::Rdf12`] was asked for on a Turtle/TriG input, but
140    /// this build of `rete-core` has the `rdf12-turtle` feature off, so the
141    /// RDF 1.2 reader was not compiled in. Refused by name rather than read as
142    /// RDF-star, which would be the wrong graph told in silence.
143    #[error(
144        "this build cannot read the RDF 1.2 Turtle/TriG surface: rete-core was compiled without \
145         the `rdf12-turtle` feature (N-Triples/N-Quads still accept `<<( s p o )>>`)"
146    )]
147    Rdf12TurtleUnavailable,
148    /// The input could not be read (streaming builds surface the path here).
149    #[error("io: {0}")]
150    Io(String),
151    /// An IRI outside the N-Triples/N-Quads `IRIREF` grammar / RFC 3987, refused
152    /// because the caller asked for a strict [`IriAudit`] (`rete build
153    /// --strict`). Without `strict` the same finding is only counted.
154    /// `location` is `line N` for the line-based syntaxes, the format name for
155    /// the others (oxttl reports its own positions).
156    #[error("{location}: invalid IRI {token} — {reason}")]
157    InvalidIri {
158        /// Where it was seen: `line 42`, or the format name.
159        location: String,
160        /// The offending term token, verbatim.
161        token: String,
162        /// Which rule it breaks ([`crate::iri::IriDefect::reason`]).
163        reason: &'static str,
164    },
165}
166
167/// Policy **and** tally for the invalid-IRI check that runs during ingest.
168///
169/// rete's reader is deliberately tolerant of IRIs a strict parser rejects (see
170/// [`crate::iri`]), which is what lets it hold every published dataset — and
171/// also what let three of the `scholar/` exports produce N-Quads Oxigraph
172/// refuses to load. Passing an audit makes the tolerance *visible*: the build
173/// still succeeds and stores exactly what it was given, but it can now say how
174/// much of what it stored is not valid RDF. `strict` turns the same finding into
175/// a refusal, on the first offending statement.
176#[derive(Debug, Default)]
177pub struct IriAudit {
178    /// Fail the parse on the first invalid IRI instead of counting it.
179    pub strict: bool,
180    /// What the parse saw. Constant memory — see [`crate::iri::IriReport`].
181    pub report: crate::iri::IriReport,
182}
183
184impl IriAudit {
185    /// Count invalid IRIs; never fail (`rete build`'s default).
186    pub fn counting() -> Self {
187        Self::default()
188    }
189
190    /// Refuse the input at the first invalid IRI (`rete build --strict`).
191    pub fn refusing() -> Self {
192        Self {
193            strict: true,
194            report: crate::iri::IriReport::default(),
195        }
196    }
197}
198
199/// Run one statement past the audit: record any invalid IRI, and in `strict`
200/// mode turn it into an error naming the offending token. `location` is only
201/// called on the error path, so the common case allocates nothing.
202fn audit_quad<F: FnOnce() -> String>(
203    audit: Option<&mut IriAudit>,
204    s: &str,
205    p: &str,
206    o: &str,
207    g: Option<&str>,
208    location: F,
209) -> Result<(), IngestError> {
210    let Some(a) = audit else { return Ok(()) };
211    let Some(defect) = a.report.observe_quad(s, p, o, g) else {
212        return Ok(());
213    };
214    if !a.strict {
215        return Ok(());
216    }
217    let token = [Some(s), Some(p), Some(o), g]
218        .into_iter()
219        .flatten()
220        .find(|t| crate::iri::term_defect(t).is_some())
221        .unwrap_or("<?>")
222        .to_string();
223    Err(IngestError::InvalidIri {
224        location: location(),
225        token,
226        reason: defect.reason(),
227    })
228}
229
230/// Estimate the statement count of an N-Triples/N-Quads text from its newline
231/// count, so the output `Vec` can be pre-sized — a big build otherwise pays
232/// repeated `Vec` doublings, each briefly holding ~2× the (large) spine. An
233/// over-estimate by a few blank/comment lines is harmless (it's a capacity hint).
234fn estimate_statements(input: &str) -> usize {
235    bytecount_newlines(input).max(1)
236}
237
238/// Count `\n` bytes — one linear pass, far cheaper than the parse it sizes.
239fn bytecount_newlines(input: &str) -> usize {
240    input.as_bytes().iter().filter(|&&b| b == b'\n').count()
241}
242
243/// Parse one N-Quads line into a quad, or `None` for a blank/comment line.
244/// Shared by the whole-text [`parse_quads`] and the streaming [`parse_reader`].
245fn parse_nq_line(
246    raw: &str,
247    lineno: usize,
248    audit: Option<&mut IriAudit>,
249) -> Result<Option<RawQuad>, IngestError> {
250    let line = raw.trim();
251    if line.is_empty() || line.starts_with('#') {
252        return Ok(None);
253    }
254    let stripped = line
255        .strip_suffix('.')
256        .ok_or(IngestError::Line(lineno, "missing trailing '.'"))?
257        .trim_end();
258    let (s, rest) = take_term(stripped).ok_or(IngestError::Line(lineno, "bad subject"))?;
259    let (p, rest) =
260        take_term(rest.trim_start()).ok_or(IngestError::Line(lineno, "bad predicate"))?;
261    let (o, rest) = take_term(rest.trim_start()).ok_or(IngestError::Line(lineno, "bad object"))?;
262    let rest = rest.trim();
263    let graph = if rest.is_empty() {
264        None
265    } else {
266        let (g, tail) = take_term(rest).ok_or(IngestError::Line(lineno, "bad graph"))?;
267        if !tail.trim().is_empty() {
268            return Err(IngestError::Line(lineno, "trailing content after graph"));
269        }
270        Some(g)
271    };
272    audit_quad(audit, &s, &p, &o, graph.as_deref(), || {
273        format!("line {lineno}")
274    })?;
275    Ok(Some((s, p, o, graph)))
276}
277
278/// Parse one N-Triples line into a triple, or `None` for a blank/comment line.
279fn parse_nt_line(
280    raw: &str,
281    lineno: usize,
282    audit: Option<&mut IriAudit>,
283) -> Result<Option<RawTriple>, IngestError> {
284    let line = raw.trim();
285    if line.is_empty() || line.starts_with('#') {
286        return Ok(None);
287    }
288    let stripped = line
289        .strip_suffix('.')
290        .ok_or(IngestError::Line(lineno, "missing trailing '.'"))?
291        .trim_end();
292    let (s, rest) = take_term(stripped).ok_or(IngestError::Line(lineno, "bad subject"))?;
293    let (p, rest) =
294        take_term(rest.trim_start()).ok_or(IngestError::Line(lineno, "bad predicate"))?;
295    let (o, rest) = take_term(rest.trim_start()).ok_or(IngestError::Line(lineno, "bad object"))?;
296    if !rest.trim().is_empty() {
297        return Err(IngestError::Line(lineno, "trailing content after object"));
298    }
299    audit_quad(audit, &s, &p, &o, None, || format!("line {lineno}"))?;
300    Ok(Some((s, p, o)))
301}
302
303/// Parse N-Quads text: `subject predicate object [graph] .` per line.
304pub fn parse_quads(input: &str) -> Result<Vec<RawQuad>, IngestError> {
305    parse_quads_audited(input, None)
306}
307
308/// [`parse_quads`] with an optional [`IriAudit`].
309pub fn parse_quads_audited(
310    input: &str,
311    mut audit: Option<&mut IriAudit>,
312) -> Result<Vec<RawQuad>, IngestError> {
313    let mut out = Vec::with_capacity(estimate_statements(input));
314    for (i, raw) in input.lines().enumerate() {
315        if let Some(q) = parse_nq_line(raw, i + 1, audit.as_deref_mut())? {
316            out.push(q);
317        }
318    }
319    Ok(out)
320}
321
322/// Parse N-Triples text into raw term-token triples, skipping blank/comment lines.
323pub fn parse(input: &str) -> Result<Vec<RawTriple>, IngestError> {
324    parse_audited(input, None)
325}
326
327/// [`parse`] with an optional [`IriAudit`].
328pub fn parse_audited(
329    input: &str,
330    mut audit: Option<&mut IriAudit>,
331) -> Result<Vec<RawTriple>, IngestError> {
332    let mut out = Vec::with_capacity(estimate_statements(input));
333    for (i, raw) in input.lines().enumerate() {
334        if let Some(t) = parse_nt_line(raw, i + 1, audit.as_deref_mut())? {
335            out.push(t);
336        }
337    }
338    Ok(out)
339}
340
341/// **Stream-parse** N-Triples (`"nt"`) or N-Quads (`"nq"`) from a reader, one
342/// line at a time, so the whole input text is **never resident** — the big-build
343/// memory win over reading the file into a `String` first (each line String is
344/// transient, freed every iteration). `cap` pre-sizes the output `Vec` (e.g.
345/// `file_len / 64`) to avoid reallocation doublings of the large spine. Turtle is
346/// not streamable here (oxttl needs the whole input); callers use the text path
347/// for `"ttl"`.
348pub fn parse_reader<R: std::io::BufRead>(
349    reader: R,
350    format: &str,
351    cap: usize,
352) -> Result<Vec<RawQuad>, IngestError> {
353    parse_reader_audited(reader, format, cap, None)
354}
355
356/// [`parse_reader`] with an optional [`IriAudit`].
357pub fn parse_reader_audited<R: std::io::BufRead>(
358    reader: R,
359    format: &str,
360    cap: usize,
361    audit: Option<&mut IriAudit>,
362) -> Result<Vec<RawQuad>, IngestError> {
363    parse_reader_audited_surface(reader, format, cap, audit, QuotedTripleSurface::default())
364}
365
366/// [`parse_reader_audited`] reading `<< … >>` in the given
367/// [`QuotedTripleSurface`] (Turtle/TriG only — see the enum).
368pub fn parse_reader_audited_surface<R: std::io::BufRead>(
369    reader: R,
370    format: &str,
371    cap: usize,
372    audit: Option<&mut IriAudit>,
373    surface: QuotedTripleSurface,
374) -> Result<Vec<RawQuad>, IngestError> {
375    let mut out = Vec::with_capacity(cap);
376    stream_reader_audited_surface(reader, format, audit, surface, &mut |q| out.push(q))?;
377    Ok(out)
378}
379
380/// **Stream** N-Triples (`"nt"`) / N-Quads (`"nq"`) from a reader, invoking `f`
381/// once per parsed quad — like [`parse_reader`] but **without collecting** into a
382/// `Vec`. Each quad's term Strings are owned by `f` (and dropped when it returns
383/// if it doesn't retain them), so a caller that only needs to *observe* every
384/// term — e.g. the two-pass [`assemble_dataset_streaming`] building its
385/// dictionary — never materializes the whole quad multiset. The big-graph,
386/// low-RAM ingest primitive. Blank/comment lines are skipped; a parse error stops
387/// the stream and is returned.
388///
389/// Every format streams: the line-based ones a line at a time, Turtle / TriG /
390/// RDF-XML through their `oxttl`/`oxrdfxml` reader parsers, which pull from the
391/// reader incrementally. Nothing here holds the input text — so a 60 GB Turtle
392/// file is as ingestible as a 60 GB `.nt` one, which is what lets the external
393/// (memory-bounded) build accept them.
394pub fn stream_reader<R: std::io::BufRead>(
395    reader: R,
396    format: &str,
397    f: &mut dyn FnMut(RawQuad),
398) -> Result<(), IngestError> {
399    stream_reader_audited(reader, format, None, f)
400}
401
402/// [`stream_reader`] with an optional [`IriAudit`].
403///
404/// Every syntax is audited, not only the line-based ones. `oxttl`/`oxrdfxml`
405/// validate IRIs themselves and reject an invalid one before it reaches us, so
406/// in practice the Turtle/TriG/RDF-XML tallies stay at zero — but auditing them
407/// too means the count is a property of the *file*, not of which reader happened
408/// to produce it, and a future parser change cannot silently reopen the hole.
409pub fn stream_reader_audited<R: std::io::BufRead>(
410    reader: R,
411    format: &str,
412    audit: Option<&mut IriAudit>,
413    f: &mut dyn FnMut(RawQuad),
414) -> Result<(), IngestError> {
415    stream_reader_audited_surface(reader, format, audit, QuotedTripleSurface::default(), f)
416}
417
418/// [`stream_reader_audited`] reading `<< … >>` in the given
419/// [`QuotedTripleSurface`].
420///
421/// The surface reaches only the `"ttl"` and `"trig"` arms: the line-based
422/// readers are rete's own and take both spellings unconditionally, and RDF/XML
423/// has no syntax for either. Under [`QuotedTripleSurface::Rdf12`] Turtle/TriG go
424/// through `oxttl` 0.2 instead of 0.1, and the RDF 1.2 triple terms it hands
425/// back (`<<( s p o )>>`) are folded to rete's stored token on the way out, so
426/// the dictionary a file lands in does not depend on which reader read it.
427///
428/// **This entry point does not carry the RDF 1.2 hint** that the text path adds
429/// to a failed RDF-star parse: it is handed a reader, not a string, and sniffing
430/// the input would mean buffering the 60 GB file this function exists to avoid
431/// buffering. `rete build --memory-budget-mb file.ttl` therefore reports the raw
432/// parser error; every other route reports the hint.
433pub fn stream_reader_audited_surface<R: std::io::BufRead>(
434    reader: R,
435    format: &str,
436    mut audit: Option<&mut IriAudit>,
437    surface: QuotedTripleSurface,
438    f: &mut dyn FnMut(RawQuad),
439) -> Result<(), IngestError> {
440    if surface == QuotedTripleSurface::Rdf12 && matches!(format, "ttl" | "trig") {
441        return rdf12::stream(reader, format, audit, f);
442    }
443    match format {
444        "nt" | "nq" => {
445            for (i, line) in reader.lines().enumerate() {
446                let line = line.map_err(|e| IngestError::Io(e.to_string()))?;
447                if format == "nq" {
448                    if let Some(q) = parse_nq_line(&line, i + 1, audit.as_deref_mut())? {
449                        f(q);
450                    }
451                } else if let Some((s, p, o)) = parse_nt_line(&line, i + 1, audit.as_deref_mut())? {
452                    f((s, p, o, None));
453                }
454            }
455            Ok(())
456        }
457        "ttl" => turtle_star(reader, audit, f),
458        "trig" => trig_star(reader, audit, f),
459        "rdfxml" => {
460            for r in oxrdfxml::RdfXmlParser::new().for_reader(reader) {
461                let t = r.map_err(|e| IngestError::RdfXml(e.to_string()))?;
462                let q = (
463                    t.subject.to_string(),
464                    t.predicate.to_string(),
465                    t.object.to_string(),
466                    None,
467                );
468                audit_quad(audit.as_deref_mut(), &q.0, &q.1, &q.2, None, || {
469                    "rdf/xml".into()
470                })?;
471                f(q);
472            }
473            Ok(())
474        }
475        other => Err(IngestError::UnknownFormat(other.to_string())),
476    }
477}
478
479/// Stream **RDF-star** Turtle: `with_quoted_triples` accepts `<< s p o >>` in
480/// subject/object position, and oxrdf 0.2's `Term::Triple` Displays as the
481/// canonical `<<…>>` token rete's own N-Triples-star tokenizer emits — so no
482/// translation is needed on the way out.
483///
484/// Its own function, and called directly from both the reader path and the text
485/// path, so that the wasm engine — which reaches only the text path and has no
486/// RDF 1.2 reader compiled in — links this and nothing else.
487fn turtle_star<R: std::io::BufRead>(
488    reader: R,
489    mut audit: Option<&mut IriAudit>,
490    f: &mut dyn FnMut(RawQuad),
491) -> Result<(), IngestError> {
492    for r in oxttl::TurtleParser::new()
493        .with_quoted_triples()
494        .for_reader(reader)
495    {
496        let t = r.map_err(|e| IngestError::Turtle(e.to_string()))?;
497        let q = (
498            t.subject.to_string(),
499            t.predicate.to_string(),
500            t.object.to_string(),
501            None,
502        );
503        audit_quad(audit.as_deref_mut(), &q.0, &q.1, &q.2, None, || {
504            "turtle".into()
505        })?;
506        f(q);
507    }
508    Ok(())
509}
510
511/// [`turtle_star`] for TriG — Turtle plus named-graph blocks (`<g> { … }`), the
512/// shape most large RDF dumps ship in.
513fn trig_star<R: std::io::BufRead>(
514    reader: R,
515    mut audit: Option<&mut IriAudit>,
516    f: &mut dyn FnMut(RawQuad),
517) -> Result<(), IngestError> {
518    for r in oxttl::TriGParser::new()
519        .with_quoted_triples()
520        .for_reader(reader)
521    {
522        let q = r.map_err(|e| IngestError::TriG(e.to_string()))?;
523        let q = (
524            q.subject.to_string(),
525            q.predicate.to_string(),
526            q.object.to_string(),
527            graph_token(&q.graph_name),
528        );
529        audit_quad(
530            audit.as_deref_mut(),
531            &q.0,
532            &q.1,
533            &q.2,
534            q.3.as_deref(),
535            || "trig".into(),
536        )?;
537        f(q);
538    }
539    Ok(())
540}
541
542/// The canonical graph token for a parsed quad — `None` for the default graph,
543/// otherwise the same `<iri>` / `_:b` spelling the N-Quads reader produces, so
544/// TriG and N-Quads inputs land identical dictionary keys. `GraphName`'s own
545/// `Display` renders the default graph as `DEFAULT`, which must never become a
546/// term, hence the explicit match.
547fn graph_token(g: &oxrdf::GraphName) -> Option<String> {
548    match g {
549        oxrdf::GraphName::DefaultGraph => None,
550        other => Some(other.to_string()),
551    }
552}
553
554/// The **RDF 1.2** Turtle/TriG reader — `oxttl` 0.2 with its `rdf-12` feature,
555/// living beside the `oxttl` 0.1 reader above rather than replacing it.
556///
557/// Everything RDF 1.2 adds over RDF 1.1 arrives through this one module:
558/// `<<( s p o )>>` triple terms, `<< s p o >>` reifiers (which expand to
559/// `_:r rdf:reifies <<( s p o )>>` plus the statement that mentioned them),
560/// `{| … |}` annotations, and `"…"@lang--dir` directional literals. None of it
561/// needs a storage change, which is the point: a reifier is a blank node, an
562/// IRI and a term, and rete has stored those since v0.
563///
564/// The one translation is at the term boundary. `oxrdf` 0.3 renders a triple
565/// term as `<<( s p o )>>`; rete's dictionary key is `<<s p o>>`. [`take_term`]
566/// already reads both — that is what made RDF 1.2 N-Quads ingestible in #262 —
567/// so the fold is a re-scan of a token oxttl has already validated.
568#[cfg(feature = "rdf12-turtle")]
569mod rdf12 {
570    use super::{audit_quad, take_term, IngestError, IriAudit, RawQuad};
571
572    /// Fold `oxrdf` 0.3's `<<( s p o )>>` rendering of a triple term into rete's
573    /// stored `<<s p o>>`, recursively (a nested triple term folds with it).
574    /// Every other token — the overwhelming majority, and all of them in a file
575    /// with no triple terms — is returned untouched, not even copied.
576    pub(super) fn fold_triple_term(t: String) -> String {
577        if !t.starts_with("<<(") {
578            return t;
579        }
580        match take_term(&t) {
581            Some((token, rest)) if rest.trim().is_empty() => token,
582            // oxttl only ever hands back a well-formed term, so this is
583            // unreachable; returning the original beats mangling it silently.
584            _ => t,
585        }
586    }
587
588    /// `super::graph_token` for `oxrdf` 0.3's `GraphName` — the same rule, a
589    /// different crate version, which is exactly why it cannot be shared.
590    fn graph_token(g: &oxrdf12::GraphName) -> Option<String> {
591        match g {
592            oxrdf12::GraphName::DefaultGraph => None,
593            other => Some(other.to_string()),
594        }
595    }
596
597    /// Stream one RDF 1.2 Turtle (`"ttl"`) or TriG (`"trig"`) input, auditing
598    /// every quad the way the RDF-star reader does.
599    pub(super) fn stream<R: std::io::BufRead>(
600        reader: R,
601        format: &str,
602        mut audit: Option<&mut IriAudit>,
603        f: &mut dyn FnMut(RawQuad),
604    ) -> Result<(), IngestError> {
605        if format == "ttl" {
606            for r in oxttl12::TurtleParser::new().for_reader(reader) {
607                let t = r.map_err(|e| IngestError::Turtle(e.to_string()))?;
608                let q = (
609                    t.subject.to_string(),
610                    t.predicate.to_string(),
611                    fold_triple_term(t.object.to_string()),
612                    None,
613                );
614                audit_quad(audit.as_deref_mut(), &q.0, &q.1, &q.2, None, || {
615                    "turtle".into()
616                })?;
617                f(q);
618            }
619        } else {
620            for r in oxttl12::TriGParser::new().for_reader(reader) {
621                let q = r.map_err(|e| IngestError::TriG(e.to_string()))?;
622                let q = (
623                    q.subject.to_string(),
624                    q.predicate.to_string(),
625                    fold_triple_term(q.object.to_string()),
626                    graph_token(&q.graph_name),
627                );
628                audit_quad(
629                    audit.as_deref_mut(),
630                    &q.0,
631                    &q.1,
632                    &q.2,
633                    q.3.as_deref(),
634                    || "trig".into(),
635                )?;
636                f(q);
637            }
638        }
639        Ok(())
640    }
641}
642
643/// The stand-in for [`rdf12`] in a build with `rdf12-turtle` off: it refuses by
644/// name. Reading the file as RDF-star instead would answer a question nobody
645/// asked, with a different graph.
646#[cfg(not(feature = "rdf12-turtle"))]
647mod rdf12 {
648    use super::{IngestError, IriAudit, RawQuad};
649
650    pub(super) fn stream<R: std::io::BufRead>(
651        _reader: R,
652        _format: &str,
653        _audit: Option<&mut IriAudit>,
654        _f: &mut dyn FnMut(RawQuad),
655    ) -> Result<(), IngestError> {
656        Err(IngestError::Rdf12TurtleUnavailable)
657    }
658}
659
660/// Parse Turtle into canonical N-Triples-token triples via oxttl, reading
661/// `<< s p o >>` as an RDF-star quoted triple. [`parse_turtle_surface`] chooses.
662pub fn parse_turtle(text: &str) -> Result<Vec<RawTriple>, IngestError> {
663    parse_turtle_surface(text, QuotedTripleSurface::default())
664}
665
666/// [`parse_turtle`] in the given [`QuotedTripleSurface`].
667pub fn parse_turtle_surface(
668    text: &str,
669    surface: QuotedTripleSurface,
670) -> Result<Vec<RawTriple>, IngestError> {
671    Ok(parse_text_surface(text, "ttl", surface)?
672        .into_iter()
673        .map(|(s, p, o, _)| (s, p, o))
674        .collect())
675}
676
677/// Parse TriG into canonical quads via oxttl — Turtle plus named-graph blocks
678/// (`<g> { … }`), the shape most large RDF dumps ship in. Reads `<< s p o >>` as
679/// an RDF-star quoted triple; [`parse_trig_surface`] chooses.
680pub fn parse_trig(text: &str) -> Result<Vec<RawQuad>, IngestError> {
681    parse_trig_surface(text, QuotedTripleSurface::default())
682}
683
684/// [`parse_trig`] in the given [`QuotedTripleSurface`].
685pub fn parse_trig_surface(
686    text: &str,
687    surface: QuotedTripleSurface,
688) -> Result<Vec<RawQuad>, IngestError> {
689    parse_text_surface(text, "trig", surface)
690}
691
692/// The shared Turtle/TriG text path: stream the string through whichever reader
693/// `surface` selects, and — this is the part only the text path can do, because
694/// only it holds the input — say so when an RDF-star parse fails on what is
695/// plainly RDF 1.2.
696///
697/// It does not audit: the Turtle/TriG text callers audit over the finished quads
698/// (see [`parse_statements_audited_surface`]), and auditing here too would count
699/// every IRI twice.
700///
701/// The RDF-star arm calls `oxttl` 0.1 **directly** rather than routing through
702/// [`stream_reader_audited_surface`]. That is not style: this is the only ingest
703/// path the wasm engine reaches, and going through the generic reader would
704/// monomorphize it over `&[u8]` and pull the whole `match format` — every
705/// syntax's reader — into a module that previously linked one Turtle parser.
706/// Measured at +17 KB of wasm for code no browser can run, since the RDF 1.2
707/// reader is not compiled there at all. Keeping the arm direct keeps the
708/// browser artifacts what they were.
709fn parse_text_surface(
710    text: &str,
711    format: &str,
712    surface: QuotedTripleSurface,
713) -> Result<Vec<RawQuad>, IngestError> {
714    let mut out = Vec::new();
715    let r = if surface == QuotedTripleSurface::Rdf12 {
716        rdf12::stream(text.as_bytes(), format, None, &mut |q| out.push(q))
717    } else if format == "ttl" {
718        turtle_star(text.as_bytes(), None, &mut |q| out.push(q))
719    } else {
720        trig_star(text.as_bytes(), None, &mut |q| out.push(q))
721    };
722    match r {
723        Ok(()) => Ok(out),
724        // The default surface met `<<(`: that is an RDF 1.2 file being read as
725        // RDF-star, the one mistake this flag exists to prevent, and the parse
726        // error alone ("unexpected character") would send the reader hunting for
727        // a typo that is not there.
728        Err(IngestError::Turtle(m))
729            if surface == QuotedTripleSurface::RdfStar && looks_like_rdf12(text) =>
730        {
731            Err(IngestError::Turtle(format!("{m}{RDF12_SURFACE_HINT}")))
732        }
733        Err(IngestError::TriG(m))
734            if surface == QuotedTripleSurface::RdfStar && looks_like_rdf12(text) =>
735        {
736            Err(IngestError::TriG(format!("{m}{RDF12_SURFACE_HINT}")))
737        }
738        Err(e) => Err(e),
739    }
740}
741
742/// Parse RDF/XML into canonical N-Triples-token triples via oxrdfxml. This is how
743/// most OWL ontologies ship (`.rdf`/`.owl`/`.xml` with an `rdf:RDF` root) — so rete
744/// ingests them directly, no external conversion. (OWL/XML — the non-RDF functional
745/// XML serialization — is a different language; convert it with owlready2 first.)
746pub fn parse_rdfxml(text: &str) -> Result<Vec<RawTriple>, IngestError> {
747    let mut out = Vec::new();
748    for r in oxrdfxml::RdfXmlParser::new().for_reader(text.as_bytes()) {
749        let t = r.map_err(|e| IngestError::RdfXml(e.to_string()))?;
750        out.push((
751            t.subject.to_string(),
752            t.predicate.to_string(),
753            t.object.to_string(),
754        ));
755    }
756    Ok(out)
757}
758
759/// Parse one text input by format name (`"nt"`, `"nq"`, `"ttl"`, `"trig"`, or
760/// `"rdfxml"`) into quads (triples land in the default graph).
761pub fn parse_statements(text: &str, format: &str) -> Result<Vec<RawQuad>, IngestError> {
762    parse_statements_audited(text, format, None)
763}
764
765/// [`parse_statements`] with an optional [`IriAudit`]. The line-based syntaxes
766/// audit as they parse (so `strict` fails on the offending line); the others are
767/// audited over the parsed quads, which is where `oxttl` has already had its say.
768pub fn parse_statements_audited(
769    text: &str,
770    format: &str,
771    audit: Option<&mut IriAudit>,
772) -> Result<Vec<RawQuad>, IngestError> {
773    parse_statements_audited_surface(text, format, audit, QuotedTripleSurface::default())
774}
775
776/// [`parse_statements_audited`] reading `<< … >>` in the given
777/// [`QuotedTripleSurface`]. This is the entry point `rete build` and
778/// `rete validate` use for everything but a streamed line-based input, and the
779/// one that turns "RDF 1.2 file, RDF-star reader" into an error that names the
780/// flag instead of a bare syntax complaint.
781pub fn parse_statements_audited_surface(
782    text: &str,
783    format: &str,
784    mut audit: Option<&mut IriAudit>,
785    surface: QuotedTripleSurface,
786) -> Result<Vec<RawQuad>, IngestError> {
787    let quads = match format {
788        "nq" => return parse_quads_audited(text, audit),
789        "nt" => {
790            return Ok(parse_audited(text, audit)?
791                .into_iter()
792                .map(|(s, p, o)| (s, p, o, None))
793                .collect())
794        }
795        // Straight to the quad path for both: going via `parse_turtle_surface`
796        // would build a `Vec<RawQuad>`, rebuild it as a `Vec<RawTriple>` to
797        // satisfy that function's signature, and rebuild it as quads again —
798        // three copies of every term String on the biggest input rete takes.
799        "ttl" | "trig" => parse_text_surface(text, format, surface)?,
800        "rdfxml" => parse_rdfxml(text)?
801            .into_iter()
802            .map(|(s, p, o)| (s, p, o, None))
803            .collect(),
804        other => return Err(IngestError::UnknownFormat(other.to_string())),
805    };
806    for (s, p, o, g) in &quads {
807        audit_quad(audit.as_deref_mut(), s, p, o, g.as_deref(), || {
808            format.to_string()
809        })?;
810    }
811    Ok(quads)
812}
813
814/// Take one term from the front of `s`, returning `(term, remainder)`.
815pub(crate) fn take_term(s: &str) -> Option<(String, &str)> {
816    let bytes = s.as_bytes();
817    let first = *bytes.first()?;
818    match first {
819        // A quoted triple / triple term, in EITHER surface — the inner terms are
820        // themselves terms (so this recurses; nesting works). `<<` starts with `<`
821        // and an IRI scan would stop at the first inner `>`, so it must be handled
822        // before the plain-IRI case.
823        //   * RDF-star:  `<< subject predicate object >>`
824        //   * RDF 1.2:   `<<( subject predicate object )>>`  (triple term)
825        // Both re-emit the SAME canonical token `<<s p o>>`, so a file written in
826        // either surface — and a query written either way — dedupe and match. This
827        // makes RDF 1.2 N-Triples interoperable with the RDF-star we already store,
828        // with no format change and no dependency swap.
829        b'<' if bytes.get(1) == Some(&b'<') => {
830            // Distinguish `<<(` (RDF 1.2) from `<<` (RDF-star) by the char after `<<`.
831            let (inner, rdf12) = match s[2..].strip_prefix('(') {
832                Some(after) => (after.trim_start(), true),
833                None => (s[2..].trim_start(), false),
834            };
835            let (subj, r) = take_term(inner)?;
836            let (pred, r) = take_term(r.trim_start())?;
837            let (obj, r) = take_term(r.trim_start())?;
838            let r = r.trim_start();
839            let rest = if rdf12 {
840                r.strip_prefix(")>>")?
841            } else {
842                r.strip_prefix(">>")?
843            };
844            // Canonical surface = oxrdf's `Triple` Display: `<<s p o>>` (tight
845            // brackets, single spaces between components) — the same token for both
846            // input surfaces, so N-Triples-star, Turtle-star, and RDF 1.2 triple
847            // terms all resolve to one dictionary entry.
848            Some((format!("<<{subj} {pred} {obj}>>"), rest))
849        }
850        b'<' => {
851            // IRI ref: up to the closing '>'.
852            let end = s.find('>')?;
853            Some((s[..=end].to_string(), &s[end + 1..]))
854        }
855        b'_' => {
856            // Blank node: up to whitespace.
857            let end = s.find(char::is_whitespace).unwrap_or(s.len());
858            Some((s[..end].to_string(), &s[end..]))
859        }
860        b'"' => {
861            // Literal: closing unescaped quote, then optional ^^<dt> or @lang.
862            let mut i = 1;
863            let b = s.as_bytes();
864            while i < b.len() {
865                match b[i] {
866                    b'\\' => i += 2, // skip escaped char
867                    b'"' => break,
868                    _ => i += 1,
869                }
870            }
871            if i >= b.len() {
872                return None; // unterminated
873            }
874            let mut end = i + 1; // past closing quote
875            if s[end..].starts_with("^^<") {
876                let close = s[end..].find('>')? + end;
877                end = close + 1;
878            } else if s[end..].starts_with('@') {
879                // Language tag: '@' then BCP-47 subtags `[a-zA-Z0-9-]+`. Stop at
880                // the first char that can't be part of a tag — normally the
881                // whitespace before the predicate, but ALSO the `>>` that closes
882                // a quoted triple when this literal is its object (`"x"@en>>`),
883                // where there is no separating whitespace.
884                let mut j = end + 1; // past '@'
885                while j < b.len() && (b[j].is_ascii_alphanumeric() || b[j] == b'-') {
886                    j += 1;
887                }
888                end = j;
889            }
890            Some((s[..end].to_string(), &s[end..]))
891        }
892        _ => None,
893    }
894}
895
896/// Split a canonical quoted-triple token `<<s p o>>` (RDF-star) into its three
897/// component term tokens, or `None` if `t` is not a quoted triple. Reuses
898/// [`take_term`] for term-boundary scanning, so nested quoting parses correctly.
899/// The inverse of the `<<…>>` construction; used by the SUBJECT/PREDICATE/OBJECT
900/// SPARQL-star builtins.
901pub(crate) fn quoted_triple_parts(t: &str) -> Option<(String, String, String)> {
902    let inner = t.strip_prefix("<<")?.strip_suffix(">>")?.trim();
903    let (s, r) = take_term(inner)?;
904    let (p, r) = take_term(r.trim_start())?;
905    let (o, r) = take_term(r.trim_start())?;
906    if !r.trim().is_empty() {
907        return None;
908    }
909    Some((s, p, o))
910}
911
912/// Counts describing an assembled file, for status lines and UIs.
913///
914/// In the [`BuildStats`] a metadata callback receives, `statements` and
915/// `default_triples` are the counts **ingested** — the input multiset, before
916/// the index deduplicates it. In the `BuildStats` an assemble function
917/// *returns*, they are the counts actually **written** (see
918/// [`FinalCounts`]), so a status line reports what the file holds. Use
919/// [`FinalCounts`] rather than these when a payload must agree with the header.
920#[derive(Debug, Clone, Copy)]
921pub struct BuildStats {
922    /// Total statements, across the default graph and every named one.
923    pub statements: usize,
924    /// Statements that landed in the default graph.
925    pub default_triples: usize,
926    /// How many named graphs the input mentioned; one index is written per graph.
927    pub named_graphs: usize,
928    /// Distinct terms in the shared dictionary.
929    pub terms: usize,
930    /// Levels in the community pyramid, or `0` when the build skipped it.
931    pub pyramid_levels: u16,
932}
933
934/// The **deduplicated** counts of the file being written — what the header
935/// records, and therefore what any embedded metadata has to report.
936///
937/// The input multiset is not the file: every permutation index sorts and
938/// deduplicates, so a harvest that pages with overlapping windows writes fewer
939/// quads than it ingested. These counts are known only once the indexes exist,
940/// which is why a payload that carries them is derived in two stages (see
941/// [`IntoMetadata`]).
942#[derive(Debug, Clone, Copy)]
943pub struct FinalCounts {
944    /// Unique triples in the default graph.
945    pub default_triples: u64,
946    /// Unique quads across the default graph and every named graph — exactly
947    /// the header's `quad_count`.
948    pub quads: u64,
949}
950
951/// What a build's metadata callback may return.
952///
953/// A `Vec<u8>` is the payload verbatim — the common case (no metadata, or a
954/// payload that carries no counts). [`DeferredMetadata`] is a payload that
955/// *does* carry counts: the callback derives everything it needs while the
956/// source quads are still resident, and returns a finalizer that the writer
957/// calls with the [`FinalCounts`], which only exist once the indexes have
958/// deduplicated the input. Without that second stage a card can only report the
959/// ingested count, which over-reports any input containing duplicates.
960pub trait IntoMetadata {
961    /// Produce the metadata section payload for a file with these counts.
962    fn into_metadata(self, counts: FinalCounts) -> Vec<u8>;
963}
964
965impl IntoMetadata for Vec<u8> {
966    fn into_metadata(self, _counts: FinalCounts) -> Vec<u8> {
967        self
968    }
969}
970
971/// A metadata payload whose final form needs the written (deduplicated) counts.
972/// See [`IntoMetadata`].
973pub struct DeferredMetadata(Box<dyn FnOnce(FinalCounts) -> Vec<u8>>);
974
975impl DeferredMetadata {
976    /// Defer `f` until the indexes are built and the true counts are known.
977    pub fn new(f: impl FnOnce(FinalCounts) -> Vec<u8> + 'static) -> Self {
978        DeferredMetadata(Box::new(f))
979    }
980
981    /// No metadata section — byte-identical to a metadata-free build. Lets one
982    /// `match` arm opt out while the other defers, without a type mismatch.
983    pub fn none() -> Self {
984        DeferredMetadata::new(|_| Vec::new())
985    }
986}
987
988impl IntoMetadata for DeferredMetadata {
989    fn into_metadata(self, counts: FinalCounts) -> Vec<u8> {
990        (self.0)(counts)
991    }
992}
993
994/// Assemble a complete `.rete` file image from parsed quads: one shared
995/// dictionary, the default-graph index, one index per named graph, and the
996/// community pyramid. `metadata` is the opaque metadata-section payload (the
997/// CLI puts a JSON Dataset Card there); pass `&[]` for none — that is
998/// byte-identical to a metadata-free build.
999pub fn assemble_dataset(quads: Vec<RawQuad>, metadata: &[u8]) -> (Vec<u8>, BuildStats) {
1000    let blob = metadata.to_vec();
1001    assemble_dataset_with(quads, move |_, _| blob)
1002}
1003
1004/// Like [`assemble_dataset`], but the metadata payload is derived from the
1005/// [`BuildStats`] and the source quads while they are still resident — for
1006/// metadata that describes the graph (the Dataset Card). Returning an empty
1007/// `Vec` is byte-identical to a metadata-free build; return a
1008/// [`DeferredMetadata`] for a payload that must also carry the file's
1009/// deduplicated counts, which exist only once the indexes are built.
1010pub fn assemble_dataset_with<M: IntoMetadata>(
1011    quads: Vec<RawQuad>,
1012    metadata: impl FnOnce(&BuildStats, &[RawQuad]) -> M,
1013) -> (Vec<u8>, BuildStats) {
1014    assemble_dataset_with_opts(quads, true, false, None, metadata)
1015}
1016
1017/// Like [`assemble_dataset_with`], but `with_pyramid = false` skips the Louvain
1018/// community pyramid entirely — no pyramid section is written (header length 0).
1019/// SPARQL / SHACL / triple / reachability queries don't use the pyramid, so a
1020/// pyramid-less file is fully queryable and markedly smaller (the pyramid is the
1021/// largest section on highly-connected graphs). Only the community / summary /
1022/// progressive paths need it.
1023pub fn assemble_dataset_with_opts<M: IntoMetadata>(
1024    quads: Vec<RawQuad>,
1025    with_pyramid: bool,
1026    with_text_index: bool,
1027    type_override: Option<&str>,
1028    metadata: impl FnOnce(&BuildStats, &[RawQuad]) -> M,
1029) -> (Vec<u8>, BuildStats) {
1030    assemble_dataset_with_opts_algo(
1031        quads,
1032        with_pyramid,
1033        with_text_index,
1034        type_override,
1035        PyramidAlgo::Louvain,
1036        metadata,
1037    )
1038}
1039
1040/// Like [`assemble_dataset_with_opts`], but selects the community [`PyramidAlgo`]
1041/// (the in-memory build path for `rete build --pyramid-algo …`).
1042#[allow(clippy::too_many_arguments)]
1043pub fn assemble_dataset_with_opts_algo<M: IntoMetadata>(
1044    quads: Vec<RawQuad>,
1045    with_pyramid: bool,
1046    with_text_index: bool,
1047    type_override: Option<&str>,
1048    algo: PyramidAlgo,
1049    metadata: impl FnOnce(&BuildStats, &[RawQuad]) -> M,
1050) -> (Vec<u8>, BuildStats) {
1051    assemble_dataset_with_perms(
1052        quads,
1053        with_pyramid,
1054        with_text_index,
1055        type_override,
1056        algo,
1057        crate::index::PermSet::ALL,
1058        metadata,
1059    )
1060}
1061
1062/// Like [`assemble_dataset_with_opts_algo`], but writes only the permutations in
1063/// `perms` (`rete build --permutations 3`). Every other byte of the file is
1064/// unchanged; [`crate::index::PermSet::ALL`] reproduces the default build
1065/// exactly, down to the header byte.
1066#[allow(clippy::too_many_arguments)]
1067pub fn assemble_dataset_with_perms<M: IntoMetadata>(
1068    quads: Vec<RawQuad>,
1069    with_pyramid: bool,
1070    with_text_index: bool,
1071    type_override: Option<&str>,
1072    algo: PyramidAlgo,
1073    perms: crate::index::PermSet,
1074    metadata: impl FnOnce(&BuildStats, &[RawQuad]) -> M,
1075) -> (Vec<u8>, BuildStats) {
1076    use std::collections::BTreeMap;
1077
1078    let mut db = DictionaryBuilder::new();
1079    for (s, p, o, _) in &quads {
1080        db.observe(s, p, o);
1081    }
1082    let dict = db.build();
1083
1084    let mut default_triples = Vec::new();
1085    let mut named: BTreeMap<String, Vec<(u32, u32, u32)>> = BTreeMap::new();
1086    for (s, p, o, g) in &quads {
1087        let t = dict.encode(s, p, o).expect("observed term");
1088        match g {
1089            None => default_triples.push(t),
1090            Some(graph) => named.entry(graph.clone()).or_default().push(t),
1091        }
1092    }
1093
1094    // Derive the metadata blob (the Dataset Card) from the raw quads NOW, while
1095    // they are resident — then DROP them before the memory-heavy pyramid + index
1096    // phases. On a big build the string quads are the largest working set (every
1097    // term an owned String, heavily duplicated) and are fully redundant with the
1098    // dictionary + id-triples once encoded, so freeing them here cuts peak RAM by
1099    // their whole size. `pyramid_levels` is not known yet (0 in the callback);
1100    // it is filled into the returned `stats` by `finish_assembly`, and no metadata
1101    // callback depends on it (the card derives from the quads + term/graph counts).
1102    let stats = BuildStats {
1103        statements: quads.len(),
1104        default_triples: default_triples.len(),
1105        named_graphs: named.len(),
1106        terms: dict.term_count() as usize,
1107        pyramid_levels: 0,
1108    };
1109    let pending = metadata(&stats, &quads);
1110    drop(quads);
1111
1112    finish_assembly(
1113        dict,
1114        default_triples,
1115        named,
1116        with_pyramid,
1117        with_text_index,
1118        type_override,
1119        algo,
1120        perms,
1121        pending,
1122        stats,
1123    )
1124}
1125
1126/// **Two-pass, low-RAM** assembly: build a `.rete` by **streaming** the input(s)
1127/// twice instead of holding every parsed quad in memory. `stream` is invoked
1128/// **twice** and MUST replay the exact same quads in the same order each time —
1129/// pass 1 observes every term into the dictionary; pass 2 encodes them to
1130/// id-triples. The raw string quads (every term an owned String, heavily
1131/// duplicated — by far the largest working set on a big graph) are **never
1132/// collected**, so peak RAM is bounded by the dictionary + id-triples + index
1133/// rather than the string-quad multiset. Output is **byte-identical** to
1134/// [`assemble_dataset_with_opts`] on the same quads (same dictionary, same
1135/// id-triples in file order, same downstream pipeline).
1136///
1137/// The metadata callback derives the Dataset Card from the built dictionary +
1138/// default-graph id-triples (resolving terms through the dictionary), since the
1139/// raw quads were never retained. `stream` propagates parse/IO errors.
1140pub fn assemble_dataset_streaming<S, M: IntoMetadata>(
1141    stream: S,
1142    with_pyramid: bool,
1143    with_text_index: bool,
1144    type_override: Option<&str>,
1145    metadata: impl FnOnce(&BuildStats, &Dictionary, &[(u32, u32, u32)]) -> M,
1146) -> Result<(Vec<u8>, BuildStats), IngestError>
1147where
1148    S: FnMut(&mut dyn FnMut(RawQuad)) -> Result<(), IngestError>,
1149{
1150    assemble_dataset_streaming_algo(
1151        stream,
1152        with_pyramid,
1153        with_text_index,
1154        type_override,
1155        PyramidAlgo::Louvain,
1156        metadata,
1157    )
1158}
1159
1160/// Like [`assemble_dataset_streaming`], but selects the community [`PyramidAlgo`]
1161/// (the streaming, low-RAM build path for `rete build --pyramid-algo …`).
1162#[allow(clippy::too_many_arguments)]
1163pub fn assemble_dataset_streaming_algo<S, M: IntoMetadata>(
1164    stream: S,
1165    with_pyramid: bool,
1166    with_text_index: bool,
1167    type_override: Option<&str>,
1168    algo: PyramidAlgo,
1169    metadata: impl FnOnce(&BuildStats, &Dictionary, &[(u32, u32, u32)]) -> M,
1170) -> Result<(Vec<u8>, BuildStats), IngestError>
1171where
1172    S: FnMut(&mut dyn FnMut(RawQuad)) -> Result<(), IngestError>,
1173{
1174    assemble_dataset_streaming_with_perms(
1175        stream,
1176        with_pyramid,
1177        with_text_index,
1178        type_override,
1179        algo,
1180        crate::index::PermSet::ALL,
1181        metadata,
1182    )
1183}
1184
1185/// Like [`assemble_dataset_streaming_algo`], but writes only the permutations in
1186/// `perms` — the two-pass low-RAM twin of [`assemble_dataset_with_perms`].
1187#[allow(clippy::too_many_arguments)]
1188pub fn assemble_dataset_streaming_with_perms<S, M: IntoMetadata>(
1189    mut stream: S,
1190    with_pyramid: bool,
1191    with_text_index: bool,
1192    type_override: Option<&str>,
1193    algo: PyramidAlgo,
1194    perms: crate::index::PermSet,
1195    metadata: impl FnOnce(&BuildStats, &Dictionary, &[(u32, u32, u32)]) -> M,
1196) -> Result<(Vec<u8>, BuildStats), IngestError>
1197where
1198    S: FnMut(&mut dyn FnMut(RawQuad)) -> Result<(), IngestError>,
1199{
1200    use std::collections::BTreeMap;
1201
1202    // Pass 1: observe every term (the dictionary dedups; the string quads are
1203    // freed line by line and never collected).
1204    let mut db = DictionaryBuilder::new();
1205    stream(&mut |(s, p, o, _g)| db.observe(&s, &p, &o))?;
1206    let dict = db.build();
1207
1208    // Pass 2: encode each quad to an id-triple, bucketing named graphs.
1209    let mut default_triples: Vec<(u32, u32, u32)> = Vec::new();
1210    let mut named: BTreeMap<String, Vec<(u32, u32, u32)>> = BTreeMap::new();
1211    stream(&mut |(s, p, o, g)| {
1212        let t = dict.encode(&s, &p, &o).expect("observed term");
1213        match g {
1214            None => default_triples.push(t),
1215            Some(graph) => named.entry(graph).or_default().push(t),
1216        }
1217    })?;
1218
1219    let statements = default_triples.len() + named.values().map(Vec::len).sum::<usize>();
1220    let stats = BuildStats {
1221        statements,
1222        default_triples: default_triples.len(),
1223        named_graphs: named.len(),
1224        terms: dict.term_count() as usize,
1225        pyramid_levels: 0,
1226    };
1227    let pending = metadata(&stats, &dict, &default_triples);
1228    Ok(finish_assembly(
1229        dict,
1230        default_triples,
1231        named,
1232        with_pyramid,
1233        with_text_index,
1234        type_override,
1235        algo,
1236        perms,
1237        pending,
1238        stats,
1239    ))
1240}
1241
1242/// The shared tail of every build path: from a finished dictionary + encoded
1243/// id-triples (default graph + named), build the community pyramid, the optional
1244/// full-text index, the permutation indexes, and serialize the file image. Takes
1245/// the id-triples **by value** so they are freed as the permutations consume them.
1246#[allow(clippy::too_many_arguments)]
1247fn finish_assembly<M: IntoMetadata>(
1248    dict: Dictionary,
1249    default_triples: Vec<(u32, u32, u32)>,
1250    named: std::collections::BTreeMap<String, Vec<(u32, u32, u32)>>,
1251    with_pyramid: bool,
1252    with_text_index: bool,
1253    type_override: Option<&str>,
1254    algo: PyramidAlgo,
1255    perms: crate::index::PermSet,
1256    pending: M,
1257    mut stats: BuildStats,
1258) -> (Vec<u8>, BuildStats) {
1259    let has_named = !named.is_empty();
1260
1261    // The pyramid and the full-text index are built FIRST, while the dictionary and
1262    // the default id-triples are both still resident — both need the two together.
1263    let (meta, levels) = if with_pyramid {
1264        build_pyramid_meta_algo(
1265            &dict,
1266            &default_triples,
1267            DEFAULT_TILE_BUDGET,
1268            type_override,
1269            algo,
1270        )
1271    } else {
1272        (Vec::new(), 0)
1273    };
1274    stats.pyramid_levels = levels;
1275
1276    let text_index = if with_text_index {
1277        crate::file::compute_text_index(&dict, &default_triples)
1278    } else {
1279        Vec::new()
1280    };
1281
1282    // From here the dictionary is only needed for its own serialized bytes. Encode
1283    // it, capture the header term count, then DROP it before building the
1284    // permutation indexes — which work purely on id-triples and never touch the
1285    // dictionary. On a large graph this frees the single biggest resident structure
1286    // (the dictionary) right when the index sort needs the headroom. The output is
1287    // byte-for-byte identical to serializing the dictionary inline.
1288    let codec = crate::file::writer_codec();
1289    let dict_container = crate::file::encode_dict_container(&dict, codec);
1290    let term_count = dict.term_count() as u64;
1291    let has_quoted_triples = dict.has_quoted_triples();
1292    drop(dict);
1293
1294    // Build each graph's six permutations one-at-a-time on a large graph (a single
1295    // permuted copy resident at a time, each sort still parallel) instead of all
1296    // six concurrently; below the threshold the faster all-permutations-parallel
1297    // build is used (its 6x transient copy is negligible there). `build_seq` is
1298    // byte-identical to `build`.
1299    let build_index = |triples: Vec<(u32, u32, u32)>| -> crate::GraphIndex {
1300        let n = triples.len();
1301        let b = GraphIndexBuilder::from_triples(triples).with_perms(perms);
1302        if n > LOWMEM_TRIPLE_THRESHOLD {
1303            b.build_seq()
1304        } else {
1305            b.build()
1306        }
1307    };
1308    let def = build_index(default_triples);
1309    let named_indexes: Vec<(String, crate::GraphIndex)> = named
1310        .into_iter()
1311        .map(|(g, ts)| (g, build_index(ts)))
1312        .collect();
1313
1314    // Only now is the DEDUPLICATED size of the file known: every index sorts and
1315    // dedups, so an input containing duplicate statements (a harvest that pages
1316    // with overlapping windows, a merge of overlapping dumps) writes fewer quads
1317    // than it ingested. `write_dataset_from_parts` stamps exactly this sum into
1318    // the header, so a metadata payload finalized with it cannot disagree with
1319    // the header — which is what an embedded Dataset Card used to do.
1320    let counts = FinalCounts {
1321        default_triples: def.triple_count() as u64,
1322        quads: def.triple_count() as u64
1323            + named_indexes
1324                .iter()
1325                .map(|(_, ix)| ix.triple_count() as u64)
1326                .sum::<u64>(),
1327    };
1328    let blob = pending.into_metadata(counts);
1329    // Report what was WRITTEN, not what was read: the returned stats drive the
1330    // CLI's "wrote …: N quads" line, which described the file, not the input.
1331    stats.statements = counts.quads as usize;
1332    stats.default_triples = counts.default_triples as usize;
1333
1334    let bytes = crate::file::write_dataset_from_parts(
1335        &dict_container,
1336        term_count,
1337        &def,
1338        &named_indexes,
1339        has_named,
1340        has_quoted_triples,
1341        &meta,
1342        levels,
1343        &blob,
1344        &text_index,
1345        codec,
1346    );
1347    (bytes, stats)
1348}
1349
1350/// Above this default-graph triple count, the index is built one permutation at a
1351/// time (low peak RAM) instead of all six in parallel. Chosen so typical builds
1352/// keep the faster parallel path while multi-hundred-million-triple graphs stay
1353/// within a bounded memory budget.
1354const LOWMEM_TRIPLE_THRESHOLD: usize = 30_000_000;
1355
1356#[cfg(test)]
1357mod tests {
1358    use super::*;
1359    use crate::Rete;
1360
1361    #[test]
1362    fn parses_iris_bnodes_literals() {
1363        let input = r#"
1364            # a comment
1365            <http://ex/Alice> <http://ex/knows> <http://ex/Bob> .
1366            <http://ex/Alice> <http://ex/age> "30"^^<http://www.w3.org/2001/XMLSchema#integer> .
1367            <http://ex/Bob> <http://ex/label> "Bob"@en .
1368            _:b0 <http://ex/p> "plain" .
1369        "#;
1370        let t = parse(input).unwrap();
1371        assert_eq!(t.len(), 4);
1372        assert_eq!(t[0].0, "<http://ex/Alice>");
1373        assert_eq!(t[0].2, "<http://ex/Bob>");
1374        assert_eq!(t[1].2, "\"30\"^^<http://www.w3.org/2001/XMLSchema#integer>");
1375        assert_eq!(t[2].2, "\"Bob\"@en");
1376        assert_eq!(t[3].0, "_:b0");
1377        assert_eq!(t[3].2, "\"plain\"");
1378    }
1379
1380    #[test]
1381    fn literal_with_spaces_and_escaped_quote() {
1382        let input = r#"<http://ex/s> <http://ex/p> "a \"quoted\" phrase here" ."#;
1383        let t = parse(input).unwrap();
1384        assert_eq!(t.len(), 1);
1385        assert_eq!(t[0].2, r#""a \"quoted\" phrase here""#);
1386    }
1387
1388    /// RDF-star: a quoted triple whose object is a **language-tagged literal**
1389    /// sits directly against the closing `>>` with no separating whitespace
1390    /// (`"name"@fr>>`). The language-tag scan must stop at `>` — a regression
1391    /// guard for the greedy scan-to-whitespace that swallowed `@fr>>` as the tag.
1392    #[test]
1393    fn quoted_triple_langtagged_object() {
1394        let s = r#"<<<http://ex/sp> <http://ex/name> "Hirondelle rustique"@fr>>"#;
1395        let (tok, rest) = take_term(s).unwrap();
1396        assert_eq!(
1397            tok,
1398            r#"<<<http://ex/sp> <http://ex/name> "Hirondelle rustique"@fr>>"#
1399        );
1400        assert_eq!(rest, "");
1401        // and it round-trips through a full annotation line
1402        let line = r#"<<<http://ex/sp> <http://ex/name> "Oreneta vulgar"@ca>> <http://purl.org/dc/terms/source> "Catalogue of Life" ."#;
1403        let t = parse(line).unwrap();
1404        assert_eq!(t.len(), 1);
1405        assert_eq!(
1406            t[0].0,
1407            r#"<<<http://ex/sp> <http://ex/name> "Oreneta vulgar"@ca>>"#
1408        );
1409        assert_eq!(t[0].2, r#""Catalogue of Life""#);
1410        // a plain lang-tagged object still terminates at whitespace
1411        assert_eq!(take_term(r#""x"@pt-BR ."#).unwrap().0, r#""x"@pt-BR"#);
1412    }
1413
1414    /// RDF 1.2 triple-term surface `<<( s p o )>>` parses to the SAME canonical
1415    /// token as the RDF-star `<< s p o >>`, so the two are interoperable (a query
1416    /// written either way matches data ingested either way).
1417    #[test]
1418    fn rdf12_triple_term_same_token_as_rdf_star() {
1419        let rdf12 = r#"<http://ex/r> <http://ex/p> <<( <http://ex/s> <http://ex/q> "o" )>> ."#;
1420        let star = r#"<http://ex/r> <http://ex/p> << <http://ex/s> <http://ex/q> "o" >> ."#;
1421        let a = parse(rdf12).unwrap();
1422        let b = parse(star).unwrap();
1423        assert_eq!(a.len(), 1);
1424        assert_eq!(a[0].2, r#"<<<http://ex/s> <http://ex/q> "o">>"#);
1425        assert_eq!(
1426            a[0].2, b[0].2,
1427            "RDF 1.2 and RDF-star must dedupe to one token"
1428        );
1429        // take_term consumes exactly the triple term (nothing left dangling).
1430        let (tok, rest) =
1431            take_term(r#"<<( <http://ex/s> <http://ex/q> <http://ex/o> )>>"#).unwrap();
1432        assert_eq!(tok, "<<<http://ex/s> <http://ex/q> <http://ex/o>>>");
1433        assert_eq!(rest, "");
1434    }
1435
1436    #[test]
1437    fn rejects_missing_dot() {
1438        assert!(parse("<a> <b> <c>").is_err());
1439    }
1440
1441    #[test]
1442    fn parses_quads_with_and_without_graph() {
1443        let input = "<http://ex/a> <http://ex/p> <http://ex/b> .\n\
1444                     <http://ex/a> <http://ex/p> <http://ex/c> <http://ex/g> .";
1445        let q = parse_quads(input).unwrap();
1446        assert_eq!(q.len(), 2);
1447        assert_eq!(q[0].3, None); // default graph
1448        assert_eq!(q[1].3, Some("<http://ex/g>".to_string()));
1449    }
1450
1451    #[test]
1452    fn parse_reader_matches_text_parse() {
1453        // The streaming reader must produce exactly the same quads as parsing the
1454        // whole text — including blank/comment skipping, a named graph, and CRLF.
1455        let nt = "<http://ex/a> <http://ex/p> <http://ex/b> .\r\n\
1456                  # comment\n\
1457                  \n\
1458                  _:b0 <http://ex/q> \"lit\"@en .\n";
1459        let via_text = parse_statements(nt, "nt").unwrap();
1460        let via_reader = parse_reader(std::io::Cursor::new(nt), "nt", 0).unwrap();
1461        assert_eq!(via_text, via_reader);
1462
1463        let nq = "<http://ex/a> <http://ex/p> <http://ex/b> .\n\
1464                  <http://ex/a> <http://ex/p> <http://ex/c> <http://ex/g> .\n";
1465        assert_eq!(
1466            parse_quads(nq).unwrap(),
1467            parse_reader(std::io::Cursor::new(nq), "nq", 0).unwrap()
1468        );
1469        // Turtle and TriG stream too — that is what lets the external build take a
1470        // multi-gigabyte `.ttl.gz` without expanding it to N-Triples first. Reading
1471        // a structured syntax from a reader must agree with parsing its whole text.
1472        let ttl = "@prefix ex: <http://ex/> .\nex:A ex:knows ex:B , ex:C .\n";
1473        assert_eq!(
1474            parse_statements(ttl, "ttl").unwrap(),
1475            parse_reader(std::io::Cursor::new(ttl), "ttl", 0).unwrap()
1476        );
1477        let trig = "@prefix ex: <http://ex/> .\nex:g { ex:A ex:knows ex:B . }\n";
1478        let streamed = parse_reader(std::io::Cursor::new(trig), "trig", 0).unwrap();
1479        assert_eq!(parse_statements(trig, "trig").unwrap(), streamed);
1480        assert_eq!(streamed[0].3.as_deref(), Some("<http://ex/g>"));
1481        // An unknown format is still a hard error, not a silent guess.
1482        assert!(parse_reader(std::io::Cursor::new(nt), "jsonld", 0).is_err());
1483    }
1484
1485    #[test]
1486    fn parse_statements_dispatches_by_format() {
1487        let nt = "<http://ex/a> <http://ex/p> <http://ex/b> .";
1488        assert_eq!(parse_statements(nt, "nt").unwrap().len(), 1);
1489        let ttl = "@prefix ex: <http://ex/> .\nex:A ex:knows ex:B , ex:C .";
1490        assert_eq!(parse_statements(ttl, "ttl").unwrap().len(), 2);
1491        // N-Triples is a subset of both Turtle and TriG, so the same text parses
1492        // under all three — and lands in the default graph under each.
1493        assert_eq!(parse_statements(nt, "trig").unwrap().len(), 1);
1494        assert!(parse_statements(nt, "trig").unwrap()[0].3.is_none());
1495        assert!(parse_statements(nt, "jsonld").is_err());
1496    }
1497
1498    /// A TriG graph block carries its graph name through as the same canonical
1499    /// token N-Quads would produce — so the two syntaxes agree on dictionary keys
1500    /// and a `.trig` shard federates with a `.nq` one.
1501    #[test]
1502    fn trig_graph_names_match_nquads() {
1503        let trig = "@prefix ex: <http://ex/> .\nex:g { ex:a ex:p ex:b . }\nex:a ex:p ex:c .\n";
1504        let nq = "<http://ex/a> <http://ex/p> <http://ex/b> <http://ex/g> .\n\
1505                  <http://ex/a> <http://ex/p> <http://ex/c> .\n";
1506        let mut from_trig = parse_statements(trig, "trig").unwrap();
1507        let mut from_nq = parse_statements(nq, "nq").unwrap();
1508        from_trig.sort();
1509        from_nq.sort();
1510        assert_eq!(from_trig, from_nq);
1511    }
1512
1513    /// RDF/XML (how most OWL ontologies ship) parses to the same canonical tokens,
1514    /// including the abbreviated typed-node syntax and `rdf:resource` references.
1515    #[test]
1516    fn parses_rdfxml_owl() {
1517        let xml = r#"<?xml version="1.0"?>
1518            <rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
1519                     xmlns:rdfs="http://www.w3.org/2000/01/rdf-schema#"
1520                     xmlns:owl="http://www.w3.org/2002/07/owl#">
1521              <owl:Class rdf:about="http://ex/Dog">
1522                <rdfs:subClassOf rdf:resource="http://ex/Animal"/>
1523                <rdfs:label>Dog</rdfs:label>
1524              </owl:Class>
1525            </rdf:RDF>"#;
1526        let triples = parse_rdfxml(xml).unwrap();
1527        // rdf:type owl:Class, rdfs:subClassOf, rdfs:label = 3 triples.
1528        assert_eq!(triples.len(), 3);
1529        assert!(triples.iter().any(|(s, p, o)| s == "<http://ex/Dog>"
1530            && p == "<http://www.w3.org/2000/01/rdf-schema#subClassOf>"
1531            && o == "<http://ex/Animal>"));
1532        // Same data through the format dispatcher (triples → default graph).
1533        assert_eq!(parse_statements(xml, "rdfxml").unwrap().len(), 3);
1534        // Malformed XML is a clear RdfXml error, not a silent empty parse.
1535        assert!(matches!(
1536            parse_statements("<rdf:RDF><not closed", "rdfxml"),
1537            Err(IngestError::RdfXml(_))
1538        ));
1539    }
1540
1541    /// Regression coverage for RUSTSEC-2026-0195: namespace declarations from
1542    /// an untrusted RDF/XML document must not make parsing panic or exhaust the
1543    /// bounded test process.
1544    #[test]
1545    fn rdfxml_namespace_fanout_is_bounded() {
1546        let declarations = (0..2_000)
1547            .map(|i| format!(r#" xmlns:p{i}="http://example.test/{i}/""#))
1548            .collect::<String>();
1549        let xml = format!(
1550            r#"<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"{declarations}/>"#
1551        );
1552        let result = parse_statements(&xml, "rdfxml");
1553        assert!(result.is_ok() || result.unwrap_err().to_string().contains("namespace"));
1554    }
1555
1556    /// Regression coverage for RUSTSEC-2026-0194: duplicate-name checking must
1557    /// complete for a large valid start tag without panicking or exhausting the
1558    /// bounded test process.
1559    #[test]
1560    fn rdfxml_attribute_fanout_completes() {
1561        let attributes = (0..2_000)
1562            .map(|i| format!(r#" p:a{i}="{i}""#))
1563            .collect::<String>();
1564        let xml = format!(
1565            r#"<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns:p="http://example.test/"><rdf:Description rdf:about="http://example.test/s"{attributes}/></rdf:RDF>"#
1566        );
1567        assert!(parse_statements(&xml, "rdfxml").is_ok());
1568    }
1569
1570    /// Text in, queryable file image out — the whole in-memory build path the
1571    /// wasm `build()` binding uses, including a named graph.
1572    #[test]
1573    fn assemble_dataset_round_trips() {
1574        let text = "<http://ex/a> <http://ex/knows> <http://ex/b> .\n\
1575                    <http://ex/b> <http://ex/knows> <http://ex/c> .\n\
1576                    <http://ex/a> <http://ex/age> \"30\"^^<http://www.w3.org/2001/XMLSchema#integer> .\n\
1577                    <http://ex/a> <http://ex/p> <http://ex/d> <http://ex/g1> .";
1578        let quads = parse_statements(text, "nq").unwrap();
1579        let (bytes, stats) = assemble_dataset(quads, &[]);
1580        assert_eq!(stats.statements, 4);
1581        assert_eq!(stats.default_triples, 3);
1582        assert_eq!(stats.named_graphs, 1);
1583        assert!(stats.terms >= 7);
1584
1585        let rete = Rete::open(&bytes).unwrap();
1586        assert_eq!(
1587            rete.query(None, Some("<http://ex/knows>"), None).len(),
1588            2,
1589            "default-graph pattern query"
1590        );
1591        assert_eq!(rete.graph_names(), vec!["<http://ex/g1>"]);
1592        let out = crate::eval_query(
1593            &rete,
1594            "SELECT ?x WHERE { ?x <http://ex/knows> ?y . ?y <http://ex/knows> ?z }",
1595        )
1596        .unwrap();
1597        match out {
1598            crate::QueryOutput::Select(_, rows) => assert_eq!(rows.len(), 1),
1599            other => panic!("expected select result, got {other:?}"),
1600        }
1601    }
1602
1603    /// The exact minimal input the playground Build tab uses: two triples, no
1604    /// literals, no rdf:type, no named graph — the smallest graph that still
1605    /// builds a community pyramid. (The wasm `build()` panic this guards was a
1606    /// `std::time::Instant::now()` in the pyramid timing path; native has a clock
1607    /// so this passes here, while the playground harness exercises the wasm path.)
1608    #[test]
1609    fn assemble_minimal_typeless_graph() {
1610        let text = "<http://ex/A> <http://ex/knows> <http://ex/B> .\n\
1611                    <http://ex/B> <http://ex/knows> <http://ex/C> .\n";
1612        let quads = parse_statements(text, "nt").unwrap();
1613        let (bytes, stats) = assemble_dataset(quads, &[]);
1614        assert_eq!(stats.default_triples, 2);
1615        let rete = Rete::open(&bytes).unwrap();
1616        assert_eq!(rete.query(None, Some("<http://ex/knows>"), None).len(), 2);
1617    }
1618}