froe 0.12.0

Reader and offline maintenance toolkit for Apache Jackrabbit Oak segment-tar (TarMK) repositories: parse archives and records, extract node data, compact, back up, and recover.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
//! The aggregate walk: the nodes and the relative properties a document
//! takes from **outside** the node it is made for.
//!
//! `docs/analysis/lucene-oak-documents.md` §4, from `oak-search`'s
//! `Aggregate` and `FulltextDocumentMaker.indexAggregates`.
//!
//! Oak combines two kinds of include into one list — the relative property
//! definitions first, then the `aggregates/<type>/include*` children — and
//! walks it per child. A matched node contributes its properties, and a
//! matched node whose own rule declares an aggregate is entered in turn,
//! as deep as `reaggregateLimit` allows.

use super::{
    DocumentMaker, DocumentState, FULLTEXT_FIELD, IndexResult, IndexingRule, Match, Matcher,
    NodeState, PropertyInclude, PropertyState, PropertyType, RELATIVE_NODE_PREFIX, values_of,
};

impl DocumentMaker<'_> {
    /// §4: the aggregates, whose values arrive in each aggregated node's
    /// property order — and, in front of them, the relative property
    /// definitions Oak turns into property includes of the same walk.
    pub(super) fn index_aggregates(
        &self,
        node: &NodeState<'_>,
        rule: &IndexingRule,
        state: &mut DocumentState,
    ) -> IndexResult<()> {
        let property_includes = rule.property_includes();
        // Before the node is read for anything: a definition with neither
        // include kind is the common one, and it owes this walk nothing.
        if !rule.aggregate.has_node_aggregates() && property_includes.is_empty() {
            return Ok(());
        }
        let walk = AggregateWalk {
            rule,
            // Oak hands `indexProperty` the **document's own** node state
            // for a property include, however deep the property lives, so
            // the binary gate is the root node's `jcr:mimeType`.
            root_has_mime_type: node.property("jcr:mimeType")?.is_some(),
            property_includes,
        };
        let level = AggregateLevel {
            nodes: rule.aggregate.matcher(),
            properties: PropertyIncludeMatcher::new(&walk.property_includes),
        };
        self.walk_aggregate(&walk, &level, node, "", state)
    }

    /// One level of the aggregate walk.
    ///
    /// The two include kinds are one list in Oak, the property includes in
    /// front of the node ones, and `collectAggregates` evaluates that list
    /// **per child** — so a child that ends both kinds contributes its
    /// property-include fields first.
    fn walk_aggregate(
        &self,
        walk: &AggregateWalk<'_>,
        level: &AggregateLevel<'_>,
        node: &NodeState<'_>,
        matched: &str,
        state: &mut DocumentState,
    ) -> IndexResult<()> {
        for (name, child) in node.child_node_entries()? {
            let (next_properties, ended_properties) = level.properties.step(&name);
            let (next_nodes, outcome) = level.nodes.step(&name, &child)?;
            for include in ended_properties {
                self.index_property_include(&child, walk, include, state)?;
            }
            // `Matcher.getMatchedPath`: the path of the matched node
            // **relative to the node the document is for**, which is what
            // the document's own rule is asked about below.
            let child_path = if matched.is_empty() {
                name.clone()
            } else {
                format!("{matched}/{name}")
            };
            match outcome {
                Match::Stop if next_properties.is_exhausted() => continue,
                Match::Stop | Match::Continue => {}
                Match::Aggregate(includes) => {
                    for include in includes {
                        let relative = include.relative_node.then(|| {
                            format!("{RELATIVE_NODE_PREFIX}{}", include.elements.join("/"))
                        });
                        let names: Vec<&str> = std::iter::once(FULLTEXT_FIELD)
                            .chain(relative.as_deref())
                            .collect();
                        self.aggregate_node(&child, walk.writing_into(&names, &child_path), state)?;
                    }
                }
            }
            let next = AggregateLevel {
                nodes: next_nodes,
                properties: next_properties,
            };
            self.walk_aggregate(walk, &next, &child, &child_path, state)?;
        }
        Ok(())
    }

    /// One property include's fields, under the **relative path** as the
    /// field name:
    ///
    /// ```java
    /// public void onResult(PropertyIncludeResult result) {
    ///     if (result.pd.ordered) addTypedOrderedFields(fields, result.propertyState, result.propertyPath, result.pd);
    ///     indexProperty(path, fields, state, result.propertyState, result.propertyPath, result.pd);
    /// }
    /// ```
    ///
    /// `propertyPath` is the definition's own parent path joined with the
    /// property's **own** name, which differ for a regular-expression
    /// definition.
    fn index_property_include(
        &self,
        node: &NodeState<'_>,
        walk: &AggregateWalk<'_>,
        include: PropertyInclude<'_>,
        state: &mut DocumentState,
    ) -> IndexResult<()> {
        let definition = include.definition;
        let parent = definition.ancestors.join("/");
        for mut property in included_properties(node, include)? {
            property.name = format!("{parent}/{}", property.name);
            if definition.ordered {
                self.index_ordered(&property, definition, state)?;
            }
            self.index_property(
                &property,
                definition,
                walk.rule,
                walk.root_has_mime_type,
                state,
            )?;
        }
        Ok(())
    }

    /// One aggregated node's properties, each a `:fulltext` value — and,
    /// for a `relativeNode` include, a `fullnode:<include path>` one
    /// **beside** it rather than instead of it.
    ///
    /// Oak's own rebuild is what says "beside": over a definition whose
    /// aggregate names one child twice, once plainly and once relatively,
    /// the page's `:fulltext` carries that child's values **twice** and
    /// `fullnode:jcr:content` carries them once — so the relative include
    /// contributed to both fields.
    ///
    /// The property definition consulted here is the one the rule covering
    /// **the aggregated node** holds, not the document's own rule:
    /// `indexAggregatedNode` resolves `getApplicableIndexingRule(result
    /// .nodeState)` and asks that rule whether the property is excluded
    /// from aggregation.
    fn aggregate_node(
        &self,
        node: &NodeState<'_>,
        into: AggregatedInto<'_>,
        state: &mut DocumentState,
    ) -> IndexResult<()> {
        let names = into.names;
        let covering = self.rules.applicable_rule(node)?;
        // `indexAggregatedNode`'s type gate is one rule's or the other's,
        // not both: the rule covering the **aggregated** node when one
        // does, and the document's own when none does.
        //
        // ```java
        // if (ruleAggNode != null) { if (!ruleAggNode.includePropertyType(tag)) continue; }
        // else if (!indexingRule.includePropertyType(tag)) continue;
        // ```
        let included_types = covering.map_or(&into.document_rule.include_property_types, |rule| {
            &rule.include_property_types
        });
        let has_mime_type = node.property("jcr:mimeType")?.is_some();
        for property in node.properties()? {
            if property.name.starts_with(':') {
                continue;
            }
            if !super::includes_property_type(included_types, &property) {
                continue;
            }
            // The **document's** rule, asked with the property's path
            // relative to the document's own node — so a
            // `jcr:content/.*` pattern of that rule reaches an aggregated
            // property, which is the shape it is written for.
            //
            // ```java
            // PropertyDefinition pd = indexingRule.getConfig(PathUtils.concat(result.nodePath, pname));
            // if (pd != null) { if (!pd.index) continue; if (pd.excludeFromAggregate) continue; }
            // ```
            let relative_path = format!("{}/{}", into.matched, property.name);
            let from_document = into.document_rule.config_of(&relative_path);
            if from_document
                .is_some_and(|definition| !definition.index || definition.exclude_from_aggregation)
            {
                continue;
            }
            if property.property_type == PropertyType::Binary {
                self.index_binary(&property, has_mime_type, names, state);
                continue;
            }
            // And the rule covering the **aggregated** node, asked with
            // the property's own name — a different rule, a different key
            // and a different flag: `nodeScopeIndex`, which a definition
            // that has one and does not set makes a skip.
            //
            // ```java
            // PropertyDefinition pdAgg = ruleAggNode != null ? ruleAggNode.getConfig(pname) : null;
            // if (pdAgg != null && !pdAgg.nodeScopeIndex) continue;
            // ```
            let definition = covering.and_then(|rule| rule.config_of(&property.name));
            if definition.is_some_and(|definition| !definition.node_scope_index) {
                continue;
            }
            for value in values_of(&property) {
                let Some(text) = value.as_text() else {
                    continue;
                };
                for name in names {
                    let position = state.fields.len();
                    self.index_fulltext(name, &text, None, state);
                    if let Some(definition) = definition
                        && let Some(field) = state.fields.get_mut(position)
                    {
                        // `if (pd != null) field.setBoost(pd.boost)`.
                        field.boost = definition.boost;
                    }
                }
            }
        }
        self.reaggregate(node, into, state)
    }

    /// The **re-aggregation**: an aggregated node whose own rule declares
    /// an aggregate contributes that aggregate's nodes too, into the same
    /// fields, after its own properties.
    ///
    /// `reaggregateLimit` — five by default — is how many levels deep that
    /// goes, and the one compared is the **root** aggregate's, however
    /// deep the walk is:
    ///
    /// ```java
    /// Aggregate nextAgg = currentInclude.getAggregate(matchedNodeState);
    /// if (nextAgg != null && aggregateStack.size() < rootState.rootAggregate.reAggregationLimit)
    /// ```
    ///
    /// The aggregate entered is the **matched node's own type's**, from
    /// the definition's `aggregates` map rather than through any rule —
    /// see [`IndexingRules::aggregate_of`]. Oak's rebuild of the interop
    /// fixture pins the rest: a `meta` child
    /// of type `sling:Folder` whose own type declares `include0 = inner`
    /// puts that grandchild's values in the page's `:fulltext` **and** in
    /// the `fullnode:meta` of the relative include that reached `meta`.
    fn reaggregate(
        &self,
        node: &NodeState<'_>,
        into: AggregatedInto<'_>,
        state: &mut DocumentState,
    ) -> IndexResult<()> {
        let Some(aggregate) = self.rules.aggregate_of(node)? else {
            return Ok(());
        };
        if !aggregate.has_node_aggregates() || into.depth >= into.limit {
            return Ok(());
        }
        self.walk_reaggregate(node, &aggregate.matcher(), into, state)
    }

    /// One level of a re-aggregation's own walk, which carries the field
    /// names of the include that reached the node it started from.
    fn walk_reaggregate(
        &self,
        node: &NodeState<'_>,
        matcher: &Matcher<'_>,
        into: AggregatedInto<'_>,
        state: &mut DocumentState,
    ) -> IndexResult<()> {
        for (name, child) in node.child_node_entries()? {
            let (next, outcome) = matcher.step(&name, &child)?;
            let child_path = format!("{}/{name}", into.matched);
            match outcome {
                Match::Stop => continue,
                Match::Continue => {}
                Match::Aggregate(includes) => {
                    // Once per include that ended here, as at the top
                    // level — a node two includes name is aggregated
                    // twice — and into the **outer** include's fields,
                    // because the field name a re-aggregated value takes
                    // is the one the walk entered with.
                    for _ended in includes {
                        self.aggregate_node(&child, into.deeper(&child_path), state)?;
                    }
                }
            }
            self.walk_reaggregate(&child, &next, into.at(&child_path), state)?;
        }
        Ok(())
    }
}

/// What a matched node's values are written into: the fields, and how
/// deep a re-aggregation from it may still go.
#[derive(Clone, Copy)]
struct AggregatedInto<'walk> {
    /// `:fulltext`, and the `fullnode:<path>` of the include that reached
    /// the node this descends from.
    names: &'walk [&'walk str],
    /// The rule the **document** is being made under, which is the type
    /// gate's fallback for an aggregated node no rule covers and the rule
    /// the `index` and `excludeFromAggregation` flags are read from.
    document_rule: &'walk IndexingRule,
    /// The matched node's path **relative to the document's own node**,
    /// which is the key that rule is asked with.
    matched: &'walk str,
    /// The **root** aggregate's `reAggregationLimit`, which is the one Oak
    /// compares its stack against however deep the stack is.
    limit: usize,
    /// How many aggregates the walk has entered, which is that stack's
    /// size.
    depth: usize,
}

impl<'walk> AggregatedInto<'walk> {
    /// The same fields, one aggregate deeper, at the node `matched`.
    fn deeper(self, matched: &'walk str) -> Self {
        Self {
            matched,
            depth: self.depth + 1,
            ..self
        }
    }

    /// The same fields at a node of the current re-aggregation level.
    const fn at(self, matched: &'walk str) -> Self {
        Self { matched, ..self }
    }
}

/// What every level of the aggregate walk carries down.
struct AggregateWalk<'rule> {
    /// The rule the document is being made under.
    rule: &'rule IndexingRule,
    /// Whether the **document's own** node declares `jcr:mimeType`, which
    /// is the gate a property include's binary is gated by however deep
    /// the property lives: Oak passes the root state to `indexProperty`.
    root_has_mime_type: bool,
    /// The relative definitions, in Oak's own combined order.
    property_includes: Vec<PropertyInclude<'rule>>,
}

impl AggregateWalk<'_> {
    /// The fields an include of this walk writes into, at the top of the
    /// aggregate stack.
    fn writing_into<'names>(
        &'names self,
        names: &'names [&'names str],
        matched: &'names str,
    ) -> AggregatedInto<'names> {
        AggregatedInto {
            names,
            document_rule: self.rule,
            matched,
            limit: usize::try_from(self.rule.aggregate.reaggregation_limit).unwrap_or(0),
            depth: 0,
        }
    }
}

/// One level of the walk: where each include kind has matched to.
struct AggregateLevel<'rule> {
    nodes: Matcher<'rule>,
    properties: PropertyIncludeMatcher<'rule>,
}

/// The property includes' half of the walk, which is the same state
/// machine over each definition's **ancestor** elements.
struct PropertyIncludeMatcher<'rule> {
    includes: &'rule [PropertyInclude<'rule>],
    /// One entry per live include: its index, and the depth it has
    /// matched to.
    live: Vec<(usize, usize)>,
}

impl<'rule> PropertyIncludeMatcher<'rule> {
    fn new(includes: &'rule [PropertyInclude<'rule>]) -> Self {
        let live = includes
            .iter()
            .enumerate()
            .filter(|(_, include)| !include.definition.ancestors.is_empty())
            .map(|(at, _)| (at, 0usize))
            .collect();
        Self { includes, live }
    }

    /// Steps into the child `name`, returning the level below and the
    /// includes whose ancestor path ends on that child.
    fn step(&self, name: &str) -> (Self, Vec<PropertyInclude<'rule>>) {
        let mut live = Vec::new();
        let mut ended = Vec::new();
        for (at, depth) in &self.live {
            let ancestors = &self.includes[*at].definition.ancestors;
            let Some(element) = ancestors.get(*depth) else {
                continue;
            };
            if element != MATCH_ALL_STEP && element != name {
                continue;
            }
            if depth + 1 == ancestors.len() {
                ended.push(self.includes[*at]);
            } else {
                live.push((*at, depth + 1));
            }
        }
        (
            Self {
                includes: self.includes,
                live,
            },
            ended,
        )
    }

    /// Whether no include can still match below this level, which is the
    /// property half of `Match::Stop`.
    const fn is_exhausted(&self) -> bool {
        self.live.is_empty()
    }
}

/// `Aggregate.MATCH_ALL`, which an ancestor element carries as well.
const MATCH_ALL_STEP: &str = "*";

/// The properties one property include contributes at the node its
/// ancestor path ended on.
///
/// ```java
/// if (pattern != null) {
///     for (PropertyState ps : nodeState.getProperties()) {
///         if (pattern.matcher(ps.getName()).matches()) results.onResult(…);
///     }
/// } else {
///     PropertyState ps = nodeState.getProperty(propertyName);
///     if (ps != null) results.onResult(…);
/// }
/// ```
///
/// The names arrive as the node state yields them, and the name the
/// pattern is matched against is the property's own — which is why the
/// field name is rebuilt from the definition's parent path rather than
/// taken from the definition's name.
///
/// **A hidden name contributes nothing here**, where the per-property
/// pass of §3.3 lets one through to the patterns as bug compatibility.
/// Oak's own rebuild of the interop fixture is what says so: a
/// `jcr:content/.*` definition over a node carrying `:childOrder` wrote
/// `full:jcr:content/jcr:primaryType` and no `:childOrder` field of
/// either kind.
fn included_properties(
    node: &NodeState<'_>,
    include: PropertyInclude<'_>,
) -> IndexResult<Vec<PropertyState>> {
    let definition = include.definition;
    let name = definition
        .name
        .rsplit('/')
        .next()
        .unwrap_or(&definition.name);
    let Some(pattern) = include.pattern else {
        if name.starts_with(':') {
            return Ok(Vec::new());
        }
        return Ok(node.property(name)?.into_iter().collect());
    };
    let parent = definition.ancestors.join("/");
    Ok(node
        .properties()?
        .into_iter()
        .filter(|property| !property.name.starts_with(':'))
        .filter(|property| pattern.matches(&format!("{parent}/{}", property.name)))
        .collect())
}