trusty_memory/kg_extract.rs
1//! Deterministic KG triple extraction from drawer content.
2//!
3//! Why: Issue #97 — `memory_remember` should populate the knowledge graph
4//! automatically so palaces with drawers always have a non-empty KG. Calling an
5//! LLM on every write would blow up latency and require network access; a
6//! deterministic heuristic stays fast and offline while still producing useful
7//! triples for tag membership, key-phrase mentions, and obvious is-a / has-a /
8//! works-at patterns. The visual graph view (the other half of #97) renders
9//! whatever shows up here, so this pass is the data source for "every palace
10//! has a graph".
11//! What: A pure function `extract_triples` that takes drawer content + tags +
12//! drawer id and returns a `Vec<Triple>` with `provenance = "auto:remember"`.
13//! The current heuristics are tag→drawer, room→drawer, hashtag→drawer, and a
14//! short pattern table (`X is a Y`, `X works at Y`, `X uses Y`, `X depends on
15//! Y`). Drawer ids are encoded as `drawer:<uuid>` so the subject keeps a
16//! stable, palace-unique identity that the graph view can dereference back
17//! to the source drawer.
18//! Test: `extract_triples_emits_tag_triples`,
19//! `extract_triples_emits_hashtag_mentions`,
20//! `extract_triples_extracts_is_a_pattern`,
21//! `extract_triples_never_panics_on_empty_input`.
22
23use crate::wordnet_pos::{self, WordNetPos};
24use chrono::Utc;
25use std::collections::HashSet;
26use trusty_common::memory_core::store::kg::Triple;
27use uuid::Uuid;
28
29/// Default tags that cause a drawer to be skipped during auto-extraction.
30///
31/// Why: Drawers tagged with these labels are by definition non-factual project
32/// knowledge (test fixtures, QA scaffolding, synthetic content) and should not
33/// pollute the KG with noise triples.
34/// What: A static slice of lowercase tag strings; matched case-insensitively
35/// during extraction.
36/// Test: `extract_triples_skips_denied_tags`.
37pub const DEFAULT_DENY_TAGS: &[&str] = &["cross-project-qa", "test", "fixture"];
38
39/// Configuration for a single extraction pass.
40///
41/// Why: Bundles per-run configuration so `extract_triples` can be called with
42/// different deny-lists (e.g. the default prod list vs. an empty list in
43/// integration tests) without changing the function signature. #5399 added the
44/// POS table here rather than to a process-wide `OnceLock`, which CLAUDE.md
45/// forbids: the table is a 16-byte `Copy` handle over `&'static str` with no
46/// build step, so passing it costs nothing and every caller can substitute a
47/// different one.
48/// What: Contains a `deny_tags` slice and a [`WordNetPos`] lookup. The
49/// extractor skips any drawer whose tags intersect the deny set, and consults
50/// the lookup when deciding which token in a noun phrase is its head.
51/// `#[non_exhaustive]` so the next field is not another breaking change.
52/// Test: `extract_triples_skips_denied_tags`, `extract_triples_empty_deny_list`,
53/// `a_caller_supplied_pos_table_is_the_one_consulted`.
54#[derive(Debug, Clone)]
55#[non_exhaustive]
56pub struct KgExtractConfig<'a> {
57 /// Tags that cause extraction to be skipped entirely. Compared
58 /// case-insensitively against the drawer's tag list.
59 pub deny_tags: &'a [&'a str],
60 /// Part-of-speech membership used to find the head of a noun phrase.
61 pub pos: WordNetPos,
62}
63
64impl Default for KgExtractConfig<'_> {
65 fn default() -> Self {
66 Self {
67 deny_tags: DEFAULT_DENY_TAGS,
68 pos: WordNetPos::shipped(),
69 }
70 }
71}
72
73/// Provenance tag stamped on every auto-extracted triple.
74///
75/// Why: Operators need a stable string to filter / retract the auto-extracted
76/// subset without scanning content. Centralising the constant keeps every
77/// emitter and the back-fill CLI in sync.
78/// What: A `&'static str` containing the literal `auto:remember`.
79/// Test: `extract_triples_stamps_provenance`.
80pub const AUTO_PROVENANCE: &str = "auto:remember";
81
82/// Confidence applied to auto-extracted triples.
83///
84/// Why: Heuristic extraction is not authoritative; downstream rankers can use
85/// the confidence to prefer explicit `kg_assert` triples over auto-extracted
86/// noise.
87/// What: `0.6` — high enough to surface in queries, low enough to be
88/// over-ridden by a manual `kg_assert` of the same `(subject, predicate)`.
89/// Test: `extract_triples_uses_reduced_confidence`.
90pub const AUTO_CONFIDENCE: f32 = 0.6;
91
92/// Subject prefix used for drawer-identity triples.
93///
94/// Why: A stable, palace-unique identifier lets the graph view dereference a
95/// node back to the source drawer (and the back-fill CLI dedupe by drawer).
96/// What: `drawer:` — concatenated with the drawer UUID hyphenated form.
97/// Test: every test in this module asserts the prefix.
98pub const DRAWER_SUBJECT_PREFIX: &str = "drawer:";
99
100/// Subject prefix used for tag entities.
101///
102/// Why: The KG enforces at most one active triple per `(subject, predicate)`,
103/// so we can't emit `drawer:X has-tag t1; drawer:X has-tag t2` — the second
104/// assert would close the first. By promoting each tag to its own subject
105/// (`tag:t1`, `tag:t2`) we keep multiple tags as distinct edges and the graph
106/// view gets natural tag-clusters around each drawer.
107/// What: `tag:` — concatenated with the lower-cased tag string.
108/// Test: `extract_triples_emits_tag_triples`.
109pub const TAG_SUBJECT_PREFIX: &str = "tag:";
110
111/// Subject prefix used for free-text mention entities.
112///
113/// Why: Same temporal-invariant reasoning as `TAG_SUBJECT_PREFIX`. Hashtag
114/// mentions and other discovered topical terms become their own subjects so
115/// multiple mentions per drawer survive the assert pipeline.
116/// What: `topic:` — concatenated with the lower-cased term.
117/// Test: `extract_triples_emits_hashtag_mentions`.
118pub const TOPIC_SUBJECT_PREFIX: &str = "topic:";
119
120/// Subject prefix used for room entities.
121///
122/// Why: A drawer can only sit in one room, but encoding the room as its own
123/// subject keeps the graph topology consistent (all "discovered metadata"
124/// entities live under prefixed namespaces) and lets multiple drawers from
125/// the same room cluster around a shared room node.
126/// What: `room:` — concatenated with the room label.
127/// Test: `extract_triples_emits_tag_triples`.
128pub const ROOM_SUBJECT_PREFIX: &str = "room:";
129
130/// Build the drawer subject string used as the (s) for every per-drawer
131/// triple emitted by this module.
132///
133/// Why: Centralises the `drawer:<uuid>` encoding so call sites cannot drift.
134/// What: Returns `format!("{DRAWER_SUBJECT_PREFIX}{id}")`.
135/// Test: covered by every extractor test.
136pub fn drawer_subject(id: Uuid) -> String {
137 format!("{DRAWER_SUBJECT_PREFIX}{id}")
138}
139
140/// Inputs to a single extraction pass.
141///
142/// Why: Bundling the inputs keeps `extract_triples` signature small and lets
143/// us add new fields (e.g. drawer_type) without breaking call sites.
144/// What: Plain data struct; all fields are borrowed so the caller keeps
145/// ownership.
146/// Test: indirectly via every test that constructs one.
147#[derive(Debug, Clone)]
148pub struct ExtractInput<'a> {
149 pub drawer_id: Uuid,
150 pub content: &'a str,
151 pub tags: &'a [String],
152 pub room: Option<&'a str>,
153}
154
155/// Run the deterministic heuristic extractor with default config.
156///
157/// Why: Convenience wrapper that uses [`KgExtractConfig::default`] (the
158/// production deny-list) so call sites that do not need a custom config
159/// remain unchanged.
160/// What: Delegates to [`extract_triples_with_config`] with a default config.
161/// Test: All existing tests call this helper and implicitly exercise the default
162/// deny-list path.
163pub fn extract_triples(input: &ExtractInput<'_>) -> Vec<Triple> {
164 extract_triples_with_config(input, &KgExtractConfig::default())
165}
166
167/// Run the deterministic heuristic extractor.
168///
169/// Why: Single entry point so `memory_remember`, `memory_note`, and the
170/// back-fill CLI all share the same logic. Pure function — no I/O, no async —
171/// so it can be unit-tested cheaply. Accepts a [`KgExtractConfig`] so callers
172/// can override the deny-list without touching the function signature.
173/// What: First checks whether any of the drawer's tags appear in
174/// `config.deny_tags` (case-insensitive); when a match is found the function
175/// returns immediately with an empty vec and logs a debug message. Otherwise
176/// walks `tags`, content tokens, and a small pattern list to emit `Triple`s;
177/// deduplicates so the same `(subject, predicate, object)` never appears twice
178/// in a single pass.
179/// Test: `extract_triples_skips_denied_tags`, `extract_triples_emits_tag_triples`,
180/// plus all other tests in this file.
181pub fn extract_triples_with_config(
182 input: &ExtractInput<'_>,
183 config: &KgExtractConfig<'_>,
184) -> Vec<Triple> {
185 // Deny-list check: if any tag on this drawer is in the deny set, skip
186 // extraction entirely. The check is case-insensitive to tolerate mixed-
187 // case tags from different clients.
188 let denied = input.tags.iter().any(|t| {
189 let lower = t.trim().to_lowercase();
190 config.deny_tags.contains(&lower.as_str())
191 });
192 if denied {
193 tracing::debug!(
194 drawer_id = %input.drawer_id,
195 tags = ?input.tags,
196 "kg_extract: skipping drawer — tag matches deny-list"
197 );
198 return Vec::new();
199 }
200 let now = Utc::now();
201 let subject = drawer_subject(input.drawer_id);
202 let mut out: Vec<Triple> = Vec::new();
203 let mut seen: HashSet<(String, String, String)> = HashSet::new();
204
205 let push = |out: &mut Vec<Triple>,
206 seen: &mut HashSet<(String, String, String)>,
207 s: String,
208 p: String,
209 o: String| {
210 let key = (s.clone(), p.clone(), o.clone());
211 if seen.insert(key) {
212 out.push(Triple {
213 subject: s,
214 predicate: p,
215 object: o,
216 valid_from: now,
217 valid_to: None,
218 confidence: AUTO_CONFIDENCE,
219 provenance: Some(AUTO_PROVENANCE.to_string()),
220 });
221 }
222 };
223
224 // Tag membership — each tag becomes its own subject so multiple tags on
225 // the same drawer don't collide under the "one active triple per
226 // (s, p)" invariant. Edge direction is `tag:<t> tags drawer:<id>` so the
227 // graph clusters drawers under their shared tag nodes.
228 for tag in input.tags {
229 let clean = tag.trim();
230 if clean.is_empty() {
231 continue;
232 }
233 push(
234 &mut out,
235 &mut seen,
236 format!("{TAG_SUBJECT_PREFIX}{}", clean.to_lowercase()),
237 "tags".to_string(),
238 subject.clone(),
239 );
240 }
241
242 // Room membership — `room:<r> contains drawer:<id>` for the same reason
243 // (multiple drawers per room must coexist).
244 if let Some(room) = input.room {
245 let clean = room.trim();
246 if !clean.is_empty() {
247 push(
248 &mut out,
249 &mut seen,
250 format!("{ROOM_SUBJECT_PREFIX}{clean}"),
251 "contains".to_string(),
252 subject.clone(),
253 );
254 }
255 }
256
257 // Hashtag-style mentions — `topic:<term> mentioned-in drawer:<id>` so
258 // multiple terms per drawer can coexist as distinct active edges.
259 for term in extract_hashtags(input.content) {
260 push(
261 &mut out,
262 &mut seen,
263 format!("{TOPIC_SUBJECT_PREFIX}{term}"),
264 "mentioned-in".to_string(),
265 subject.clone(),
266 );
267 }
268
269 // Simple natural-language patterns. Each yields a free-form
270 // `<subject> <predicate> <object>` triple anchored to entities found in
271 // the content (not the drawer subject), so the graph develops topical
272 // edges over time.
273 for (s, p, o) in extract_patterns(input.content, &config.pos) {
274 push(&mut out, &mut seen, s, p, o);
275 }
276
277 out
278}
279
280/// Pull `#hashtag`-style tokens out of free-form content.
281///
282/// Why: Hashtags are a cheap, intentional signal — when a user writes `#rust`
283/// or `#design-doc` we should record the mention so the graph picks it up.
284/// What: Walks the string, captures runs of `[a-zA-Z0-9_-]` following a `#`,
285/// lower-cases and deduplicates. Skips empty captures (a lone `#`).
286/// Test: `extract_triples_emits_hashtag_mentions`.
287fn extract_hashtags(content: &str) -> Vec<String> {
288 let mut out: Vec<String> = Vec::new();
289 let mut seen: HashSet<String> = HashSet::new();
290 let mut iter = content.char_indices().peekable();
291 while let Some((_, c)) = iter.next() {
292 if c != '#' {
293 continue;
294 }
295 let mut term = String::new();
296 while let Some(&(_, nc)) = iter.peek() {
297 if nc.is_ascii_alphanumeric() || nc == '_' || nc == '-' {
298 term.push(nc.to_ascii_lowercase());
299 iter.next();
300 } else {
301 break;
302 }
303 }
304 if term.is_empty() {
305 continue;
306 }
307 if seen.insert(term.clone()) {
308 out.push(term);
309 }
310 }
311 out
312}
313
314/// Pattern dictionary used by `extract_patterns`.
315///
316/// Why: A small, predictable set of (predicate, marker phrases) keeps the
317/// extractor explicable and deterministic. Each entry maps a predicate to one
318/// or more space-padded marker phrases; when the marker appears in the lower-
319/// cased content we split on it and read the entity tokens immediately to
320/// each side.
321/// What: A static slice of `(predicate, &[marker, ...])`. Markers must be
322/// lower-case and surrounded by whatever whitespace the input has — we add
323/// the padding ourselves.
324/// Test: `extract_triples_extracts_is_a_pattern`.
325const PATTERN_TABLE: &[(&str, &[&str])] = &[
326 ("is-a", &[" is a ", " is an "]),
327 ("works-at", &[" works at "]),
328 ("uses", &[" uses ", " using "]),
329 ("depends-on", &[" depends on ", " requires "]),
330];
331
332/// Function words that can never be a KG entity.
333///
334/// Why: #4678 — a [`PATTERN_TABLE`] marker hit says a marker phrase appeared,
335/// not that the tokens either side of it name anything. "calling them is a
336/// no-op" put `them --is-a--> no-op` into the live palace. Grouped by part of
337/// speech because the boundary is grammatical, not statistical: these are the
338/// closed word classes English never coins new members of, so the list is
339/// finite and stable rather than a frequency cut-off that needs re-tuning.
340/// What: a flat lower-case slice, matched exactly against a normalised token
341/// by [`is_stop_token`]. Membership is checked with a linear scan — it runs at
342/// most twice per pattern hit and at most four hits exist per drawer.
343/// Test: `stopwords_are_unique`, `is_stop_token_rejects_every_stopword`.
344const STOPWORDS: &[&str] = &[
345 // Articles and determiners.
346 "a",
347 "an",
348 "the",
349 "this",
350 "that",
351 "these",
352 "those",
353 "some",
354 "any",
355 "each",
356 "every",
357 "all",
358 "both",
359 "either",
360 "neither",
361 "another",
362 "such",
363 "same",
364 "no",
365 "none",
366 // Pronouns.
367 "i",
368 "me",
369 "my",
370 "mine",
371 "myself",
372 "we",
373 "us",
374 "our",
375 "ours",
376 "ourselves",
377 "you",
378 "your",
379 "yours",
380 "yourself",
381 "he",
382 "him",
383 "his",
384 "himself",
385 "she",
386 "her",
387 "hers",
388 "herself",
389 "it",
390 "its",
391 "itself",
392 "they",
393 "them",
394 "their",
395 "theirs",
396 "themselves",
397 "who",
398 "whom",
399 "whose",
400 "which",
401 "what",
402 "one",
403 "ones",
404 "someone",
405 "something",
406 "anyone",
407 "anything",
408 "everyone",
409 "everything",
410 "nobody",
411 "nothing",
412 "others",
413 // Prepositions.
414 "of",
415 "in",
416 "on",
417 "at",
418 "to",
419 "for",
420 "with",
421 "by",
422 "from",
423 "about",
424 "into",
425 "onto",
426 "over",
427 "under",
428 "below",
429 "above",
430 "between",
431 "among",
432 "through",
433 "during",
434 "before",
435 "after",
436 "across",
437 "against",
438 "within",
439 "without",
440 "upon",
441 "per",
442 "via",
443 "than",
444 "toward",
445 "towards",
446 "off",
447 "out",
448 "up",
449 "down",
450 "near",
451 "around",
452 "behind",
453 "beyond",
454 "beside",
455 "along",
456 "past",
457 "throughout",
458 // #5399: WordNet lists these two as nouns ("the inside of the box"), so
459 // without the closed-class list they pass the POS check and continue a
460 // noun phrase they actually close.
461 "inside",
462 "outside",
463 // Conjunctions and subordinators.
464 "and",
465 "or",
466 "but",
467 "nor",
468 "so",
469 "yet",
470 "if",
471 "then",
472 "else",
473 "because",
474 "while",
475 "when",
476 "where",
477 "whether",
478 "though",
479 "although",
480 "since",
481 "as",
482 "unless",
483 "until",
484 "whereas",
485 // Copulas, auxiliaries and modals.
486 "is",
487 "are",
488 "was",
489 "were",
490 "be",
491 "been",
492 "being",
493 "am",
494 "do",
495 "does",
496 "did",
497 "done",
498 "doing",
499 "has",
500 "have",
501 "had",
502 "having",
503 "can",
504 "could",
505 "will",
506 "would",
507 "shall",
508 "should",
509 "may",
510 "might",
511 "must",
512 "ought",
513 "need",
514 "needs",
515 "let",
516 "lets",
517 "get",
518 "gets",
519 "got",
520 "gotten",
521 // Degree, negation and discourse adverbs — closed-class fillers that sit
522 // next to a marker often and name nothing.
523 "not",
524 "only",
525 "just",
526 "also",
527 "very",
528 "too",
529 "still",
530 "already",
531 "always",
532 "never",
533 "often",
534 "again",
535 "here",
536 "there",
537 "now",
538 "actually",
539 "really",
540 "simply",
541 "merely",
542 "quite",
543 "rather",
544 "even",
545 "ever",
546 "once",
547 "yes",
548 "well",
549 "much",
550 "many",
551 "more",
552 "most",
553 "less",
554 "least",
555 "few",
556 "several",
557 "enough",
558 "almost",
559 "perhaps",
560 "maybe",
561 "instead",
562 "otherwise",
563 "hence",
564 "therefore",
565 "however",
566 "thus",
567 "moreover",
568];
569
570/// Minimum character length for a token to be accepted as an entity.
571///
572/// Why: one- and two-character tokens are overwhelmingly punctuation debris,
573/// initials, or list markers rather than entities.
574/// What: `3`. Tokens shorter than this are rejected unless they appear in
575/// [`SHORT_ENTITY_ALLOWLIST`].
576/// Test: `extract_triples_rejects_short_token_off_allowlist`.
577pub const MIN_ENTITY_TOKEN_LEN: usize = 3;
578
579/// Short tokens that are real entities in this workspace's subject matter.
580///
581/// Why: [`MIN_ENTITY_TOKEN_LEN`] would otherwise reject the languages, crate
582/// aliases, and infrastructure abbreviations these palaces actually discuss —
583/// a precision filter that silently drops `Go` and `C` costs recall for
584/// nothing. The crate aliases (`tm`, `ts`, `tc`, `ta`) come from this repo's
585/// own abbreviation table.
586/// What: lower-case tokens of fewer than [`MIN_ENTITY_TOKEN_LEN`] characters
587/// that bypass the length floor. The stopword check still applies first, so an
588/// entry here cannot resurrect a function word.
589/// Test: `extract_triples_keeps_allowlisted_go`,
590/// `extract_triples_keeps_allowlisted_c`,
591/// `short_entity_allowlist_entries_survive_the_length_floor`.
592pub const SHORT_ENTITY_ALLOWLIST: &[&str] = &[
593 // Languages and language-shaped tokens.
594 "go", "c", "c#", "js", "ts", "py", "ml", // Domains and infrastructure.
595 "ai", "kg", "db", "ui", "os", "io", "vm", "ip", // Process and workspace aliases.
596 "pr", "ci", "qa", "pm", "tm", "tc", "ta",
597];
598
599/// Punctuation stripped from a token's edges before it is classified.
600///
601/// Why: `last_token` / `first_token` only strip a trailing run, so a token can
602/// still arrive as `("the` or `` `redb` ``. Comparing that against
603/// [`STOPWORDS`] would miss, and the filter would leak exactly the tokens it
604/// exists to catch.
605/// What: the edge characters [`is_stop_token`] trims before matching. Interior
606/// characters are untouched, so `no-op` and `c#` survive intact.
607/// Test: `is_stop_token_normalises_surrounding_punctuation`.
608const TOKEN_EDGE_PUNCT: &[char] = &[
609 '(', ')', '[', ']', '{', '}', '<', '>', '"', '\'', '`', ',', '.', ';', ':', '!', '?', '*', '_',
610 '-', '/', '\\', '|', '—', '–', '…', '“', '”', '‘', '’',
611];
612
613/// Whether `tok` must be refused as a KG entity.
614///
615/// Why: #4678 — the pattern pass treated any whitespace-delimited token beside
616/// a marker as an entity, which is how `them --is-a--> no-op`,
617/// `exhaustiveness --is-a--> hard`, and `squash --is-a--> ancestor` reached the
618/// live graph. This is the single gate both the extractor and the
619/// `--purge-stale-subjects` back-fill consult, so forward extraction and
620/// historical clean-up can never disagree about what counts as garbage.
621/// What: normalises `tok` through [`clean_token`], lower-cases it, and rejects
622/// it when the result is empty, appears in [`STOPWORDS`], or is shorter than
623/// [`MIN_ENTITY_TOKEN_LEN`] without being in [`SHORT_ENTITY_ALLOWLIST`].
624/// Length is counted in `char`s, not bytes, so a multi-byte token is not
625/// mis-measured. Purely lexical: it judges the token alone and knows nothing of
626/// the sentence, so it cannot catch a triple whose tokens are both ordinary
627/// words (see #4678 for the residue).
628///
629/// The normalisation step is NOT redundant with `clean_token`'s use in
630/// `first_token` / `last_token`. Those clean tokens on the way IN, which only
631/// helps content extracted from now on. The `--purge-stale-subjects` path calls
632/// this on subjects read back out of redb, and those were written by the old
633/// extractor with the punctuation already welded on — normalising here is what
634/// lets a stored `("the` be recognised as the stopword it is.
635/// Test: `is_stop_token_rejects_every_stopword`,
636/// `is_stop_token_accepts_ordinary_entities`,
637/// `is_stop_token_normalises_surrounding_punctuation`,
638/// `purge_selects_a_legacy_subject_with_welded_punctuation`.
639pub fn is_stop_token(tok: &str) -> bool {
640 let norm = clean_token(tok).to_lowercase();
641 if norm.is_empty() {
642 return true;
643 }
644 if STOPWORDS.contains(&norm.as_str()) {
645 return true;
646 }
647 norm.chars().count() < MIN_ENTITY_TOKEN_LEN && !SHORT_ENTITY_ALLOWLIST.contains(&norm.as_str())
648}
649
650/// The cleaned spelling a stored entity term must be merged onto, or `None`
651/// when the term is already canonical.
652///
653/// Why: #4678's edge trim fixed extraction going forward and split every
654/// pre-fix entity in two — a palace holds both `` `redb` `` and `redb`, and
655/// each `kg-rebuild` over the same drawer content widens the gap (#5401).
656/// Deciding the canonical spelling next to the filter that produced the split
657/// is what stops the merge pass from inventing a second normalisation that can
658/// drift away from the extractor's.
659/// What: `Some(cleaned)` when [`clean_token`] changes `term`; `None` when it
660/// does not, or when [`is_stop_token`] rejects the term. That second arm is
661/// what keeps this disjoint from `--purge-stale-subjects`: a punctuated
662/// stopword such as `("the` is garbage the purge deletes, never a twin some
663/// real node should absorb.
664///
665/// The merge therefore inherits [`clean_token`]'s normalisation whole,
666/// dotfiles included: `.env` names `env`, `--verbose` names `verbose`, and the
667/// leading punctuation that IS the identity is collapsed with the decorative
668/// kind. That is the point of reusing `clean_token` — the extractor already
669/// emits `env` for content saying `.env`, so the merge aligns stored nodes
670/// with current extraction instead of preserving a spelling nothing produces
671/// any more. `close_active_row` keeps the punctuated row under its `hist:`
672/// key, so the original spelling is not lost.
673/// Test: `canonical_entity_names_the_cleaned_twin`.
674pub fn canonical_entity(term: &str) -> Option<&str> {
675 // #5401: the merge pass and the extractor must agree on one spelling.
676 if is_stop_token(term) {
677 return None;
678 }
679 let cleaned = clean_token(term);
680 (cleaned != term).then_some(cleaned)
681}
682
683/// Apply the pattern table to a single content blob.
684///
685/// Why: Keeps the matching loop out of `extract_triples` so the dispatcher
686/// stays readable.
687/// What: For every `(predicate, markers)` row, scan every marker against the
688/// lower-cased content; on the first hit emit `(left_token, predicate,
689/// right_token)` and move on to the next predicate. Only the first hit per
690/// predicate is taken to avoid combinatorial output on long texts. A hit whose
691/// subject or object fails [`is_stop_token`] emits nothing and still consumes
692/// the predicate's turn, so one rejected hit never cascades into a scan for a
693/// second, lower-quality match later in the blob.
694/// Test: `extract_triples_extracts_is_a_pattern`,
695/// `extract_triples_rejects_pronoun_subject`,
696/// `extract_triples_rejects_stopword_object`.
697fn extract_patterns(content: &str, pos: &WordNetPos) -> Vec<(String, String, String)> {
698 let lower = content.to_lowercase();
699 let mut out: Vec<(String, String, String)> = Vec::new();
700 for (predicate, markers) in PATTERN_TABLE {
701 for marker in *markers {
702 if let Some(idx) = lower.find(marker) {
703 // #5399: production hands this whole multi-line drawer bodies
704 // (`auto_extract_and_assert`, `kg_rebuild`), so an unbounded
705 // walk joins two unrelated sentences across a line break. A
706 // newline closes the phrase exactly as a period does.
707 let line_start = lower[..idx].rfind('\n').map_or(0, |p| p + 1);
708 let left = lower[line_start..idx].trim();
709 let right_start = idx + marker.len();
710 let line_end = lower[right_start..]
711 .find('\n')
712 .map_or(lower.len(), |p| right_start + p);
713 let right = lower[right_start..line_end].trim();
714 // #4678: a marker hit is not evidence of two entities — reject
715 // the whole triple when either side is a function word or too
716 // short, never half of it. #5399: and take the HEAD of each
717 // noun phrase rather than the token nearest the marker.
718 if let (Some(subject_tok), Some(object_tok)) =
719 (select_subject(left, pos), select_object(right, pos))
720 {
721 out.push((subject_tok, (*predicate).to_string(), object_tok));
722 }
723 break;
724 }
725 }
726 }
727 out
728}
729
730/// Prepositions that continue a noun phrase rather than closing it.
731///
732/// Why: #5399 — `ancestor of origin/main` and `member of the process group`
733/// are single noun phrases, so grabbing `ancestor` or `member` alone truncates
734/// a relation into a bogus type. The stopping token is what reveals it. The set
735/// is deliberately just `of`: it is the genitive linker and is almost always
736/// NP-internal, whereas every other preposition attaches ambiguously —
737/// `trusty-memory uses redb for persistence` has `for` closing the object and
738/// opening a purpose adjunct on the verb, and that triple must survive.
739/// What: matched against the cleaned token immediately after the extracted
740/// noun-phrase run; a hit rejects the whole triple.
741/// Test: `row2_rejects_ancestor_truncated_before_of`,
742/// `row6_rejects_member_of_the_process_group`,
743/// `row5_keeps_uses_object_before_a_non_genitive_preposition`.
744const NP_CONTINUING_PREPOSITIONS: &[&str] = &["of"];
745
746/// Characters that close a noun phrase when welded to a token's trailing edge.
747///
748/// Why: `is a parser, and ...` must not walk the run into the next clause.
749/// What: consulted on the RAW token, through [`ends_noun_phrase`], before
750/// [`clean_token`] strips it.
751/// Test: `stops_the_noun_phrase_run_at_a_comma`,
752/// `stops_the_run_at_a_terminator_behind_markdown_emphasis`.
753const NP_TERMINATING_PUNCT: &[char] = &['.', ',', ';', ':', '!', '?', ')'];
754
755/// Longest noun-phrase run either walk will consider.
756///
757/// Why: a bound keeps a pathological line (a long unpunctuated list of unknown
758/// tokens) from letting the head drift far from the marker. Four covers
759/// `[adj] [noun] [noun]` plus one, which is past the length of any real
760/// technical compound in drawer prose.
761/// Test: `caps_the_noun_phrase_run`.
762const NP_RUN_MAX: usize = 4;
763
764/// Whether a raw token's trailing punctuation closes the noun phrase.
765///
766/// Why: #5399 — the check used to read only the token's LAST character, so
767/// `**MCP is a thin proxy.**` (raw token `proxy.**`) ended in `*`, missed the
768/// `.`, and let the run walk into the next clause. Markdown emphasis is the
769/// normal shape of drawer content, not an edge case, so the terminator is
770/// almost always behind one or two more punctuation characters.
771/// What: scans the whole trailing punctuation run — every trailing character
772/// that [`clean_token`] would strip — and reports whether any of it terminates
773/// a sentence or clause. Interior punctuation is not consulted, so `Node.js`
774/// and `src/main.rs` still read as one unterminated token.
775/// Test: `stops_the_run_at_a_terminator_behind_markdown_emphasis`,
776/// `interior_punctuation_does_not_terminate_the_run`.
777fn ends_noun_phrase(raw: &str) -> bool {
778 raw.trim_end()
779 .chars()
780 .rev()
781 .take_while(|c| TOKEN_EDGE_PUNCT.contains(c))
782 .any(|c| NP_TERMINATING_PUNCT.contains(&c))
783}
784
785/// Which way a noun-phrase walk moves away from the marker.
786///
787/// Why: the two sides are mirror images in one respect that matters. Walking
788/// RIGHT, a token's trailing `.` closes the phrase we are building, so the
789/// token belongs to it. Walking LEFT, that same `.` ended the PREVIOUS
790/// sentence, so the token belongs to that one and must not be taken.
791/// Test: `stops_the_noun_phrase_run_at_a_comma`,
792/// `subject_walk_stops_before_a_previous_sentence`.
793#[derive(Debug, Clone, Copy, PartialEq, Eq)]
794enum Walk {
795 /// Away from the marker into the object phrase.
796 Right,
797 /// Away from the marker into the subject phrase.
798 Left,
799}
800
801/// Collect the noun-phrase run adjacent to a marker.
802///
803/// Why: both sides of a pattern hit need the same walk — "which tokens belong
804/// to this phrase" is one question with one answer, and writing it twice is how
805/// the two sides drift.
806/// What: consumes `toks` in walk order (the caller reverses them for
807/// [`Walk::Left`]) and stops at a function word, a verb-only or adverb-only
808/// token, sentence punctuation, an unknown token, or [`NP_RUN_MAX`]. Returns
809/// the accepted tokens in READING order regardless of direction, whether
810/// punctuation closed the phrase, and how many raw tokens were consumed — the
811/// caller needs the last two to decide about a following genitive.
812/// Test: `caps_the_noun_phrase_run`, `stops_the_noun_phrase_run_at_a_comma`,
813/// `subject_walk_stops_before_a_previous_sentence`.
814fn noun_phrase_run<'a>(
815 toks: &[&'a str],
816 pos: &WordNetPos,
817 walk: Walk,
818) -> (Vec<&'a str>, bool, usize) {
819 let mut run: Vec<&str> = Vec::new();
820 let mut idx = 0usize;
821 let mut terminated = false;
822 while idx < toks.len() && run.len() < NP_RUN_MAX {
823 let raw = toks[idx];
824 let tok = clean_token(raw);
825 if tok.is_empty() || is_stop_token(tok) {
826 break;
827 }
828 let closes = ends_noun_phrase(raw);
829 // Walking left, the punctuation sits between this token and the phrase
830 // we are collecting, so the token is on the far side of the boundary.
831 if closes && walk == Walk::Left {
832 terminated = true;
833 break;
834 }
835 let mask = pos.mask(tok);
836 // A word WordNet knows only as a verb or only as an adverb cannot sit
837 // inside a noun phrase, so the phrase ended before it.
838 if mask != 0 && mask & (wordnet_pos::NOUN | wordnet_pos::ADJ) == 0 {
839 break;
840 }
841 run.push(tok);
842 idx += 1;
843 if closes {
844 terminated = true;
845 break;
846 }
847 // A name heads its phrase rather than modifying the next word.
848 if mask == 0 {
849 break;
850 }
851 }
852 if walk == Walk::Left {
853 run.reverse();
854 }
855 (run, terminated, idx)
856}
857
858/// Pick the head of a noun-phrase run.
859///
860/// Why: #5399 — English compounds are right-headed, so the head is the
861/// rightmost token of the phrase that can name a thing. An adjective-only token
862/// names a property, so it is skipped rather than taken. This RE-WALK is the
863/// whole policy: the earlier spike instead REJECTED the triple when the token
864/// nearest the marker was adjective-only, and that dropped 304 pairs over 306k
865/// lines of repo markdown, 196 of them (64%) with a perfectly good head noun
866/// sitting one token further along. See the fixture comment on
867/// `row1_rewalks_past_an_adjective_only_modifier` for why the reject rule was
868/// wrong on its own terms and not merely expensive.
869/// What: `run` is in reading order, so this returns its last non-adjective-only
870/// token — or `None` when every token in it is adjective-only (`a robust` names
871/// nothing) or the run is empty.
872/// Test: `row1_rewalks_past_an_adjective_only_modifier`,
873/// `row4_rewalks_past_the_adjective_to_the_head_noun`,
874/// `an_all_adjective_subject_yields_no_triple`.
875fn phrase_head(run: &[&str], pos: &WordNetPos) -> Option<String> {
876 run.iter()
877 .rev()
878 .find(|t| !pos.is_adjective_only(t))
879 .map(|t| (*t).to_string())
880}
881
882/// Pick the subject entity out of the text preceding a pattern marker.
883///
884/// Why: the head of the subject phrase is normally the token immediately before
885/// the marker — `a fast parser is a tool` — so this agrees with the old
886/// `last_token` in every ordinary case. It differs only when that token cannot
887/// head a phrase, where `last_token` emitted it as an entity anyway. That is
888/// the #5399 defect (`exhaustiveness --is-a--> hard`) mirrored onto the other
889/// side of the marker, so it gets the same re-walk rather than a reject rule.
890/// What: walks leftward from the marker through [`noun_phrase_run`] and takes
891/// the phrase head. Returns `None` when the phrase has no head — `anything
892/// robust is a compiler` names no entity to be the subject of anything.
893/// Test: `subject_side_rewalks_past_an_adjective_only_token`,
894/// `an_all_adjective_subject_yields_no_triple`,
895/// `extract_triples_rejects_pronoun_subject`.
896fn select_subject(left: &str, pos: &WordNetPos) -> Option<String> {
897 let mut toks: Vec<&str> = left.split_whitespace().collect();
898 toks.reverse();
899 let (run, _, _) = noun_phrase_run(&toks, pos, Walk::Left);
900 phrase_head(&run, pos)
901}
902
903/// Pick the object entity out of the text following a pattern marker.
904///
905/// Why: #5399 — taking one token is wrong three ways at once. It takes the
906/// modifier instead of the head (`a fast parser` -> `fast`), it emits a bare
907/// property as a type (`a hard requirement` -> `hard`), and it silently
908/// truncates a relation into a type (`an ancestor of origin main` ->
909/// `ancestor`). All three need to see more than one token and need to know each
910/// token's part of speech.
911/// What: walks rightward from the marker through [`noun_phrase_run`], rejects
912/// the triple when the token that STOPPED the run is in
913/// [`NP_CONTINUING_PREPOSITIONS`] (the phrase was not finished, so its head is
914/// not in what we collected), and otherwise returns the phrase head. Every
915/// WordNet miss widens the accepted set rather than narrowing it, so an unknown
916/// crate name is never rejected for being unknown.
917/// Test: `row4_rewalks_past_the_adjective_to_the_head_noun`,
918/// `row2_rejects_ancestor_truncated_before_of`,
919/// `unknown_subject_and_object_both_fail_open`.
920fn select_object(right: &str, pos: &WordNetPos) -> Option<String> {
921 let toks: Vec<&str> = right.split_whitespace().collect();
922 let (run, terminated, consumed) = noun_phrase_run(&toks, pos, Walk::Right);
923 if !terminated
924 && toks
925 .get(consumed)
926 .is_some_and(|next| NP_CONTINUING_PREPOSITIONS.contains(&clean_token(next)))
927 {
928 return None;
929 }
930 phrase_head(&run, pos)
931}
932
933/// Strip surrounding punctuation from one raw token.
934///
935/// Why: #4678 — `first_token` trimmed a TRAILING run while its doc promised a
936/// leading one, and both helpers used a set that omitted the characters drawer
937/// content actually wraps names in (backticks, brackets, asterisks). So
938/// `` `redb` `` reached the graph verbatim and became a second node for an
939/// entity that already had one. Which SIDE of the marker a token sits on says
940/// nothing about which edge carries punctuation, so both helpers clean both
941/// edges through this one function rather than each guessing.
942/// What: trims whitespace, then [`TOKEN_EDGE_PUNCT`] from both ends. Interior
943/// characters are untouched, so `no-op`, `c#`, and `src/main.rs` survive whole.
944/// Test: `extract_triples_strips_punctuation_from_both_token_edges`.
945fn clean_token(raw: &str) -> &str {
946 raw.trim().trim_matches(TOKEN_EDGE_PUNCT)
947}
948
949#[cfg(test)]
950mod tests {
951 use super::*;
952
953 fn input_for(content: &str, tags: &[&str], room: Option<&str>) -> (Uuid, Vec<String>) {
954 let id = Uuid::new_v4();
955 let owned_tags: Vec<String> = tags.iter().map(|s| s.to_string()).collect();
956 let _ = content; // silence unused warning if test ignores content
957 let _ = room;
958 (id, owned_tags)
959 }
960
961 /// Why: Tag-derived triples are the lowest-hanging extraction and the
962 /// graph view's first signal when no patterns fire. The KG's temporal
963 /// model only allows one active triple per `(subject, predicate)`, so
964 /// each tag becomes its own subject (`tag:<name>`) with a `tags`
965 /// predicate pointing at the drawer.
966 /// What: One `tag:<t> tags drawer:<id>` per non-empty tag, plus
967 /// `room:<r> contains drawer:<id>` when a room is supplied.
968 /// Test: This test.
969 #[test]
970 fn extract_triples_emits_tag_triples() {
971 let (id, tags) = input_for("hello world", &["rust", "design"], Some("Backend"));
972 let triples = extract_triples(&ExtractInput {
973 drawer_id: id,
974 content: "hello world",
975 tags: &tags,
976 room: Some("Backend"),
977 });
978 let object = drawer_subject(id);
979 assert!(triples
980 .iter()
981 .any(|t| t.subject == "tag:rust" && t.predicate == "tags" && t.object == object));
982 assert!(triples
983 .iter()
984 .any(|t| t.subject == "tag:design" && t.predicate == "tags" && t.object == object));
985 assert!(triples.iter().any(|t| t.subject == "room:Backend"
986 && t.predicate == "contains"
987 && t.object == object));
988 }
989
990 /// Why: Hashtag tokens are a cheap user signal; the extractor must catch
991 /// them so the graph picks up topical entities.
992 /// What: `#rust` and `#design-doc` both become `topic:<term>
993 /// mentioned-in drawer:<id>` triples, lower-cased and deduplicated.
994 /// Test: This test.
995 #[test]
996 fn extract_triples_emits_hashtag_mentions() {
997 let (id, tags) = input_for("see #Rust and #design-doc and #rust again", &[], None);
998 let triples = extract_triples(&ExtractInput {
999 drawer_id: id,
1000 content: "see #Rust and #design-doc and #rust again",
1001 tags: &tags,
1002 room: None,
1003 });
1004 let mention_subjects: Vec<&str> = triples
1005 .iter()
1006 .filter(|t| t.predicate == "mentioned-in")
1007 .map(|t| t.subject.as_str())
1008 .collect();
1009 assert!(mention_subjects.contains(&"topic:rust"));
1010 assert!(mention_subjects.contains(&"topic:design-doc"));
1011 // Dedupe — `#rust` and `#Rust` collapse.
1012 assert_eq!(
1013 mention_subjects
1014 .iter()
1015 .filter(|s| **s == "topic:rust")
1016 .count(),
1017 1
1018 );
1019 }
1020
1021 /// Why: `is a` is the simplest NL pattern and the most common idiom in
1022 /// quick notes ("rustc is a compiler").
1023 /// What: Pattern fires once per content blob; subject and object are the
1024 /// nouns either side of the marker.
1025 /// Test: This test.
1026 #[test]
1027 fn extract_triples_extracts_is_a_pattern() {
1028 let (id, _) = input_for("rustc is a compiler for rust", &[], None);
1029 let triples = extract_triples(&ExtractInput {
1030 drawer_id: id,
1031 content: "rustc is a compiler for rust",
1032 tags: &[],
1033 room: None,
1034 });
1035 assert!(triples
1036 .iter()
1037 .any(|t| t.subject == "rustc" && t.predicate == "is-a" && t.object == "compiler"));
1038 }
1039
1040 /// Why: Confidence and provenance are guard-rails — extracted triples
1041 /// must be recognisable and over-ridable.
1042 /// What: Every triple carries `provenance = Some("auto:remember")` and
1043 /// `confidence == AUTO_CONFIDENCE`.
1044 /// Test: This test.
1045 #[test]
1046 fn extract_triples_stamps_provenance() {
1047 let (id, tags) = input_for("anything", &["x"], None);
1048 let triples = extract_triples(&ExtractInput {
1049 drawer_id: id,
1050 content: "anything",
1051 tags: &tags,
1052 room: None,
1053 });
1054 assert!(!triples.is_empty());
1055 for t in &triples {
1056 assert_eq!(t.provenance.as_deref(), Some(AUTO_PROVENANCE));
1057 assert!((t.confidence - AUTO_CONFIDENCE).abs() < f32::EPSILON);
1058 }
1059 }
1060
1061 /// Why: Reduced confidence is the contract a manual `kg_assert` of the
1062 /// same `(subject, predicate)` needs in order to "win" against the
1063 /// auto-extracted edge.
1064 /// What: Every triple carries `confidence == AUTO_CONFIDENCE` (currently
1065 /// 0.6); the constant is asserted to stay strictly below 1.0 so manual
1066 /// asserts always rank higher.
1067 /// Test: This test.
1068 #[test]
1069 #[allow(clippy::assertions_on_constants)]
1070 fn extract_triples_uses_reduced_confidence() {
1071 // Why: both bounds are static facts about the AUTO_CONFIDENCE
1072 // constant; the assertion is documentation for future tweakers.
1073 assert!(AUTO_CONFIDENCE < 1.0);
1074 assert!(AUTO_CONFIDENCE > 0.0);
1075 }
1076
1077 /// Why: Empty / whitespace-only content must not panic or emit garbage.
1078 /// What: No tags, no room, no content → empty vec.
1079 /// Test: This test.
1080 #[test]
1081 fn extract_triples_never_panics_on_empty_input() {
1082 let id = Uuid::new_v4();
1083 let triples = extract_triples(&ExtractInput {
1084 drawer_id: id,
1085 content: "",
1086 tags: &[],
1087 room: None,
1088 });
1089 assert!(triples.is_empty());
1090 }
1091
1092 /// Why: Edge-case test — content with no patterns but tags should still
1093 /// produce the tag triples (the graph view's primary signal).
1094 /// What: Single tag, no room, prose with no pattern hits → exactly one
1095 /// triple shaped as `tag:meeting tags drawer:<id>`.
1096 /// Test: This test.
1097 #[test]
1098 fn extract_triples_tags_only_path() {
1099 let id = Uuid::new_v4();
1100 let tags = vec!["meeting".to_string()];
1101 let triples = extract_triples(&ExtractInput {
1102 drawer_id: id,
1103 content: "Discussed roadmap.",
1104 tags: &tags,
1105 room: None,
1106 });
1107 assert_eq!(triples.len(), 1);
1108 assert_eq!(triples[0].subject, "tag:meeting");
1109 assert_eq!(triples[0].predicate, "tags");
1110 assert_eq!(triples[0].object, drawer_subject(id));
1111 }
1112
1113 /// Why: Drawers tagged with deny-listed labels (test fixtures, QA scaffolding)
1114 /// must not pollute the KG with non-factual content.
1115 /// What: A drawer with the `test` tag must produce zero triples even when
1116 /// it also has a room and content with extractable patterns.
1117 /// Test: This test.
1118 #[test]
1119 fn extract_triples_skips_denied_tags() {
1120 let id = Uuid::new_v4();
1121 let tags = vec!["test".to_string(), "rust".to_string()];
1122 let triples = extract_triples(&ExtractInput {
1123 drawer_id: id,
1124 content: "rustc is a compiler",
1125 tags: &tags,
1126 room: Some("Backend"),
1127 });
1128 assert!(
1129 triples.is_empty(),
1130 "a drawer with a deny-list tag must produce zero triples, got {triples:?}"
1131 );
1132 }
1133
1134 /// Why: Deny-list matching is case-insensitive so `TEST` and `Test` are
1135 /// blocked the same as `test`.
1136 /// What: A drawer tagged `FIXTURE` (upper-case) must still produce zero
1137 /// triples.
1138 /// Test: This test.
1139 #[test]
1140 fn extract_triples_deny_list_is_case_insensitive() {
1141 let id = Uuid::new_v4();
1142 let tags = vec!["FIXTURE".to_string()];
1143 let triples = extract_triples(&ExtractInput {
1144 drawer_id: id,
1145 content: "some content",
1146 tags: &tags,
1147 room: None,
1148 });
1149 assert!(
1150 triples.is_empty(),
1151 "upper-cased deny tag must still be blocked"
1152 );
1153 }
1154
1155 /// Collect only the triples produced by the [`PATTERN_TABLE`] pass.
1156 ///
1157 /// Why: every assertion about extraction precision is about the pattern
1158 /// pass; the tag / room / hashtag passes always fire and would otherwise
1159 /// mask a "zero triples" assertion.
1160 /// What: filters by predicate against [`PATTERN_TABLE`].
1161 /// Test: used by every precision test below.
1162 fn pattern_triples(triples: &[Triple]) -> Vec<(String, String, String)> {
1163 let predicates: Vec<&str> = PATTERN_TABLE.iter().map(|(p, _)| *p).collect();
1164 triples
1165 .iter()
1166 .filter(|t| predicates.contains(&t.predicate.as_str()))
1167 .map(|t| (t.subject.clone(), t.predicate.clone(), t.object.clone()))
1168 .collect()
1169 }
1170
1171 /// Run the extractor over bare content with no tags and no room.
1172 ///
1173 /// Why: the precision fixtures care only about content; tags and rooms
1174 /// would add noise triples to every assertion.
1175 /// What: builds an `ExtractInput` with empty tags and no room.
1176 /// Test: used by every precision test below.
1177 fn patterns_for(content: &str) -> Vec<(String, String, String)> {
1178 let triples = extract_triples(&ExtractInput {
1179 drawer_id: Uuid::new_v4(),
1180 content,
1181 tags: &[],
1182 room: None,
1183 });
1184 pattern_triples(&triples)
1185 }
1186
1187 /// Why: `them --is-a--> no-op` is live in the real trusty-tools palace
1188 /// (asserted 2026-08-04). A marker hit is not evidence that the tokens
1189 /// either side of it are entities; a pronoun never is one.
1190 /// What: "calling them is a no-op ..." must yield zero pattern triples.
1191 /// Test: This test.
1192 #[test]
1193 fn extract_triples_rejects_pronoun_subject() {
1194 let got = patterns_for("calling them is a no-op when the flag is off");
1195 assert!(
1196 got.is_empty(),
1197 "pronoun subject must reject the whole triple, got {got:?}"
1198 );
1199 }
1200
1201 /// Why: a stopword can land on either side of a marker, so filtering only
1202 /// the subject would still admit `libpq --depends-on--> the`.
1203 /// What: "libpq depends on the license" must yield zero pattern triples —
1204 /// the triple is rejected whole, never truncated to a half-triple.
1205 /// Test: This test.
1206 #[test]
1207 fn extract_triples_rejects_stopword_object() {
1208 let got = patterns_for("libpq depends on the license header");
1209 assert!(
1210 got.is_empty(),
1211 "stopword object must reject the whole triple, got {got:?}"
1212 );
1213 }
1214
1215 /// Why: a one- or two-character token is almost never an entity, and the
1216 /// extractor had no length floor at all.
1217 /// What: "x is a thing" must yield zero pattern triples.
1218 /// Test: This test.
1219 #[test]
1220 fn extract_triples_rejects_short_token_off_allowlist() {
1221 let got = patterns_for("x is a thing worth recording");
1222 assert!(
1223 got.is_empty(),
1224 "short token off the allowlist must be rejected, got {got:?}"
1225 );
1226 }
1227
1228 /// Why: the length floor must not swallow the short names this repo
1229 /// genuinely discusses — without the allowlist, `Go` and `C` would be
1230 /// rejected as noise and the filter would cost real recall.
1231 /// What: "Go is a language" still extracts `go --is-a--> language`.
1232 /// Test: This test.
1233 #[test]
1234 fn extract_triples_keeps_allowlisted_go() {
1235 let got = patterns_for("Go is a language with green threads");
1236 assert!(
1237 got.contains(&("go".into(), "is-a".into(), "language".into())),
1238 "allowlisted `Go` must survive the length floor, got {got:?}"
1239 );
1240 }
1241
1242 /// Why: `C` is the single-character case — the length floor's worst
1243 /// false positive if the allowlist is not consulted.
1244 /// What: "C is a language" still extracts `c --is-a--> language`.
1245 /// Test: This test.
1246 #[test]
1247 fn extract_triples_keeps_allowlisted_c() {
1248 let got = patterns_for("C is a language without a runtime");
1249 assert!(
1250 got.contains(&("c".into(), "is-a".into(), "language".into())),
1251 "allowlisted `C` must survive the length floor, got {got:?}"
1252 );
1253 }
1254
1255 /// Why: the filter must not cost recall on ordinary, well-formed content.
1256 /// A rejection filter that also rejects the good cases is a regression,
1257 /// so every predicate in [`PATTERN_TABLE`] keeps a worked example.
1258 /// What: table-driven — each row is `(content, expected triple)` and must
1259 /// appear in the extracted pattern set.
1260 /// Test: This test.
1261 #[test]
1262 fn extract_triples_keeps_real_entities_across_all_predicates() {
1263 let cases: &[(&str, (&str, &str, &str))] = &[
1264 (
1265 "tokio is an executor for async rust",
1266 ("tokio", "is-a", "executor"),
1267 ),
1268 (
1269 "alice works at initech today",
1270 ("alice", "works-at", "initech"),
1271 ),
1272 (
1273 "trusty-memory uses redb for persistence",
1274 ("trusty-memory", "uses", "redb"),
1275 ),
1276 (
1277 "trusty-search depends on trusty-common for shared helpers",
1278 ("trusty-search", "depends-on", "trusty-common"),
1279 ),
1280 (
1281 "the embedder requires onnxruntime at startup",
1282 ("embedder", "depends-on", "onnxruntime"),
1283 ),
1284 ];
1285 for (content, (s, p, o)) in cases {
1286 let got = patterns_for(content);
1287 let want = ((*s).to_string(), (*p).to_string(), (*o).to_string());
1288 assert!(
1289 got.contains(&want),
1290 "content {content:?} must still extract {want:?}, got {got:?}"
1291 );
1292 }
1293 }
1294
1295 /// Why: `first_token` trimmed only a TRAILING run while its doc promised it
1296 /// stripped the leading one, so an entity quoted the way drawer content
1297 /// actually quotes things — markdown backticks, brackets, an opening
1298 /// quote — entered the graph with the punctuation welded on: `` `redb ``
1299 /// rather than `redb`. Two spellings of one entity are two nodes that never
1300 /// join up.
1301 /// What: pins the emitted strings. Both helpers strip punctuation from BOTH
1302 /// edges, so the subject side (`last_token`) is covered too, and no emitted
1303 /// token may retain an edge character.
1304 /// Test: This test.
1305 #[test]
1306 fn extract_triples_strips_punctuation_from_both_token_edges() {
1307 let cases: &[(&str, (&str, &str, &str))] = &[
1308 // Object side, markdown backticks — the common shape in drawers.
1309 (
1310 "trusty-memory uses `redb` for persistence",
1311 ("trusty-memory", "uses", "redb"),
1312 ),
1313 // Object side, parenthesised.
1314 (
1315 "the daemon uses (tantivy) for search",
1316 ("daemon", "uses", "tantivy"),
1317 ),
1318 // Subject side, opening quote with no closing one before the marker.
1319 (
1320 "he said \"rustc is a compiler for rust",
1321 ("rustc", "is-a", "compiler"),
1322 ),
1323 // Object side, bold markdown.
1324 (
1325 "trusty-search depends on **trusty-common** for helpers",
1326 ("trusty-search", "depends-on", "trusty-common"),
1327 ),
1328 ];
1329 for (content, (s, p, o)) in cases {
1330 let got = patterns_for(content);
1331 let want = ((*s).to_string(), (*p).to_string(), (*o).to_string());
1332 assert!(
1333 got.contains(&want),
1334 "content {content:?} must emit {want:?}, got {got:?}"
1335 );
1336 for (subject, _, object) in &got {
1337 for tok in [subject, object] {
1338 assert!(
1339 !tok.starts_with(TOKEN_EDGE_PUNCT) && !tok.ends_with(TOKEN_EDGE_PUNCT),
1340 "emitted token {tok:?} still carries edge punctuation"
1341 );
1342 }
1343 }
1344 }
1345 }
1346
1347 /// Why: a duplicated entry is dead weight and a sign the categories drifted
1348 /// apart during editing.
1349 /// What: every [`STOPWORDS`] entry appears exactly once and is already
1350 /// lower-case, which is what [`is_stop_token`] compares against.
1351 /// Test: This test.
1352 #[test]
1353 fn stopwords_are_unique() {
1354 let mut seen: HashSet<&str> = HashSet::new();
1355 for w in STOPWORDS {
1356 assert!(seen.insert(w), "duplicate stopword {w:?}");
1357 assert_eq!(*w, w.to_lowercase(), "stopword {w:?} must be lower-case");
1358 }
1359 }
1360
1361 /// Why: the rejection contract is only as good as the list behind it.
1362 /// What: every [`STOPWORDS`] entry is rejected, in its own case and
1363 /// upper-cased, since content reaches the filter lower-cased but the purge
1364 /// path reads subjects straight out of the store.
1365 /// Test: This test.
1366 #[test]
1367 fn is_stop_token_rejects_every_stopword() {
1368 for w in STOPWORDS {
1369 assert!(is_stop_token(w), "{w:?} must be rejected");
1370 assert!(
1371 is_stop_token(&w.to_uppercase()),
1372 "{w:?} must be rejected case-insensitively"
1373 );
1374 }
1375 }
1376
1377 /// Why: an over-broad filter costs recall, which is the failure mode that
1378 /// would make this change a net loss.
1379 /// What: ordinary multi-character entity names are accepted.
1380 /// Test: This test.
1381 #[test]
1382 fn is_stop_token_accepts_ordinary_entities() {
1383 for tok in [
1384 "rustc",
1385 "compiler",
1386 "trusty-memory",
1387 "redb",
1388 "no-op",
1389 "onnxruntime",
1390 "initech",
1391 ] {
1392 assert!(!is_stop_token(tok), "{tok:?} must be accepted");
1393 }
1394 }
1395
1396 /// Why: `first_token` / `last_token` strip only a trailing run, so a token
1397 /// can still carry an opening bracket or backtick. Comparing that raw
1398 /// against the list would miss the stopword it is meant to catch.
1399 /// What: edge punctuation is stripped before matching, interior characters
1400 /// are not, and a token that is nothing but punctuation is rejected.
1401 /// Test: This test.
1402 #[test]
1403 fn is_stop_token_normalises_surrounding_punctuation() {
1404 assert!(
1405 is_stop_token("(the"),
1406 "leading bracket must not hide a stopword"
1407 );
1408 assert!(is_stop_token("`it`"), "backticks must not hide a stopword");
1409 assert!(
1410 is_stop_token("---"),
1411 "punctuation-only token must be rejected"
1412 );
1413 assert!(
1414 is_stop_token(" "),
1415 "whitespace-only token must be rejected"
1416 );
1417 assert!(
1418 !is_stop_token("`redb`"),
1419 "interior name must survive trimming"
1420 );
1421 assert!(
1422 !is_stop_token("no-op"),
1423 "interior hyphen must survive trimming"
1424 );
1425 }
1426
1427 /// Why: #5401's merge asks this function which node a stored term belongs
1428 /// under, and its three answers are the pass's whole selection rule.
1429 /// What: a punctuated real entity names its cleaned twin; an already-clean
1430 /// term and a punctuated stopword both name nothing — the second because it
1431 /// is `--purge-stale-subjects`'s to delete, not this pass's to re-point. A
1432 /// name whose leading punctuation IS the identity — `.env`, `--verbose` —
1433 /// names the same twin as decorative punctuation does, because the merge
1434 /// inherits [`clean_token`] rather than re-deciding.
1435 /// Test: This test.
1436 #[test]
1437 fn canonical_entity_names_the_cleaned_twin() {
1438 assert_eq!(canonical_entity("`redb`"), Some("redb"));
1439 assert_eq!(canonical_entity("(sled)"), Some("sled"));
1440 assert_eq!(canonical_entity("*trusty-memory*"), Some("trusty-memory"));
1441 assert_eq!(
1442 canonical_entity("src/main.rs,"),
1443 Some("src/main.rs"),
1444 "only edge punctuation moves; the interior path must survive"
1445 );
1446
1447 assert_eq!(canonical_entity("redb"), None, "a clean term is canonical");
1448 assert_eq!(canonical_entity("no-op"), None);
1449
1450 assert_eq!(
1451 canonical_entity("(\"the"),
1452 None,
1453 "a punctuated stopword belongs to the purge, not the merge"
1454 );
1455 assert_eq!(canonical_entity("`it`"), None);
1456 assert_eq!(canonical_entity("---"), None);
1457
1458 // The dotfile class: the leading punctuation is part of the name, and
1459 // it is collapsed anyway. Deliberate — today's extractor already emits
1460 // `env` for a drawer that says `.env`, so the merge aligns the stored
1461 // node with what extraction now produces rather than inventing a
1462 // second normalisation that could drift from it.
1463 assert_eq!(canonical_entity(".env"), Some("env"));
1464 assert_eq!(canonical_entity(".git"), Some("git"));
1465 assert_eq!(canonical_entity(".gitignore"), Some("gitignore"));
1466 assert_eq!(canonical_entity("_private"), Some("private"));
1467 assert_eq!(canonical_entity("--verbose"), Some("verbose"));
1468 }
1469
1470 /// Why: the allowlist is the length floor's escape hatch; if an entry stops
1471 /// working the floor silently starts eating real entities.
1472 /// What: every [`SHORT_ENTITY_ALLOWLIST`] entry is accepted, and no entry
1473 /// collides with [`STOPWORDS`] (which is checked first and would win).
1474 /// Test: This test.
1475 #[test]
1476 fn short_entity_allowlist_entries_survive_the_length_floor() {
1477 for tok in SHORT_ENTITY_ALLOWLIST {
1478 assert!(
1479 !STOPWORDS.contains(tok),
1480 "{tok:?} is on both lists; the stopword check runs first and wins"
1481 );
1482 assert!(!is_stop_token(tok), "allowlisted {tok:?} must be accepted");
1483 }
1484 }
1485
1486 /// Why: #4678 named three live regressions and could only reach the first.
1487 /// Its filter is lexical — it judges one token at a time — and
1488 /// `exhaustiveness`, `hard`, `squash` and `ancestor` are all ordinary
1489 /// content words, indistinguishable token-wise from the `rustc --is-a-->
1490 /// compiler` the extractor is supposed to keep. So #4678 shipped this test
1491 /// ASSERTING the residue, under the name
1492 /// `lexical_filter_does_not_reach_the_two_content_word_regressions`, as the
1493 /// standing record of an open gate.
1494 ///
1495 /// #5399 closed that gate with WordNet head-noun selection, so the
1496 /// assertion is INVERTED rather than deleted and the name is kept
1497 /// recognisable — `git log -S` on either name finds both states. The
1498 /// lexical half is unchanged and still passes: it is what proves the fix
1499 /// landed in the POS pass and not by quietly widening the stopword list,
1500 /// which would have broken `rustc --is-a--> compiler` too.
1501 /// What: asserts the four words still survive the lexical filter, AND that
1502 /// neither #4678 residue triple is produced any more.
1503 /// Test: This test.
1504 #[test]
1505 fn lexical_filter_still_cannot_reach_the_two_content_word_regressions_but_wordnet_does() {
1506 assert!(
1507 !is_stop_token("exhaustiveness")
1508 && !is_stop_token("hard")
1509 && !is_stop_token("squash")
1510 && !is_stop_token("ancestor"),
1511 "these are ordinary words; no lexical filter may reject them"
1512 );
1513 let exhaustiveness = patterns_for("match exhaustiveness is a hard requirement here");
1514 assert!(
1515 !exhaustiveness.contains(&("exhaustiveness".into(), "is-a".into(), "hard".into())),
1516 "#4678 residue must be gone; got {exhaustiveness:?}"
1517 );
1518 let squash = patterns_for("confirm the squash is an ancestor of origin main");
1519 assert!(
1520 !squash.contains(&("squash".into(), "is-a".into(), "ancestor".into())),
1521 "#4678 residue must be gone; got {squash:?}"
1522 );
1523 }
1524
1525 /// Why: An empty deny-list (e.g. in integration tests that want to exercise
1526 /// extraction regardless of tags) must not suppress any triples.
1527 /// What: Calling `extract_triples_with_config` with `deny_tags = &[]` on a
1528 /// drawer tagged `test` must produce the normal tag triple.
1529 /// Test: This test.
1530 #[test]
1531 fn extract_triples_empty_deny_list_passes_through() {
1532 let id = Uuid::new_v4();
1533 let tags = vec!["test".to_string()];
1534 let config = KgExtractConfig {
1535 deny_tags: &[],
1536 ..Default::default()
1537 };
1538 let triples = extract_triples_with_config(
1539 &ExtractInput {
1540 drawer_id: id,
1541 content: "anything",
1542 tags: &tags,
1543 room: None,
1544 },
1545 &config,
1546 );
1547 // "test" tag should produce a tag triple when the deny-list is empty.
1548 assert!(
1549 !triples.is_empty(),
1550 "empty deny-list must not suppress extraction"
1551 );
1552 }
1553}
1554
1555/// The #5399 evaluation set for the shipped WordNet policy, plus the cases that
1556/// probe where it behaves surprisingly.
1557///
1558/// Why: these rows decided the bake-off between the WordNet lane and a spaCy
1559/// lane on identical inputs, so they are the record of what was chosen. The
1560/// `surprising_*` tests are deliberately included even where they record a
1561/// WRONG answer — an eval set built only from passing rows tells the next
1562/// reader nothing about where the approach runs out.
1563/// What: each test drives `extract_triples` end to end and asserts on the
1564/// pattern-derived triples only (tag/room/hashtag triples are unrelated).
1565#[cfg(test)]
1566mod wordnet_eval {
1567 use super::*;
1568
1569 /// Pattern triples only, as `(subject, predicate, object)`.
1570 fn pattern_triples(content: &str) -> Vec<(String, String, String)> {
1571 extract_triples(&ExtractInput {
1572 drawer_id: Uuid::new_v4(),
1573 content,
1574 tags: &[],
1575 room: None,
1576 })
1577 .into_iter()
1578 .filter(|t| PATTERN_TABLE.iter().any(|(p, _)| *p == t.predicate))
1579 .map(|t| (t.subject, t.predicate, t.object))
1580 .collect()
1581 }
1582
1583 fn assert_none(content: &str) {
1584 let got = pattern_triples(content);
1585 assert!(
1586 got.is_empty(),
1587 "expected no triple from {content:?}, got {got:?}"
1588 );
1589 }
1590
1591 fn assert_one(content: &str, s: &str, p: &str, o: &str) {
1592 let got = pattern_triples(content);
1593 assert_eq!(
1594 got,
1595 vec![(s.to_string(), p.to_string(), o.to_string())],
1596 "wrong extraction from {content:?}"
1597 );
1598 }
1599
1600 // ---- Row 1: adjective-only modifier. `hard` is ADJ|ADV, never NOUN. ----
1601 //
1602 // 🔴 THIS ROW'S EXPECTATION WAS CHANGED, AND THE OLD ONE WAS WRONG. The
1603 // lane-A spike specified "NO triple" here, on the theory that a phrase
1604 // whose modifier is adjective-only is unsalvageable. That misread the
1605 // defect. #5399 is about extracting the ADJECTIVE `hard` as though it were
1606 // a type; it is not about the noun `requirement`, and
1607 // `exhaustiveness --is-a--> requirement` is a TRUE statement about this
1608 // sentence. Rejecting the whole triple threw away a good fact to avoid a
1609 // bad one.
1610 //
1611 // The corpus settled it: over 306k lines of repo markdown the reject rule
1612 // dropped 304 pairs, and 196 of them (64%) had a recoverable head noun one
1613 // token further along. 43 of 78 ordinary technical adjectives — `robust`,
1614 // `concurrent`, `embedded`, `distributed`, `immutable`, `scalable` — are
1615 // adjective-only in WordNet, so the rule fired constantly, and every time
1616 // it fired it destroyed a usable triple.
1617 //
1618 // Do NOT "fix" this back to `assert_none`. Re-walking to the head noun IS
1619 // the shipped policy; a test asserting no triple here asserts the bug.
1620 #[test]
1621 fn row1_rewalks_past_an_adjective_only_modifier() {
1622 assert_one(
1623 "match exhaustiveness is a hard requirement here",
1624 "exhaustiveness",
1625 "is-a",
1626 "requirement",
1627 );
1628 }
1629
1630 // ---- Row 2: genitive boundary. The NP is `ancestor of origin main`. ----
1631 #[test]
1632 fn row2_rejects_ancestor_truncated_before_of() {
1633 assert_none("confirm the squash is an ancestor of origin main");
1634 }
1635
1636 // ---- Row 3: the baseline good triple must survive untouched. ----
1637 #[test]
1638 fn row3_keeps_rustc_is_a_compiler() {
1639 assert_one("rustc is a compiler", "rustc", "is-a", "compiler");
1640 }
1641
1642 // ---- Row 4: head-noun re-walk past a modifier. ----
1643 #[test]
1644 fn row4_rewalks_past_the_adjective_to_the_head_noun() {
1645 assert_one("librs is a fast parser", "librs", "is-a", "parser");
1646 }
1647
1648 // ---- Row 5: `for` closes the object; it is a verb adjunct, not a NP. ----
1649 #[test]
1650 fn row5_keeps_uses_object_before_a_non_genitive_preposition() {
1651 assert_one(
1652 "trusty-memory uses redb for persistence",
1653 "trusty-memory",
1654 "uses",
1655 "redb",
1656 );
1657 }
1658
1659 // ---- Row 6: same genitive shape as row 2, different noun. ----
1660 #[test]
1661 fn row6_rejects_member_of_the_process_group() {
1662 assert_none("the daemon is a member of the process group");
1663 }
1664
1665 // ---- Row 7: right-headed compound; the head is the LAST noun. ----
1666 #[test]
1667 fn row7_takes_the_head_of_a_noun_noun_compound() {
1668 assert_one("tantivy is a search library", "tantivy", "is-a", "library");
1669 }
1670
1671 // ================== the re-walk, on the cases that drove it =============
1672
1673 /// The adjectives with no WordNet noun sense. Under the spike's reject rule
1674 /// every one of these produced nothing; under the shipped policy each
1675 /// re-walks to its head noun.
1676 #[test]
1677 fn adjective_only_modifiers_re_walk_instead_of_destroying_the_triple() {
1678 assert_one("librs is a robust parser", "librs", "is-a", "parser");
1679 assert_one(
1680 "tantivy is an embedded library",
1681 "tantivy",
1682 "is-a",
1683 "library",
1684 );
1685 assert_one("sled is a concurrent database", "sled", "is-a", "database");
1686 // A modifier WordNet DOES give a noun sense now behaves identically,
1687 // which is the point: the lexicographic accident no longer changes the
1688 // outcome, only which token the re-walk skips.
1689 assert_one("librs is a fast parser", "librs", "is-a", "parser");
1690 }
1691
1692 /// The lexicographic split the re-walk removed, measured rather than
1693 /// asserted by feel.
1694 ///
1695 /// Why: 43 of these 78 ordinary technical adjectives have no noun sense in
1696 /// WordNet 3.1. That is a property of Princeton's lexicography, not of
1697 /// anything tunable here, and it is the size of the blast radius the reject
1698 /// rule had. Pinned so a WordNet upgrade that moves it is visible.
1699 #[test]
1700 fn measures_the_adjective_only_population() {
1701 let wn = WordNetPos::shipped();
1702 let sample: &[&str] = &[
1703 "fast",
1704 "small",
1705 "simple",
1706 "modern",
1707 "lightweight",
1708 "portable",
1709 "generic",
1710 "old",
1711 "good",
1712 "great",
1713 "better",
1714 "new",
1715 "robust",
1716 "minimal",
1717 "tiny",
1718 "huge",
1719 "concurrent",
1720 "embedded",
1721 "reliable",
1722 "lazy",
1723 "secure",
1724 "rusty",
1725 "async",
1726 "asynchronous",
1727 "synchronous",
1728 "distributed",
1729 "persistent",
1730 "immutable",
1731 "mutable",
1732 "functional",
1733 "relational",
1734 "hierarchical",
1735 "incremental",
1736 "deterministic",
1737 "idempotent",
1738 "scalable",
1739 "extensible",
1740 "pluggable",
1741 "configurable",
1742 "optional",
1743 "required",
1744 "deprecated",
1745 "experimental",
1746 "stable",
1747 "unstable",
1748 "legacy",
1749 "native",
1750 "remote",
1751 "local",
1752 "static",
1753 "dynamic",
1754 "public",
1755 "private",
1756 "internal",
1757 "external",
1758 "open",
1759 "closed",
1760 "free",
1761 "paid",
1762 "commercial",
1763 "fancy",
1764 "neat",
1765 "clean",
1766 "dirty",
1767 "quick",
1768 "slow",
1769 "heavy",
1770 "light",
1771 "cheap",
1772 "expensive",
1773 "strict",
1774 "lenient",
1775 "safe",
1776 "unsafe",
1777 "correct",
1778 "incorrect",
1779 "complete",
1780 "partial",
1781 ];
1782 let adj_only = sample.iter().filter(|w| wn.is_adjective_only(w)).count();
1783 assert_eq!(
1784 (sample.len(), adj_only),
1785 (78, 43),
1786 "adjective-only population moved; re-measure before trusting the #5399 numbers"
1787 );
1788 }
1789
1790 /// A phrase with no head names nothing, so no triple — and this is the
1791 /// re-walk returning `None`, not a surviving reject rule.
1792 #[test]
1793 fn an_all_adjective_subject_yields_no_triple() {
1794 assert_none("anything robust is a compiler");
1795 }
1796
1797 /// The subject side gets the same re-walk. Without it, the old `last_token`
1798 /// would emit whichever token sat against the marker.
1799 #[test]
1800 fn subject_side_rewalks_past_an_adjective_only_token() {
1801 assert_one("the fast parser is a tool", "parser", "is-a", "tool");
1802 }
1803
1804 // ====================== markdown and phrase boundaries ==================
1805
1806 /// #5399: the terminator check used to read only the raw token's LAST
1807 /// character. `**MCP is a thin proxy.**` ends in `*`, so the `.` was missed
1808 /// and the run walked on into the next sentence, yielding
1809 /// `mcp --is-a--> sessions`. Markdown emphasis is the ordinary shape of
1810 /// drawer content, so this was never an edge case.
1811 #[test]
1812 fn stops_the_run_at_a_terminator_behind_markdown_emphasis() {
1813 assert_one(
1814 "**MCP is a thin proxy.** Sessions are cheap here",
1815 "mcp",
1816 "is-a",
1817 "proxy",
1818 );
1819 }
1820
1821 /// The trailing-punctuation scan must not fire on interior punctuation, or
1822 /// every dotted name would close its own phrase.
1823 #[test]
1824 fn interior_punctuation_does_not_terminate_the_run() {
1825 for raw in ["parser", "node.js", "src/main.rs", "**bold**", "v1.2"] {
1826 assert!(!ends_noun_phrase(raw), "{raw:?} must not terminate");
1827 }
1828 for raw in ["proxy.**", "compiler,", "thing.", "group)", "list;", "end?"] {
1829 assert!(ends_noun_phrase(raw), "{raw:?} must terminate");
1830 }
1831 }
1832
1833 /// Walking LEFT, a trailing `.` ended the PREVIOUS sentence, so the token
1834 /// carrying it belongs to that sentence and cannot join this phrase.
1835 /// Without the direction rule the head jumps the boundary and asserts
1836 /// `tantivy --is-a--> compiler` from two unrelated clauses.
1837 #[test]
1838 fn subject_walk_stops_before_a_previous_sentence() {
1839 assert_none("we shipped tantivy. robust is a compiler");
1840 }
1841
1842 // ================= multi-line bodies, the production shape ==============
1843 // 🔴 Every other fixture in this module is a SINGLE LINE, and so is the
1844 // corpus harness (`examples/kg_dump.rs` reads one line per extraction).
1845 // Production is not: `tools::helpers::auto_extract_and_assert` and
1846 // `commands::kg_rebuild::rebuild_one` both pass a whole drawer body. The
1847 // walk had no newline boundary, so it ran off the end of its own sentence
1848 // and took a head from the next line — a correct-to-wrong regression that
1849 // no single-line fixture could see. Use `--whole-file` when measuring.
1850
1851 /// A bare newline closes the phrase exactly as a period does.
1852 ///
1853 /// Against `8402bd8b` every row here produced the token from line two:
1854 /// `builds`, `builds`, `runs`, `sled`.
1855 #[test]
1856 fn the_object_walk_stops_at_a_line_break() {
1857 assert_one(
1858 "trusty-search is a daemon\ncargo builds it",
1859 "trusty-search",
1860 "is-a",
1861 "daemon",
1862 );
1863 assert_one(
1864 "rustc is a compiler\ncargo builds it",
1865 "rustc",
1866 "is-a",
1867 "compiler",
1868 );
1869 assert_one(
1870 "the parser is a tool\ncargo runs fine",
1871 "parser",
1872 "is-a",
1873 "tool",
1874 );
1875 assert_one(
1876 "redb is a database\nsled is another one",
1877 "redb",
1878 "is-a",
1879 "database",
1880 );
1881 }
1882
1883 /// The subject side takes the same boundary: a preceding line is a
1884 /// previous sentence, whether or not it ended in punctuation.
1885 #[test]
1886 fn the_subject_walk_stops_at_a_line_break() {
1887 assert_one(
1888 "we shipped tantivy\nrobust compilers are a myth\nredb is a database",
1889 "redb",
1890 "is-a",
1891 "database",
1892 );
1893 }
1894
1895 /// A trailing period already worked; it must keep working.
1896 #[test]
1897 fn a_terminated_line_is_unaffected_by_the_newline_rule() {
1898 assert_one(
1899 "trusty-search is a daemon.\ncargo builds it",
1900 "trusty-search",
1901 "is-a",
1902 "daemon",
1903 );
1904 }
1905
1906 // ============ an inflected or unknown token must not take the head ======
1907
1908 /// A participle is not a noun, and WordNet indexes only its base form.
1909 ///
1910 /// Against `8402bd8b` `containing` was unknown, and "unknown" meant
1911 /// "eligible to head a phrase", so this asserted
1912 /// `skill --is-a--> containing`. Resolving `containing` to `contain`
1913 /// (VERB-only) ends the phrase before it, as it always should have.
1914 #[test]
1915 fn a_participle_does_not_head_the_phrase() {
1916 assert_one(
1917 "each skill is a directory containing:",
1918 "skill",
1919 "is-a",
1920 "directory",
1921 );
1922 assert_one(
1923 "trusty-search is a daemon parsing every file",
1924 "trusty-search",
1925 "is-a",
1926 "daemon",
1927 );
1928 }
1929
1930 /// An unknown token may OPEN a phrase but never joins one that already has
1931 /// a head — against `8402bd8b` this asserted the whole file path as a type.
1932 ///
1933 /// The closed-class `inside` is what stops the run here: it is a
1934 /// preposition, so the phrase ends before the path regardless of whether
1935 /// WordNet has heard of the path.
1936 #[test]
1937 fn an_unknown_token_does_not_displace_an_established_head() {
1938 assert_one(
1939 "tree is a comment inside crates/trusty-search/src/allowlist/tests.rs",
1940 "tree",
1941 "is-a",
1942 "comment",
1943 );
1944 }
1945
1946 /// A plural subject survives, including one whose singular ends in `e`.
1947 ///
1948 /// Both of these yielded a triple at `8402bd8b` and nothing at `ba579925`:
1949 /// `base_form_candidates` chopped `es` before `s`, so `notes` resolved to
1950 /// the adverb `not` and `sites` to the verb `sit`, and a verb-or-adverb
1951 /// token ends the phrase. The sibilant rule in
1952 /// [`crate::wordnet_pos::base_form_candidates`] is what restores them.
1953 #[test]
1954 fn a_plural_subject_whose_singular_ends_in_e_still_heads_its_phrase() {
1955 assert_one("notes is a drawer", "notes", "is-a", "drawer");
1956 assert_one("sites is a directory", "sites", "is-a", "directory");
1957 }
1958
1959 /// An unknown identifier still heads its phrase when the token before it is
1960 /// a modifier WordNet also lists as a noun.
1961 ///
1962 /// 🔴 DO NOT "fix" this by stopping the walk at an unknown token once the
1963 /// run is non-empty. That rule was written and then removed here, measured
1964 /// over 1,277 markdown files: it changed 156 heads, and 83 of those
1965 /// replaced the real head with an adjective-capable modifier —
1966 /// `tcode|is-a|local`, `persistence|is-a|redb` -> `single`,
1967 /// `file|is-a|no-op` -> `true`. That is the #5399 defect itself arriving by
1968 /// a different route, and [`phrase_head`]'s adjective-only skip cannot
1969 /// catch it, because WordNet lists every one of those words as a noun as
1970 /// well. It also dropped 76 triples outright, several of them real facts
1971 /// (`server|uses|rocksdb`, `codebase|uses|redb`).
1972 /// Test: this is the isolating test — reintroducing the rule fails it.
1973 #[test]
1974 fn an_unknown_token_heads_its_phrase_over_an_adjective_capable_modifier() {
1975 // `local` is NOUN|ADJ, so it passes both the POS check and the
1976 // adjective-only skip; only walking on to `app` finds the real head.
1977 assert_one("tcode is a local app", "tcode", "is-a", "app");
1978 assert_one(
1979 "persistence is a single redb",
1980 "persistence",
1981 "is-a",
1982 "redb",
1983 );
1984 }
1985
1986 #[test]
1987 fn stops_the_noun_phrase_run_at_a_comma() {
1988 assert_one(
1989 "rustc is a compiler, and cargo is the build tool",
1990 "rustc",
1991 "is-a",
1992 "compiler",
1993 );
1994 }
1995
1996 #[test]
1997 fn caps_the_noun_phrase_run() {
1998 // Five noun/adjective tokens; the head must come from the first four.
1999 assert_one(
2000 "librs is a fast small modular parser core",
2001 "librs",
2002 "is-a",
2003 "parser",
2004 );
2005 }
2006
2007 // ============================ fail-open =================================
2008
2009 #[test]
2010 fn unknown_subject_and_object_both_fail_open() {
2011 assert_one("tantivy uses fst", "tantivy", "uses", "fst");
2012 }
2013
2014 /// WordNet knows ordinary English, and ordinary English words are also
2015 /// language and tool names. Nothing here rejects a known word, so these
2016 /// work — but the fail-open rule is what saves them, and any future
2017 /// "reject known common nouns" tightening would break exactly these.
2018 #[test]
2019 fn common_english_words_that_are_real_entity_names_survive() {
2020 assert_one("rust is a language", "rust", "is-a", "language");
2021 assert_one("go is a language", "go", "is-a", "language");
2022 assert_one("python is a language", "python", "is-a", "language");
2023 }
2024
2025 /// The caller's table is the one consulted — the property that replaced the
2026 /// `OnceLock`. Against a table that knows nothing, every token fails open,
2027 /// so the walk stops at the first one and reproduces the #4678 residue on
2028 /// demand. That is the proof the shipped table is what removes it.
2029 #[test]
2030 fn a_caller_supplied_pos_table_is_the_one_consulted() {
2031 let config = KgExtractConfig {
2032 pos: WordNetPos::from_table("zzz\t1\n"),
2033 ..Default::default()
2034 };
2035 let got: Vec<_> = extract_triples_with_config(
2036 &ExtractInput {
2037 drawer_id: Uuid::new_v4(),
2038 content: "match exhaustiveness is a hard requirement here",
2039 tags: &[],
2040 room: None,
2041 },
2042 &config,
2043 )
2044 .into_iter()
2045 .filter(|t| PATTERN_TABLE.iter().any(|(p, _)| *p == t.predicate))
2046 .map(|t| (t.subject, t.predicate, t.object))
2047 .collect();
2048 assert_eq!(
2049 got,
2050 vec![(
2051 "exhaustiveness".to_string(),
2052 "is-a".to_string(),
2053 "hard".to_string()
2054 )]
2055 );
2056 }
2057
2058 // ================= where this approach behaves surprisingly =============
2059 // Each of these is a KNOWN limitation, filed separately and deliberately
2060 // NOT fixed in #5399. They are pinned so that a future change altering them
2061 // does so on purpose.
2062
2063 /// Only `of` is treated as NP-continuing, so the same truncation slips
2064 /// through with every other preposition. Widening the set breaks row 5
2065 /// (`uses redb for persistence`); what actually separates the two is
2066 /// relational-noun complement requirements, which WordNet does not encode.
2067 #[test]
2068 fn surprising_non_genitive_relational_phrase_still_truncates() {
2069 assert_one(
2070 "the daemon is a participant in the process group",
2071 "daemon",
2072 "is-a",
2073 "participant",
2074 );
2075 }
2076
2077 /// A regular inflection is POS-checked through its base form.
2078 ///
2079 /// 🔴 THIS EXPECTATION WAS CHANGED, AND THE OLD ONE DESCRIBED A DEFECT.
2080 /// It used to assert `mask("parsers") == 0` and file that under "known
2081 /// limitations, benign" — WordNet indexes base forms, so an inflected word
2082 /// simply took the fail-open path. That reading missed what fail-open
2083 /// MEANS inside the walk: unknown is what makes a token eligible to head a
2084 /// phrase, so every participle became a candidate head and won the slot by
2085 /// sitting rightmost. `a directory containing:` asserted `containing` as
2086 /// the type. So the limitation was not benign, and the fix is to resolve
2087 /// the base form rather than to refuse unknown tokens — refusing them
2088 /// outright would take `parsers` with it, and `parsers` is a real head.
2089 #[test]
2090 fn inflected_forms_resolve_through_their_base_form() {
2091 let wn = WordNetPos::shipped();
2092 assert!(
2093 wn.is_noun("parsers"),
2094 "a regular plural keeps its noun sense"
2095 );
2096 assert_eq!(wn.mask("containing") & wordnet_pos::NOUN, 0);
2097 assert_one("librs is a fast parsers", "librs", "is-a", "parsers");
2098 }
2099
2100 /// A gerund that WordNet lists as a noun in its own right still takes the
2101 /// head, because membership cannot tell a nominalisation from a participle.
2102 ///
2103 /// `running` is NOUN|ADJ and `clearing` is NOUN, so no base-form retry
2104 /// fires and both read as ordinary nouns. Stopping the run at any `-ing`
2105 /// word whose base is a verb WOULD fix these two, and it was measured over
2106 /// this repo's markdown before being rejected. The count first recorded
2107 /// here — "repairs ~25, breaks ~18" — was wrong and flattered the decision;
2108 /// the break side is the larger one. Over the 1,277-file corpus the rule
2109 /// reaches 71 head positions ending in `-ing` (47 distinct), and most of
2110 /// them are ordinary nouns it would destroy: `running`, `mapping`,
2111 /// `warning`, `tooling`, `ranking`, `understanding`. Telling a
2112 /// nominalisation from a participle is syntax, not lexical membership, so
2113 /// it needs a tagger rather than a wider suffix table.
2114 #[test]
2115 fn surprising_a_gerund_noun_still_takes_the_head() {
2116 assert_one(
2117 "session creation depends on daemon running",
2118 "creation",
2119 "depends-on",
2120 "running",
2121 );
2122 assert_one(
2123 "budget_tokens is an estimate clearing a budget",
2124 "budget_tokens",
2125 "is-a",
2126 "clearing",
2127 );
2128 }
2129
2130 /// Two unknown tokens in a row: the run stops at the first, so the head is
2131 /// the first name rather than the last. Right for `uses redb for X`, wrong
2132 /// for a genuine unknown-unknown compound.
2133 #[test]
2134 fn surprising_run_stops_at_the_first_unknown_token() {
2135 assert_one(
2136 "trusty-memory uses redb sled",
2137 "trusty-memory",
2138 "uses",
2139 "redb",
2140 );
2141 }
2142
2143 /// A single-token head truncates a compound.
2144 #[test]
2145 fn surprising_single_token_head_truncates_a_compound() {
2146 assert_one(
2147 "brew is a system package managers",
2148 "brew",
2149 "is-a",
2150 "managers",
2151 );
2152 }
2153}