use crate::wordnet_pos::{self, WordNetPos};
use chrono::Utc;
use std::collections::HashSet;
use trusty_common::memory_core::store::kg::Triple;
use uuid::Uuid;
pub const DEFAULT_DENY_TAGS: &[&str] = &["cross-project-qa", "test", "fixture"];
#[derive(Debug, Clone)]
#[non_exhaustive]
pub struct KgExtractConfig<'a> {
pub deny_tags: &'a [&'a str],
pub pos: WordNetPos,
}
impl Default for KgExtractConfig<'_> {
fn default() -> Self {
Self {
deny_tags: DEFAULT_DENY_TAGS,
pos: WordNetPos::shipped(),
}
}
}
pub const AUTO_PROVENANCE: &str = "auto:remember";
pub const AUTO_CONFIDENCE: f32 = 0.6;
pub const DRAWER_SUBJECT_PREFIX: &str = "drawer:";
pub const TAG_SUBJECT_PREFIX: &str = "tag:";
pub const TOPIC_SUBJECT_PREFIX: &str = "topic:";
pub const ROOM_SUBJECT_PREFIX: &str = "room:";
pub fn drawer_subject(id: Uuid) -> String {
format!("{DRAWER_SUBJECT_PREFIX}{id}")
}
#[derive(Debug, Clone)]
pub struct ExtractInput<'a> {
pub drawer_id: Uuid,
pub content: &'a str,
pub tags: &'a [String],
pub room: Option<&'a str>,
}
pub fn extract_triples(input: &ExtractInput<'_>) -> Vec<Triple> {
extract_triples_with_config(input, &KgExtractConfig::default())
}
pub fn extract_triples_with_config(
input: &ExtractInput<'_>,
config: &KgExtractConfig<'_>,
) -> Vec<Triple> {
let denied = input.tags.iter().any(|t| {
let lower = t.trim().to_lowercase();
config.deny_tags.contains(&lower.as_str())
});
if denied {
tracing::debug!(
drawer_id = %input.drawer_id,
tags = ?input.tags,
"kg_extract: skipping drawer — tag matches deny-list"
);
return Vec::new();
}
let now = Utc::now();
let subject = drawer_subject(input.drawer_id);
let mut out: Vec<Triple> = Vec::new();
let mut seen: HashSet<(String, String, String)> = HashSet::new();
let push = |out: &mut Vec<Triple>,
seen: &mut HashSet<(String, String, String)>,
s: String,
p: String,
o: String| {
let key = (s.clone(), p.clone(), o.clone());
if seen.insert(key) {
out.push(Triple {
subject: s,
predicate: p,
object: o,
valid_from: now,
valid_to: None,
confidence: AUTO_CONFIDENCE,
provenance: Some(AUTO_PROVENANCE.to_string()),
});
}
};
for tag in input.tags {
let clean = tag.trim();
if clean.is_empty() {
continue;
}
push(
&mut out,
&mut seen,
format!("{TAG_SUBJECT_PREFIX}{}", clean.to_lowercase()),
"tags".to_string(),
subject.clone(),
);
}
if let Some(room) = input.room {
let clean = room.trim();
if !clean.is_empty() {
push(
&mut out,
&mut seen,
format!("{ROOM_SUBJECT_PREFIX}{clean}"),
"contains".to_string(),
subject.clone(),
);
}
}
for term in extract_hashtags(input.content) {
push(
&mut out,
&mut seen,
format!("{TOPIC_SUBJECT_PREFIX}{term}"),
"mentioned-in".to_string(),
subject.clone(),
);
}
for (s, p, o) in extract_patterns(input.content, &config.pos) {
push(&mut out, &mut seen, s, p, o);
}
out
}
fn extract_hashtags(content: &str) -> Vec<String> {
let mut out: Vec<String> = Vec::new();
let mut seen: HashSet<String> = HashSet::new();
let mut iter = content.char_indices().peekable();
while let Some((_, c)) = iter.next() {
if c != '#' {
continue;
}
let mut term = String::new();
while let Some(&(_, nc)) = iter.peek() {
if nc.is_ascii_alphanumeric() || nc == '_' || nc == '-' {
term.push(nc.to_ascii_lowercase());
iter.next();
} else {
break;
}
}
if term.is_empty() {
continue;
}
if seen.insert(term.clone()) {
out.push(term);
}
}
out
}
const PATTERN_TABLE: &[(&str, &[&str])] = &[
("is-a", &[" is a ", " is an "]),
("works-at", &[" works at "]),
("uses", &[" uses ", " using "]),
("depends-on", &[" depends on ", " requires "]),
];
const STOPWORDS: &[&str] = &[
"a",
"an",
"the",
"this",
"that",
"these",
"those",
"some",
"any",
"each",
"every",
"all",
"both",
"either",
"neither",
"another",
"such",
"same",
"no",
"none",
"i",
"me",
"my",
"mine",
"myself",
"we",
"us",
"our",
"ours",
"ourselves",
"you",
"your",
"yours",
"yourself",
"he",
"him",
"his",
"himself",
"she",
"her",
"hers",
"herself",
"it",
"its",
"itself",
"they",
"them",
"their",
"theirs",
"themselves",
"who",
"whom",
"whose",
"which",
"what",
"one",
"ones",
"someone",
"something",
"anyone",
"anything",
"everyone",
"everything",
"nobody",
"nothing",
"others",
"of",
"in",
"on",
"at",
"to",
"for",
"with",
"by",
"from",
"about",
"into",
"onto",
"over",
"under",
"below",
"above",
"between",
"among",
"through",
"during",
"before",
"after",
"across",
"against",
"within",
"without",
"upon",
"per",
"via",
"than",
"toward",
"towards",
"off",
"out",
"up",
"down",
"near",
"around",
"behind",
"beyond",
"beside",
"along",
"past",
"throughout",
"inside",
"outside",
"and",
"or",
"but",
"nor",
"so",
"yet",
"if",
"then",
"else",
"because",
"while",
"when",
"where",
"whether",
"though",
"although",
"since",
"as",
"unless",
"until",
"whereas",
"is",
"are",
"was",
"were",
"be",
"been",
"being",
"am",
"do",
"does",
"did",
"done",
"doing",
"has",
"have",
"had",
"having",
"can",
"could",
"will",
"would",
"shall",
"should",
"may",
"might",
"must",
"ought",
"need",
"needs",
"let",
"lets",
"get",
"gets",
"got",
"gotten",
"not",
"only",
"just",
"also",
"very",
"too",
"still",
"already",
"always",
"never",
"often",
"again",
"here",
"there",
"now",
"actually",
"really",
"simply",
"merely",
"quite",
"rather",
"even",
"ever",
"once",
"yes",
"well",
"much",
"many",
"more",
"most",
"less",
"least",
"few",
"several",
"enough",
"almost",
"perhaps",
"maybe",
"instead",
"otherwise",
"hence",
"therefore",
"however",
"thus",
"moreover",
];
pub const MIN_ENTITY_TOKEN_LEN: usize = 3;
pub const SHORT_ENTITY_ALLOWLIST: &[&str] = &[
"go", "c", "c#", "js", "ts", "py", "ml", "ai", "kg", "db", "ui", "os", "io", "vm", "ip", "pr", "ci", "qa", "pm", "tm", "tc", "ta",
];
const TOKEN_EDGE_PUNCT: &[char] = &[
'(', ')', '[', ']', '{', '}', '<', '>', '"', '\'', '`', ',', '.', ';', ':', '!', '?', '*', '_',
'-', '/', '\\', '|', '—', '–', '…', '“', '”', '‘', '’',
];
pub fn is_stop_token(tok: &str) -> bool {
let norm = clean_token(tok).to_lowercase();
if norm.is_empty() {
return true;
}
if STOPWORDS.contains(&norm.as_str()) {
return true;
}
norm.chars().count() < MIN_ENTITY_TOKEN_LEN && !SHORT_ENTITY_ALLOWLIST.contains(&norm.as_str())
}
fn extract_patterns(content: &str, pos: &WordNetPos) -> Vec<(String, String, String)> {
let lower = content.to_lowercase();
let mut out: Vec<(String, String, String)> = Vec::new();
for (predicate, markers) in PATTERN_TABLE {
for marker in *markers {
if let Some(idx) = lower.find(marker) {
let line_start = lower[..idx].rfind('\n').map_or(0, |p| p + 1);
let left = lower[line_start..idx].trim();
let right_start = idx + marker.len();
let line_end = lower[right_start..]
.find('\n')
.map_or(lower.len(), |p| right_start + p);
let right = lower[right_start..line_end].trim();
if let (Some(subject_tok), Some(object_tok)) =
(select_subject(left, pos), select_object(right, pos))
{
out.push((subject_tok, (*predicate).to_string(), object_tok));
}
break;
}
}
}
out
}
const NP_CONTINUING_PREPOSITIONS: &[&str] = &["of"];
const NP_TERMINATING_PUNCT: &[char] = &['.', ',', ';', ':', '!', '?', ')'];
const NP_RUN_MAX: usize = 4;
fn ends_noun_phrase(raw: &str) -> bool {
raw.trim_end()
.chars()
.rev()
.take_while(|c| TOKEN_EDGE_PUNCT.contains(c))
.any(|c| NP_TERMINATING_PUNCT.contains(&c))
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum Walk {
Right,
Left,
}
fn noun_phrase_run<'a>(
toks: &[&'a str],
pos: &WordNetPos,
walk: Walk,
) -> (Vec<&'a str>, bool, usize) {
let mut run: Vec<&str> = Vec::new();
let mut idx = 0usize;
let mut terminated = false;
while idx < toks.len() && run.len() < NP_RUN_MAX {
let raw = toks[idx];
let tok = clean_token(raw);
if tok.is_empty() || is_stop_token(tok) {
break;
}
let closes = ends_noun_phrase(raw);
if closes && walk == Walk::Left {
terminated = true;
break;
}
let mask = pos.mask(tok);
if mask != 0 && mask & (wordnet_pos::NOUN | wordnet_pos::ADJ) == 0 {
break;
}
run.push(tok);
idx += 1;
if closes {
terminated = true;
break;
}
if mask == 0 {
break;
}
}
if walk == Walk::Left {
run.reverse();
}
(run, terminated, idx)
}
fn phrase_head(run: &[&str], pos: &WordNetPos) -> Option<String> {
run.iter()
.rev()
.find(|t| !pos.is_adjective_only(t))
.map(|t| (*t).to_string())
}
fn select_subject(left: &str, pos: &WordNetPos) -> Option<String> {
let mut toks: Vec<&str> = left.split_whitespace().collect();
toks.reverse();
let (run, _, _) = noun_phrase_run(&toks, pos, Walk::Left);
phrase_head(&run, pos)
}
fn select_object(right: &str, pos: &WordNetPos) -> Option<String> {
let toks: Vec<&str> = right.split_whitespace().collect();
let (run, terminated, consumed) = noun_phrase_run(&toks, pos, Walk::Right);
if !terminated
&& toks
.get(consumed)
.is_some_and(|next| NP_CONTINUING_PREPOSITIONS.contains(&clean_token(next)))
{
return None;
}
phrase_head(&run, pos)
}
fn clean_token(raw: &str) -> &str {
raw.trim().trim_matches(TOKEN_EDGE_PUNCT)
}
#[cfg(test)]
mod tests {
use super::*;
fn input_for(content: &str, tags: &[&str], room: Option<&str>) -> (Uuid, Vec<String>) {
let id = Uuid::new_v4();
let owned_tags: Vec<String> = tags.iter().map(|s| s.to_string()).collect();
let _ = content; let _ = room;
(id, owned_tags)
}
#[test]
fn extract_triples_emits_tag_triples() {
let (id, tags) = input_for("hello world", &["rust", "design"], Some("Backend"));
let triples = extract_triples(&ExtractInput {
drawer_id: id,
content: "hello world",
tags: &tags,
room: Some("Backend"),
});
let object = drawer_subject(id);
assert!(triples
.iter()
.any(|t| t.subject == "tag:rust" && t.predicate == "tags" && t.object == object));
assert!(triples
.iter()
.any(|t| t.subject == "tag:design" && t.predicate == "tags" && t.object == object));
assert!(triples.iter().any(|t| t.subject == "room:Backend"
&& t.predicate == "contains"
&& t.object == object));
}
#[test]
fn extract_triples_emits_hashtag_mentions() {
let (id, tags) = input_for("see #Rust and #design-doc and #rust again", &[], None);
let triples = extract_triples(&ExtractInput {
drawer_id: id,
content: "see #Rust and #design-doc and #rust again",
tags: &tags,
room: None,
});
let mention_subjects: Vec<&str> = triples
.iter()
.filter(|t| t.predicate == "mentioned-in")
.map(|t| t.subject.as_str())
.collect();
assert!(mention_subjects.contains(&"topic:rust"));
assert!(mention_subjects.contains(&"topic:design-doc"));
assert_eq!(
mention_subjects
.iter()
.filter(|s| **s == "topic:rust")
.count(),
1
);
}
#[test]
fn extract_triples_extracts_is_a_pattern() {
let (id, _) = input_for("rustc is a compiler for rust", &[], None);
let triples = extract_triples(&ExtractInput {
drawer_id: id,
content: "rustc is a compiler for rust",
tags: &[],
room: None,
});
assert!(triples
.iter()
.any(|t| t.subject == "rustc" && t.predicate == "is-a" && t.object == "compiler"));
}
#[test]
fn extract_triples_stamps_provenance() {
let (id, tags) = input_for("anything", &["x"], None);
let triples = extract_triples(&ExtractInput {
drawer_id: id,
content: "anything",
tags: &tags,
room: None,
});
assert!(!triples.is_empty());
for t in &triples {
assert_eq!(t.provenance.as_deref(), Some(AUTO_PROVENANCE));
assert!((t.confidence - AUTO_CONFIDENCE).abs() < f32::EPSILON);
}
}
#[test]
#[allow(clippy::assertions_on_constants)]
fn extract_triples_uses_reduced_confidence() {
assert!(AUTO_CONFIDENCE < 1.0);
assert!(AUTO_CONFIDENCE > 0.0);
}
#[test]
fn extract_triples_never_panics_on_empty_input() {
let id = Uuid::new_v4();
let triples = extract_triples(&ExtractInput {
drawer_id: id,
content: "",
tags: &[],
room: None,
});
assert!(triples.is_empty());
}
#[test]
fn extract_triples_tags_only_path() {
let id = Uuid::new_v4();
let tags = vec!["meeting".to_string()];
let triples = extract_triples(&ExtractInput {
drawer_id: id,
content: "Discussed roadmap.",
tags: &tags,
room: None,
});
assert_eq!(triples.len(), 1);
assert_eq!(triples[0].subject, "tag:meeting");
assert_eq!(triples[0].predicate, "tags");
assert_eq!(triples[0].object, drawer_subject(id));
}
#[test]
fn extract_triples_skips_denied_tags() {
let id = Uuid::new_v4();
let tags = vec!["test".to_string(), "rust".to_string()];
let triples = extract_triples(&ExtractInput {
drawer_id: id,
content: "rustc is a compiler",
tags: &tags,
room: Some("Backend"),
});
assert!(
triples.is_empty(),
"a drawer with a deny-list tag must produce zero triples, got {triples:?}"
);
}
#[test]
fn extract_triples_deny_list_is_case_insensitive() {
let id = Uuid::new_v4();
let tags = vec!["FIXTURE".to_string()];
let triples = extract_triples(&ExtractInput {
drawer_id: id,
content: "some content",
tags: &tags,
room: None,
});
assert!(
triples.is_empty(),
"upper-cased deny tag must still be blocked"
);
}
fn pattern_triples(triples: &[Triple]) -> Vec<(String, String, String)> {
let predicates: Vec<&str> = PATTERN_TABLE.iter().map(|(p, _)| *p).collect();
triples
.iter()
.filter(|t| predicates.contains(&t.predicate.as_str()))
.map(|t| (t.subject.clone(), t.predicate.clone(), t.object.clone()))
.collect()
}
fn patterns_for(content: &str) -> Vec<(String, String, String)> {
let triples = extract_triples(&ExtractInput {
drawer_id: Uuid::new_v4(),
content,
tags: &[],
room: None,
});
pattern_triples(&triples)
}
#[test]
fn extract_triples_rejects_pronoun_subject() {
let got = patterns_for("calling them is a no-op when the flag is off");
assert!(
got.is_empty(),
"pronoun subject must reject the whole triple, got {got:?}"
);
}
#[test]
fn extract_triples_rejects_stopword_object() {
let got = patterns_for("libpq depends on the license header");
assert!(
got.is_empty(),
"stopword object must reject the whole triple, got {got:?}"
);
}
#[test]
fn extract_triples_rejects_short_token_off_allowlist() {
let got = patterns_for("x is a thing worth recording");
assert!(
got.is_empty(),
"short token off the allowlist must be rejected, got {got:?}"
);
}
#[test]
fn extract_triples_keeps_allowlisted_go() {
let got = patterns_for("Go is a language with green threads");
assert!(
got.contains(&("go".into(), "is-a".into(), "language".into())),
"allowlisted `Go` must survive the length floor, got {got:?}"
);
}
#[test]
fn extract_triples_keeps_allowlisted_c() {
let got = patterns_for("C is a language without a runtime");
assert!(
got.contains(&("c".into(), "is-a".into(), "language".into())),
"allowlisted `C` must survive the length floor, got {got:?}"
);
}
#[test]
fn extract_triples_keeps_real_entities_across_all_predicates() {
let cases: &[(&str, (&str, &str, &str))] = &[
(
"tokio is an executor for async rust",
("tokio", "is-a", "executor"),
),
(
"alice works at initech today",
("alice", "works-at", "initech"),
),
(
"trusty-memory uses redb for persistence",
("trusty-memory", "uses", "redb"),
),
(
"trusty-search depends on trusty-common for shared helpers",
("trusty-search", "depends-on", "trusty-common"),
),
(
"the embedder requires onnxruntime at startup",
("embedder", "depends-on", "onnxruntime"),
),
];
for (content, (s, p, o)) in cases {
let got = patterns_for(content);
let want = ((*s).to_string(), (*p).to_string(), (*o).to_string());
assert!(
got.contains(&want),
"content {content:?} must still extract {want:?}, got {got:?}"
);
}
}
#[test]
fn extract_triples_strips_punctuation_from_both_token_edges() {
let cases: &[(&str, (&str, &str, &str))] = &[
(
"trusty-memory uses `redb` for persistence",
("trusty-memory", "uses", "redb"),
),
(
"the daemon uses (tantivy) for search",
("daemon", "uses", "tantivy"),
),
(
"he said \"rustc is a compiler for rust",
("rustc", "is-a", "compiler"),
),
(
"trusty-search depends on **trusty-common** for helpers",
("trusty-search", "depends-on", "trusty-common"),
),
];
for (content, (s, p, o)) in cases {
let got = patterns_for(content);
let want = ((*s).to_string(), (*p).to_string(), (*o).to_string());
assert!(
got.contains(&want),
"content {content:?} must emit {want:?}, got {got:?}"
);
for (subject, _, object) in &got {
for tok in [subject, object] {
assert!(
!tok.starts_with(TOKEN_EDGE_PUNCT) && !tok.ends_with(TOKEN_EDGE_PUNCT),
"emitted token {tok:?} still carries edge punctuation"
);
}
}
}
}
#[test]
fn stopwords_are_unique() {
let mut seen: HashSet<&str> = HashSet::new();
for w in STOPWORDS {
assert!(seen.insert(w), "duplicate stopword {w:?}");
assert_eq!(*w, w.to_lowercase(), "stopword {w:?} must be lower-case");
}
}
#[test]
fn is_stop_token_rejects_every_stopword() {
for w in STOPWORDS {
assert!(is_stop_token(w), "{w:?} must be rejected");
assert!(
is_stop_token(&w.to_uppercase()),
"{w:?} must be rejected case-insensitively"
);
}
}
#[test]
fn is_stop_token_accepts_ordinary_entities() {
for tok in [
"rustc",
"compiler",
"trusty-memory",
"redb",
"no-op",
"onnxruntime",
"initech",
] {
assert!(!is_stop_token(tok), "{tok:?} must be accepted");
}
}
#[test]
fn is_stop_token_normalises_surrounding_punctuation() {
assert!(
is_stop_token("(the"),
"leading bracket must not hide a stopword"
);
assert!(is_stop_token("`it`"), "backticks must not hide a stopword");
assert!(
is_stop_token("---"),
"punctuation-only token must be rejected"
);
assert!(
is_stop_token(" "),
"whitespace-only token must be rejected"
);
assert!(
!is_stop_token("`redb`"),
"interior name must survive trimming"
);
assert!(
!is_stop_token("no-op"),
"interior hyphen must survive trimming"
);
}
#[test]
fn short_entity_allowlist_entries_survive_the_length_floor() {
for tok in SHORT_ENTITY_ALLOWLIST {
assert!(
!STOPWORDS.contains(tok),
"{tok:?} is on both lists; the stopword check runs first and wins"
);
assert!(!is_stop_token(tok), "allowlisted {tok:?} must be accepted");
}
}
#[test]
fn lexical_filter_still_cannot_reach_the_two_content_word_regressions_but_wordnet_does() {
assert!(
!is_stop_token("exhaustiveness")
&& !is_stop_token("hard")
&& !is_stop_token("squash")
&& !is_stop_token("ancestor"),
"these are ordinary words; no lexical filter may reject them"
);
let exhaustiveness = patterns_for("match exhaustiveness is a hard requirement here");
assert!(
!exhaustiveness.contains(&("exhaustiveness".into(), "is-a".into(), "hard".into())),
"#4678 residue must be gone; got {exhaustiveness:?}"
);
let squash = patterns_for("confirm the squash is an ancestor of origin main");
assert!(
!squash.contains(&("squash".into(), "is-a".into(), "ancestor".into())),
"#4678 residue must be gone; got {squash:?}"
);
}
#[test]
fn extract_triples_empty_deny_list_passes_through() {
let id = Uuid::new_v4();
let tags = vec!["test".to_string()];
let config = KgExtractConfig {
deny_tags: &[],
..Default::default()
};
let triples = extract_triples_with_config(
&ExtractInput {
drawer_id: id,
content: "anything",
tags: &tags,
room: None,
},
&config,
);
assert!(
!triples.is_empty(),
"empty deny-list must not suppress extraction"
);
}
}
#[cfg(test)]
mod wordnet_eval {
use super::*;
fn pattern_triples(content: &str) -> Vec<(String, String, String)> {
extract_triples(&ExtractInput {
drawer_id: Uuid::new_v4(),
content,
tags: &[],
room: None,
})
.into_iter()
.filter(|t| PATTERN_TABLE.iter().any(|(p, _)| *p == t.predicate))
.map(|t| (t.subject, t.predicate, t.object))
.collect()
}
fn assert_none(content: &str) {
let got = pattern_triples(content);
assert!(
got.is_empty(),
"expected no triple from {content:?}, got {got:?}"
);
}
fn assert_one(content: &str, s: &str, p: &str, o: &str) {
let got = pattern_triples(content);
assert_eq!(
got,
vec![(s.to_string(), p.to_string(), o.to_string())],
"wrong extraction from {content:?}"
);
}
#[test]
fn row1_rewalks_past_an_adjective_only_modifier() {
assert_one(
"match exhaustiveness is a hard requirement here",
"exhaustiveness",
"is-a",
"requirement",
);
}
#[test]
fn row2_rejects_ancestor_truncated_before_of() {
assert_none("confirm the squash is an ancestor of origin main");
}
#[test]
fn row3_keeps_rustc_is_a_compiler() {
assert_one("rustc is a compiler", "rustc", "is-a", "compiler");
}
#[test]
fn row4_rewalks_past_the_adjective_to_the_head_noun() {
assert_one("librs is a fast parser", "librs", "is-a", "parser");
}
#[test]
fn row5_keeps_uses_object_before_a_non_genitive_preposition() {
assert_one(
"trusty-memory uses redb for persistence",
"trusty-memory",
"uses",
"redb",
);
}
#[test]
fn row6_rejects_member_of_the_process_group() {
assert_none("the daemon is a member of the process group");
}
#[test]
fn row7_takes_the_head_of_a_noun_noun_compound() {
assert_one("tantivy is a search library", "tantivy", "is-a", "library");
}
#[test]
fn adjective_only_modifiers_re_walk_instead_of_destroying_the_triple() {
assert_one("librs is a robust parser", "librs", "is-a", "parser");
assert_one(
"tantivy is an embedded library",
"tantivy",
"is-a",
"library",
);
assert_one("sled is a concurrent database", "sled", "is-a", "database");
assert_one("librs is a fast parser", "librs", "is-a", "parser");
}
#[test]
fn measures_the_adjective_only_population() {
let wn = WordNetPos::shipped();
let sample: &[&str] = &[
"fast",
"small",
"simple",
"modern",
"lightweight",
"portable",
"generic",
"old",
"good",
"great",
"better",
"new",
"robust",
"minimal",
"tiny",
"huge",
"concurrent",
"embedded",
"reliable",
"lazy",
"secure",
"rusty",
"async",
"asynchronous",
"synchronous",
"distributed",
"persistent",
"immutable",
"mutable",
"functional",
"relational",
"hierarchical",
"incremental",
"deterministic",
"idempotent",
"scalable",
"extensible",
"pluggable",
"configurable",
"optional",
"required",
"deprecated",
"experimental",
"stable",
"unstable",
"legacy",
"native",
"remote",
"local",
"static",
"dynamic",
"public",
"private",
"internal",
"external",
"open",
"closed",
"free",
"paid",
"commercial",
"fancy",
"neat",
"clean",
"dirty",
"quick",
"slow",
"heavy",
"light",
"cheap",
"expensive",
"strict",
"lenient",
"safe",
"unsafe",
"correct",
"incorrect",
"complete",
"partial",
];
let adj_only = sample.iter().filter(|w| wn.is_adjective_only(w)).count();
assert_eq!(
(sample.len(), adj_only),
(78, 43),
"adjective-only population moved; re-measure before trusting the #5399 numbers"
);
}
#[test]
fn an_all_adjective_subject_yields_no_triple() {
assert_none("anything robust is a compiler");
}
#[test]
fn subject_side_rewalks_past_an_adjective_only_token() {
assert_one("the fast parser is a tool", "parser", "is-a", "tool");
}
#[test]
fn stops_the_run_at_a_terminator_behind_markdown_emphasis() {
assert_one(
"**MCP is a thin proxy.** Sessions are cheap here",
"mcp",
"is-a",
"proxy",
);
}
#[test]
fn interior_punctuation_does_not_terminate_the_run() {
for raw in ["parser", "node.js", "src/main.rs", "**bold**", "v1.2"] {
assert!(!ends_noun_phrase(raw), "{raw:?} must not terminate");
}
for raw in ["proxy.**", "compiler,", "thing.", "group)", "list;", "end?"] {
assert!(ends_noun_phrase(raw), "{raw:?} must terminate");
}
}
#[test]
fn subject_walk_stops_before_a_previous_sentence() {
assert_none("we shipped tantivy. robust is a compiler");
}
#[test]
fn the_object_walk_stops_at_a_line_break() {
assert_one(
"trusty-search is a daemon\ncargo builds it",
"trusty-search",
"is-a",
"daemon",
);
assert_one(
"rustc is a compiler\ncargo builds it",
"rustc",
"is-a",
"compiler",
);
assert_one(
"the parser is a tool\ncargo runs fine",
"parser",
"is-a",
"tool",
);
assert_one(
"redb is a database\nsled is another one",
"redb",
"is-a",
"database",
);
}
#[test]
fn the_subject_walk_stops_at_a_line_break() {
assert_one(
"we shipped tantivy\nrobust compilers are a myth\nredb is a database",
"redb",
"is-a",
"database",
);
}
#[test]
fn a_terminated_line_is_unaffected_by_the_newline_rule() {
assert_one(
"trusty-search is a daemon.\ncargo builds it",
"trusty-search",
"is-a",
"daemon",
);
}
#[test]
fn a_participle_does_not_head_the_phrase() {
assert_one(
"each skill is a directory containing:",
"skill",
"is-a",
"directory",
);
assert_one(
"trusty-search is a daemon parsing every file",
"trusty-search",
"is-a",
"daemon",
);
}
#[test]
fn an_unknown_token_does_not_displace_an_established_head() {
assert_one(
"tree is a comment inside crates/trusty-search/src/allowlist/tests.rs",
"tree",
"is-a",
"comment",
);
}
#[test]
fn a_plural_subject_whose_singular_ends_in_e_still_heads_its_phrase() {
assert_one("notes is a drawer", "notes", "is-a", "drawer");
assert_one("sites is a directory", "sites", "is-a", "directory");
}
#[test]
fn an_unknown_token_heads_its_phrase_over_an_adjective_capable_modifier() {
assert_one("tcode is a local app", "tcode", "is-a", "app");
assert_one(
"persistence is a single redb",
"persistence",
"is-a",
"redb",
);
}
#[test]
fn stops_the_noun_phrase_run_at_a_comma() {
assert_one(
"rustc is a compiler, and cargo is the build tool",
"rustc",
"is-a",
"compiler",
);
}
#[test]
fn caps_the_noun_phrase_run() {
assert_one(
"librs is a fast small modular parser core",
"librs",
"is-a",
"parser",
);
}
#[test]
fn unknown_subject_and_object_both_fail_open() {
assert_one("tantivy uses fst", "tantivy", "uses", "fst");
}
#[test]
fn common_english_words_that_are_real_entity_names_survive() {
assert_one("rust is a language", "rust", "is-a", "language");
assert_one("go is a language", "go", "is-a", "language");
assert_one("python is a language", "python", "is-a", "language");
}
#[test]
fn a_caller_supplied_pos_table_is_the_one_consulted() {
let config = KgExtractConfig {
pos: WordNetPos::from_table("zzz\t1\n"),
..Default::default()
};
let got: Vec<_> = extract_triples_with_config(
&ExtractInput {
drawer_id: Uuid::new_v4(),
content: "match exhaustiveness is a hard requirement here",
tags: &[],
room: None,
},
&config,
)
.into_iter()
.filter(|t| PATTERN_TABLE.iter().any(|(p, _)| *p == t.predicate))
.map(|t| (t.subject, t.predicate, t.object))
.collect();
assert_eq!(
got,
vec![(
"exhaustiveness".to_string(),
"is-a".to_string(),
"hard".to_string()
)]
);
}
#[test]
fn surprising_non_genitive_relational_phrase_still_truncates() {
assert_one(
"the daemon is a participant in the process group",
"daemon",
"is-a",
"participant",
);
}
#[test]
fn inflected_forms_resolve_through_their_base_form() {
let wn = WordNetPos::shipped();
assert!(
wn.is_noun("parsers"),
"a regular plural keeps its noun sense"
);
assert_eq!(wn.mask("containing") & wordnet_pos::NOUN, 0);
assert_one("librs is a fast parsers", "librs", "is-a", "parsers");
}
#[test]
fn surprising_a_gerund_noun_still_takes_the_head() {
assert_one(
"session creation depends on daemon running",
"creation",
"depends-on",
"running",
);
assert_one(
"budget_tokens is an estimate clearing a budget",
"budget_tokens",
"is-a",
"clearing",
);
}
#[test]
fn surprising_run_stops_at_the_first_unknown_token() {
assert_one(
"trusty-memory uses redb sled",
"trusty-memory",
"uses",
"redb",
);
}
#[test]
fn surprising_single_token_head_truncates_a_compound() {
assert_one(
"brew is a system package managers",
"brew",
"is-a",
"managers",
);
}
}