trusty_memory/wordnet_pos.rs
1//! Part-of-speech membership lookup backed by Princeton WordNet 3.1.
2//!
3//! Why: #5399 — the #4678 lexical filter judges a token against a closed-class
4//! stopword list, so it cannot tell `hard` from `requirement` or `ancestor`
5//! from `compiler`. Those are open-class words and no hand-written list can
6//! separate them; what separates them is which part of speech WordNet records
7//! for each. This module is the smallest thing that answers "which POS can this
8//! word be" without a tagger, a model, or a runtime service.
9//! What: binary-searches a byte-sorted `<lemma>\t<mask>` table embedded with
10//! `include_str!`. Membership is the ONLY fact exposed — no senses, glosses, or
11//! synset offsets are shipped; masks are read with the `NOUN`/`VERB`/`ADJ`/`ADV`
12//! bitmask constants. There is no load step, no cache, and no shared
13//! mutable state: [`WordNetPos`] is a 16-byte `Copy` handle over `&'static str`,
14//! so constructing one is free and every caller can own its own.
15//! Test: `mod tests` at the bottom of this file.
16//!
17//! Provenance and licence: WordNet 3.1, Princeton University, 2011. SPDX
18//! `WordNet` — permissive, no copyleft, requires the copyright notice to travel
19//! with every copy. The notice is preserved verbatim in `wordnet/LICENSE` and
20//! carried again in the `#` header of `wordnet/lemma-pos.txt`, which this
21//! module skips at lookup time rather than stripping from the file.
22//! Regeneration: `wordnet/README.md`.
23//!
24//! [`WordNetPos`]: crate::wordnet_pos::WordNetPos
25
26use std::cmp::Ordering;
27
28/// Bit for "this lemma has a noun sense".
29pub const NOUN: u8 = 1 << 0;
30/// Bit for "this lemma has a verb sense".
31pub const VERB: u8 = 1 << 1;
32/// Bit for "this lemma has an adjective sense".
33pub const ADJ: u8 = 1 << 2;
34/// Bit for "this lemma has an adverb sense".
35pub const ADV: u8 = 1 << 3;
36
37/// The vendored lemma/POS projection, embedded at compile time.
38///
39/// Why: reading this from disk at runtime would make the daemon depend on an
40/// install layout and break the self-contained-binary property `cargo install`
41/// gives us. It is the PROJECTION rather than WordNet's own `index.*` files
42/// because the extractor reads one field per line and the upstream files carry
43/// six more — 6,305,332 bytes of source collapses to 979,462 here (#5399).
44/// What: `#`-prefixed licence/provenance header, then one `<lemma>\t<mask>`
45/// record per line, byte-sorted by lemma. The sort is a correctness property:
46/// [`WordNetPos::lookup`] binary-searches it in place.
47/// Test: `the_shipped_table_is_sorted_and_parseable`.
48const TABLE: &str = include_str!("../wordnet/lemma-pos.txt");
49
50/// Byte offset of the first data line in [`TABLE`], resolved at compile time.
51const TABLE_DATA_START: usize = data_start(TABLE.as_bytes());
52
53/// Find the first line that is not part of the `#` header.
54///
55/// Why: the licence header must travel with the data (see the module docs), but
56/// a binary search that could land inside it would compare a prose line as if
57/// it were a lemma. Resolving the boundary as a `const` means the search starts
58/// past the header at zero runtime cost.
59/// What: skips whole lines while they begin with `#`; returns the offset of the
60/// first line that does not, or the length when every line is header.
61/// Test: `data_start_skips_the_whole_header`, `data_start_handles_no_header`.
62const fn data_start(bytes: &[u8]) -> usize {
63 let mut i = 0;
64 while i < bytes.len() {
65 if bytes[i] != b'#' {
66 return i;
67 }
68 while i < bytes.len() && bytes[i] != b'\n' {
69 i += 1;
70 }
71 i += 1;
72 }
73 bytes.len()
74}
75
76/// Regular-inflection base forms to try when a word is not in the table.
77///
78/// Why: this is deliberately a suffix table and not a stemmer. A stemmer
79/// over-generates (`caching` -> `cach`) and its extra reach buys nothing here:
80/// the only question asked of the result is which POS bits the base form
81/// carries, and a wrong stem simply misses the table and falls back to
82/// "unknown" — the behaviour that already existed. Six rules cover the
83/// inflections that appear in drawer prose.
84/// What: yields candidates in most-likely-first order for plural `-s` / `-es` /
85/// `-ies`, participial `-ing` (bare, restored `-e`, and undoubled final
86/// consonant), and past `-ed`. Returns nothing for a word that is not entirely
87/// ASCII letters, which keeps code identifiers, paths, and hyphenated crate
88/// names off this path entirely.
89/// Test: `base_forms_cover_the_regular_inflections`,
90/// `base_forms_skip_non_words`, `es_and_s_are_ordered_by_the_sibilant_rule`.
91fn base_form_candidates(word: &str) -> Vec<String> {
92 if word.len() < 3 || !word.bytes().all(|b| b.is_ascii_lowercase()) {
93 return Vec::new();
94 }
95 let mut out: Vec<String> = Vec::new();
96 let mut push = |s: String| {
97 if s.len() >= 2 && !out.contains(&s) {
98 out.push(s);
99 }
100 };
101 if let Some(stem) = word.strip_suffix("ies") {
102 push(format!("{stem}y"));
103 }
104 // Both `-es` and `-s` fit a word ending in `es`, and whichever is probed
105 // first wins outright when its stem is also a WordNet word. Order therefore
106 // decides the answer, and English decides the order: `-es` is the suffix
107 // only after a sibilant. `attaches` -> `attach`, `passes` -> `pass`, but
108 // `notes` -> `note` and `sites` -> `site` (#5399). The loser stays in the
109 // list, so a stem that misses still falls through to the other reading.
110 let es_stem = word.strip_suffix("es");
111 let s_stem = word.strip_suffix('s').filter(|s| !s.ends_with('s'));
112 let (first, second) = if es_stem.is_some_and(ends_in_sibilant) {
113 (es_stem, s_stem)
114 } else {
115 (s_stem, es_stem)
116 };
117 for stem in [first, second].into_iter().flatten() {
118 push(stem.to_string());
119 }
120 if let Some(stem) = word.strip_suffix("ing") {
121 push(stem.to_string());
122 push(format!("{stem}e"));
123 if let Some(undoubled) = undouble(stem) {
124 push(undoubled);
125 }
126 }
127 if let Some(stem) = word.strip_suffix("ed") {
128 push(stem.to_string());
129 push(format!("{stem}e"));
130 if let Some(undoubled) = undouble(stem) {
131 push(undoubled);
132 }
133 }
134 out
135}
136
137/// Whether a stem ends in the sibilant that forces the `-es` spelling.
138///
139/// Why: this is the whole of the `-es` / `-s` disambiguation — English writes
140/// `-es` after a sibilant and a bare `-s` everywhere else, so a word ending in
141/// `es` whose `-es` stem is NOT a sibilant kept its own `e`.
142/// What: `ss`, `x`, `z`, `ch`, `sh`. A SINGLE final `s` is deliberately absent:
143/// `uses` and `passes` share the `-ses` surface, and the `-se` base (`use`,
144/// `case`, `release`) is the common one, so `-ses` takes the `-s` reading
145/// first. A genuine single-`s` base still resolves, because its `-s` stem is
146/// not a word and the probe falls through — `buses` -> `buse` (miss) -> `bus`.
147/// Test: `es_and_s_are_ordered_by_the_sibilant_rule`.
148fn ends_in_sibilant(stem: &str) -> bool {
149 stem.ends_with("ss")
150 || stem.ends_with('x')
151 || stem.ends_with('z')
152 || stem.ends_with("ch")
153 || stem.ends_with("sh")
154}
155
156/// Drop a doubled final consonant, so `runn` offers `run`.
157fn undouble(stem: &str) -> Option<String> {
158 let mut chars = stem.chars().rev();
159 let last = chars.next()?;
160 if last != chars.next()? {
161 return None;
162 }
163 Some(stem[..stem.len() - last.len_utf8()].to_string())
164}
165
166/// Lemma-to-POS membership lookup.
167///
168/// Why: #5399 rejected a process-wide `OnceLock<HashMap>` — CLAUDE.md permits
169/// global state only for the tracing subscriber. Binary-searching the sorted
170/// table directly removes the reason the global existed: there is nothing to
171/// build, so there is nothing to share. The type is `Copy` and 16 bytes, so
172/// threading it through [`crate::kg_extract::KgExtractConfig`] costs a pointer
173/// pair rather than an `Arc`.
174/// What: holds the table text and the offset its data starts at. Every lookup
175/// is an O(log n) probe over `&'static str`; no allocation, no interior
176/// mutability, no teardown.
177/// Test: `shipped_table_answers_the_four_pos_classes`, `mask_is_case_insensitive`.
178#[derive(Debug, Clone, Copy)]
179pub struct WordNetPos {
180 table: &'static str,
181 data_start: usize,
182}
183
184impl Default for WordNetPos {
185 fn default() -> Self {
186 Self::shipped()
187 }
188}
189
190impl WordNetPos {
191 /// The vendored WordNet 3.1 table.
192 ///
193 /// Why: `const` so a caller that wants the shipped data pays nothing —
194 /// this is what lets `KgExtractConfig::default()` stay free.
195 /// What: pairs [`TABLE`] with its precomputed [`TABLE_DATA_START`].
196 /// Test: `shipped_table_answers_the_four_pos_classes`.
197 pub const fn shipped() -> Self {
198 Self {
199 table: TABLE,
200 data_start: TABLE_DATA_START,
201 }
202 }
203
204 /// Build a lookup over a caller-supplied table in the shipped format.
205 ///
206 /// Why: the binary search's edge cases (first record, last record, absent
207 /// key either side of the range) are invisible against 83k real lemmas but
208 /// obvious against six synthetic ones.
209 /// What: same contract as [`Self::shipped`]; the caller owes byte-sorted
210 /// `<lemma>\t<mask>` lines and an optional `#` header.
211 /// Test: `lookup_finds_the_first_and_last_records`.
212 pub fn from_table(table: &'static str) -> Self {
213 Self {
214 table,
215 data_start: data_start(table.as_bytes()),
216 }
217 }
218
219 /// POS bitmask for `word`, or `0` when WordNet has never heard of it.
220 ///
221 /// Why: #5399 requires unknown words to FAIL OPEN. Returning `0` rather
222 /// than an error or a default makes every caller's "unknown" branch
223 /// explicit at the call site instead of hidden here.
224 /// What: probes as given first — the extractor lower-cases its content up
225 /// front, so that path allocates nothing — and retries lower-cased only
226 /// when the input actually contains an upper-case character. WordNet index
227 /// lemmas are all lower-case. A final retry strips a regular inflection
228 /// ([`base_form_candidates`]), which is what lets `containing` read as the
229 /// verb it is instead of as an unknown word eligible to head a phrase.
230 /// Test: `mask_returns_zero_for_unknown_words`, `mask_is_case_insensitive`,
231 /// `mask_resolves_regular_inflections_to_their_base_form`.
232 pub fn mask(&self, word: &str) -> u8 {
233 if let Some(m) = self.lookup(word.as_bytes()) {
234 return m;
235 }
236 if word.chars().any(char::is_uppercase) {
237 let lowered = word.to_lowercase();
238 if let Some(m) = self.lookup(lowered.as_bytes()) {
239 return m;
240 }
241 return self.inflected_mask(&lowered);
242 }
243 self.inflected_mask(word)
244 }
245
246 /// POS bitmask for `word`'s base form, or `0` when no regular inflection of
247 /// it is in the table either.
248 ///
249 /// Why: WordNet indexes base forms only, so every `-s` / `-ing` / `-ed`
250 /// token reads as unknown. #5399 made "unknown" mean "eligible to head a
251 /// noun phrase", which turned each participle into a false head — `a
252 /// directory containing:` asserted `containing` as the type. Recovering the
253 /// base form makes `containing` resolve to `contain`, a verb, so the phrase
254 /// correctly ends before it.
255 /// What: probes each candidate from [`base_form_candidates`] in order and
256 /// returns the first hit. Every miss returns `0`, so the fail-open contract
257 /// of [`Self::mask`] is unchanged.
258 /// Test: `mask_resolves_regular_inflections_to_their_base_form`,
259 /// `mask_leaves_non_words_and_names_unknown`.
260 fn inflected_mask(&self, word: &str) -> u8 {
261 for candidate in base_form_candidates(word) {
262 if let Some(m) = self.lookup(candidate.as_bytes()) {
263 return m;
264 }
265 }
266 0
267 }
268
269 /// Binary-search the table for one lemma.
270 ///
271 /// Why: split out so the two `mask` probes share one implementation and so
272 /// a malformed table degrades to "unknown" (fail open) rather than
273 /// panicking inside the daemon's write path.
274 /// What: standard bisection, except the midpoint is walked back to its
275 /// line start before comparing — `lo` and `hi` are therefore always
276 /// line-aligned, which is what makes the forward scan for the line end
277 /// safe. Both branches strictly narrow the range, so it always terminates.
278 /// Test: `lookup_finds_the_first_and_last_records`,
279 /// `lookup_misses_outside_the_table_range`, `lookup_tolerates_a_bad_line`.
280 fn lookup(&self, needle: &[u8]) -> Option<u8> {
281 let bytes = self.table.as_bytes();
282 let mut lo = self.data_start;
283 let mut hi = bytes.len();
284 while lo < hi {
285 let mut start = lo + (hi - lo) / 2;
286 while start > lo && bytes[start - 1] != b'\n' {
287 start -= 1;
288 }
289 let mut end = start;
290 while end < hi && bytes[end] != b'\n' {
291 end += 1;
292 }
293 let line = &bytes[start..end];
294 let tab = line.iter().position(|b| *b == b'\t')?;
295 match line[..tab].cmp(needle) {
296 Ordering::Less => lo = end + 1,
297 Ordering::Greater => hi = start,
298 Ordering::Equal => {
299 return std::str::from_utf8(&line[tab + 1..])
300 .ok()?
301 .trim()
302 .parse::<u8>()
303 .ok();
304 }
305 }
306 }
307 None
308 }
309
310 /// Whether WordNet lists `word` under any part of speech.
311 pub fn is_known(&self, word: &str) -> bool {
312 self.mask(word) != 0
313 }
314
315 /// Whether `word` can be a noun.
316 pub fn is_noun(&self, word: &str) -> bool {
317 self.mask(word) & NOUN != 0
318 }
319
320 /// Whether `word` is an adjective and nothing else.
321 ///
322 /// Why: this is the head-eligibility test the noun-phrase walk uses. A word
323 /// that can ONLY be an adjective names a property, so it cannot be the head
324 /// of the phrase — `hard` in `a hard requirement` modifies, it does not
325 /// name. #5399 uses that to SKIP such a token when picking the head, not to
326 /// reject the triple: the re-walk lands on `requirement`, which is what the
327 /// sentence actually asserts.
328 /// What: true when the ADJ bit is set and the NOUN bit is not. An unknown
329 /// word has mask `0` and is therefore never adjective-only — the fail-open
330 /// direction, which is what keeps unknown crate names eligible as heads.
331 /// Test: `adjective_only_catches_hard_and_spares_fast`.
332 pub fn is_adjective_only(&self, word: &str) -> bool {
333 let m = self.mask(word);
334 m & ADJ != 0 && m & NOUN == 0
335 }
336
337 /// Number of records in the table.
338 ///
339 /// Why: the measurement harness and the table's own sanity floor need it.
340 /// What: counts data lines — an O(n) scan, so it is not a hot-path call.
341 /// Test: `shipped_table_answers_the_four_pos_classes`.
342 pub fn lemma_count(&self) -> usize {
343 self.table[self.data_start..]
344 .lines()
345 .filter(|l| !l.is_empty())
346 .count()
347 }
348}
349
350#[cfg(test)]
351mod tests {
352 use super::*;
353
354 /// Six records with a header, exercising both range ends.
355 const TINY: &str = "# notice line\n# another\nalpha\t1\nbeta\t4\ndelta\t2\nomega\t15\n";
356
357 #[test]
358 fn data_start_skips_the_whole_header() {
359 assert_eq!(
360 &TINY[data_start(TINY.as_bytes())..],
361 "alpha\t1\nbeta\t4\ndelta\t2\nomega\t15\n"
362 );
363 }
364
365 #[test]
366 fn data_start_handles_no_header() {
367 assert_eq!(data_start(b"alpha\t1\n"), 0);
368 assert_eq!(data_start(b"# only header\n"), 14);
369 }
370
371 #[test]
372 fn lookup_finds_the_first_and_last_records() {
373 let wn = WordNetPos::from_table(TINY);
374 assert_eq!(wn.mask("alpha"), 1);
375 assert_eq!(wn.mask("beta"), 4);
376 assert_eq!(wn.mask("delta"), 2);
377 assert_eq!(wn.mask("omega"), 15);
378 }
379
380 #[test]
381 fn lookup_misses_outside_the_table_range() {
382 let wn = WordNetPos::from_table(TINY);
383 // Before the first record, after the last, and in the gaps between.
384 // `alphas` USED TO SIT IN THIS LIST as a near-miss of `alpha`, and that
385 // expectation is now wrong rather than merely stale: `mask` resolves a
386 // regular plural to its base form, so `alphas` legitimately answers
387 // `alpha`. The near-miss this list still needs is a PREFIX, which no
388 // suffix rule can reach — `alph` covers it, and the plural moved to
389 // `mask_resolves_regular_inflections_to_their_base_form`.
390 for w in ["aardvark", "zulu", "carrot", "epsilon", "alph"] {
391 assert_eq!(wn.mask(w), 0, "{w} should not be found");
392 }
393 }
394
395 /// The plural that used to read as a miss now answers its singular.
396 #[test]
397 fn lookup_retries_an_inflected_form() {
398 let wn = WordNetPos::from_table(TINY);
399 assert_eq!(wn.mask("alphas"), wn.mask("alpha"));
400 assert_eq!(wn.mask("alph"), 0, "a prefix is still a miss");
401 }
402
403 /// Why: this is the whole point of the retry — a participle must read as
404 /// the verb it inflects, so the noun-phrase walk ends before it instead of
405 /// treating it as an unknown word eligible to head the phrase.
406 #[test]
407 fn mask_resolves_regular_inflections_to_their_base_form() {
408 let wn = WordNetPos::shipped();
409 // `containing` is absent; `contain` is VERB-only, which is what stops
410 // the run in `a directory containing:`.
411 assert_eq!(wn.mask("containing"), wn.mask("contain"));
412 assert_eq!(
413 wn.mask("containing") & NOUN,
414 0,
415 "a participle is not a noun"
416 );
417 // -ing with a restored `e`, and with an undoubled final consonant.
418 // Both words are genuinely absent from the table; many other `-ing`
419 // forms (`mapping`, `shipping`, `running`) are WordNet nouns in their
420 // own right, so the direct probe answers and the retry never fires.
421 assert_eq!(wn.mask("parsing"), wn.mask("parse"));
422 assert_eq!(wn.mask("committing"), wn.mask("commit"));
423 // Plurals keep an unknown-looking token eligible as a head: this is the
424 // case a bare "refuse unknown tokens" rule would have broken.
425 assert_eq!(wn.mask("parsers"), wn.mask("parser"));
426 assert!(wn.is_noun("parsers"));
427 assert_eq!(wn.mask("libraries"), wn.mask("library"));
428 assert_eq!(wn.mask("indexed"), wn.mask("index"));
429 }
430
431 /// Why: the retry must not start inventing words. Fail-open means an
432 /// unrecognised token stays mask `0`, so widening the probe set must not
433 /// widen what counts as KNOWN for names, paths, or code identifiers.
434 #[test]
435 fn mask_leaves_non_words_and_names_unknown() {
436 let wn = WordNetPos::shipped();
437 for w in [
438 "rustc",
439 "librs",
440 "tantivy",
441 "redb",
442 "trusty-memory",
443 "crates/trusty-search/src/allowlist/tests.rs",
444 "budget_tokens",
445 ] {
446 assert_eq!(wn.mask(w), 0, "{w} must stay unknown");
447 assert!(!wn.is_adjective_only(w), "{w} must fail open");
448 }
449 }
450
451 #[test]
452 fn base_forms_cover_the_regular_inflections() {
453 assert!(base_form_candidates("parsers").contains(&"parser".to_string()));
454 assert!(base_form_candidates("libraries").contains(&"library".to_string()));
455 assert!(base_form_candidates("boxes").contains(&"box".to_string()));
456 assert!(base_form_candidates("containing").contains(&"contain".to_string()));
457 assert!(base_form_candidates("parsing").contains(&"parse".to_string()));
458 assert!(base_form_candidates("stopping").contains(&"stop".to_string()));
459 assert!(base_form_candidates("indexed").contains(&"index".to_string()));
460 // A double-`s` ending is not a plural marker.
461 assert!(!base_form_candidates("class").contains(&"clas".to_string()));
462 }
463
464 /// A plural of a word ending in `e` resolves to that word, not to the stem
465 /// left by chopping `es` off it.
466 ///
467 /// 🔴 DO NOT SIMPLIFY THIS BACK TO A FIXED ORDER. `-es` was
468 /// tried first unconditionally, so `notes` answered `not` (ADV) instead of
469 /// `note` (NOUN|VERB) and `sites` answered `sit` (VERB) instead of `site`.
470 /// Both then failed the walk's `NOUN|ADJ` check and ended the phrase, so
471 /// `notes is a drawer` and `sites is a directory` yielded nothing at all.
472 /// Flipping the order unconditionally just moves the damage: `attaches`
473 /// would answer the noun `attache` rather than the verb `attach`, and
474 /// `passes` the adjective-only `passe` rather than `pass`. The sibilant is
475 /// what separates the two cases, so it is what the order keys on.
476 #[test]
477 fn es_and_s_are_ordered_by_the_sibilant_rule() {
478 let wn = WordNetPos::shipped();
479 // Stem keeps its `e`: the plural marker is a bare `-s`.
480 for (inflected, base) in [
481 ("notes", "note"),
482 ("sites", "site"),
483 ("writes", "write"),
484 ("rides", "ride"),
485 ("envelopes", "envelope"),
486 ("uses", "use"),
487 ("houses", "house"),
488 ("releases", "release"),
489 ("cases", "case"),
490 ] {
491 assert_eq!(
492 wn.mask(inflected),
493 wn.mask(base),
494 "{inflected} must resolve to {base}"
495 );
496 }
497 // Sibilant stem: `-es` is the marker, and the `e` is not the stem's.
498 for (inflected, base) in [
499 ("attaches", "attach"),
500 ("passes", "pass"),
501 ("boxes", "box"),
502 ("dishes", "dish"),
503 ("matches", "match"),
504 ("indexes", "index"),
505 ("classes", "class"),
506 ("buses", "bus"),
507 ] {
508 assert_eq!(
509 wn.mask(inflected),
510 wn.mask(base),
511 "{inflected} must resolve to {base}"
512 );
513 }
514 // The wrong reading is a real word in each of these, which is why the
515 // order decides the answer rather than merely the probe count.
516 assert_ne!(wn.mask("not"), wn.mask("note"));
517 assert_ne!(wn.mask("attach"), wn.mask("attache"));
518 }
519
520 /// Anything that is not a plain lower-case word is off this path entirely,
521 /// so a path or an identifier never probes the table at all.
522 #[test]
523 fn base_forms_skip_non_words() {
524 for w in ["trusty-memory", "budget_tokens", "src/main.rs", "c#", "ab"] {
525 assert!(
526 base_form_candidates(w).is_empty(),
527 "{w} should generate no candidates"
528 );
529 }
530 }
531
532 #[test]
533 fn lookup_tolerates_a_bad_line() {
534 // A line with no tab is malformed; the probe must fail open, not panic.
535 let wn = WordNetPos::from_table("alpha\t1\nbroken-line\nomega\t15\n");
536 assert_eq!(wn.mask("nonsense"), 0);
537 }
538
539 #[test]
540 fn shipped_table_answers_the_four_pos_classes() {
541 let wn = WordNetPos::shipped();
542 assert_eq!(wn.lemma_count(), 83_253);
543 assert!(wn.is_noun("compiler"));
544 assert!(wn.mask("run") & VERB != 0);
545 assert!(wn.mask("hard") & ADJ != 0);
546 assert!(wn.mask("quickly") & ADV != 0);
547 }
548
549 /// The projection's invariants, checked against the committed file rather
550 /// than trusted from the generator's last run.
551 ///
552 /// Why: the table is regenerated by hand (`wordnet/README.md`), so nothing
553 /// mechanical guarantees a re-run stayed sorted or kept the `\t<mask>`
554 /// shape. An unsorted table does not fail loudly — it silently returns 0
555 /// for arbitrary words, which reads as "WordNet does not know this" and
556 /// would quietly disable the whole filter.
557 /// What: walks every data line once, asserting byte-ascending lemma order,
558 /// a parseable non-zero mask, and no multi-word lemma.
559 ///
560 /// 🔴 This used to `continue` past an empty line, and `lemma_count` filters
561 /// them out, so a stray blank line satisfied BOTH guards while silently
562 /// breaking lookups: `lookup` bisects onto that line, finds no tab, and the
563 /// `?` aborts the whole probe — every needle whose search path crosses it
564 /// reads as "WordNet does not know this word". A blank line is therefore a
565 /// table defect, not something to skip, and this asserts against it.
566 #[test]
567 fn the_shipped_table_is_sorted_and_parseable() {
568 let wn = WordNetPos::shipped();
569 let mut prev: &str = "";
570 let mut n = 0usize;
571 for line in wn.table[wn.data_start..].lines() {
572 assert!(
573 !line.is_empty(),
574 "blank data line after {prev:?} — it aborts any lookup that bisects onto it"
575 );
576 let (lemma, mask) = line.split_once('\t').expect("every data line has a tab");
577 assert!(
578 lemma.as_bytes() > prev.as_bytes(),
579 "table out of order at {lemma:?} (after {prev:?}) — binary search is invalid"
580 );
581 assert!(
582 !lemma.contains('_'),
583 "multi-word lemma {lemma:?} is dead weight"
584 );
585 let m: u8 = mask.parse().expect("mask parses");
586 assert!(
587 m > 0 && m <= (NOUN | VERB | ADJ | ADV),
588 "bad mask {m} for {lemma:?}"
589 );
590 prev = lemma;
591 n += 1;
592 }
593 assert_eq!(n, wn.lemma_count());
594 }
595
596 #[test]
597 fn multiword_lemmas_are_absent() {
598 let wn = WordNetPos::shipped();
599 assert_eq!(wn.mask("hot_dog"), 0);
600 assert!(wn.is_noun("dog"));
601 }
602
603 #[test]
604 fn mask_returns_zero_for_unknown_words() {
605 let wn = WordNetPos::shipped();
606 for w in ["rustc", "librs", "tantivy", "redb", "trusty-memory"] {
607 assert_eq!(wn.mask(w), 0, "{w} should be unknown to WordNet");
608 assert!(!wn.is_adjective_only(w), "{w} must fail open");
609 }
610 }
611
612 #[test]
613 fn mask_reports_every_pos_for_a_four_way_lemma() {
614 let wn = WordNetPos::shipped();
615 assert_eq!(wn.mask("fast"), NOUN | VERB | ADJ | ADV);
616 }
617
618 #[test]
619 fn adjective_only_catches_hard_and_spares_fast() {
620 let wn = WordNetPos::shipped();
621 assert!(wn.is_adjective_only("hard"));
622 assert!(!wn.is_adjective_only("fast"));
623 assert!(!wn.is_adjective_only("parser"));
624 }
625
626 #[test]
627 fn mask_is_case_insensitive() {
628 let wn = WordNetPos::shipped();
629 assert_eq!(wn.mask("Compiler"), wn.mask("compiler"));
630 assert_eq!(wn.mask("HARD"), wn.mask("hard"));
631 }
632
633 /// Two handles must agree without sharing anything — this is the property
634 /// that made the `OnceLock` unnecessary.
635 #[test]
636 fn independent_handles_agree() {
637 let a = WordNetPos::shipped();
638 let b = WordNetPos::default();
639 for w in ["compiler", "hard", "fast", "unknownium"] {
640 assert_eq!(a.mask(w), b.mask(w));
641 }
642 }
643}