Skip to main content

trusty_memory/
wordnet_pos.rs

1//! Part-of-speech membership lookup backed by Princeton WordNet 3.1.
2//!
3//! Why: #5399 — the #4678 lexical filter judges a token against a closed-class
4//! stopword list, so it cannot tell `hard` from `requirement` or `ancestor`
5//! from `compiler`. Those are open-class words and no hand-written list can
6//! separate them; what separates them is which part of speech WordNet records
7//! for each. This module is the smallest thing that answers "which POS can this
8//! word be" without a tagger, a model, or a runtime service.
9//! What: binary-searches a byte-sorted `<lemma>\t<mask>` table embedded with
10//! `include_str!`. Membership is the ONLY fact exposed — no senses, glosses, or
11//! synset offsets are shipped; masks are read with the `NOUN`/`VERB`/`ADJ`/`ADV`
12//! bitmask constants. There is no load step, no cache, and no shared
13//! mutable state: [`WordNetPos`] is a 16-byte `Copy` handle over `&'static str`,
14//! so constructing one is free and every caller can own its own.
15//! Test: `mod tests` at the bottom of this file.
16//!
17//! Provenance and licence: WordNet 3.1, Princeton University, 2011. SPDX
18//! `WordNet` — permissive, no copyleft, requires the copyright notice to travel
19//! with every copy. The notice is preserved verbatim in `wordnet/LICENSE` and
20//! carried again in the `#` header of `wordnet/lemma-pos.txt`, which this
21//! module skips at lookup time rather than stripping from the file.
22//! Regeneration: `wordnet/README.md`.
23//!
24//! [`WordNetPos`]: crate::wordnet_pos::WordNetPos
25
26use std::cmp::Ordering;
27
28/// Bit for "this lemma has a noun sense".
29pub const NOUN: u8 = 1 << 0;
30/// Bit for "this lemma has a verb sense".
31pub const VERB: u8 = 1 << 1;
32/// Bit for "this lemma has an adjective sense".
33pub const ADJ: u8 = 1 << 2;
34/// Bit for "this lemma has an adverb sense".
35pub const ADV: u8 = 1 << 3;
36
37/// The vendored lemma/POS projection, embedded at compile time.
38///
39/// Why: reading this from disk at runtime would make the daemon depend on an
40/// install layout and break the self-contained-binary property `cargo install`
41/// gives us. It is the PROJECTION rather than WordNet's own `index.*` files
42/// because the extractor reads one field per line and the upstream files carry
43/// six more — 6,305,332 bytes of source collapses to 979,462 here (#5399).
44/// What: `#`-prefixed licence/provenance header, then one `<lemma>\t<mask>`
45/// record per line, byte-sorted by lemma. The sort is a correctness property:
46/// [`WordNetPos::lookup`] binary-searches it in place.
47/// Test: `the_shipped_table_is_sorted_and_parseable`.
48const TABLE: &str = include_str!("../wordnet/lemma-pos.txt");
49
50/// Byte offset of the first data line in [`TABLE`], resolved at compile time.
51const TABLE_DATA_START: usize = data_start(TABLE.as_bytes());
52
53/// Find the first line that is not part of the `#` header.
54///
55/// Why: the licence header must travel with the data (see the module docs), but
56/// a binary search that could land inside it would compare a prose line as if
57/// it were a lemma. Resolving the boundary as a `const` means the search starts
58/// past the header at zero runtime cost.
59/// What: skips whole lines while they begin with `#`; returns the offset of the
60/// first line that does not, or the length when every line is header.
61/// Test: `data_start_skips_the_whole_header`, `data_start_handles_no_header`.
62const fn data_start(bytes: &[u8]) -> usize {
63    let mut i = 0;
64    while i < bytes.len() {
65        if bytes[i] != b'#' {
66            return i;
67        }
68        while i < bytes.len() && bytes[i] != b'\n' {
69            i += 1;
70        }
71        i += 1;
72    }
73    bytes.len()
74}
75
76/// Regular-inflection base forms to try when a word is not in the table.
77///
78/// Why: this is deliberately a suffix table and not a stemmer. A stemmer
79/// over-generates (`caching` -> `cach`) and its extra reach buys nothing here:
80/// the only question asked of the result is which POS bits the base form
81/// carries, and a wrong stem simply misses the table and falls back to
82/// "unknown" — the behaviour that already existed. Six rules cover the
83/// inflections that appear in drawer prose.
84/// What: yields candidates in most-likely-first order for plural `-s` / `-es` /
85/// `-ies`, participial `-ing` (bare, restored `-e`, and undoubled final
86/// consonant), and past `-ed`. Returns nothing for a word that is not entirely
87/// ASCII letters, which keeps code identifiers, paths, and hyphenated crate
88/// names off this path entirely.
89/// Test: `base_forms_cover_the_regular_inflections`,
90/// `base_forms_skip_non_words`, `es_and_s_are_ordered_by_the_sibilant_rule`.
91fn base_form_candidates(word: &str) -> Vec<String> {
92    if word.len() < 3 || !word.bytes().all(|b| b.is_ascii_lowercase()) {
93        return Vec::new();
94    }
95    let mut out: Vec<String> = Vec::new();
96    let mut push = |s: String| {
97        if s.len() >= 2 && !out.contains(&s) {
98            out.push(s);
99        }
100    };
101    if let Some(stem) = word.strip_suffix("ies") {
102        push(format!("{stem}y"));
103    }
104    // Both `-es` and `-s` fit a word ending in `es`, and whichever is probed
105    // first wins outright when its stem is also a WordNet word. Order therefore
106    // decides the answer, and English decides the order: `-es` is the suffix
107    // only after a sibilant. `attaches` -> `attach`, `passes` -> `pass`, but
108    // `notes` -> `note` and `sites` -> `site` (#5399). The loser stays in the
109    // list, so a stem that misses still falls through to the other reading.
110    let es_stem = word.strip_suffix("es");
111    let s_stem = word.strip_suffix('s').filter(|s| !s.ends_with('s'));
112    let (first, second) = if es_stem.is_some_and(ends_in_sibilant) {
113        (es_stem, s_stem)
114    } else {
115        (s_stem, es_stem)
116    };
117    for stem in [first, second].into_iter().flatten() {
118        push(stem.to_string());
119    }
120    if let Some(stem) = word.strip_suffix("ing") {
121        push(stem.to_string());
122        push(format!("{stem}e"));
123        if let Some(undoubled) = undouble(stem) {
124            push(undoubled);
125        }
126    }
127    if let Some(stem) = word.strip_suffix("ed") {
128        push(stem.to_string());
129        push(format!("{stem}e"));
130        if let Some(undoubled) = undouble(stem) {
131            push(undoubled);
132        }
133    }
134    out
135}
136
137/// Whether a stem ends in the sibilant that forces the `-es` spelling.
138///
139/// Why: this is the whole of the `-es` / `-s` disambiguation — English writes
140/// `-es` after a sibilant and a bare `-s` everywhere else, so a word ending in
141/// `es` whose `-es` stem is NOT a sibilant kept its own `e`.
142/// What: `ss`, `x`, `z`, `ch`, `sh`. A SINGLE final `s` is deliberately absent:
143/// `uses` and `passes` share the `-ses` surface, and the `-se` base (`use`,
144/// `case`, `release`) is the common one, so `-ses` takes the `-s` reading
145/// first. A genuine single-`s` base still resolves, because its `-s` stem is
146/// not a word and the probe falls through — `buses` -> `buse` (miss) -> `bus`.
147/// Test: `es_and_s_are_ordered_by_the_sibilant_rule`.
148fn ends_in_sibilant(stem: &str) -> bool {
149    stem.ends_with("ss")
150        || stem.ends_with('x')
151        || stem.ends_with('z')
152        || stem.ends_with("ch")
153        || stem.ends_with("sh")
154}
155
156/// Drop a doubled final consonant, so `runn` offers `run`.
157fn undouble(stem: &str) -> Option<String> {
158    let mut chars = stem.chars().rev();
159    let last = chars.next()?;
160    if last != chars.next()? {
161        return None;
162    }
163    Some(stem[..stem.len() - last.len_utf8()].to_string())
164}
165
166/// Lemma-to-POS membership lookup.
167///
168/// Why: #5399 rejected a process-wide `OnceLock<HashMap>` — CLAUDE.md permits
169/// global state only for the tracing subscriber. Binary-searching the sorted
170/// table directly removes the reason the global existed: there is nothing to
171/// build, so there is nothing to share. The type is `Copy` and 16 bytes, so
172/// threading it through [`crate::kg_extract::KgExtractConfig`] costs a pointer
173/// pair rather than an `Arc`.
174/// What: holds the table text and the offset its data starts at. Every lookup
175/// is an O(log n) probe over `&'static str`; no allocation, no interior
176/// mutability, no teardown.
177/// Test: `shipped_table_answers_the_four_pos_classes`, `mask_is_case_insensitive`.
178#[derive(Debug, Clone, Copy)]
179pub struct WordNetPos {
180    table: &'static str,
181    data_start: usize,
182}
183
184impl Default for WordNetPos {
185    fn default() -> Self {
186        Self::shipped()
187    }
188}
189
190impl WordNetPos {
191    /// The vendored WordNet 3.1 table.
192    ///
193    /// Why: `const` so a caller that wants the shipped data pays nothing —
194    /// this is what lets `KgExtractConfig::default()` stay free.
195    /// What: pairs [`TABLE`] with its precomputed [`TABLE_DATA_START`].
196    /// Test: `shipped_table_answers_the_four_pos_classes`.
197    pub const fn shipped() -> Self {
198        Self {
199            table: TABLE,
200            data_start: TABLE_DATA_START,
201        }
202    }
203
204    /// Build a lookup over a caller-supplied table in the shipped format.
205    ///
206    /// Why: the binary search's edge cases (first record, last record, absent
207    /// key either side of the range) are invisible against 83k real lemmas but
208    /// obvious against six synthetic ones.
209    /// What: same contract as [`Self::shipped`]; the caller owes byte-sorted
210    /// `<lemma>\t<mask>` lines and an optional `#` header.
211    /// Test: `lookup_finds_the_first_and_last_records`.
212    pub fn from_table(table: &'static str) -> Self {
213        Self {
214            table,
215            data_start: data_start(table.as_bytes()),
216        }
217    }
218
219    /// POS bitmask for `word`, or `0` when WordNet has never heard of it.
220    ///
221    /// Why: #5399 requires unknown words to FAIL OPEN. Returning `0` rather
222    /// than an error or a default makes every caller's "unknown" branch
223    /// explicit at the call site instead of hidden here.
224    /// What: probes as given first — the extractor lower-cases its content up
225    /// front, so that path allocates nothing — and retries lower-cased only
226    /// when the input actually contains an upper-case character. WordNet index
227    /// lemmas are all lower-case. A final retry strips a regular inflection
228    /// ([`base_form_candidates`]), which is what lets `containing` read as the
229    /// verb it is instead of as an unknown word eligible to head a phrase.
230    /// Test: `mask_returns_zero_for_unknown_words`, `mask_is_case_insensitive`,
231    /// `mask_resolves_regular_inflections_to_their_base_form`.
232    pub fn mask(&self, word: &str) -> u8 {
233        if let Some(m) = self.lookup(word.as_bytes()) {
234            return m;
235        }
236        if word.chars().any(char::is_uppercase) {
237            let lowered = word.to_lowercase();
238            if let Some(m) = self.lookup(lowered.as_bytes()) {
239                return m;
240            }
241            return self.inflected_mask(&lowered);
242        }
243        self.inflected_mask(word)
244    }
245
246    /// POS bitmask for `word`'s base form, or `0` when no regular inflection of
247    /// it is in the table either.
248    ///
249    /// Why: WordNet indexes base forms only, so every `-s` / `-ing` / `-ed`
250    /// token reads as unknown. #5399 made "unknown" mean "eligible to head a
251    /// noun phrase", which turned each participle into a false head — `a
252    /// directory containing:` asserted `containing` as the type. Recovering the
253    /// base form makes `containing` resolve to `contain`, a verb, so the phrase
254    /// correctly ends before it.
255    /// What: probes each candidate from [`base_form_candidates`] in order and
256    /// returns the first hit. Every miss returns `0`, so the fail-open contract
257    /// of [`Self::mask`] is unchanged.
258    /// Test: `mask_resolves_regular_inflections_to_their_base_form`,
259    /// `mask_leaves_non_words_and_names_unknown`.
260    fn inflected_mask(&self, word: &str) -> u8 {
261        for candidate in base_form_candidates(word) {
262            if let Some(m) = self.lookup(candidate.as_bytes()) {
263                return m;
264            }
265        }
266        0
267    }
268
269    /// Binary-search the table for one lemma.
270    ///
271    /// Why: split out so the two `mask` probes share one implementation and so
272    /// a malformed table degrades to "unknown" (fail open) rather than
273    /// panicking inside the daemon's write path.
274    /// What: standard bisection, except the midpoint is walked back to its
275    /// line start before comparing — `lo` and `hi` are therefore always
276    /// line-aligned, which is what makes the forward scan for the line end
277    /// safe. Both branches strictly narrow the range, so it always terminates.
278    /// Test: `lookup_finds_the_first_and_last_records`,
279    /// `lookup_misses_outside_the_table_range`, `lookup_tolerates_a_bad_line`.
280    fn lookup(&self, needle: &[u8]) -> Option<u8> {
281        let bytes = self.table.as_bytes();
282        let mut lo = self.data_start;
283        let mut hi = bytes.len();
284        while lo < hi {
285            let mut start = lo + (hi - lo) / 2;
286            while start > lo && bytes[start - 1] != b'\n' {
287                start -= 1;
288            }
289            let mut end = start;
290            while end < hi && bytes[end] != b'\n' {
291                end += 1;
292            }
293            let line = &bytes[start..end];
294            let tab = line.iter().position(|b| *b == b'\t')?;
295            match line[..tab].cmp(needle) {
296                Ordering::Less => lo = end + 1,
297                Ordering::Greater => hi = start,
298                Ordering::Equal => {
299                    return std::str::from_utf8(&line[tab + 1..])
300                        .ok()?
301                        .trim()
302                        .parse::<u8>()
303                        .ok();
304                }
305            }
306        }
307        None
308    }
309
310    /// Whether WordNet lists `word` under any part of speech.
311    pub fn is_known(&self, word: &str) -> bool {
312        self.mask(word) != 0
313    }
314
315    /// Whether `word` can be a noun.
316    pub fn is_noun(&self, word: &str) -> bool {
317        self.mask(word) & NOUN != 0
318    }
319
320    /// Whether `word` is an adjective and nothing else.
321    ///
322    /// Why: this is the head-eligibility test the noun-phrase walk uses. A word
323    /// that can ONLY be an adjective names a property, so it cannot be the head
324    /// of the phrase — `hard` in `a hard requirement` modifies, it does not
325    /// name. #5399 uses that to SKIP such a token when picking the head, not to
326    /// reject the triple: the re-walk lands on `requirement`, which is what the
327    /// sentence actually asserts.
328    /// What: true when the ADJ bit is set and the NOUN bit is not. An unknown
329    /// word has mask `0` and is therefore never adjective-only — the fail-open
330    /// direction, which is what keeps unknown crate names eligible as heads.
331    /// Test: `adjective_only_catches_hard_and_spares_fast`.
332    pub fn is_adjective_only(&self, word: &str) -> bool {
333        let m = self.mask(word);
334        m & ADJ != 0 && m & NOUN == 0
335    }
336
337    /// Number of records in the table.
338    ///
339    /// Why: the measurement harness and the table's own sanity floor need it.
340    /// What: counts data lines — an O(n) scan, so it is not a hot-path call.
341    /// Test: `shipped_table_answers_the_four_pos_classes`.
342    pub fn lemma_count(&self) -> usize {
343        self.table[self.data_start..]
344            .lines()
345            .filter(|l| !l.is_empty())
346            .count()
347    }
348}
349
350#[cfg(test)]
351mod tests {
352    use super::*;
353
354    /// Six records with a header, exercising both range ends.
355    const TINY: &str = "# notice line\n# another\nalpha\t1\nbeta\t4\ndelta\t2\nomega\t15\n";
356
357    #[test]
358    fn data_start_skips_the_whole_header() {
359        assert_eq!(
360            &TINY[data_start(TINY.as_bytes())..],
361            "alpha\t1\nbeta\t4\ndelta\t2\nomega\t15\n"
362        );
363    }
364
365    #[test]
366    fn data_start_handles_no_header() {
367        assert_eq!(data_start(b"alpha\t1\n"), 0);
368        assert_eq!(data_start(b"# only header\n"), 14);
369    }
370
371    #[test]
372    fn lookup_finds_the_first_and_last_records() {
373        let wn = WordNetPos::from_table(TINY);
374        assert_eq!(wn.mask("alpha"), 1);
375        assert_eq!(wn.mask("beta"), 4);
376        assert_eq!(wn.mask("delta"), 2);
377        assert_eq!(wn.mask("omega"), 15);
378    }
379
380    #[test]
381    fn lookup_misses_outside_the_table_range() {
382        let wn = WordNetPos::from_table(TINY);
383        // Before the first record, after the last, and in the gaps between.
384        // `alphas` USED TO SIT IN THIS LIST as a near-miss of `alpha`, and that
385        // expectation is now wrong rather than merely stale: `mask` resolves a
386        // regular plural to its base form, so `alphas` legitimately answers
387        // `alpha`. The near-miss this list still needs is a PREFIX, which no
388        // suffix rule can reach — `alph` covers it, and the plural moved to
389        // `mask_resolves_regular_inflections_to_their_base_form`.
390        for w in ["aardvark", "zulu", "carrot", "epsilon", "alph"] {
391            assert_eq!(wn.mask(w), 0, "{w} should not be found");
392        }
393    }
394
395    /// The plural that used to read as a miss now answers its singular.
396    #[test]
397    fn lookup_retries_an_inflected_form() {
398        let wn = WordNetPos::from_table(TINY);
399        assert_eq!(wn.mask("alphas"), wn.mask("alpha"));
400        assert_eq!(wn.mask("alph"), 0, "a prefix is still a miss");
401    }
402
403    /// Why: this is the whole point of the retry — a participle must read as
404    /// the verb it inflects, so the noun-phrase walk ends before it instead of
405    /// treating it as an unknown word eligible to head the phrase.
406    #[test]
407    fn mask_resolves_regular_inflections_to_their_base_form() {
408        let wn = WordNetPos::shipped();
409        // `containing` is absent; `contain` is VERB-only, which is what stops
410        // the run in `a directory containing:`.
411        assert_eq!(wn.mask("containing"), wn.mask("contain"));
412        assert_eq!(
413            wn.mask("containing") & NOUN,
414            0,
415            "a participle is not a noun"
416        );
417        // -ing with a restored `e`, and with an undoubled final consonant.
418        // Both words are genuinely absent from the table; many other `-ing`
419        // forms (`mapping`, `shipping`, `running`) are WordNet nouns in their
420        // own right, so the direct probe answers and the retry never fires.
421        assert_eq!(wn.mask("parsing"), wn.mask("parse"));
422        assert_eq!(wn.mask("committing"), wn.mask("commit"));
423        // Plurals keep an unknown-looking token eligible as a head: this is the
424        // case a bare "refuse unknown tokens" rule would have broken.
425        assert_eq!(wn.mask("parsers"), wn.mask("parser"));
426        assert!(wn.is_noun("parsers"));
427        assert_eq!(wn.mask("libraries"), wn.mask("library"));
428        assert_eq!(wn.mask("indexed"), wn.mask("index"));
429    }
430
431    /// Why: the retry must not start inventing words. Fail-open means an
432    /// unrecognised token stays mask `0`, so widening the probe set must not
433    /// widen what counts as KNOWN for names, paths, or code identifiers.
434    #[test]
435    fn mask_leaves_non_words_and_names_unknown() {
436        let wn = WordNetPos::shipped();
437        for w in [
438            "rustc",
439            "librs",
440            "tantivy",
441            "redb",
442            "trusty-memory",
443            "crates/trusty-search/src/allowlist/tests.rs",
444            "budget_tokens",
445        ] {
446            assert_eq!(wn.mask(w), 0, "{w} must stay unknown");
447            assert!(!wn.is_adjective_only(w), "{w} must fail open");
448        }
449    }
450
451    #[test]
452    fn base_forms_cover_the_regular_inflections() {
453        assert!(base_form_candidates("parsers").contains(&"parser".to_string()));
454        assert!(base_form_candidates("libraries").contains(&"library".to_string()));
455        assert!(base_form_candidates("boxes").contains(&"box".to_string()));
456        assert!(base_form_candidates("containing").contains(&"contain".to_string()));
457        assert!(base_form_candidates("parsing").contains(&"parse".to_string()));
458        assert!(base_form_candidates("stopping").contains(&"stop".to_string()));
459        assert!(base_form_candidates("indexed").contains(&"index".to_string()));
460        // A double-`s` ending is not a plural marker.
461        assert!(!base_form_candidates("class").contains(&"clas".to_string()));
462    }
463
464    /// A plural of a word ending in `e` resolves to that word, not to the stem
465    /// left by chopping `es` off it.
466    ///
467    /// 🔴 DO NOT SIMPLIFY THIS BACK TO A FIXED ORDER. `-es` was
468    /// tried first unconditionally, so `notes` answered `not` (ADV) instead of
469    /// `note` (NOUN|VERB) and `sites` answered `sit` (VERB) instead of `site`.
470    /// Both then failed the walk's `NOUN|ADJ` check and ended the phrase, so
471    /// `notes is a drawer` and `sites is a directory` yielded nothing at all.
472    /// Flipping the order unconditionally just moves the damage: `attaches`
473    /// would answer the noun `attache` rather than the verb `attach`, and
474    /// `passes` the adjective-only `passe` rather than `pass`. The sibilant is
475    /// what separates the two cases, so it is what the order keys on.
476    #[test]
477    fn es_and_s_are_ordered_by_the_sibilant_rule() {
478        let wn = WordNetPos::shipped();
479        // Stem keeps its `e`: the plural marker is a bare `-s`.
480        for (inflected, base) in [
481            ("notes", "note"),
482            ("sites", "site"),
483            ("writes", "write"),
484            ("rides", "ride"),
485            ("envelopes", "envelope"),
486            ("uses", "use"),
487            ("houses", "house"),
488            ("releases", "release"),
489            ("cases", "case"),
490        ] {
491            assert_eq!(
492                wn.mask(inflected),
493                wn.mask(base),
494                "{inflected} must resolve to {base}"
495            );
496        }
497        // Sibilant stem: `-es` is the marker, and the `e` is not the stem's.
498        for (inflected, base) in [
499            ("attaches", "attach"),
500            ("passes", "pass"),
501            ("boxes", "box"),
502            ("dishes", "dish"),
503            ("matches", "match"),
504            ("indexes", "index"),
505            ("classes", "class"),
506            ("buses", "bus"),
507        ] {
508            assert_eq!(
509                wn.mask(inflected),
510                wn.mask(base),
511                "{inflected} must resolve to {base}"
512            );
513        }
514        // The wrong reading is a real word in each of these, which is why the
515        // order decides the answer rather than merely the probe count.
516        assert_ne!(wn.mask("not"), wn.mask("note"));
517        assert_ne!(wn.mask("attach"), wn.mask("attache"));
518    }
519
520    /// Anything that is not a plain lower-case word is off this path entirely,
521    /// so a path or an identifier never probes the table at all.
522    #[test]
523    fn base_forms_skip_non_words() {
524        for w in ["trusty-memory", "budget_tokens", "src/main.rs", "c#", "ab"] {
525            assert!(
526                base_form_candidates(w).is_empty(),
527                "{w} should generate no candidates"
528            );
529        }
530    }
531
532    #[test]
533    fn lookup_tolerates_a_bad_line() {
534        // A line with no tab is malformed; the probe must fail open, not panic.
535        let wn = WordNetPos::from_table("alpha\t1\nbroken-line\nomega\t15\n");
536        assert_eq!(wn.mask("nonsense"), 0);
537    }
538
539    #[test]
540    fn shipped_table_answers_the_four_pos_classes() {
541        let wn = WordNetPos::shipped();
542        assert_eq!(wn.lemma_count(), 83_253);
543        assert!(wn.is_noun("compiler"));
544        assert!(wn.mask("run") & VERB != 0);
545        assert!(wn.mask("hard") & ADJ != 0);
546        assert!(wn.mask("quickly") & ADV != 0);
547    }
548
549    /// The projection's invariants, checked against the committed file rather
550    /// than trusted from the generator's last run.
551    ///
552    /// Why: the table is regenerated by hand (`wordnet/README.md`), so nothing
553    /// mechanical guarantees a re-run stayed sorted or kept the `\t<mask>`
554    /// shape. An unsorted table does not fail loudly — it silently returns 0
555    /// for arbitrary words, which reads as "WordNet does not know this" and
556    /// would quietly disable the whole filter.
557    /// What: walks every data line once, asserting byte-ascending lemma order,
558    /// a parseable non-zero mask, and no multi-word lemma.
559    ///
560    /// 🔴 This used to `continue` past an empty line, and `lemma_count` filters
561    /// them out, so a stray blank line satisfied BOTH guards while silently
562    /// breaking lookups: `lookup` bisects onto that line, finds no tab, and the
563    /// `?` aborts the whole probe — every needle whose search path crosses it
564    /// reads as "WordNet does not know this word". A blank line is therefore a
565    /// table defect, not something to skip, and this asserts against it.
566    #[test]
567    fn the_shipped_table_is_sorted_and_parseable() {
568        let wn = WordNetPos::shipped();
569        let mut prev: &str = "";
570        let mut n = 0usize;
571        for line in wn.table[wn.data_start..].lines() {
572            assert!(
573                !line.is_empty(),
574                "blank data line after {prev:?} — it aborts any lookup that bisects onto it"
575            );
576            let (lemma, mask) = line.split_once('\t').expect("every data line has a tab");
577            assert!(
578                lemma.as_bytes() > prev.as_bytes(),
579                "table out of order at {lemma:?} (after {prev:?}) — binary search is invalid"
580            );
581            assert!(
582                !lemma.contains('_'),
583                "multi-word lemma {lemma:?} is dead weight"
584            );
585            let m: u8 = mask.parse().expect("mask parses");
586            assert!(
587                m > 0 && m <= (NOUN | VERB | ADJ | ADV),
588                "bad mask {m} for {lemma:?}"
589            );
590            prev = lemma;
591            n += 1;
592        }
593        assert_eq!(n, wn.lemma_count());
594    }
595
596    #[test]
597    fn multiword_lemmas_are_absent() {
598        let wn = WordNetPos::shipped();
599        assert_eq!(wn.mask("hot_dog"), 0);
600        assert!(wn.is_noun("dog"));
601    }
602
603    #[test]
604    fn mask_returns_zero_for_unknown_words() {
605        let wn = WordNetPos::shipped();
606        for w in ["rustc", "librs", "tantivy", "redb", "trusty-memory"] {
607            assert_eq!(wn.mask(w), 0, "{w} should be unknown to WordNet");
608            assert!(!wn.is_adjective_only(w), "{w} must fail open");
609        }
610    }
611
612    #[test]
613    fn mask_reports_every_pos_for_a_four_way_lemma() {
614        let wn = WordNetPos::shipped();
615        assert_eq!(wn.mask("fast"), NOUN | VERB | ADJ | ADV);
616    }
617
618    #[test]
619    fn adjective_only_catches_hard_and_spares_fast() {
620        let wn = WordNetPos::shipped();
621        assert!(wn.is_adjective_only("hard"));
622        assert!(!wn.is_adjective_only("fast"));
623        assert!(!wn.is_adjective_only("parser"));
624    }
625
626    #[test]
627    fn mask_is_case_insensitive() {
628        let wn = WordNetPos::shipped();
629        assert_eq!(wn.mask("Compiler"), wn.mask("compiler"));
630        assert_eq!(wn.mask("HARD"), wn.mask("hard"));
631    }
632
633    /// Two handles must agree without sharing anything — this is the property
634    /// that made the `OnceLock` unnecessary.
635    #[test]
636    fn independent_handles_agree() {
637        let a = WordNetPos::shipped();
638        let b = WordNetPos::default();
639        for w in ["compiler", "hard", "fast", "unknownium"] {
640            assert_eq!(a.mask(w), b.mask(w));
641        }
642    }
643}