Skip to main content

surrealdb_expr/expr/
tokenizer.rs

1use std::fmt;
2use std::fmt::Display;
3
4use revision::revisioned;
5
6/// The language a [`Tokenizer::Segment`] segments with.
7///
8/// Separate from [`crate::expr::Language`], which selects a Snowball stemming
9/// algorithm: the two vocabularies do not overlap. Snowball strips suffixes and
10/// has no CJK algorithms, while these languages are segmented by
11/// dictionary-driven morphological analysis and have no stemmer. Keeping them
12/// apart makes `SNOWBALL(KOREAN)` and `SEGMENT(ENGLISH)` unrepresentable rather
13/// than merely rejected.
14#[revisioned(revision = 1)]
15#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)]
16pub enum SegmentLanguage {
17	Chinese,
18	Japanese,
19	Korean,
20}
21
22impl SegmentLanguage {
23	pub fn as_str(self) -> &'static str {
24		match self {
25			Self::Chinese => "CHINESE",
26			Self::Japanese => "JAPANESE",
27			Self::Korean => "KOREAN",
28		}
29	}
30}
31
32impl Display for SegmentLanguage {
33	fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
34		f.write_str(self.as_str())
35	}
36}
37
38#[revisioned(revision = 2)]
39#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)]
40pub enum Tokenizer {
41	Blank,
42	Camel,
43	Class,
44	Punct,
45	/// Splits a span into words using a morphological dictionary.
46	///
47	/// Unlike the character-class tokenizers, which decide a boundary from the
48	/// pair of characters around it, this one needs the whole span at once: the
49	/// boundaries come from a lattice search over dictionary entries. It
50	/// therefore runs after the character-class tokenizers have produced spans,
51	/// segmenting each one further.
52	#[revision(start = 2)]
53	Segment(SegmentLanguage),
54}
55
56impl Display for Tokenizer {
57	fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
58		match self {
59			Self::Blank => f.write_str("BLANK"),
60			Self::Camel => f.write_str("CAMEL"),
61			Self::Class => f.write_str("CLASS"),
62			Self::Punct => f.write_str("PUNCT"),
63			Self::Segment(l) => write!(f, "SEGMENT({l})"),
64		}
65	}
66}