Skip to main content

surrealdb_sql/
tokenizer.rs

1use std::fmt;
2use std::fmt::Display;
3
4use surrealdb_types::{SqlFormat, ToSql, write_sql};
5
6/// The language a [`Tokenizer::Segment`] segments with.
7///
8/// Separate from [`crate::language::Language`], which selects a Snowball
9/// stemming algorithm: the two vocabularies do not overlap. Snowball strips
10/// suffixes and has no CJK algorithms, while these languages are segmented by
11/// dictionary-driven morphological analysis and have no stemmer. Keeping them
12/// apart makes `SNOWBALL(KOREAN)` and `SEGMENT(ENGLISH)` unrepresentable
13/// rather than merely rejected.
14#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)]
15#[cfg_attr(feature = "arbitrary", derive(arbitrary::Arbitrary))]
16pub enum SegmentLanguage {
17	Chinese,
18	Japanese,
19	Korean,
20}
21
22impl SegmentLanguage {
23	pub fn as_str(self) -> &'static str {
24		match self {
25			Self::Chinese => "CHINESE",
26			Self::Japanese => "JAPANESE",
27			Self::Korean => "KOREAN",
28		}
29	}
30}
31
32impl Display for SegmentLanguage {
33	fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
34		f.write_str(self.as_str())
35	}
36}
37
38impl ToSql for SegmentLanguage {
39	fn fmt_sql(&self, f: &mut String, _fmt: SqlFormat) {
40		f.push_str(self.as_str())
41	}
42}
43
44#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)]
45#[cfg_attr(feature = "arbitrary", derive(arbitrary::Arbitrary))]
46pub enum Tokenizer {
47	Blank,
48	Camel,
49	Class,
50	Punct,
51	/// Splits a span into words using a morphological dictionary.
52	///
53	/// Unlike the character-class tokenizers, which decide a boundary from the
54	/// pair of characters around it, this one needs the whole span at once: the
55	/// boundaries come from a lattice search over dictionary entries. It
56	/// therefore runs after the character-class tokenizers have produced spans,
57	/// segmenting each one further.
58	Segment(SegmentLanguage),
59}
60
61impl Display for Tokenizer {
62	fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
63		match self {
64			Self::Blank => f.write_str("BLANK"),
65			Self::Camel => f.write_str("CAMEL"),
66			Self::Class => f.write_str("CLASS"),
67			Self::Punct => f.write_str("PUNCT"),
68			Self::Segment(l) => write!(f, "SEGMENT({l})"),
69		}
70	}
71}
72
73impl ToSql for Tokenizer {
74	fn fmt_sql(&self, f: &mut String, sql_fmt: SqlFormat) {
75		match self {
76			Self::Blank => f.push_str("BLANK"),
77			Self::Camel => f.push_str("CAMEL"),
78			Self::Class => f.push_str("CLASS"),
79			Self::Punct => f.push_str("PUNCT"),
80			Self::Segment(l) => write_sql!(f, sql_fmt, "SEGMENT({})", l),
81		}
82	}
83}
84
85/// Writes tokenizer keywords separated by commas (no spaces), as in `TOKENIZERS BLANK,CAMEL`.
86pub fn write_tokenizers_sql<I>(f: &mut String, sql_fmt: SqlFormat, tokenizers: I)
87where
88	I: IntoIterator<Item = Tokenizer>,
89{
90	for (i, t) in tokenizers.into_iter().enumerate() {
91		if i > 0 {
92			f.push(',');
93		}
94		t.fmt_sql(f, sql_fmt);
95	}
96}