Skip to main content

icydb_core/value/ops/
text.rs

1//! Module: value::ops::text
2//!
3//! Responsibility: text and casefolded identifier operations for `Value`.
4//! Does not own: collection membership or predicate-level coercion policy.
5//! Boundary: representation-local text helpers used by query operators.
6
7use crate::value::{TextMode, Value};
8use std::borrow::Cow;
9
10/// Apply the canonical case-insensitive text fold.
11///
12/// Casefolding remains a distinct semantic contract from SQL `LOWER`, even
13/// while both use Unicode lowercase conversion today. A future full Unicode
14/// casefold must not silently change `LOWER` or persisted index expressions.
15#[must_use]
16pub(crate) fn casefold_text(input: &str) -> String {
17    lowercase_text(input)
18}
19
20/// Apply the canonical `LOWER` transform used by query and index expressions.
21#[must_use]
22pub(crate) fn lower_text(input: &str) -> String {
23    lowercase_text(input)
24}
25
26/// Conservative construction allowance for `lower_text` on the pinned Rust
27/// 1.98.1 implementation: cumulative requested backing and byte-work units.
28/// This is not allocator telemetry or an IC instruction estimate.
29#[must_use]
30pub(crate) fn lower_text_construction_allowance(input_len: usize) -> (u64, u64) {
31    let len = input_len as u64;
32    if len <= 1 {
33        // Empty/single-byte UTF-8 is necessarily ASCII and never grows:
34        // inspect, copy, then lowercase the copied bytes in place.
35        return (len, len.saturating_mul(3));
36    }
37    // Unicode lowercase output is at most twice the input byte length (the
38    // exhaustive mapping test pins this). Rust starts at `len`, then at most
39    // once grows to max(2 * len, 8), including its small byte-vector minimum.
40    let backing = len.saturating_add(len.saturating_mul(2).max(8));
41    // Input work: ASCII check (n), ASCII-prefix conversion (<=2n), scalar walk
42    // (n), and final-sigma context scans (<=2n: Sigma is not case-ignorable).
43    // Backing also covers output fills and the retained-prefix copy on growth.
44    (backing, len.saturating_mul(6).saturating_add(backing))
45}
46
47/// Apply the canonical `UPPER` transform used by query and index expressions.
48#[must_use]
49pub(crate) fn upper_text(input: &str) -> String {
50    if input.is_ascii() {
51        return input.to_ascii_uppercase();
52    }
53
54    input.to_uppercase()
55}
56
57fn lowercase_text(input: &str) -> String {
58    if input.is_ascii() {
59        return input.to_ascii_lowercase();
60    }
61
62    input.to_lowercase()
63}
64
65fn text_with_mode(s: &'_ str, mode: TextMode) -> Cow<'_, str> {
66    match mode {
67        TextMode::Cs => Cow::Borrowed(s),
68        TextMode::Ci => Cow::Owned(casefold_text(s)),
69    }
70}
71
72fn text_op(
73    left: &Value,
74    right: &Value,
75    mode: TextMode,
76    f: impl Fn(&str, &str) -> bool,
77) -> Option<bool> {
78    let (a, b) = (left.as_text()?, right.as_text()?);
79    let a = text_with_mode(a, mode);
80    let b = text_with_mode(b, mode);
81    Some(f(&a, &b))
82}
83
84fn ci_key(value: &Value) -> Option<String> {
85    match value {
86        Value::Text(s) => Some(casefold_text(s)),
87        Value::Ulid(u) => Some(u.to_string().to_ascii_lowercase()),
88        Value::Principal(p) => Some(p.to_string().to_ascii_lowercase()),
89        Value::Account(a) => Some(a.to_string().to_ascii_lowercase()),
90        _ => None,
91    }
92}
93
94pub(super) fn eq_ci(left: &Value, right: &Value) -> bool {
95    if let (Some(left_key), Some(right_key)) = (ci_key(left), ci_key(right)) {
96        return left_key == right_key;
97    }
98
99    left == right
100}
101
102/// Case-sensitive/insensitive equality check for text-like values.
103#[must_use]
104fn text_eq(left: &Value, right: &Value, mode: TextMode) -> Option<bool> {
105    text_op(left, right, mode, |a, b| a == b)
106}
107
108/// Check whether `needle` is a substring of `value` under the given text mode.
109#[must_use]
110fn text_contains(value: &Value, needle: &Value, mode: TextMode) -> Option<bool> {
111    text_op(value, needle, mode, |a, b| a.contains(b))
112}
113
114/// Check whether `value` starts with `needle` under the given text mode.
115#[must_use]
116fn text_starts_with(value: &Value, needle: &Value, mode: TextMode) -> Option<bool> {
117    text_op(value, needle, mode, |a, b| a.starts_with(b))
118}
119
120/// Check whether `value` ends with `needle` under the given text mode.
121#[must_use]
122fn text_ends_with(value: &Value, needle: &Value, mode: TextMode) -> Option<bool> {
123    text_op(value, needle, mode, |a, b| a.ends_with(b))
124}
125
126impl Value {
127    /// Case-sensitive/insensitive equality check for text-like values.
128    #[must_use]
129    pub fn text_eq(&self, other: &Self, mode: TextMode) -> Option<bool> {
130        text_eq(self, other, mode)
131    }
132
133    /// Check whether `other` is a substring of `self` under the given text mode.
134    #[must_use]
135    pub fn text_contains(&self, needle: &Self, mode: TextMode) -> Option<bool> {
136        text_contains(self, needle, mode)
137    }
138
139    /// Check whether `self` starts with `other` under the given text mode.
140    #[must_use]
141    pub fn text_starts_with(&self, needle: &Self, mode: TextMode) -> Option<bool> {
142        text_starts_with(self, needle, mode)
143    }
144
145    /// Check whether `self` ends with `other` under the given text mode.
146    #[must_use]
147    pub fn text_ends_with(&self, needle: &Self, mode: TextMode) -> Option<bool> {
148        text_ends_with(self, needle, mode)
149    }
150}
151
152#[cfg(test)]
153mod tests {
154    use super::{casefold_text, lower_text, lower_text_construction_allowance, upper_text};
155
156    #[test]
157    fn lowercase_allowance_covers_unicode_expansion_and_output_growth() {
158        // Allocation-free qualification of every scalar mapping in the pinned
159        // toolchain. Contextual sigma changes spelling, not UTF-8 width.
160        for scalar in (0..=u32::from(char::MAX)).filter_map(char::from_u32) {
161            let output_bytes: usize = scalar.to_lowercase().map(char::len_utf8).sum();
162            assert!(output_bytes <= 2 * scalar.len_utf8(), "{scalar:?}");
163        }
164        for text in ["", "A", "İ", "Aİ", "İΣ", "ΟΣ\u{301}", "ΣΑ", "ASCII"] {
165            for repeats in [1, 2, 16, 1024] {
166                let input = text.repeat(repeats);
167                let output = lower_text(&input);
168                let (backing, _) = lower_text_construction_allowance(input.len());
169                let requested = if output.capacity() > input.len() {
170                    input.len() + output.capacity()
171                } else {
172                    input.len()
173                };
174                assert!(requested as u64 <= backing);
175                assert_eq!(output, input.to_lowercase());
176            }
177        }
178    }
179
180    #[test]
181    fn canonical_text_transforms_preserve_current_ascii_and_unicode_semantics() {
182        assert_eq!(casefold_text("IcYDB"), "icydb");
183        assert_eq!(lower_text("IcYDB"), "icydb");
184        assert_eq!(upper_text("IcYDB"), "ICYDB");
185
186        assert_eq!(casefold_text("Straße"), "straße");
187        assert_eq!(lower_text("Straße"), "straße");
188        assert_eq!(upper_text("Straße"), "STRASSE");
189    }
190}