use std::sync::Arc;
const INLINE_CAPACITY: usize = 22;
#[derive(Clone)]
pub(crate) enum InlineText {
Inline {
len: u8,
bytes: [u8; INLINE_CAPACITY],
},
Shared(Arc<str>),
}
impl InlineText {
pub(crate) fn new(text: &str) -> Self {
if text.len() <= INLINE_CAPACITY {
let mut bytes = [0u8; INLINE_CAPACITY];
bytes[..text.len()].copy_from_slice(text.as_bytes());
InlineText::Inline {
len: text.len() as u8,
bytes,
}
} else {
InlineText::Shared(Arc::from(text))
}
}
pub(crate) fn shared(text: Arc<str>) -> Self {
InlineText::Shared(text)
}
pub(crate) fn as_str(&self) -> &str {
match self {
InlineText::Inline { len, bytes } => unsafe {
std::str::from_utf8_unchecked(&bytes[..*len as usize])
},
InlineText::Shared(text) => text,
}
}
}
impl Default for InlineText {
fn default() -> Self {
InlineText::Inline {
len: 0,
bytes: [0u8; INLINE_CAPACITY],
}
}
}
#[derive(Clone, Default)]
pub(crate) struct CharSet {
ascii: [u64; 2],
other: Vec<char>,
}
impl CharSet {
pub(crate) fn new(chars: &str) -> Self {
Self::build(chars.chars())
}
pub(crate) fn with_extra(chars: &str, extra: &[char]) -> Self {
Self::build(chars.chars().chain(extra.iter().copied()))
}
fn build(chars: impl Iterator<Item = char>) -> Self {
let mut ascii = [0u64; 2];
let mut other = Vec::new();
for c in chars {
let u = c as u32;
if u < 128 {
ascii[(u >> 6) as usize] |= 1u64 << (u & 63);
} else {
other.push(c);
}
}
other.sort_unstable();
other.dedup();
CharSet { ascii, other }
}
#[inline]
pub(crate) fn contains(&self, c: char) -> bool {
let u = c as u32;
if u < 128 {
(self.ascii[(u >> 6) as usize] >> (u & 63)) & 1 != 0
} else {
!self.other.is_empty() && self.other.binary_search(&c).is_ok()
}
}
}
#[derive(Clone, Default)]
pub(crate) struct CharSets {
pub(crate) space: CharSet,
pub(crate) line_ends: CharSet,
pub(crate) line: CharSet,
pub(crate) row: CharSet,
pub(crate) string: CharSet,
}
#[cfg(test)]
mod char_set_tests {
use super::{CharSet, CharSets};
#[test]
fn char_set_agrees_with_the_scan_it_replaced() {
for class in [
"",
" ",
" \t",
"\n\r",
"\"'`",
"\u{0}\u{1f}\u{7f}",
"?@ABab",
"\u{2028}\u{2029}\u{a0}",
"αβγ",
] {
let set = CharSet::new(class);
let probes = ('\u{0}'..='\u{ff}').chain([
'\u{2027}',
'\u{2028}',
'\u{2029}',
'α',
'β',
'δ',
'\u{10000}',
]);
for c in probes {
assert_eq!(
set.contains(c),
class.contains(c),
"class {class:?}, char {c:?} (U+{:04X})",
c as u32
);
}
}
}
#[test]
fn with_extra_merges_both_sources() {
let set = CharSet::with_extra("\n\r", &['\u{b}', '\u{2028}']);
for c in ['\n', '\r', '\u{b}', '\u{2028}'] {
assert!(set.contains(c), "{c:?} should be in the merged set");
}
for c in ['\t', ' ', 'a', '\u{2029}'] {
assert!(!set.contains(c), "{c:?} should not be in the merged set");
}
}
#[test]
fn line_and_line_ends_stay_distinct() {
let sets = CharSets {
line: CharSet::new("\n"),
line_ends: CharSet::with_extra("\n", &['\u{b}']),
..Default::default()
};
assert!(sets.line_ends.contains('\u{b}'));
assert!(!sets.line.contains('\u{b}'));
assert!(sets.line.contains('\n') && sets.line_ends.contains('\n'));
}
#[test]
fn a_large_non_ascii_class_answers_for_every_member() {
let mut members: Vec<char> = Vec::new();
for i in 0..150u32 {
members.push(char::from_u32(0x2000 + i * 3).unwrap());
members.push(char::from_u32(0x30A0 - i * 5).unwrap());
}
let class: String = members.iter().collect();
let set = CharSet::new(&class);
for &c in &members {
assert!(
set.contains(c),
"{c:?} (U+{:04X}) is in the class",
c as u32
);
}
for &c in &members {
for delta in [-1i32, 1] {
let probe = char::from_u32((c as i32 + delta) as u32).unwrap();
if !members.contains(&probe) {
assert!(
!set.contains(probe),
"{probe:?} (U+{:04X}) is not in the class",
probe as u32
);
}
}
}
}
#[test]
fn an_empty_class_contains_nothing() {
let set = CharSet::default();
for c in ['\u{0}', ' ', 'a', '\u{7f}', '\u{2028}'] {
assert!(!set.contains(c));
}
}
}