#[derive(Clone, Debug, PartialEq, Eq)]
pub enum BytePat {
Empty,
Byte(u8),
Any,
Class(Vec<(char, char)>, bool),
Property(UnicodeClass, bool),
Builtin(ByteClassKind),
Concat(Vec<BytePat>),
Alt(Vec<BytePat>),
Star(Box<BytePat>),
Plus(Box<BytePat>),
Opt(Box<BytePat>),
Repeat(Box<BytePat>, usize, Option<usize>),
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum UnicodeClass {
Letter,
Number,
Alphanumeric,
Uppercase,
Lowercase,
Whitespace,
Control,
}
impl UnicodeClass {
#[must_use]
pub fn parse(name: &str) -> Option<UnicodeClass> {
Some(match name {
"L" | "Letter" | "Alphabetic" => UnicodeClass::Letter,
"N" | "Number" | "Numeric" => UnicodeClass::Number,
"Alnum" | "Alphanumeric" => UnicodeClass::Alphanumeric,
"Lu" | "Uppercase" => UnicodeClass::Uppercase,
"Ll" | "Lowercase" => UnicodeClass::Lowercase,
"White_Space" | "Whitespace" | "Z" => UnicodeClass::Whitespace,
"C" | "Control" => UnicodeClass::Control,
_ => return None,
})
}
pub(crate) const NAMES: &'static str =
"L, N, Alnum, Lu, Ll, White_Space, C (with long aliases)";
#[must_use]
fn holds(self, c: char) -> bool {
match self {
UnicodeClass::Letter => c.is_alphabetic(),
UnicodeClass::Number => c.is_numeric(),
UnicodeClass::Alphanumeric => c.is_alphanumeric(),
UnicodeClass::Uppercase => c.is_uppercase(),
UnicodeClass::Lowercase => c.is_lowercase(),
UnicodeClass::Whitespace => c.is_whitespace(),
UnicodeClass::Control => c.is_control(),
}
}
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum ByteClassKind {
Digit,
NotDigit,
Word,
NotWord,
Space,
NotSpace,
}
impl BytePat {
#[must_use]
pub fn matches_whole(&self, text: &[u8]) -> bool {
if text.len() <= BITSET_TEXT_MAX {
return (reach_bits(self, text, 1) & (1 << text.len())) != 0;
}
reach(self, text, vec![0]).contains(&text.len())
}
#[must_use]
pub fn literal_prefix(&self) -> Vec<u8> {
let mut out = Vec::new();
self.push_literal_prefix(&mut out);
out
}
fn push_literal_prefix(&self, out: &mut Vec<u8>) -> bool {
match self {
BytePat::Empty => true,
BytePat::Byte(b) => {
out.push(*b);
true
}
BytePat::Concat(v) => v.iter().all(|p| p.push_literal_prefix(out)),
_ => false,
}
}
#[must_use]
pub fn max_len(&self) -> Option<usize> {
match self {
BytePat::Empty => Some(0),
BytePat::Byte(_) | BytePat::Builtin(_) => Some(1),
BytePat::Any | BytePat::Class(..) | BytePat::Property(..) => Some(MAX_CHAR_BYTES),
BytePat::Concat(parts) => {
let mut total = 0usize;
for p in parts {
total = total.checked_add(p.max_len()?)?;
}
Some(total)
}
BytePat::Alt(parts) => {
let mut mx = 0usize;
for p in parts {
mx = mx.max(p.max_len()?);
}
Some(mx)
}
BytePat::Opt(inner) => inner.max_len(),
BytePat::Star(_) | BytePat::Plus(_) | BytePat::Repeat(_, _, None) => None,
BytePat::Repeat(inner, _, Some(hi)) => inner.max_len()?.checked_mul(*hi),
}
}
#[must_use]
pub fn longest_prefix(&self, text: &[u8]) -> Option<usize> {
if text.len() <= BITSET_TEXT_MAX {
let ends = reach_bits(self, text, 1);
return (ends != 0).then(|| (Bits::BITS - 1 - ends.leading_zeros()) as usize);
}
reach(self, text, vec![0]).into_iter().max()
}
#[must_use]
pub fn prefix_ends(&self, text: &[u8]) -> Vec<usize> {
if text.len() <= BITSET_TEXT_MAX {
let ends = reach_bits(self, text, 1);
return (0..=text.len()).filter(|&i| (ends & (1 << i)) != 0).collect();
}
let mut ends = reach(self, text, vec![0]);
ends.sort_unstable();
ends.dedup();
ends
}
}
pub struct ByteAutomaton {
accept: Box<[u64; 256]>,
follow: Vec<u64>,
first: u64,
last: u64,
nullable: bool,
}
const MAX_POSITIONS: usize = 64;
enum ByteTest {
Byte(u8),
Builtin(ByteClassKind),
Ranges(Vec<(u8, u8)>),
}
struct Positions {
tests: Vec<ByteTest>,
follow: Vec<u64>,
}
impl Positions {
fn leaf(&mut self, test: ByteTest) -> Option<(bool, u64, u64)> {
if self.tests.len() == MAX_POSITIONS {
return None;
}
let bit = 1u64 << self.tests.len();
self.tests.push(test);
self.follow.push(0);
Some((false, bit, bit))
}
}
fn positions_of(p: &BytePat, b: &mut Positions) -> Option<(bool, u64, u64)> {
match p {
BytePat::Empty => Some((true, 0, 0)),
BytePat::Byte(x) if *x < 0x80 => b.leaf(ByteTest::Byte(*x)),
BytePat::Builtin(k) => b.leaf(ByteTest::Builtin(*k)),
BytePat::Class(ranges, false) => {
let mut rs = Vec::with_capacity(ranges.len());
for &(lo, hi) in ranges {
if lo > '\u{7f}' || hi > '\u{7f}' {
return None;
}
rs.push((lo as u8, hi as u8));
}
b.leaf(ByteTest::Ranges(rs))
}
BytePat::Concat(v) => {
let (mut nullable, mut first, mut last) = (true, 0u64, 0u64);
for part in v {
let (n, f, l) = positions_of(part, b)?;
for i in set_bits(last) {
b.follow[i] |= f;
}
if nullable {
first |= f;
}
last = if n { last | l } else { l };
nullable &= n;
}
Some((nullable, first, last))
}
BytePat::Alt(v) => {
let (mut nullable, mut first, mut last) = (false, 0u64, 0u64);
for part in v {
let (n, f, l) = positions_of(part, b)?;
nullable |= n;
first |= f;
last |= l;
}
Some((nullable, first, last))
}
BytePat::Star(p) => {
let (_, f, l) = positions_of(p, b)?;
for i in set_bits(l) {
b.follow[i] |= f;
}
Some((true, f, l))
}
BytePat::Plus(p) => {
let (n, f, l) = positions_of(p, b)?;
for i in set_bits(l) {
b.follow[i] |= f;
}
Some((n, f, l))
}
BytePat::Opt(p) => {
let (_, f, l) = positions_of(p, b)?;
Some((true, f, l))
}
BytePat::Repeat(p, lo, hi) => {
let mut parts: Vec<BytePat> = Vec::new();
for _ in 0..*lo {
parts.push((**p).clone());
}
match hi {
Some(hi) => {
for _ in *lo..*hi {
parts.push(BytePat::Opt(p.clone()));
}
}
None => parts.push(BytePat::Star(p.clone())),
}
positions_of(&BytePat::Concat(parts), b)
}
BytePat::Byte(_) | BytePat::Any | BytePat::Property(..) | BytePat::Class(_, true) => None,
}
}
fn set_bits(mask: u64) -> impl Iterator<Item = usize> {
std::iter::successors((mask != 0).then_some(mask), |m| {
let next = *m & (*m - 1);
(next != 0).then_some(next)
})
.map(|m| m.trailing_zeros() as usize)
}
impl ByteAutomaton {
#[must_use]
pub fn matches_whole(&self, text: &[u8]) -> bool {
let Some((&last_byte, head)) = text.split_last() else {
return self.nullable;
};
let mut cur = self.first;
for &byte in head {
cur &= self.accept[byte as usize];
if cur == 0 {
return false;
}
cur = self.step(cur);
}
cur & self.accept[last_byte as usize] & self.last != 0
}
#[inline]
fn step(&self, cur: u64) -> u64 {
let (mut m, mut out) = (cur, 0u64);
while m != 0 {
out |= self.follow[m.trailing_zeros() as usize];
m &= m - 1;
}
out
}
}
impl BytePat {
#[must_use]
pub fn automaton(&self) -> Option<ByteAutomaton> {
let mut b = Positions { tests: Vec::new(), follow: Vec::new() };
let (nullable, first, last) = positions_of(self, &mut b)?;
let mut accept = Box::new([0u64; 256]);
for (i, test) in b.tests.iter().enumerate() {
let bit = 1u64 << i;
for (byte, slot) in accept.iter_mut().enumerate() {
let byte = byte as u8;
let takes = match test {
ByteTest::Byte(x) => byte == *x,
ByteTest::Builtin(k) => builtin_match(*k, byte),
ByteTest::Ranges(rs) => {
byte < 0x80 && rs.iter().any(|&(lo, hi)| byte >= lo && byte <= hi)
}
};
if takes {
*slot |= bit;
}
}
}
Some(ByteAutomaton { accept, follow: b.follow, first, last, nullable })
}
}
fn builtin_match(kind: ByteClassKind, b: u8) -> bool {
match kind {
ByteClassKind::Digit => b.is_ascii_digit(),
ByteClassKind::NotDigit => !b.is_ascii_digit(),
ByteClassKind::Word => b == b'_' || b.is_ascii_alphanumeric(),
ByteClassKind::NotWord => !(b == b'_' || b.is_ascii_alphanumeric()),
ByteClassKind::Space => b.is_ascii_whitespace(),
ByteClassKind::NotSpace => !b.is_ascii_whitespace(),
}
}
fn class_match(c: char, ranges: &[(char, char)], negated: bool) -> bool {
let inside = ranges.iter().any(|&(lo, hi)| c >= lo && c <= hi);
inside != negated
}
fn next_char(text: &[u8], at: usize) -> Option<(char, usize)> {
let lead = *text.get(at)?;
let (width, mut code) = match lead {
0x00..=0x7f => return Some((char::from(lead), 1)),
0xc2..=0xdf => (2usize, u32::from(lead & 0x1f)),
0xe0..=0xef => (3, u32::from(lead & 0x0f)),
0xf0..=0xf4 => (4, u32::from(lead & 0x07)),
_ => return None,
};
for k in 1..width {
let b = *text.get(at + k)?;
if b & 0xc0 != 0x80 {
return None;
}
code = (code << 6) | u32::from(b & 0x3f);
}
char::from_u32(code).map(|c| (c, width))
}
const MAX_CHAR_BYTES: usize = 4;
fn dedup(mut positions: Vec<usize>) -> Vec<usize> {
positions.sort_unstable();
positions.dedup();
positions
}
fn reach(p: &BytePat, text: &[u8], starts: Vec<usize>) -> Vec<usize> {
match p {
BytePat::Empty => dedup(starts),
BytePat::Byte(b) => one_byte(starts, text, |c| c == *b),
BytePat::Any => one_char(starts, text, |_| true),
BytePat::Class(ranges, neg) => one_char(starts, text, |c| class_match(c, ranges, *neg)),
BytePat::Property(class, neg) => {
one_char(starts, text, |c| class.holds(c) != *neg)
}
BytePat::Builtin(kind) => one_byte(starts, text, |c| builtin_match(*kind, c)),
BytePat::Concat(parts) => {
parts.iter().fold(dedup(starts), |acc, part| reach(part, text, acc))
}
BytePat::Alt(parts) => {
let mut out = Vec::new();
for part in parts {
out.extend(reach(part, text, starts.clone()));
}
dedup(out)
}
BytePat::Opt(inner) => {
let mut out = starts.clone();
out.extend(reach(inner, text, starts));
dedup(out)
}
BytePat::Star(inner) => closure(inner, text, starts),
BytePat::Plus(inner) => {
let once = reach(inner, text, starts);
closure(inner, text, once)
}
BytePat::Repeat(inner, m, n) => repeat(inner, *m, *n, text, starts),
}
}
fn one_byte(starts: Vec<usize>, text: &[u8], pred: impl Fn(u8) -> bool) -> Vec<usize> {
let mut out = Vec::new();
for s in starts {
if s < text.len() && pred(text[s]) {
out.push(s + 1);
}
}
dedup(out)
}
fn one_char(starts: Vec<usize>, text: &[u8], pred: impl Fn(char) -> bool) -> Vec<usize> {
let mut out = Vec::new();
for s in starts {
if let Some((c, width)) = next_char(text, s)
&& pred(c)
{
out.push(s + width);
}
}
dedup(out)
}
fn closure(inner: &BytePat, text: &[u8], starts: Vec<usize>) -> Vec<usize> {
let mut seen = dedup(starts);
let mut frontier = seen.clone();
loop {
let next = reach(inner, text, frontier);
let fresh: Vec<usize> = next.into_iter().filter(|q| !seen.contains(q)).collect();
if fresh.is_empty() {
break;
}
for q in &fresh {
seen.push(*q);
}
frontier = fresh;
}
dedup(seen)
}
fn repeat(
inner: &BytePat,
m: usize,
n: Option<usize>,
text: &[u8],
starts: Vec<usize>,
) -> Vec<usize> {
let mut cur = dedup(starts);
for _ in 0..m {
cur = reach(inner, text, cur);
if cur.is_empty() {
return cur;
}
}
match n {
None => closure(inner, text, cur),
Some(nn) => {
let mut acc = cur.clone();
let mut frontier = cur;
for _ in m..nn {
let next = reach(inner, text, frontier);
let fresh: Vec<usize> = next.into_iter().filter(|q| !acc.contains(q)).collect();
if fresh.is_empty() {
break;
}
for q in &fresh {
acc.push(*q);
}
frontier = fresh;
}
dedup(acc)
}
}
}
type Bits = u128;
const BITSET_TEXT_MAX: usize = (Bits::BITS - 1) as usize;
fn reach_bits(p: &BytePat, text: &[u8], starts: Bits) -> Bits {
if starts == 0 {
return 0;
}
match p {
BytePat::Empty => starts,
BytePat::Byte(b) => step_byte(starts, text, |c| c == *b),
BytePat::Any => step_char(starts, text, |_| true),
BytePat::Class(ranges, neg) => step_char(starts, text, |c| class_match(c, ranges, *neg)),
BytePat::Property(class, neg) => step_char(starts, text, |c| class.holds(c) != *neg),
BytePat::Builtin(kind) => step_byte(starts, text, |c| builtin_match(*kind, c)),
BytePat::Concat(parts) => parts.iter().fold(starts, |acc, part| reach_bits(part, text, acc)),
BytePat::Alt(parts) => parts.iter().fold(0, |acc, part| acc | reach_bits(part, text, starts)),
BytePat::Opt(inner) => starts | reach_bits(inner, text, starts),
BytePat::Star(inner) => closure_bits(inner, text, starts),
BytePat::Plus(inner) => {
let once = reach_bits(inner, text, starts);
closure_bits(inner, text, once)
}
BytePat::Repeat(inner, m, n) => {
let mut cur = starts;
for _ in 0..*m {
cur = reach_bits(inner, text, cur);
if cur == 0 {
return 0;
}
}
match n {
None => closure_bits(inner, text, cur),
Some(nn) => {
let mut acc = cur;
let mut frontier = cur;
for _ in *m..*nn {
let fresh = reach_bits(inner, text, frontier) & !acc;
if fresh == 0 {
break;
}
acc |= fresh;
frontier = fresh;
}
acc
}
}
}
}
}
fn step_byte(starts: Bits, text: &[u8], pred: impl Fn(u8) -> bool) -> Bits {
let mut out = 0;
let mut rest = starts;
while rest != 0 {
let s = rest.trailing_zeros() as usize;
rest &= rest - 1;
if s < text.len() && pred(text[s]) {
out |= 1 << (s + 1);
}
}
out
}
fn step_char(starts: Bits, text: &[u8], pred: impl Fn(char) -> bool) -> Bits {
let mut out = 0;
let mut rest = starts;
while rest != 0 {
let s = rest.trailing_zeros() as usize;
rest &= rest - 1;
if let Some((c, width)) = next_char(text, s)
&& pred(c)
{
out |= 1 << (s + width);
}
}
out
}
fn closure_bits(inner: &BytePat, text: &[u8], starts: Bits) -> Bits {
let mut seen = starts;
let mut frontier = starts;
loop {
let fresh = reach_bits(inner, text, frontier) & !seen;
if fresh == 0 {
return seen;
}
seen |= fresh;
frontier = fresh;
}
}
pub fn parse(src: &[u8]) -> Result<BytePat, String> {
let mut p = BParser { s: src, i: 0 };
let pat = p.alt()?;
if p.i != p.s.len() {
return Err(format!("unexpected byte at {} in byte-pattern", p.i));
}
Ok(pat)
}
struct BParser<'a> {
s: &'a [u8],
i: usize,
}
impl BParser<'_> {
fn peek(&self) -> Option<u8> {
self.s.get(self.i).copied()
}
fn bump(&mut self) -> Option<u8> {
let b = self.peek();
if b.is_some() {
self.i += 1;
}
b
}
fn alt(&mut self) -> Result<BytePat, String> {
let mut alts = vec![self.concat()?];
while self.peek() == Some(b'|') {
self.bump();
alts.push(self.concat()?);
}
Ok(if alts.len() == 1 { alts.pop().unwrap() } else { BytePat::Alt(alts) })
}
fn concat(&mut self) -> Result<BytePat, String> {
let mut items = Vec::new();
while !matches!(self.peek(), None | Some(b'|') | Some(b')')) {
items.push(self.postfix()?);
}
Ok(match items.len() {
0 => BytePat::Empty,
1 => items.pop().unwrap(),
_ => BytePat::Concat(items),
})
}
fn postfix(&mut self) -> Result<BytePat, String> {
let atom = self.atom()?;
match self.peek() {
Some(b'*') => {
self.bump();
Ok(BytePat::Star(Box::new(atom)))
}
Some(b'+') => {
self.bump();
Ok(BytePat::Plus(Box::new(atom)))
}
Some(b'?') => {
self.bump();
Ok(BytePat::Opt(Box::new(atom)))
}
Some(b'{') => self.repeat(atom),
_ => Ok(atom),
}
}
fn repeat(&mut self, inner: BytePat) -> Result<BytePat, String> {
self.bump(); let m = self.number()?;
let n = if self.peek() == Some(b',') {
self.bump();
if self.peek() == Some(b'}') { None } else { Some(self.number()?) }
} else {
Some(m)
};
if self.bump() != Some(b'}') {
return Err("expected '}' in byte-pattern quantifier".to_string());
}
Ok(BytePat::Repeat(Box::new(inner), m, n))
}
fn number(&mut self) -> Result<usize, String> {
let start = self.i;
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
self.bump();
}
if self.i == start {
return Err("expected a number in byte-pattern".to_string());
}
std::str::from_utf8(&self.s[start..self.i])
.ok()
.and_then(|t| t.parse().ok())
.ok_or_else(|| "invalid number in byte-pattern".to_string())
}
fn atom(&mut self) -> Result<BytePat, String> {
match self.peek() {
Some(b'\\') => {
self.bump();
let c = self.bump().ok_or("dangling backslash in byte-pattern")?;
Ok(match c {
b'd' => BytePat::Builtin(ByteClassKind::Digit),
b'D' => BytePat::Builtin(ByteClassKind::NotDigit),
b'w' => BytePat::Builtin(ByteClassKind::Word),
b'W' => BytePat::Builtin(ByteClassKind::NotWord),
b's' => BytePat::Builtin(ByteClassKind::Space),
b'S' => BytePat::Builtin(ByteClassKind::NotSpace),
b'p' | b'P' => return self.property(c == b'P'),
other => BytePat::Byte(other),
})
}
Some(b'.') => {
self.bump();
Ok(BytePat::Any)
}
Some(b'[') => self.class(),
Some(b'(') => {
self.bump();
let inner = self.alt()?;
if self.bump() != Some(b')') {
return Err("expected ')' in byte-pattern".to_string());
}
Ok(inner)
}
Some(c) => {
self.bump();
Ok(BytePat::Byte(c))
}
None => Err("expected a byte-pattern atom".to_string()),
}
}
fn property(&mut self, negated: bool) -> Result<BytePat, String> {
if self.bump() != Some(b'{') {
return Err("expected '{' after \\p in byte-pattern".to_string());
}
let start = self.i;
while matches!(self.peek(), Some(c) if c != b'}') {
self.bump();
}
let raw = &self.s[start..self.i];
if self.bump() != Some(b'}') {
return Err("unterminated \\p{...} in byte-pattern".to_string());
}
let name = match std::str::from_utf8(raw) {
Ok(n) => n,
Err(e) => return Err(format!("\\p{{...}} name is not UTF-8: {e}")),
};
match UnicodeClass::parse(name) {
Some(class) => Ok(BytePat::Property(class, negated)),
None => Err(format!(
"unknown Unicode class {name:?} in byte-pattern (known: {})",
UnicodeClass::NAMES
)),
}
}
fn class_char(&mut self) -> Option<char> {
let (c, width) = next_char(self.s, self.i)?;
self.i += width;
Some(c)
}
fn class(&mut self) -> Result<BytePat, String> {
self.bump(); let negated = if self.peek() == Some(b'^') {
self.bump();
true
} else {
false
};
let mut ranges = Vec::new();
while let Some(c) = self.peek() {
if c == b']' {
self.bump();
return Ok(BytePat::Class(ranges, negated));
}
let Some(lo) = self.class_char() else {
return Err("malformed character in byte-pattern class".to_string());
};
if self.peek() == Some(b'-') && self.s.get(self.i + 1).is_some_and(|&n| n != b']') {
self.bump(); let Some(hi) = self.class_char() else {
return Err("malformed character in byte-pattern class".to_string());
};
ranges.push((lo.min(hi), lo.max(hi)));
} else {
ranges.push((lo, lo));
}
}
Err("unterminated character class in byte-pattern".to_string())
}
}
#[cfg(test)]
mod tests {
use super::*;
fn m(pat: &str, text: &str) -> bool {
parse(pat.as_bytes()).unwrap().matches_whole(text.as_bytes())
}
#[test]
fn the_bitset_reach_is_the_vector_reach() {
let pats = [
"cond_[0-9]+", "a*b", "(ab|a)(c|bcd)", "x{2,3}", "x{2,}", "x{0,2}y", ".+", "\\w+\\d",
"[a-c\u{e9}]+", "a?b*", "(a|ab)(bc|c)?", "", "\\s*\\S+", "[^x]{1,3}",
];
let texts: &[&[u8]] = &[
b"", b"a", b"ab", b"abc", b"abcd", b"cond_42", b"cond_", b"xx", b"xxx", b"xxxxy",
b"aaab", b"ba", "\u{e9}a\u{e9}".as_bytes(), b"a\xffb", b"snake_case2", b" two words ",
];
for pat in pats {
let p = parse(pat.as_bytes()).unwrap();
for text in texts {
let vector = reach(&p, text, vec![0]);
let bits = reach_bits(&p, text, 1);
let from_bits: Vec<usize> = (0..=text.len()).filter(|&i| (bits & (1 << i)) != 0).collect();
assert_eq!(from_bits, vector, "{pat} over {text:?}");
assert_eq!(p.longest_prefix(text), vector.iter().copied().max(), "{pat} over {text:?}");
}
}
}
#[test]
fn classes_and_quantifiers() {
assert!(m("[A-Z][a-z]+", "Title"));
assert!(!m("[A-Z][a-z]+", "title"));
assert!(!m("[A-Z][a-z]+", "TITLE"));
assert!(m("\\d{4}", "2026"));
assert!(!m("\\d{4}", "202"));
assert!(m("\\w+", "snake_case2"));
assert!(m("a|bc|d", "bc"));
assert!(m("(ab)+", "ababab"));
assert!(!m("(ab)+", "aba"));
assert!(m("[^0-9]+", "letters"));
assert!(!m("[^0-9]+", "ha2"));
}
#[test]
fn a_class_range_spans_characters_not_bytes() {
assert!(m("[\u{3b1}-\u{3c9}]+", "\u{3b1}\u{3b2}\u{3b3}"), "a greek range");
assert!(!m("[\u{3b1}-\u{3c9}]+", "abc"), "and it excludes latin");
assert!(m("[\u{430}-\u{44f}]+", "\u{434}\u{430}"), "a cyrillic range");
assert!(m("[a-z]+", "abc"));
}
#[test]
fn a_property_class_reads_unicode_where_the_ascii_builtins_do_not() {
assert!(m("\\p{L}+", "\u{3b1}\u{3b2}"), "greek letters are letters");
assert!(m("\\p{L}+", "abc"));
assert!(m("\\p{N}+", "123"));
assert!(m("\\p{Lu}\\p{Ll}+", "Title"));
assert!(m("\\P{L}+", "123"), "the negation excludes letters");
assert!(!m("\\P{L}+", "abc"));
assert!(!m("\\w+", "\u{3b1}\u{3b2}"), "\\w is ASCII by decision");
assert!(m("\\p{L}+", "\u{3b1}\u{3b2}"), "and \\p{{L}} is the wide spelling");
}
#[test]
fn an_unknown_property_is_refused_rather_than_matching_nothing() {
let err = parse(b"\\p{Nonesuch}").expect_err("unknown category is refused");
assert!(err.contains("Nonesuch"), "the diagnostic names it: {err}");
assert!(err.contains('L'), "and lists what is known: {err}");
assert!(parse(b"\\p{L").is_err(), "unterminated");
assert!(parse(b"\\pL").is_err(), "missing brace");
}
#[test]
fn dot_steps_a_character_not_a_byte() {
assert!(m("...", "\u{3b1}\u{3b2}\u{3b3}"));
assert!(!m("......", "\u{3b1}\u{3b2}\u{3b3}"));
assert_eq!(parse(b".").unwrap().max_len(), Some(MAX_CHAR_BYTES));
assert_eq!(parse(b"[a-z]").unwrap().max_len(), Some(MAX_CHAR_BYTES));
assert_eq!(parse(b"a").unwrap().max_len(), Some(1));
assert_eq!(parse(b"\\d").unwrap().max_len(), Some(1));
}
#[test]
fn malformed_bytes_match_nothing_rather_than_panicking() {
let pat = parse(b".").expect("parses");
for bad in [&[0x80u8][..], &[0xc3][..], &[0xff][..], &[0xe2, 0x28][..]] {
assert!(!pat.matches_whole(bad), "no match on {bad:?}");
}
let cls = parse("[\u{3b1}-\u{3c9}]".as_bytes()).expect("parses");
assert!(!cls.matches_whole(&[0xce]), "a truncated greek lead byte");
}
#[test]
fn the_byte_machine_answers_what_the_tree_walk_answers() {
let pats = [
"cond_[0-9]+", "a*b", "(ab|a)(c|bcd)", "x{2,3}", "x{2,}", "x{0,2}y", "\\w+\\d",
"a?b*", "(a|ab)(bc|c)?", "", "\\s*\\S+", "[0-9a-f]{2,}", "v[0-9]+\\.[0-9]+",
".+", "[a-c\u{e9}]+", "\\p{L}+", "[^x]{1,3}",
];
let texts: &[&[u8]] = &[
b"", b"a", b"ab", b"abc", b"abcd", b"cond_42", b"cond_", b"xx", b"xxx", b"xxxxy",
b"aaab", b"ba", "\u{e9}a\u{e9}".as_bytes(), b"a\xffb", b"snake_case2",
b" two words ", b"deadbeef", b"v1.20", b"cond_", b"_0",
];
let mut taken = 0usize;
for src in pats {
let p = parse(src.as_bytes()).expect("parses");
let Some(machine) = p.automaton() else { continue };
taken += 1;
for text in texts {
assert_eq!(
machine.matches_whole(text),
p.matches_whole(text),
"{src:?} over {text:?}"
);
}
}
assert!(taken >= 12, "the machine took {taken} of the patterns");
for src in [".+", "[a-c\u{e9}]+", "\\p{L}+", "[^x]{1,3}", "[\u{3b1}-\u{3c9}]"] {
let p = parse(src.as_bytes()).expect("parses");
assert!(p.automaton().is_none(), "{src:?} reads characters and must be refused");
}
}
#[test]
fn whole_anchored() {
assert!(!m("\\d+", "12a"));
assert!(m(".*\\d.*", "12a"));
}
}