use crate::byte_nfa::{Builder, ByteClass, ByteNfa, Piece};
pub mod id {
pub const URL: u32 = 0;
pub const JWT: u32 = 1;
pub const PHONE: u32 = 2;
pub const UUID: u32 = 4;
pub const MAC: u32 = 5;
pub const HASH_DIGEST: u32 = 6;
pub const CIDR: u32 = 7;
pub const EMAIL: u32 = 8;
pub const IP: u32 = 9;
pub const CREDIT_CARD: u32 = 11;
pub const GEO: u32 = 12;
pub const MONEY: u32 = 13;
pub const VERSION: u32 = 14;
pub const BYTE_SIZE: u32 = 15;
pub const PERCENT: u32 = 16;
pub const DURATION: u32 = 17;
pub const QUANTITY: u32 = 18;
pub const HEX_COLOR: u32 = 19;
pub const PATH: u32 = 20;
pub const BASE64: u32 = 21;
pub const TIMESTAMP: u32 = 22;
pub const NUMBER: u32 = 23;
pub const WORD: u32 = 24;
}
#[must_use]
pub fn number() -> ByteNfa {
let mut b = Builder::new();
let all = number_piece(&mut b);
b.accept(all, id::NUMBER)
}
fn number_piece(b: &mut Builder) -> Piece {
let digit = ByteClass::range(b'0', b'9');
let lead = b.class(digit);
let whole = b.plus(lead);
let point = b.class(ByteClass::just(b'.'));
let after = b.class(digit);
let after_run = b.plus(after);
let frac = b.then(point, after_run);
let maybe_frac = b.maybe(frac);
b.then(whole, maybe_frac)
}
fn word_byte() -> ByteClass {
ByteClass::range(b'a', b'z')
.union(ByteClass::range(b'A', b'Z'))
.union(ByteClass::range(b'0', b'9'))
.union(ByteClass::just(b'_'))
}
#[must_use]
pub fn word() -> ByteNfa {
let mut b = Builder::new();
let all = word_piece(&mut b);
b.accept(all, id::WORD)
}
fn word_piece(b: &mut Builder) -> Piece {
let lead = b.class(word_lead());
let rest = b.class(word_byte());
let tail = b.star(rest);
b.then(lead, tail)
}
fn word_lead() -> ByteClass {
ByteClass::range(b'a', b'z').union(ByteClass::range(b'A', b'Z')).union(ByteClass::just(b'_'))
}
fn magnitude() -> ByteClass {
let mut c = ByteClass::none();
for m in b"kmgtpeKMGTPE" {
c.add(*m);
}
c
}
fn alphanumeric() -> ByteClass {
ByteClass::range(b'a', b'z')
.union(ByteClass::range(b'A', b'Z'))
.union(ByteClass::range(b'0', b'9'))
}
fn letter() -> ByteClass {
ByteClass::range(b'a', b'z').union(ByteClass::range(b'A', b'Z'))
}
fn scheme_byte() -> ByteClass {
alphanumeric()
.union(ByteClass::just(b'+'))
.union(ByteClass::just(b'.'))
.union(ByteClass::just(b'-'))
}
fn url_body_byte() -> ByteClass {
let mut stop = ByteClass::none();
for b in b" \t\n\x0c\r\"<>" {
stop.add(*b);
}
stop.negate()
}
fn base64url_byte() -> ByteClass {
alphanumeric().union(ByteClass::just(b'-')).union(ByteClass::just(b'_'))
}
fn hex() -> ByteClass {
ByteClass::range(b'0', b'9')
.union(ByteClass::range(b'a', b'f'))
.union(ByteClass::range(b'A', b'F'))
}
fn hex_letter() -> ByteClass {
ByteClass::range(b'a', b'f').union(ByteClass::range(b'A', b'F'))
}
#[must_use]
pub fn hash_digest() -> ByteNfa {
let follow = alphanumeric().union(ByteClass::just(b'_')).negate();
let shape = {
let mut b = Builder::new();
let mut lengths = exactly(&mut b, hex(), 32);
for len in [40usize, 64] {
let next = exactly(&mut b, hex(), len);
lengths = b.or(lengths, next);
}
b.accept_if(lengths, id::HASH_DIGEST, follow, true)
};
let holds_a_letter = {
let mut b = Builder::new();
let before = b.class(hex());
let head = b.star(before);
let letter = b.class(hex_letter());
let after = b.class(hex());
let tail = b.star(after);
let led = b.then(head, letter);
let all = b.then(led, tail);
b.accept(all, id::HASH_DIGEST)
};
shape.intersect(&holds_a_letter, id::HASH_DIGEST)
}
fn base64_byte() -> ByteClass {
alphanumeric().union(ByteClass::just(b'+')).union(ByteClass::just(b'/'))
}
fn base64_follow() -> ByteClass {
base64_byte().union(ByteClass::just(b'=')).negate()
}
fn base64_holding(class: ByteClass) -> ByteNfa {
let mut b = Builder::new();
let before = b.class(base64_byte());
let head = b.star(before);
let wanted = b.class(class);
let after = b.class(base64_byte());
let tail = b.star(after);
let led = b.then(head, wanted);
let all = b.then(led, tail);
b.accept(all, id::BASE64)
}
fn base64_shape(b: &mut Builder, lead: usize, pad: usize) -> Piece {
let head = exactly(b, base64_byte(), lead);
let four = exactly(b, base64_byte(), 4);
let more = b.star(four);
let body = b.then(head, more);
if pad == 0 {
return body;
}
let tail = exactly(b, ByteClass::just(b'='), pad);
b.then(body, tail)
}
#[must_use]
pub fn base64() -> ByteNfa {
let unpadded = {
let mut b = Builder::new();
let shape = base64_shape(&mut b, 16, 0);
b.accept_if(shape, id::BASE64, base64_follow(), true)
};
let marked = base64_holding(ByteClass::just(b'+').union(ByteClass::just(b'/')));
let mixed = base64_holding(ByteClass::range(b'0', b'9'))
.intersect(&base64_holding(ByteClass::range(b'A', b'Z')), id::BASE64)
.intersect(&base64_holding(ByteClass::range(b'a', b'z')), id::BASE64);
let by_mark = unpadded.intersect(&marked, id::BASE64);
let by_mix = unpadded.intersect(&mixed, id::BASE64);
let mut b = Builder::new();
let one = base64_shape(&mut b, 15, 1);
let one = b.accepting_if(one, id::BASE64, base64_follow(), true);
let two = base64_shape(&mut b, 14, 2);
let two = b.accepting_if(two, id::BASE64, base64_follow(), true);
let mark = b.adopt(&by_mark);
let mix = b.adopt(&by_mix);
let padded = b.or(one, two);
let plain = b.or(mark, mix);
let all = b.or(padded, plain);
b.build(all)
}
fn exactly(b: &mut Builder, class: ByteClass, n: usize) -> Piece {
let mut run = b.class(class);
for _ in 1..n {
let next = b.class(class);
run = b.then(run, next);
}
run
}
#[must_use]
pub fn uuid() -> ByteNfa {
let mut b = Builder::new();
let all = uuid_piece(&mut b);
let follow = alphanumeric().union(ByteClass::just(b'-')).negate();
b.accept_if(all, id::UUID, follow, true)
}
fn uuid_piece(b: &mut Builder) -> Piece {
let mut all = exactly(b, hex(), 8);
for len in [4usize, 4, 4, 12] {
let sep = b.class(ByteClass::just(b'-'));
let group = exactly(b, hex(), len);
let joined = b.then(sep, group);
all = b.then(all, joined);
}
all
}
fn mac_separators() -> ByteClass {
ByteClass::just(b':').union(ByteClass::just(b'-'))
}
#[must_use]
pub fn mac() -> ByteNfa {
let mut b = Builder::new();
let all = mac_piece(&mut b);
let follow = alphanumeric().union(mac_separators()).negate();
b.accept_if(all, id::MAC, follow, true)
}
fn mac_piece(b: &mut Builder) -> Piece {
let colon = mac_groups(b, b':');
let hyphen = mac_groups(b, b'-');
b.or(colon, hyphen)
}
fn mac_groups(b: &mut Builder, sep: u8) -> Piece {
let mut all = exactly(b, hex(), 2);
for _ in 1..6 {
let joiner = b.class(ByteClass::just(sep));
let group = exactly(b, hex(), 2);
let joined = b.then(joiner, group);
all = b.then(all, joined);
}
all
}
#[must_use]
pub fn hexcolor() -> ByteNfa {
let mut b = Builder::new();
let all = hexcolor_piece(&mut b);
let follow = alphanumeric().union(ByteClass::just(b'_')).negate();
b.accept_if(all, id::HEX_COLOR, follow, true)
}
fn hexcolor_piece(b: &mut Builder) -> Piece {
let hash = b.class(ByteClass::just(b'#'));
let six = exactly(b, hex(), 6);
let three = exactly(b, hex(), 3);
let digits = b.or(six, three);
b.then(hash, digits)
}
#[must_use]
pub fn bytesize() -> ByteNfa {
let mut b = Builder::new();
let all = bytesize_piece(&mut b);
b.accept_if(all, id::BYTE_SIZE, alphanumeric().negate(), true)
}
fn duration_letter() -> ByteClass {
let mut c = ByteClass::none();
for u in b"smhdwy" {
c.add(*u);
}
c
}
fn duration_pair_lead() -> ByteClass {
let mut c = ByteClass::none();
for u in b"num" {
c.add(*u);
}
c
}
#[must_use]
pub fn duration() -> ByteNfa {
let mut b = Builder::new();
let all = duration_piece(&mut b);
b.accept_if(all, id::DURATION, alphanumeric().negate(), true)
}
fn duration_piece(b: &mut Builder) -> Piece {
let num = number_piece(b);
let lead = b.class(duration_pair_lead());
let tail = b.class(ByteClass::just(b's'));
let pair = b.then(lead, tail);
let one = b.class(duration_letter());
let unit = b.or(pair, one);
let segment = b.then(num, unit);
b.plus(segment)
}
#[must_use]
pub fn percent() -> ByteNfa {
let mut b = Builder::new();
let all = percent_piece(&mut b);
b.accept(all, id::PERCENT)
}
fn percent_piece(b: &mut Builder) -> Piece {
let num = number_piece(b);
let sign = b.class(ByteClass::just(b'%'));
b.then(num, sign)
}
#[must_use]
pub fn url() -> ByteNfa {
let mut b = Builder::new();
let all = url_piece(&mut b);
b.accept(all, id::URL)
}
fn url_piece(b: &mut Builder) -> Piece {
let lead = b.class(letter());
let more = b.class(scheme_byte());
let more_run = b.star(more);
let scheme = b.then(lead, more_run);
let colon = b.class(ByteClass::just(b':'));
let first_slash = b.class(ByteClass::just(b'/'));
let second_slash = b.class(ByteClass::just(b'/'));
let opener = b.then(colon, first_slash);
let opener = b.then(opener, second_slash);
let opened = b.then(scheme, opener);
let byte = b.class(url_body_byte());
let body = b.plus(byte);
b.then(opened, body)
}
#[must_use]
pub fn jwt() -> ByteNfa {
let mut b = Builder::new();
let all = jwt_piece(&mut b);
b.accept(all, id::JWT)
}
fn jwt_piece(b: &mut Builder) -> Piece {
let e = b.class(ByteClass::just(b'e'));
let y = b.class(ByteClass::just(b'y'));
let j = b.class(ByteClass::just(b'J'));
let ey = b.then(e, y);
let marker = b.then(ey, j);
let rest = b.class(base64url_byte());
let rest_run = b.star(rest);
let header = b.then(marker, rest_run);
let first_dot = b.class(ByteClass::just(b'.'));
let payload = jwt_segment(b);
let second_dot = b.class(ByteClass::just(b'.'));
let signature = jwt_segment(b);
let through_first = b.then(header, first_dot);
let through_payload = b.then(through_first, payload);
let through_second = b.then(through_payload, second_dot);
b.then(through_second, signature)
}
fn jwt_segment(b: &mut Builder) -> Piece {
let byte = b.class(base64url_byte());
b.plus(byte)
}
fn bytesize_piece(b: &mut Builder) -> Piece {
let num = number_piece(b);
let mag = b.class(magnitude());
let binary = b.class(ByteClass::just(b'i'));
let maybe_binary = b.maybe(binary);
let with_binary = b.then(mag, maybe_binary);
let maybe_mag = b.maybe(with_binary);
let unit = b.class(ByteClass::just(b'b').union(ByteClass::just(b'B')));
let head = b.then(num, maybe_mag);
b.then(head, unit)
}
#[must_use]
pub fn tokens() -> ByteNfa {
let mut b = Builder::new();
let num = number_piece(&mut b);
let num = b.accepting(num, id::NUMBER);
let wrd = word_piece(&mut b);
let wrd = b.accepting(wrd, id::WORD);
let size = bytesize_piece(&mut b);
let size = b.accepting_if(size, id::BYTE_SIZE, alphanumeric().negate(), true);
let pct = percent_piece(&mut b);
let pct = b.accepting(pct, id::PERCENT);
let span = duration_piece(&mut b);
let span = b.accepting_if(span, id::DURATION, alphanumeric().negate(), true);
let uid = uuid_piece(&mut b);
let uid_follow = alphanumeric().union(ByteClass::just(b'-')).negate();
let uid = b.accepting_if(uid, id::UUID, uid_follow, true);
let hardware = mac_piece(&mut b);
let hardware_follow = alphanumeric().union(mac_separators()).negate();
let hardware = b.accepting_if(hardware, id::MAC, hardware_follow, true);
let color = hexcolor_piece(&mut b);
let color_follow = alphanumeric().union(ByteClass::just(b'_')).negate();
let color = b.accepting_if(color, id::HEX_COLOR, color_follow, true);
let digest = hash_digest();
let digest = b.adopt(&digest);
let blob = base64();
let blob = b.adopt(&blob);
let link = url_piece(&mut b);
let link = b.accepting(link, id::URL);
let token = jwt_piece(&mut b);
let token = b.accepting(token, id::JWT);
let mut all = num;
for piece in [wrd, size, pct, span, uid, hardware, color, digest, blob, link, token] {
all = b.or(all, piece);
}
b.build(all)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::token::TokenKind;
fn corpus() -> Vec<u8> {
let mut s = String::new();
for i in 0..200 {
s.push_str(&format!("let value_{i} = {} ;\n", i * 37));
s.push_str(&format!("call_{i}(alpha, {}.{}, beta) ;\n", i, i + 1));
s.push_str(&format!("size_{i} = {}KB ;\n", i * 3));
s.push_str(&format!("share_{i} = {}.{}% ;\n", i % 100, i));
s.push_str(&format!("wait_{i} = {}ms ;\n", i * 7));
s.push_str(&format!("id_{i} = {i:08x}-{i:04x}-{i:04x}-{i:04x}-{i:012x} ;\n"));
let o = i % 256;
s.push_str(&format!("mac_{i} = {o:02x}:{o:02x}:{o:02x}:{o:02x}:{o:02x}:{o:02x} ;\n"));
s.push_str(&format!("tint_{i} = #{:06x} ;\n", i * 1234));
s.push_str(&format!("link_{i} = http://h{i}.example/p/{i} ;\n"));
s.push_str(&format!("auth_{i} = eyJh{i:04x}.eyJz{i:04x}.Sfl-{i:04x} ;\n"));
match i % 3 {
0 => s.push_str(&format!("sum_{i} = a{i:031x} ;\n")),
1 => s.push_str(&format!("sum_{i} = b{i:039x} ;\n")),
_ => s.push_str(&format!("sum_{i} = c{i:063x} ;\n")),
}
s.push_str(&format!("blob_{i} = aB{i:012x}Yz ;\n"));
}
s.push_str("edge 1. 2.. .3 4.5.6 007 0 9999999999 x1 1x\n");
s.push_str("sizes 10B 512KB 1.5GiB 2TB 10MB 3b 10M 10MBx 1.5GIB 10iB\n");
s.push_str("shares 50% 3.5% 100% 50%x .5% 0%\n");
s.push_str("spans 1500ms 2.5s 3h20m 90s 5m 5us 5ns 5x 5string 3h20 5msx\n");
s.push_str("fixed 550e8400-e29b-41d4-a716-446655440000 aa:bb:cc:dd:ee:ff\n");
s.push_str("more aa-bb-cc-dd-ee-ff #abc #abcdef #abcd aa:bb-cc:dd-ee:ff\n");
s.into_bytes()
}
#[test]
fn the_automaton_ends_a_number_where_the_lexer_does() {
let input = corpus();
let toks = crate::lexer::lex(&input);
let nfa = number();
let mut seen = 0usize;
for t in &toks {
if t.kind != TokenKind::Number {
continue;
}
seen += 1;
let at = t.start();
let got = nfa.recognize(&input[at..]);
assert_eq!(
got,
Some((t.end() - at, id::NUMBER)),
"the lexer read {:?} at {at} and the automaton read {got:?}",
String::from_utf8_lossy(&input[at..t.end()])
);
}
assert!(seen > 400, "the corpus holds numbers to check: {seen}");
}
#[test]
fn the_byte_size_reads_the_units_the_lexer_reads() {
let nfa = bytesize();
let cases: [(&[u8], Option<usize>); 11] = [
(b"10B", Some(3)),
(b"512KB", Some(5)),
(b"1.5GiB", Some(6)),
(b"2TB", Some(3)),
(b"3b", Some(2)),
(b"10MB x", Some(4)),
(b"1.5GIB", None),
(b"10MBx", None),
(b"10MB7", None),
(b"10M", None),
(b"10iB", None),
];
for (run, want) in cases {
assert_eq!(
nfa.recognize(run).map(|(end, _)| end),
want,
"{}",
String::from_utf8_lossy(run)
);
}
}
#[test]
fn a_url_needs_its_marker_and_a_body() {
let nfa = url();
let cases: [(&[u8], Option<usize>); 9] = [
(b"http://x.com/p", Some(14)),
(b"https://a.b", Some(11)),
(b"git+ssh://h/r.git", Some(17)),
(b"http://x y", Some(8)),
(b"http://x\"y", Some(8)),
(b"http://x<y", Some(8)),
(b"http://", None),
(b"http:/x", None),
(b"://x", None),
];
for (run, want) in cases {
assert_eq!(
nfa.recognize(run).map(|(end, _)| end),
want,
"{}",
String::from_utf8_lossy(run)
);
}
}
#[test]
fn the_automaton_ends_a_url_where_the_lexer_does() {
let mut s = String::new();
for i in 0..40 {
s.push_str(&format!("see http://host{i}.example/p/{i}?q={i} ;\n"));
s.push_str(&format!("src=\"https://cdn{i}.example/a.js\" ;\n"));
s.push_str(&format!("<ftp://files{i}.example/x> ;\n"));
s.push_str(&format!("git+ssh://git{i}.example/r.git ;\n"));
}
let input = s.into_bytes();
let toks = crate::lexer::lex(&input);
let nfa = url();
let mut seen = 0usize;
for t in &toks {
if t.kind != TokenKind::Url {
continue;
}
seen += 1;
let at = t.start();
let got = nfa.recognize(&input[at..]);
assert_eq!(
got,
Some((t.end() - at, id::URL)),
"the lexer read {:?} at {at} and the automaton read {got:?}",
String::from_utf8_lossy(&input[at..t.end()])
);
}
assert!(seen > 100, "the input holds URLs to check: {seen}");
}
#[test]
fn a_token_needs_its_marker_and_three_segments() {
let nfa = jwt();
let cases: [(&[u8], Option<usize>); 8] = [
(b"eyJa.b.c", Some(8)),
(b"eyJhbGci.eyJzdWIi.SflKxw", Some(24)),
(b"eyJa-b_c.d.e", Some(12)),
(b"eyJa.b.c.d", Some(8)),
(b"abc.def.ghi", None),
(b"eyJa.b.", None),
(b"eyJa.b", None),
(b"eyj.b.c", None),
];
for (run, want) in cases {
assert_eq!(
nfa.recognize(run).map(|(end, _)| end),
want,
"{}",
String::from_utf8_lossy(run)
);
}
}
#[test]
fn the_automaton_ends_a_token_where_the_lexer_does() {
let mut s = String::new();
for i in 0..40 {
s.push_str(&format!("auth = eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiJ{i}9.Sfl-Kx_w{i} ;\n"));
}
let input = s.into_bytes();
let toks = crate::lexer::lex(&input);
let nfa = jwt();
let mut seen = 0usize;
for t in &toks {
if t.kind != TokenKind::Jwt {
continue;
}
seen += 1;
let at = t.start();
let got = nfa.recognize(&input[at..]);
assert_eq!(
got,
Some((t.end() - at, id::JWT)),
"the lexer read {:?} at {at} and the automaton read {got:?}",
String::from_utf8_lossy(&input[at..t.end()])
);
}
assert_eq!(seen, 40, "the input holds one token a line");
}
#[test]
fn a_percentage_needs_no_trailing_condition_where_a_byte_size_does() {
let pct = percent();
let cases: [(&[u8], Option<usize>); 5] = [
(b"50%", Some(3)),
(b"3.5%", Some(4)),
(b"0%", Some(2)),
(b"50%x", Some(3)),
(b"50", None),
];
for (run, want) in cases {
assert_eq!(
pct.recognize(run).map(|(end, _)| end),
want,
"{}",
String::from_utf8_lossy(run)
);
}
assert_eq!(
bytesize().recognize(b"10MBx"),
None,
"the byte size refuses the letter the percentage admits"
);
}
#[test]
fn the_duration_reads_the_segments_the_lexer_reads() {
let nfa = duration();
let cases: [(&[u8], Option<usize>); 11] = [
(b"1500ms", Some(6)),
(b"2.5s", Some(4)),
(b"3h20m", Some(5)),
(b"90s", Some(3)),
(b"5m", Some(2)),
(b"5us", Some(3)),
(b"5ns", Some(3)),
(b"5x", None),
(b"5string", None),
(b"3h20", None),
(b"5msx", None),
];
for (run, want) in cases {
assert_eq!(
nfa.recognize(run).map(|(end, _)| end),
want,
"{}",
String::from_utf8_lossy(run)
);
}
}
#[test]
fn the_lexer_takes_the_shorter_reading_where_precedence_says_so() {
let mut s = String::from("blob a");
for _ in 0..39 {
s.push('0');
}
s.push_str("+AAAAAAA end");
let input = s.into_bytes();
let toks = crate::lexer::lex(&input);
let read: Vec<(TokenKind, usize)> =
toks.iter().map(|t| (t.kind, t.end() - t.start())).collect();
assert_eq!(
read,
[
(TokenKind::Word, 4),
(TokenKind::Whitespace, 1),
(TokenKind::HashDigest, 40),
(TokenKind::Punct, 1),
(TokenKind::Word, 7),
(TokenKind::Whitespace, 1),
(TokenKind::Word, 3),
],
"the lexer no longer prefers the digest to the longer base64 run"
);
let digest = toks
.iter()
.find(|t| t.kind == TokenKind::HashDigest)
.expect("the lexer read a digest in it");
assert_eq!(
tokens().recognize(&input[digest.start()..]),
Some((40, id::HASH_DIGEST)),
"the gate took the forty-eight byte base64 run the lexer passed over"
);
}
#[test]
fn the_base64_blob_needs_its_length_and_its_charset() {
let nfa = base64();
assert_eq!(nfa.recognize(b"aB1dEfGhIjKlMnOp"), Some((16, id::BASE64)), "a mixed run");
assert_eq!(nfa.recognize(b"abcdefghijklmnop"), None, "one case and no digit");
assert_eq!(nfa.recognize(b"abcdefghijklmno+"), Some((16, id::BASE64)), "a base64 byte");
assert_eq!(nfa.recognize(b"abcdefghijklmno="), Some((16, id::BASE64)), "one pad");
assert_eq!(nfa.recognize(b"abcdefghijklmn=="), Some((16, id::BASE64)), "two pads");
assert_eq!(nfa.recognize(b"aB1dEfGhIjKlMnO"), None, "fifteen is no length");
assert_eq!(nfa.recognize(b"aB1dEfGhIjKlMnOpQrSt"), Some((20, id::BASE64)), "the next one");
assert_eq!(nfa.recognize(b"aB1dEfGhIjKlMnOpQr"), None, "eighteen is no length");
}
#[test]
fn the_fixed_shapes_read_what_their_scanners_read() {
let u = uuid();
assert_eq!(u.recognize(b"550e8400-e29b-41d4-a716-446655440000"), Some((36, id::UUID)));
assert_eq!(u.recognize(b"00000000-0000-0000-0000-000000000000"), Some((36, id::UUID)));
assert_eq!(u.recognize(b"550e8400-e29b-41d4-a716-44665544000"), None, "a short group");
let m = mac();
assert_eq!(m.recognize(b"aa:bb:cc:dd:ee:ff"), Some((17, id::MAC)));
assert_eq!(m.recognize(b"aa-bb-cc-dd-ee-ff"), Some((17, id::MAC)));
assert_eq!(m.recognize(b"aa:bb-cc:dd-ee:ff"), None, "one separator throughout");
let c = hexcolor();
assert_eq!(c.recognize(b"#abc"), Some((4, id::HEX_COLOR)));
assert_eq!(c.recognize(b"#abcdef"), Some((7, id::HEX_COLOR)));
assert_eq!(c.recognize(b"#abcd"), None, "four is not a length a color has");
}
#[test]
fn the_digest_needs_a_letter_and_one_of_three_lengths() {
let nfa = hash_digest();
let md5 = "d41d8cd98f00b204e9800998ecf8427e";
let sha1 = "da39a3ee5e6b4b0d3255bfef95601890afd80709";
let sha256 = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855";
assert_eq!(nfa.recognize(md5.as_bytes()), Some((32, id::HASH_DIGEST)));
assert_eq!(nfa.recognize(sha1.as_bytes()), Some((40, id::HASH_DIGEST)));
assert_eq!(nfa.recognize(sha256.as_bytes()), Some((64, id::HASH_DIGEST)));
assert_eq!(
nfa.recognize(format!("{sha1} rest").as_bytes()),
Some((40, id::HASH_DIGEST)),
"a boundary after it"
);
assert_eq!(nfa.recognize("0".repeat(40).as_bytes()), None, "no hex letter in it");
assert_eq!(nfa.recognize(&sha1.as_bytes()[..39]), None, "thirty-nine is not a length");
assert_eq!(nfa.recognize(format!("{sha1}0").as_bytes()), None, "a hex byte runs it on");
}
#[test]
fn the_gate_reads_inside_a_string_where_the_lexer_reads_one_token() {
let input = b"let s = \"hello world\" ;".to_vec();
let toks = crate::lexer::lex(&input);
let quoted = toks
.iter()
.find(|t| t.kind == TokenKind::Quoted)
.expect("the lexer took the string as one token");
assert_eq!(
&input[quoted.start()..quoted.end()],
b"\"hello world\"",
"the whole string, quotes included"
);
let inside = quoted.start() + 1;
assert_eq!(
tokens().recognize(&input[inside..]),
Some((5, id::WORD)),
"the gate reads `hello` where the lexer is inside a string"
);
assert!(
!toks.iter().any(|t| t.start() == inside),
"the lexer begins no token there, which is the whole of the difference"
);
}
#[test]
fn the_gate_divides_a_stream_the_way_the_lexer_divides_it() {
let input = corpus();
let toks = crate::lexer::lex(&input);
let nfa = tokens();
let (mut words, mut numbers) = (0usize, 0usize);
let (mut sizes, mut shares, mut spans) = (0usize, 0usize, 0usize);
let (mut uuids, mut macs, mut colors) = (0usize, 0usize, 0usize);
let (mut digests, mut blobs) = (0usize, 0usize);
let (mut links, mut webtokens) = (0usize, 0usize);
for t in &toks {
let want = match t.kind {
TokenKind::Number => {
numbers += 1;
id::NUMBER
}
TokenKind::Word => {
words += 1;
id::WORD
}
TokenKind::ByteSize => {
sizes += 1;
id::BYTE_SIZE
}
TokenKind::Percent => {
shares += 1;
id::PERCENT
}
TokenKind::Duration => {
spans += 1;
id::DURATION
}
TokenKind::Uuid => {
uuids += 1;
id::UUID
}
TokenKind::Mac => {
macs += 1;
id::MAC
}
TokenKind::HexColor => {
colors += 1;
id::HEX_COLOR
}
TokenKind::HashDigest => {
digests += 1;
id::HASH_DIGEST
}
TokenKind::Base64 => {
blobs += 1;
id::BASE64
}
TokenKind::Url => {
links += 1;
id::URL
}
TokenKind::Jwt => {
webtokens += 1;
id::JWT
}
_ => continue,
};
let at = t.start();
assert_eq!(
nfa.recognize(&input[at..]),
Some((t.end() - at, want)),
"the lexer read {:?} at {at} as {:?}",
String::from_utf8_lossy(&input[at..t.end()]),
t.kind
);
}
assert!(
words > 400
&& numbers > 400
&& sizes > 100
&& shares > 100
&& spans > 100
&& uuids > 100
&& macs > 100
&& colors > 100
&& digests > 100
&& blobs > 100
&& links > 100
&& webtokens > 100,
"the corpus holds every kind the gate reads: {words}, {numbers}, {sizes}, \
{shares}, {spans}, {uuids}, {macs}, {colors}, {digests}, {blobs}, {links} \
and {webtokens}"
);
}
#[test]
fn a_digit_beside_a_letter_divides_where_the_lexer_divides_it() {
let input = b"x1 1x a_2 3b".to_vec();
let toks = crate::lexer::lex(&input);
let nfa = tokens();
let mut read = Vec::new();
for t in &toks {
let want = match t.kind {
TokenKind::Number => id::NUMBER,
TokenKind::Word => id::WORD,
TokenKind::ByteSize => id::BYTE_SIZE,
_ => continue,
};
let at = t.start();
assert_eq!(
nfa.recognize(&input[at..]),
Some((t.end() - at, want)),
"the lexer read {:?} at {at} as {:?}",
String::from_utf8_lossy(&input[at..t.end()]),
t.kind
);
read.push(String::from_utf8_lossy(&input[at..t.end()]).into_owned());
}
assert_eq!(read, ["x1", "1", "x", "a_2", "3b"], "the runs the three branches reach");
}
}