use std::ops::Range;
#[non_exhaustive]
#[derive(Copy, Clone, Debug, PartialEq, Eq, Hash)]
pub enum DataKind {
Key,
String,
Number,
Literal,
Comment,
Punct,
Tag,
}
#[derive(Copy, Clone, Debug, Default)]
pub struct JsonLexer;
impl JsonLexer {
pub fn new() -> JsonLexer {
JsonLexer
}
pub fn matches_lang(label: &str) -> bool {
let first = label.split_whitespace().next().unwrap_or("");
["json", "jsonc", "json5", "jsonl", "ndjson"]
.iter()
.any(|l| first.eq_ignore_ascii_case(l))
}
pub fn spans(&self, line: &str) -> Vec<(Range<usize>, DataKind)> {
let mut out = Vec::new();
let b = line.as_bytes();
let mut i = 0;
while i < b.len() {
let c = b[i];
if c == b'/' && b.get(i + 1) == Some(&b'/') {
out.push((i..b.len(), DataKind::Comment));
break;
}
if c == b'/' && b.get(i + 1) == Some(&b'*') {
let end = find_sub(b, i + 2, b"*/").map(|p| p + 2).unwrap_or(b.len());
out.push((i..end, DataKind::Comment));
i = end;
continue;
}
if c == b'"' {
let end = scan_dquote(b, i + 1);
let kind = if json_key_follows(b, end) {
DataKind::Key
} else {
DataKind::String
};
out.push((i..end, kind));
i = end;
continue;
}
if c.is_ascii_digit() || (c == b'-' && b.get(i + 1).is_some_and(u8::is_ascii_digit)) {
let end = scan_number(b, i + 1);
out.push((i..end, DataKind::Number));
i = end;
continue;
}
if c.is_ascii_alphabetic() {
let end = scan_word(b, i + 1);
if matches!(&line[i..end], "true" | "false" | "null") {
out.push((i..end, DataKind::Literal));
}
i = end;
continue;
}
if matches!(c, b'{' | b'}' | b'[' | b']' | b',' | b':') {
out.push((i..i + 1, DataKind::Punct));
i += 1;
continue;
}
i += utf8_len(c);
}
out
}
}
#[derive(Copy, Clone, Debug, Default)]
pub struct YamlLexer;
impl YamlLexer {
pub fn new() -> YamlLexer {
YamlLexer
}
pub fn matches_lang(label: &str) -> bool {
let first = label.split_whitespace().next().unwrap_or("");
first.eq_ignore_ascii_case("yaml") || first.eq_ignore_ascii_case("yml")
}
pub fn spans(&self, line: &str) -> Vec<(Range<usize>, DataKind)> {
let mut out = Vec::new();
let b = line.as_bytes();
let trimmed = line.trim();
if trimmed == "---" || trimmed == "..." {
let start = line.len() - line.trim_start().len();
out.push((start..start + 3, DataKind::Tag));
return out;
}
let mut i = 0;
let mut prev_space = true;
while i < b.len() {
let c = b[i];
if c == b'#' && prev_space {
out.push((i..b.len(), DataKind::Comment));
break;
}
if c == b'"' || c == b'\'' {
let end = if c == b'"' {
scan_dquote(b, i + 1)
} else {
scan_squote(b, i + 1)
};
out.push((i..end, key_or_string(b, end)));
prev_space = false;
i = end;
continue;
}
if c == b'-' && prev_space && matches!(b.get(i + 1), None | Some(b' ')) {
out.push((i..i + 1, DataKind::Punct));
prev_space = true; i += 1;
if b.get(i) == Some(&b' ') {
i += 1;
}
continue;
}
if matches!(c, b'&' | b'*' | b'!') && prev_space {
let mut j = i + 1;
while j < b.len()
&& !b[j].is_ascii_whitespace()
&& !matches!(b[j], b',' | b'}' | b']')
{
j += 1;
}
if j > i + 1 || c == b'!' {
out.push((i..j, DataKind::Tag));
prev_space = false;
i = j;
continue;
}
}
if c.is_ascii_digit() || (c == b'-' && b.get(i + 1).is_some_and(u8::is_ascii_digit)) {
let end = scan_number(b, i + if c == b'-' { 2 } else { 1 });
if key_follows(b, end) {
out.push((i..end, DataKind::Key));
} else if matches!(
b.get(end),
None | Some(b' ') | Some(b',') | Some(b'}') | Some(b']') | Some(b'#')
) {
out.push((i..end, DataKind::Number));
}
prev_space = false;
i = end.max(i + 1);
continue;
}
if c.is_ascii_alphabetic() || c == b'_' {
let end = scan_yaml_word(b, i + 1);
let word = &line[i..end];
if key_follows(b, end) {
out.push((i..end, DataKind::Key));
} else if is_yaml_literal(word) {
out.push((i..end, DataKind::Literal));
}
prev_space = false;
i = end;
continue;
}
if c == b'~' {
out.push((i..i + 1, DataKind::Literal));
prev_space = false;
i += 1;
continue;
}
if matches!(
c,
b'{' | b'}' | b'[' | b']' | b',' | b':' | b'|' | b'>' | b'?'
) {
out.push((i..i + 1, DataKind::Punct));
prev_space = matches!(c, b'{' | b'[' | b',' | b':');
i += 1;
continue;
}
prev_space = c.is_ascii_whitespace();
i += utf8_len(c);
}
out
}
}
fn json_key_follows(b: &[u8], end: usize) -> bool {
let mut j = end;
while j < b.len() && b[j] == b' ' {
j += 1;
}
b.get(j) == Some(&b':')
}
fn key_follows(b: &[u8], end: usize) -> bool {
let mut j = end;
while j < b.len() && b[j] == b' ' {
j += 1;
}
if b.get(j) != Some(&b':') {
return false;
}
matches!(
b.get(j + 1),
None | Some(b' ') | Some(b'\t') | Some(b',') | Some(b'}') | Some(b']')
)
}
fn key_or_string(b: &[u8], end: usize) -> DataKind {
if key_follows(b, end) {
DataKind::Key
} else {
DataKind::String
}
}
fn scan_dquote(b: &[u8], from: usize) -> usize {
let mut j = from;
while j < b.len() {
match b[j] {
b'\\' => j = (j + 2).min(b.len()),
b'"' => return j + 1,
_ => j += 1,
}
}
b.len()
}
fn scan_squote(b: &[u8], from: usize) -> usize {
let mut j = from;
while j < b.len() {
if b[j] == b'\'' {
if b.get(j + 1) == Some(&b'\'') {
j += 2;
continue;
}
return j + 1;
}
j += 1;
}
b.len()
}
fn scan_number(b: &[u8], from: usize) -> usize {
let mut j = from;
while j < b.len() && (b[j].is_ascii_alphanumeric() || b[j] == b'_' || b[j] == b'.') {
j += 1;
}
j
}
fn scan_word(b: &[u8], from: usize) -> usize {
let mut j = from;
while j < b.len() && (b[j].is_ascii_alphanumeric() || b[j] == b'_') {
j += 1;
}
j
}
fn scan_yaml_word(b: &[u8], from: usize) -> usize {
let mut j = from;
while j < b.len() && (b[j].is_ascii_alphanumeric() || b[j] == b'_' || b[j] == b'-') {
j += 1;
}
j
}
fn is_yaml_literal(word: &str) -> bool {
["true", "false", "null", "yes", "no", "on", "off"]
.iter()
.any(|l| word.eq_ignore_ascii_case(l))
}
fn find_sub(hay: &[u8], from: usize, needle: &[u8]) -> Option<usize> {
hay.get(from..)?
.windows(needle.len())
.position(|w| w == needle)
.map(|p| p + from)
}
fn utf8_len(first: u8) -> usize {
match first {
0x00..=0x7F => 1,
0xC0..=0xDF => 2,
0xE0..=0xEF => 3,
_ => 4,
}
}
#[cfg(test)]
#[path = "data_tests.rs"]
mod tests;