fn escape_core(s: &str, apos: Option<&str>) -> String {
let mut out = String::with_capacity(s.len());
for c in s.chars() {
match c {
'&' => out.push_str("&"),
'<' => out.push_str("<"),
'>' => out.push_str(">"),
'"' => out.push_str("""),
'\'' => match apos {
Some(entity) => out.push_str(entity),
None => out.push('\''),
},
_ => out.push(c),
}
}
out
}
pub fn html_escape(s: &str) -> String {
escape_core(s, Some("'"))
}
pub fn html_enc_legacy(s: &str) -> String {
escape_core(s, None)
}
pub fn xml_escape(s: &str) -> String {
escape_core(s, Some("'"))
}
pub fn java_trim(s: &str) -> &str {
s.trim_matches(|c: char| c <= '\u{20}')
}
pub fn glob_to_regex(glob: &str, case_insensitive: bool) -> Result<regex::Regex, String> {
let mut regex = String::with_capacity(glob.len() * 6);
let chars: Vec<char> = glob.chars().collect();
let ln = chars.len();
let mut next_start = 0;
let mut escaped = false;
let mut idx = 0;
while idx < ln {
let c = chars[idx];
if !escaped {
if c == '?' {
push_literal_glob_section(&mut regex, &chars, next_start, idx);
regex.push_str("[^/]");
next_start = idx + 1;
} else if c == '*' {
push_literal_glob_section(&mut regex, &chars, next_start, idx);
if idx + 1 < ln && chars[idx + 1] == '*' {
if !(idx == 0 || chars[idx - 1] == '/') {
return Err(format!(
"The \"**\" wildcard must be directly after a \"/\" or it must be at the beginning, in this glob: {glob}"
));
}
if idx + 2 == ln {
regex.push_str(".*");
idx += 1;
} else {
if !(idx + 2 < ln && chars[idx + 2] == '/') {
return Err(format!(
"The \"**\" wildcard must be followed by \"/\", or must be at the end, in this glob: {glob}"
));
}
regex.push_str("(.*?/)*");
idx += 2; }
} else {
regex.push_str("[^/]*");
}
next_start = idx + 1;
} else if c == '\\' {
escaped = true;
} else if c == '[' || c == '{' {
return Err(format!(
"The \"{c}\" glob operator is currently unsupported (precede it with \\ for literal matching), in this glob: {glob}"
));
}
} else {
escaped = false;
}
idx += 1;
}
push_literal_glob_section(&mut regex, &chars, next_start, ln);
let mut pattern = String::with_capacity(regex.len() + 16);
if case_insensitive {
pattern.push_str("(?iu)");
}
pattern.push_str("^(?:");
pattern.push_str(®ex);
pattern.push_str(")$");
regex::Regex::new(&pattern).map_err(|e| format!("invalid glob regex: {e}"))
}
fn push_literal_glob_section(out: &mut String, chars: &[char], start: usize, end: usize) {
if start == end {
return;
}
let part: String = unescape_literal_glob_section(&chars[start..end]);
out.push_str(®ex::escape(&part));
}
fn unescape_literal_glob_section(s: &[char]) -> String {
let mut out = String::with_capacity(s.len());
let mut escaped = false;
for &c in s {
if !escaped && c == '\\' {
escaped = true;
} else {
out.push(c);
escaped = false;
}
}
if escaped {
out.push('\\'); }
out
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn html_escape_basic() {
assert_eq!(
html_escape("<a href=\"x\">&'"),
"<a href="x">&'"
);
}
#[test]
fn xml_escape_apos() {
assert_eq!(xml_escape("a'b"), "a'b");
}
#[test]
fn java_trim_only_ascii_space() {
assert_eq!(java_trim(" \t x \n"), "x");
assert_eq!(java_trim("\u{a0}x\u{a0}"), "\u{a0}x\u{a0}");
}
}