#[inline]
pub(crate) fn is_bidi_char(c: char) -> bool {
matches!(
c,
'\u{061C}' | '\u{200E}' | '\u{200F}' | '\u{202A}' | '\u{202B}' | '\u{202C}' | '\u{202D}' | '\u{202E}' | '\u{2066}' | '\u{2067}' | '\u{2068}' | '\u{2069}' )
}
pub fn strip_bidi_formatting(s: &str) -> String {
if !s.chars().any(is_bidi_char) {
return s.to_string();
}
s.chars().filter(|c| !is_bidi_char(*c)).collect()
}
pub fn fix_html_comment_fences(s: &str) -> String {
if !s.contains("-->") {
return s.to_string();
}
let mut result = String::with_capacity(s.len() + 16);
let mut current_pos = 0;
while let Some(open_idx) = s[current_pos..].find("<!--") {
let abs_open = current_pos + open_idx;
if let Some(close_idx) = s[abs_open..].find("-->") {
let abs_close = abs_open + close_idx;
let mut after_fence = abs_close + 3;
let opener_has_extra_hyphen = s
.get(abs_open + 4..)
.is_some_and(|rest| rest.starts_with('-'));
if opener_has_extra_hyphen
&& s.get(after_fence..)
.is_some_and(|rest| rest.starts_with('-'))
{
after_fence += 1;
}
result.push_str(&s[current_pos..after_fence]);
let after_content = &s[after_fence..];
let needs_newline = if after_content.is_empty()
|| after_content.starts_with('\n')
|| after_content.starts_with("\r\n")
{
false
} else {
let next_newline = after_content.find('\n');
let until_newline = match next_newline {
Some(pos) => &after_content[..pos],
None => after_content,
};
!until_newline.trim().is_empty()
};
if needs_newline {
result.push('\n');
}
current_pos = after_fence;
} else {
result.push_str(&s[current_pos..]);
current_pos = s.len();
break;
}
}
if current_pos < s.len() {
result.push_str(&s[current_pos..]);
}
result
}
pub fn normalize_markdown(markdown: &str) -> String {
let cleaned = normalize_line_endings(markdown);
let cleaned = strip_bidi_formatting(&cleaned);
fix_html_comment_fences(&cleaned)
}
fn normalize_line_endings(s: &str) -> String {
if !s.contains('\r') {
return s.to_string();
}
let mut out = String::with_capacity(s.len());
let mut chars = s.chars().peekable();
while let Some(c) = chars.next() {
if c == '\r' {
if chars.peek() == Some(&'\n') {
chars.next();
}
out.push('\n');
} else {
out.push(c);
}
}
out
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_strip_bidi_formatting_cases() {
let cases: &[(&str, &str)] = &[
("hello world", "hello world"),
("", ""),
("**bold** text", "**bold** text"),
("he\u{202D}llo", "hello"),
("**asdf** or \u{202D}**(1234**", "**asdf** or **(1234**"),
("a\u{200E}b\u{200F}c", "abc"),
("\u{202A}text\u{202B}more\u{202C}", "textmore"),
("\u{2066}a\u{2067}b\u{2068}c\u{2069}", "abc"),
(
"\u{061C}\u{200E}\u{200F}\u{202A}\u{202B}\u{202C}\u{202D}\u{202E}\u{2066}\u{2067}\u{2068}\u{2069}",
"",
),
("hello\u{061C}world", "helloworld"),
("\u{061C}**bold**", "**bold**"),
("你好世界", "你好世界"),
("مرحبا", "مرحبا"),
("🎉", "🎉"),
];
for (input, expected) in cases {
assert_eq!(strip_bidi_formatting(input), *expected, "input: {:?}", input);
}
}
#[test]
fn test_normalize_markdown_basic() {
assert_eq!(normalize_markdown("hello"), "hello");
assert_eq!(
normalize_markdown("**bold** \u{202D}**more**"),
"**bold** **more**"
);
}
#[test]
fn test_normalize_markdown_html_comment() {
assert_eq!(
normalize_markdown("<!-- comment -->Some text"),
"<!-- comment -->\nSome text"
);
}
#[test]
fn test_fix_html_comment_fences_cases() {
let cases: &[(&str, &str)] = &[
("hello world", "hello world"),
("**bold** text", "**bold** text"),
("", ""),
(
"<!-- comment -->Same line text",
"<!-- comment -->\nSame line text",
),
(
"<!-- comment -->\nNext line text",
"<!-- comment -->\nNext line text",
),
(
"<!-- comment --> \nSome text",
"<!-- comment --> \nSome text",
),
(
"<!--\nmultiline\ncomment\n-->Trailing text",
"<!--\nmultiline\ncomment\n-->\nTrailing text",
),
(
"<!--\nmultiline\n-->\n\nParagraph text",
"<!--\nmultiline\n-->\n\nParagraph text",
),
(
"<!-- first -->Text\n\n<!-- second -->More text",
"<!-- first -->\nText\n\n<!-- second -->\nMore text",
),
(
"Some text before <!-- comment -->",
"Some text before <!-- comment -->",
),
("-->some text", "-->some text"),
("<!-- <!-- -->Trailing", "<!-- <!-- -->\nTrailing"),
(
"<!-- valid -->FixMe\ntext --> Ignore\n<!-- valid2 -->FixMe2",
"<!-- valid -->\nFixMe\ntext --> Ignore\n<!-- valid2 -->\nFixMe2",
),
(
"<!-- comment -->\r\nSome text",
"<!-- comment -->\r\nSome text",
),
(
"<!--- comment --->Trailing text",
"<!--- comment --->\nTrailing text",
),
];
for (input, expected) in cases {
assert_eq!(
fix_html_comment_fences(input),
*expected,
"input: {:?}",
input
);
}
}
}