use htmd::HtmlToMarkdownBuilder;
use htmd::options::{BrStyle, Options};
use super::sanitize::find_sub;
const MAX_HTML_DEPTH: usize = 100;
pub fn html_to_markdown(html: &str) -> String {
if max_open_tag_depth(html) > MAX_HTML_DEPTH {
return tidy(&strip_tags_to_text(html));
}
let converter = HtmlToMarkdownBuilder::new()
.options(Options {
br_style: BrStyle::Backslash,
ul_bullet_spacing: 1,
ol_number_spacing: 1,
..Options::default()
})
.skip_tags(vec!["script", "style"])
.build();
let md = converter.convert(html).unwrap_or_default();
tidy(&md)
}
fn tidy(md: &str) -> String {
let mut out = String::with_capacity(md.len());
for line in md.lines() {
out.push_str(line.trim_end());
out.push('\n');
}
out.truncate(out.trim_end().len());
let start = out.len() - out.trim_start().len();
out.drain(..start);
out
}
const VOID_ELEMENTS: &[&str] = &[
"area", "base", "basefont", "bgsound", "br", "embed", "hr", "img", "input", "keygen", "link",
"meta", "param", "source", "track", "wbr",
];
fn is_void_element(tag: &str) -> bool {
VOID_ELEMENTS.iter().any(|v| tag.eq_ignore_ascii_case(v))
}
fn implied_end_pops(incoming: &str, top: &str) -> bool {
let is = |t: &str, n: &str| t.eq_ignore_ascii_case(n);
let cell = |t: &str| is(t, "td") || is(t, "th");
let definition = |t: &str| is(t, "dd") || is(t, "dt");
if is(top, "p") {
return is(incoming, "p")
|| is(incoming, "li")
|| definition(incoming)
|| cell(incoming)
|| is(incoming, "tr");
}
if is(top, "li") {
return is(incoming, "li");
}
if definition(top) {
return definition(incoming);
}
if cell(top) {
return cell(incoming) || is(incoming, "tr");
}
if is(top, "tr") {
return is(incoming, "tr");
}
if is(top, "option") {
return is(incoming, "option");
}
false
}
fn is_foreign_root(tag: &str) -> bool {
tag.eq_ignore_ascii_case("svg") || tag.eq_ignore_ascii_case("math")
}
fn max_open_tag_depth(html: &str) -> usize {
let bytes = html.as_bytes();
let mut i = 0;
let mut open: Vec<&str> = Vec::with_capacity(16);
let mut max_depth: usize = 0;
let mut foreign: usize = 0;
while i < bytes.len() {
if bytes[i] != b'<' {
i += 1;
continue;
}
if let Some(end) = skip_comment_or_declaration(bytes, i) {
i = end;
continue;
}
let closing = bytes.get(i + 1) == Some(&b'/');
let name_start = if closing { i + 2 } else { i + 1 };
if !bytes.get(name_start).is_some_and(u8::is_ascii_alphabetic) {
i += 1;
continue;
}
let name_end = tag_name_end(bytes, name_start);
let tag = &html[name_start..name_end];
let Some((gt, self_closing)) = find_tag_end(bytes, name_end) else {
break; };
i = gt + 1;
if foreign == 0 && is_void_element(tag) {
continue;
}
if closing {
if open.last().is_some_and(|top| top.eq_ignore_ascii_case(tag)) {
let top = open.pop();
if top.is_some_and(is_foreign_root) {
foreign -= 1;
}
}
} else {
if self_closing && is_foreign_root(tag) {
continue;
}
if is_foreign_root(tag) {
foreign += 1;
}
if foreign == 0 {
while open.last().is_some_and(|top| implied_end_pops(tag, top)) {
open.pop();
}
}
open.push(tag);
max_depth = max_depth.max(open.len());
if max_depth > MAX_HTML_DEPTH {
return max_depth;
}
}
}
max_depth
}
fn strip_tags_to_text(html: &str) -> String {
let bytes = html.as_bytes();
let mut out = String::with_capacity(html.len());
let mut i = 0;
while i < bytes.len() {
if bytes[i] != b'<' {
let next_lt = bytes[i..]
.iter()
.position(|&b| b == b'<')
.map_or(bytes.len(), |p| i + p);
out.push_str(&html[i..next_lt]);
i = next_lt;
continue;
}
if let Some(end) = skip_comment_or_declaration(bytes, i) {
i = end;
continue;
}
let closing = bytes.get(i + 1) == Some(&b'/');
let name_start = if closing { i + 2 } else { i + 1 };
if !bytes.get(name_start).is_some_and(u8::is_ascii_alphabetic) {
i += 1;
continue;
}
let name_end = tag_name_end(bytes, name_start);
let tag = &html[name_start..name_end];
let Some((gt, _)) = find_tag_end(bytes, name_end) else {
break;
};
if !closing && (tag.eq_ignore_ascii_case("script") || tag.eq_ignore_ascii_case("style")) {
i = skip_element_body(bytes, gt + 1, tag);
} else {
i = gt + 1;
}
}
out
}
fn is_tag_space(c: u8) -> bool {
matches!(c, b'\t' | b'\n' | b'\r' | b'\x0C' | b' ')
}
fn tag_name_end(bytes: &[u8], start: usize) -> usize {
let mut end = start;
while end < bytes.len() && !is_tag_space(bytes[end]) && !matches!(bytes[end], b'/' | b'>') {
end += 1;
}
end
}
fn find_tag_end(bytes: &[u8], from: usize) -> Option<(usize, bool)> {
#[derive(Clone, Copy)]
enum S {
BeforeName,
Name,
AfterName,
UnquotedValue,
}
let mut j = from;
let mut state = S::BeforeName;
let mut self_closing = false;
while j < bytes.len() {
let c = bytes[j];
if c == b'>' {
return Some((j, self_closing));
}
self_closing = c == b'/' && !matches!(state, S::UnquotedValue);
j += 1;
state = match (state, c) {
(S::Name | S::AfterName, b'=') => {
while bytes.get(j).is_some_and(|&c| is_tag_space(c)) {
j += 1;
}
match bytes.get(j) {
Some(&q @ (b'"' | b'\'')) => {
j = find_byte(bytes, j + 1, q)? + 1;
S::BeforeName
}
_ => S::UnquotedValue,
}
}
(S::UnquotedValue, c) if is_tag_space(c) => S::BeforeName,
(S::UnquotedValue, b'&') => {
let run = bytes[j..]
.iter()
.take_while(|b| b.is_ascii_alphanumeric() || **b == b'#')
.count();
if bytes.get(j + run) == Some(&b';') {
j += run + 1;
}
S::UnquotedValue
}
(S::UnquotedValue, _) => S::UnquotedValue,
(S::BeforeName | S::AfterName, c) if is_tag_space(c) || c == b'/' => state,
(S::Name, c) if is_tag_space(c) => S::AfterName,
(S::Name, b'/') => S::BeforeName,
_ => S::Name,
};
}
None
}
fn skip_comment_or_declaration(bytes: &[u8], at: usize) -> Option<usize> {
if bytes[at..].starts_with(b"<!--") {
return Some(comment_end(bytes, at + 4));
}
if bytes[at..].starts_with(b"<!") || bytes[at..].starts_with(b"<?") {
return Some(find_byte(bytes, at, b'>').map_or(bytes.len(), |p| p + 1));
}
None
}
fn comment_end(bytes: &[u8], body: usize) -> usize {
if bytes.get(body) == Some(&b'>') {
return body + 1;
}
if bytes.get(body) == Some(&b'-') && bytes.get(body + 1) == Some(&b'>') {
return body + 2;
}
let close = find_sub(&bytes[body..], b"-->").map(|p| body + p + 3);
let close_bang = find_sub(&bytes[body..], b"--!>").map(|p| body + p + 4);
match (close, close_bang) {
(Some(a), Some(b)) => a.min(b),
(Some(e), None) | (None, Some(e)) => e,
(None, None) => bytes.len(),
}
}
fn find_byte(bytes: &[u8], from: usize, b: u8) -> Option<usize> {
bytes[from..].iter().position(|&c| c == b).map(|p| from + p)
}
fn skip_element_body(bytes: &[u8], from: usize, tag: &str) -> usize {
let mut i = from;
while i < bytes.len() {
if bytes[i] == b'<' && bytes.get(i + 1) == Some(&b'/') {
let name_end = tag_name_end(bytes, i + 2);
if bytes[i + 2..name_end].eq_ignore_ascii_case(tag.as_bytes()) {
return find_tag_end(bytes, name_end).map_or(bytes.len(), |(gt, _)| gt + 1);
}
}
i += 1;
}
bytes.len()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn structural_elements_convert_to_markdown() {
let html = r#"<h2>Title</h2><p>Some <em>emphasis</em> and <strong>bold</strong>.</p>
<ul><li>one</li><li>two</li></ul>
<p><a href="https://example.com/a">link</a> and <img src="https://example.com/i.png" alt="alt text"></p>
<pre><code>let x = 1;</code></pre>"#;
let md = html_to_markdown(html);
assert!(md.contains("## Title"), "{md}");
assert!(
md.contains("*emphasis*") || md.contains("_emphasis_"),
"{md}"
);
assert!(md.contains("**bold**"), "{md}");
assert!(md.contains("- one") || md.contains("* one"), "{md}");
assert!(md.contains("[link](https://example.com/a)"), "{md}");
assert!(
md.contains(""),
"{md}"
);
assert!(md.contains("let x = 1;"), "{md}");
}
#[test]
fn script_and_style_are_dropped_wholesale() {
let html =
r#"<p>keep</p><script>alert("x")</script><style>p{color:red}</style><!-- comment -->"#;
let md = html_to_markdown(html);
assert!(md.contains("keep"));
assert!(!md.contains("alert"), "{md}");
assert!(!md.contains("color"), "{md}");
assert!(!md.contains("comment"), "{md}");
}
#[test]
fn unknown_markup_reduces_to_text_content_no_raw_html_survives() {
let html = r#"<article data-x="1"><custom-widget>inner text</custom-widget><video controls>fallback</video></article>"#;
let md = html_to_markdown(html);
assert!(md.contains("inner text"));
assert!(md.contains("fallback"));
assert!(!md.contains('<'), "raw HTML survived: {md}");
}
#[test]
fn conversion_is_deterministic() {
let html = include_str!("fixtures/golden_probe.html");
let a = html_to_markdown(html);
let b = html_to_markdown(html);
assert_eq!(a, b);
assert!(!a.is_empty(), "the probe fixture must produce output");
assert!(!a.contains("leak"), "script/style/comment leaked: {a}");
assert!(a.contains("widget text"), "unknown element text lost: {a}");
}
#[test]
fn javascript_href_is_preserved_as_data_not_executed_markup() {
let md = html_to_markdown(r#"<a href="javascript:alert(1)">x</a>"#);
assert!(!md.contains('<'));
}
#[test]
fn tables_convert_or_reduce_to_text() {
let html = r#"<table><thead><tr><th>h1</th><th>h2</th></tr></thead>
<tbody><tr><td>a1</td><td>a2</td></tr><tr><td>b1</td><td>b2</td></tr></tbody></table>"#;
let md = html_to_markdown(html);
for cell in ["h1", "h2", "a1", "a2", "b1", "b2"] {
assert!(md.contains(cell), "cell {cell} lost: {md}");
}
assert!(!md.contains('<'), "raw HTML survived: {md}");
assert!(md.contains('|'), "expected a pipe table: {md}");
}
#[test]
fn empty_and_whitespace_input_yield_empty() {
assert_eq!(html_to_markdown(""), "");
assert_eq!(html_to_markdown(" \n\t "), "");
assert_eq!(html_to_markdown("<p></p>"), "");
}
#[test]
fn entities_decode_to_text_not_markup() {
let md = html_to_markdown("<p><b>not bold</b> & ©</p>");
assert!(md.contains("not bold"), "{md}");
assert!(md.contains('&') || md.contains("amp"), "{md}");
assert!(
md.contains('©'),
"numeric character reference decoded: {md}"
);
let unescaped = md
.char_indices()
.filter(|(i, c)| *c == '<' && !md[..*i].ends_with('\\'))
.count();
assert_eq!(unescaped, 0, "unescaped `<` survived: {md}");
}
#[test]
fn output_has_no_trailing_whitespace_and_hard_breaks_survive() {
let md = html_to_markdown("<p>first line<br>second line</p>");
assert!(md.contains("first line"), "{md}");
assert!(md.contains("second line"), "{md}");
assert!(
md.lines().all(|l| l == l.trim_end()),
"line kept trailing whitespace: {md:?}"
);
assert!(
md.lines().count() > 1,
"the hard break must survive trimming: {md:?}"
);
assert_eq!(md, md.trim(), "outer whitespace trimmed: {md:?}");
}
#[test]
fn beyond_max_depth_degrades_to_tag_stripped_text_not_markdown() {
let depth = 1000;
let html = format!(
"{}<em>innermost</em>{}",
"<div>".repeat(depth),
"</div>".repeat(depth)
);
let md = html_to_markdown(&html);
assert!(md.contains("innermost"), "{md}");
assert!(
!md.contains("*innermost*") && !md.contains("_innermost_"),
"output looks markdown-converted, not tag-stripped: {md}"
);
}
#[test]
fn the_depth_ceiling_converts_at_100_and_degrades_at_101() {
let nested = |depth: usize| {
format!(
"{}<em>innermost</em>{}",
"<div>".repeat(depth - 1),
"</div>".repeat(depth - 1)
)
};
assert_eq!(max_open_tag_depth(&nested(100)), 100);
assert_eq!(max_open_tag_depth(&nested(101)), 101);
let at = html_to_markdown(&nested(100));
assert!(
at.contains("*innermost*") || at.contains("_innermost_"),
"depth 100 is the last depth htmd still converts, and this one degraded \
to tag-stripped text: {at}"
);
let past = html_to_markdown(&nested(101));
assert!(past.contains("innermost"), "text lost at depth 101: {past}");
assert!(
!past.contains("*innermost*") && !past.contains("_innermost_"),
"depth 101 is past the ceiling and must degrade to tag-stripped text: {past}"
);
}
fn deep(prefix: &str, unit: &str) -> String {
format!("{prefix}{}<em>innermost</em>", unit.repeat(1000))
}
fn assert_degraded_to_text(html: &str, case: &str) {
let md = html_to_markdown(html);
assert!(md.contains("innermost"), "{case}: text lost: {md}");
assert!(
!md.contains("*innermost*") && !md.contains("_innermost_"),
"{case}: output is markdown-converted, not tag-stripped: {md}"
);
}
#[test]
fn mismatched_close_tags_do_not_reduce_the_scanned_depth() {
assert_degraded_to_text(&deep("", "<div></b>"), "</b>");
assert_degraded_to_text(&deep("", "<div></p>"), "</p>");
}
#[test]
fn abrupt_closing_empty_comment_does_not_hide_the_nesting() {
assert_degraded_to_text(&deep("<!-->", "<div>"), "<!-->");
assert_degraded_to_text(&deep("<!--->", "<div>"), "<!--->");
}
#[test]
fn bang_comment_end_terminates_the_comment() {
assert_degraded_to_text(&deep("<!--x--!>", "<div>"), "--!>");
}
#[test]
fn question_mark_opens_a_bogus_comment_ending_at_the_first_gt() {
assert_degraded_to_text(&deep("<?>", "<div>"), "<?>");
assert_degraded_to_text(&deep("<?php ", "<div>"), "<?php");
}
#[test]
fn a_quote_outside_an_attribute_value_does_not_swallow_the_document() {
assert_degraded_to_text(&deep("<b '>", "<div>"), "<b '>");
assert_degraded_to_text(&deep("<b ='>", "<div>"), "<b ='>");
}
#[test]
fn void_elements_nest_inside_foreign_content() {
assert_degraded_to_text(&deep("<svg>", "<wbr>"), "svg/wbr");
assert_degraded_to_text(&deep("<math>", "<input>"), "math/input");
}
#[test]
fn a_self_closing_tag_under_foreign_content_still_counts_as_nesting() {
assert_degraded_to_text(&deep("<svg>", "<i/>"), "svg/self-closing i");
assert_degraded_to_text(
&deep("<svg><foreignObject>", "<div/>"),
"svg/foreignObject/self-closing div",
);
}
#[test]
fn tag_names_are_compared_whole_not_truncated() {
assert_degraded_to_text(&deep("", "<div_></div>"), "div_");
}
#[test]
fn balanced_content_far_past_the_ceiling_still_converts() {
let md = html_to_markdown(&format!("{}<em>innermost</em>", "<p>x</p>".repeat(1000)));
assert!(
md.contains("*innermost*") || md.contains("_innermost_"),
"1000 balanced paragraphs are depth 1, not 1000: {md}"
);
}
#[test]
fn unclosed_paragraph_runs_are_siblings_not_nesting() {
let html = "<p>para".repeat(150);
assert_eq!(max_open_tag_depth(&html), 1);
let md = html_to_markdown(&html);
assert!(
md.contains("para\n\npara"),
"unclosed paragraphs must convert as paragraphs, not degrade to a \
single run of text: {md:?}"
);
}
#[test]
fn unclosed_sibling_runs_do_not_trip_the_ceiling() {
for (name, html) in [
("li", format!("<ul>{}</ul>", "<li>item".repeat(150))),
("dd/dt", format!("<dl>{}</dl>", "<dt>t<dd>d".repeat(150))),
(
"tr/td",
format!("<table>{}</table>", "<tr><td>a<td>b".repeat(150)),
),
(
"option",
format!("<select>{}</select>", "<option>o".repeat(150)),
),
("li over p", format!("<ul>{}</ul>", "<li><p>x".repeat(150))),
] {
let depth = max_open_tag_depth(&html);
assert!(
depth <= 10,
"{name}: unclosed sibling runs are flat, scanned {depth}"
);
}
}
#[test]
fn genuinely_nested_lists_still_count_and_still_degrade() {
let html = "<ul><li>".repeat(60);
assert!(
max_open_tag_depth(&html) > MAX_HTML_DEPTH,
"60 nested ul/li pairs are real nesting, not siblings"
);
}
#[test]
fn crlf_inside_a_start_tag_does_not_inflate_the_scanned_depth() {
let html = format!(
"{}<em>innermost</em>",
"<p\r\nclass=\"lede\">x</p>".repeat(101)
);
let md = html_to_markdown(&html);
assert!(
md.contains("*innermost*") || md.contains("_innermost_"),
"CRLF in a start tag degraded a depth-1 document: {md}"
);
let lf = html_to_markdown(&format!(
"{}<em>innermost</em>",
"<p\nclass=\"lede\">x</p>".repeat(101)
));
assert_eq!(md, lf, "CRLF and LF spellings converted differently");
}
#[test]
fn many_void_elements_in_a_row_do_not_trip_the_depth_ceiling() {
let html = format!("line{}", "<br>".repeat(200));
let md = html_to_markdown(&html);
assert!(md.contains("line"), "{md}");
assert!(
md.contains('\\'),
"expected htmd's real <br> (BrStyle::Backslash) conversion: {md:?}"
);
}
#[test]
fn no_tag_survives_but_a_bare_lt_can() {
let md = html_to_markdown(r#""><<>>"#);
assert_eq!(md, r#""><<>>"#);
}
#[test]
fn attribute_value_lt_survives_unescaped_in_the_output() {
let md = html_to_markdown(r##"<a href="#" title="<script>">t</a>"##);
assert_eq!(md, r##"[t](# "<script>")"##);
}
#[test]
fn noscript_fallback_text_is_preserved_now_that_head_is_not_skipped() {
let md = html_to_markdown("<noscript><p>ns fallback text</p></noscript>");
assert!(md.contains("ns fallback text"), "{md}");
}
#[test]
fn leading_title_text_is_preserved_not_silently_dropped() {
let md = html_to_markdown("<title>TITLETEXT</title><p>body</p>");
assert!(md.contains("TITLETEXT"), "{md}");
assert!(md.contains("body"), "{md}");
}
#[test]
fn template_content_is_lost_not_reduced_to_text() {
let md = html_to_markdown("<template><p>tpl</p></template>");
assert_eq!(md, "", "{md}");
}
}