use crate::converter::utility::content::{collect_link_label_text, escape_link_label, normalize_link_label};
use crate::converter::utility::preprocessing::sanitize_markdown_url;
use crate::options::ConversionOptions;
use std::borrow::Cow;
#[cfg(feature = "metadata")]
use std::collections::BTreeMap;
use tl::{NodeHandle, Parser};
type Context = crate::converter::Context;
type DomContext = crate::converter::DomContext;
pub fn handle(
node_handle: &NodeHandle,
parser: &Parser,
output: &mut String,
options: &ConversionOptions,
ctx: &Context,
depth: usize,
dom_ctx: &DomContext,
) {
use crate::converter::block::heading::{heading_allows_inline_images, push_heading};
use crate::converter::utility::content::normalized_tag_name;
#[allow(unused_imports)]
use crate::converter::utility::serialization::serialize_node;
use crate::converter::{find_single_heading_child, get_text_content, walk_node};
let Some(node) = node_handle.get(parser) else {
return;
};
let tl::Node::Tag(tag) = node else {
return;
};
let href_attr = tag.attributes().get("href").flatten().map(|v| {
let decoded = crate::text::decode_html_entities(&v.as_utf8_str());
sanitize_markdown_url(&decoded).into_owned()
});
let title = tag.attributes().get("title").flatten().map(|v| v.as_utf8_str());
if let Some(href) = href_attr {
if ctx.in_link {
let children = tag.children();
for child_handle in children.top().iter() {
walk_node(child_handle, parser, output, options, ctx, depth + 1, dom_ctx);
}
return;
}
let owned_children: Vec<tl::NodeHandle>;
let children: &[tl::NodeHandle] = if let Some(c) = dom_ctx.children_of(node_handle.get_inner()) {
c.as_slice()
} else {
owned_children = tag.children().top().iter().copied().collect();
owned_children.as_slice()
};
let (inline_label, _block_nodes, saw_block) = collect_link_label_text(children, parser, dom_ctx);
let text_source: Cow<'_, str> = if saw_block {
Cow::Owned(get_text_content(node_handle, parser, dom_ctx))
} else {
Cow::Borrowed(inline_label.as_str())
};
let normalized_text = crate::text::normalize_whitespace_cow(text_source.as_ref());
let raw_text = normalized_text.trim();
let is_autolink = options.autolinks
&& !options.default_title
&& !href.is_empty()
&& has_uri_scheme(href.as_str())
&& (raw_text == href || (href.starts_with("mailto:") && raw_text == &href[7..]));
if is_autolink {
output.push('<');
if href.starts_with("mailto:") && raw_text == &href[7..] {
output.push_str(raw_text);
} else {
output.push_str(&href);
}
output.push('>');
return;
}
if let Some((heading_level, heading_handle)) = find_single_heading_child(*node_handle, parser) {
if let Some(heading_node) = heading_handle.get(parser) {
if let tl::Node::Tag(heading_tag) = heading_node {
let heading_name = normalized_tag_name(heading_tag.name().as_utf8_str()).into_owned();
let mut heading_text = String::new();
let heading_ctx = Context {
in_heading: true,
convert_as_inline: true,
heading_allow_inline_images: heading_allows_inline_images(
&heading_name,
&ctx.keep_inline_images_in,
),
..ctx.clone()
};
walk_node(
&heading_handle,
parser,
&mut heading_text,
options,
&heading_ctx,
depth + 1,
dom_ctx,
);
let trimmed_heading = heading_text.trim();
if !trimmed_heading.is_empty() {
let escaped_label = escape_link_label(trimmed_heading);
let mut link_buffer = String::new();
append_markdown_link(
&mut link_buffer,
&escaped_label,
href.as_str(),
title.as_deref(),
raw_text,
options,
ctx.reference_collector.as_ref(),
);
push_heading(output, ctx, options, heading_level, link_buffer.as_str());
return;
}
}
}
}
let mut label = if saw_block {
let mut content = String::new();
let link_ctx = Context {
inline_depth: ctx.inline_depth + 1,
convert_as_inline: true,
in_link: true,
..ctx.clone()
};
for child_handle in children {
let mut child_buf = String::new();
walk_node(
child_handle,
parser,
&mut child_buf,
options,
&link_ctx,
depth + 1,
dom_ctx,
);
if !child_buf.trim().is_empty()
&& !content.is_empty()
&& !content.chars().last().is_none_or(char::is_whitespace)
&& !child_buf.chars().next().is_none_or(char::is_whitespace)
{
content.push(' ');
}
content.push_str(&child_buf);
}
if content.trim().is_empty() {
normalize_link_label(&inline_label)
} else {
normalize_link_label(&content)
}
} else {
let mut content = String::new();
let link_ctx = Context {
inline_depth: ctx.inline_depth + 1,
in_link: true,
..ctx.clone()
};
for child_handle in children {
walk_node(
child_handle,
parser,
&mut content,
options,
&link_ctx,
depth + 1,
dom_ctx,
);
}
normalize_link_label(&content)
};
if label.is_empty() && !raw_text.is_empty() {
label = normalize_link_label(raw_text);
}
if label.is_empty() && !href.is_empty() && !children.is_empty() {
label.clone_from(&href);
}
let escaped_label = escape_link_label(&label);
#[cfg(feature = "visitor")]
if let Some(ref visitor_handle) = ctx.visitor {
use crate::visitor::{NodeContext, NodeType, VisitResult};
let node_id = node_handle.get_inner();
let parent_tag = dom_ctx.parent_tag_name(node_id, parser);
let index_in_parent = dom_ctx.get_sibling_index(node_id).unwrap_or(0);
let node_ctx = NodeContext::with_lazy_attributes(
NodeType::Link,
Cow::Borrowed("a"),
tag,
depth,
index_in_parent,
parent_tag.map(Cow::Borrowed),
true,
);
let visit_result = {
let mut visitor = visitor_handle.lock().expect("visitor mutex poisoned");
visitor.visit_link(&node_ctx, &href, &label, title.as_deref())
};
match visit_result {
VisitResult::Continue => append_markdown_link(
output,
&escaped_label,
href.as_str(),
title.as_deref(),
label.as_str(),
options,
ctx.reference_collector.as_ref(),
),
VisitResult::Custom(custom) => output.push_str(&custom),
VisitResult::Skip => {}
VisitResult::Error(err) => {
if ctx.visitor_error.borrow().is_none() {
*ctx.visitor_error.borrow_mut() = Some(err);
}
}
VisitResult::PreserveHtml => output.push_str(&serialize_node(node_handle, parser)),
}
} else {
append_markdown_link(
output,
&escaped_label,
href.as_str(),
title.as_deref(),
label.as_str(),
options,
ctx.reference_collector.as_ref(),
);
}
#[cfg(not(feature = "visitor"))]
append_markdown_link(
output,
&escaped_label,
href.as_str(),
title.as_deref(),
label.as_str(),
options,
ctx.reference_collector.as_ref(),
);
#[cfg(feature = "metadata")]
if ctx.metadata_wants_links {
if let Some(ref collector) = ctx.metadata_collector {
let rel_attr = tag
.attributes()
.get("rel")
.flatten()
.map(|v| v.as_utf8_str().to_string());
let mut attributes_map = BTreeMap::new();
for (key, value_opt) in tag.attributes().iter() {
let key_str = key.to_string();
if key_str == "href" {
continue;
}
let value = value_opt.map(|v| v.to_string()).unwrap_or_default();
attributes_map.insert(key_str, value);
}
collector.borrow_mut().add_link(
href.clone(),
label,
title.as_deref().map(str::to_string),
rel_attr,
attributes_map,
);
}
}
} else {
let children = tag.children();
for child_handle in children.top().iter() {
walk_node(child_handle, parser, output, options, ctx, depth + 1, dom_ctx);
}
}
}
#[must_use]
pub fn has_uri_scheme(href: &str) -> bool {
let mut bytes = href.bytes();
match bytes.next() {
Some(b) if b.is_ascii_alphabetic() => {}
_ => return false,
}
for b in bytes {
match b {
b':' => return true,
b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'+' | b'-' | b'.' => {}
_ => return false,
}
}
false
}
#[must_use]
pub fn percent_encode_url(url: &str) -> String {
let mut encoded = String::with_capacity(url.len() * 2);
for byte in url.bytes() {
match byte {
b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' | b'/' => {
encoded.push(byte as char);
}
other => {
encoded.push('%');
let hi = char::from_digit(u32::from(other >> 4), 16)
.unwrap_or('0')
.to_ascii_uppercase();
let lo = char::from_digit(u32::from(other & 0x0f), 16)
.unwrap_or('0')
.to_ascii_uppercase();
encoded.push(hi);
encoded.push(lo);
}
}
}
encoded
}
#[must_use]
fn parens_are_balanced(href: &str) -> bool {
let mut depth: i32 = 0;
for c in href.chars() {
match c {
'(' => depth += 1,
')' => {
depth -= 1;
if depth < 0 {
return false;
}
}
_ => {}
}
}
depth == 0
}
#[must_use]
pub fn escape_markdown_title(text: &str) -> std::borrow::Cow<'_, str> {
if !text.contains('\\') && !text.contains('"') {
return std::borrow::Cow::Borrowed(text);
}
std::borrow::Cow::Owned(text.replace('\\', "\\\\").replace('"', "\\\""))
}
pub fn append_url_destination(
output: &mut String,
dest: &str,
url_escape_style: crate::options::validation::UrlEscapeStyle,
) {
if dest.is_empty() {
output.push_str("<>");
} else if url_escape_style == crate::options::validation::UrlEscapeStyle::Percent {
let encoded = percent_encode_url(dest);
output.push_str(&encoded);
} else if dest.contains(' ') || dest.contains('\n') {
output.push('<');
for c in dest.chars() {
match c {
'\\' => output.push_str("\\\\"),
'<' => output.push_str("\\<"),
'>' => output.push_str("\\>"),
other => output.push(other),
}
}
output.push('>');
} else if parens_are_balanced(dest) {
output.push_str(dest);
} else {
let escaped_dest = dest.replace('(', "\\(").replace(')', "\\)");
output.push_str(&escaped_dest);
}
}
pub fn append_markdown_link(
output: &mut String,
label: &str,
href: &str,
title: Option<&str>,
raw_text: &str,
options: &ConversionOptions,
reference_collector: Option<&crate::converter::reference_collector::ReferenceCollectorHandle>,
) {
if options.link_style == crate::options::validation::LinkStyle::Reference && !href.is_empty() {
if let Some(collector) = reference_collector {
let ref_num = collector.borrow_mut().get_or_insert(href, title);
output.push('[');
output.push_str(label);
output.push_str("][");
output.push_str(&ref_num.to_string());
output.push(']');
return;
}
}
output.push('[');
output.push_str(label);
output.push_str("](");
append_url_destination(output, href, options.url_escape_style);
if let Some(title_text) = title {
output.push_str(" \"");
output.push_str(&escape_markdown_title(title_text));
output.push('"');
} else if options.default_title && raw_text == href {
output.push_str(" \"");
output.push_str(&escape_markdown_title(href));
output.push('"');
}
output.push(')');
}
#[cfg(test)]
mod tests {
use super::*;
use crate::options::validation::UrlEscapeStyle;
fn opts_with_style(style: UrlEscapeStyle) -> ConversionOptions {
ConversionOptions::builder().url_escape_style(style).build()
}
#[test]
fn has_uri_scheme_accepts_http() {
assert!(has_uri_scheme("http://example.com"));
assert!(has_uri_scheme("https://example.com/path"));
}
#[test]
fn has_uri_scheme_accepts_mailto() {
assert!(has_uri_scheme("mailto:a@b.com"));
}
#[test]
fn has_uri_scheme_accepts_uncommon_schemes() {
assert!(has_uri_scheme("ftp://host"));
assert!(has_uri_scheme("ssh://host"));
assert!(has_uri_scheme("data:text/plain,foo"));
assert!(has_uri_scheme("file:///etc/hosts"));
}
#[test]
fn has_uri_scheme_rejects_bare_paths() {
assert!(!has_uri_scheme("foobar.png"));
assert!(!has_uri_scheme("/relative/path"));
assert!(!has_uri_scheme("../up.html"));
assert!(!has_uri_scheme("#fragment"));
}
#[test]
fn has_uri_scheme_rejects_leading_digit_or_punct() {
assert!(!has_uri_scheme("9scheme:foo"));
assert!(!has_uri_scheme(":no-scheme"));
assert!(!has_uri_scheme(""));
}
#[test]
fn issue_397_filename_with_extension_is_not_autolinked() {
assert!(!has_uri_scheme("foobar.png"));
}
#[test]
fn percent_encode_url_leaves_unreserved_chars_unchanged() {
let result = percent_encode_url("/path-to_file.html~");
assert_eq!(result, "/path-to_file.html~");
}
#[test]
fn percent_encode_url_encodes_spaces() {
assert_eq!(percent_encode_url("/file (1).pdf"), "/file%20%281%29.pdf");
}
#[test]
fn percent_encode_url_encodes_angle_brackets() {
assert_eq!(percent_encode_url("/file <draft>.pdf"), "/file%20%3Cdraft%3E.pdf");
}
#[test]
fn percent_encode_url_full_issue_example() {
assert_eq!(
percent_encode_url("/file (1) <draft>.pdf"),
"/file%20%281%29%20%3Cdraft%3E.pdf"
);
}
#[test]
fn append_markdown_link_angle_plain_url_unchanged() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Angle);
append_markdown_link(&mut out, "text", "/file.pdf", None, "text", &options, None);
assert_eq!(out, "[text](/file.pdf)");
}
#[test]
fn append_markdown_link_angle_wraps_space_in_angle_brackets() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Angle);
append_markdown_link(&mut out, "file", "/file (1).pdf", None, "file", &options, None);
assert_eq!(out, "[file](</file (1).pdf>)");
}
#[test]
fn append_markdown_link_percent_encodes_spaces() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Percent);
append_markdown_link(&mut out, "file", "/file (1).pdf", None, "file", &options, None);
assert_eq!(out, "[file](/file%20%281%29.pdf)");
}
#[test]
fn append_markdown_link_percent_encodes_angle_brackets() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Percent);
append_markdown_link(&mut out, "file", "/file <draft>.pdf", None, "file", &options, None);
assert_eq!(out, "[file](/file%20%3Cdraft%3E.pdf)");
}
#[test]
fn append_markdown_link_percent_full_issue_example() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Percent);
append_markdown_link(&mut out, "file", "/file (1) <draft>.pdf", None, "file", &options, None);
assert_eq!(out, "[file](/file%20%281%29%20%3Cdraft%3E.pdf)");
}
#[test]
fn parens_are_balanced_accepts_nested_parens() {
assert!(parens_are_balanced("wiki/Rust_(programming_language)"));
assert!(parens_are_balanced("no/parens/here"));
}
#[test]
fn parens_are_balanced_rejects_equal_counts_out_of_order() {
assert!(!parens_are_balanced("a)b(c"));
}
#[test]
fn parens_are_balanced_rejects_unmatched_open_or_close() {
assert!(!parens_are_balanced("a(b"));
assert!(!parens_are_balanced("a)b"));
}
#[test]
fn append_markdown_link_angle_leaves_balanced_parens_unescaped_when_href_has_parens() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Angle);
append_markdown_link(
&mut out,
"Rust",
"https://en.wikipedia.org/wiki/Rust_(programming_language)",
None,
"Rust",
&options,
None,
);
assert_eq!(out, "[Rust](https://en.wikipedia.org/wiki/Rust_(programming_language))");
}
#[test]
fn append_markdown_link_angle_escapes_out_of_order_parens_when_href_has_parens() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Angle);
append_markdown_link(
&mut out,
"link",
"http://example.com/a)(b",
None,
"link",
&options,
None,
);
assert_eq!(out, "[link](http://example.com/a\\)\\(b)");
}
#[test]
fn append_markdown_link_angle_escapes_gt_inside_wrap_when_href_has_space_and_gt() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Angle);
append_markdown_link(&mut out, "text", "/my file >.pdf", None, "text", &options, None);
assert_eq!(out, "[text](</my file \\>.pdf>)");
}
#[test]
fn append_markdown_link_angle_produces_empty_angle_brackets_when_href_is_empty() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Angle);
append_markdown_link(&mut out, "text", "", None, "text", &options, None);
assert_eq!(out, "[text](<>)");
}
#[test]
fn append_markdown_link_percent_preserves_title() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Percent);
append_markdown_link(
&mut out,
"link",
"/path with spaces",
Some("My Title"),
"link",
&options,
None,
);
assert_eq!(out, "[link](/path%20with%20spaces \"My Title\")");
}
#[test]
fn escape_markdown_title_escapes_backslash_before_quote_so_the_closing_quote_is_not_swallowed() {
assert_eq!(escape_markdown_title("foo\\"), "foo\\\\");
assert_eq!(escape_markdown_title("say \"hi\"\\"), "say \\\"hi\\\"\\\\");
}
#[test]
fn append_markdown_link_escapes_a_trailing_backslash_in_title_so_the_closing_quote_is_not_swallowed() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Angle);
append_markdown_link(&mut out, "text", "/url", Some("foo\\"), "text", &options, None);
assert_eq!(out, "[text](/url \"foo\\\\\")");
}
#[test]
fn append_markdown_link_escapes_a_trailing_backslash_in_default_title_so_the_closing_quote_is_not_swallowed() {
let mut out = String::new();
let mut options = opts_with_style(UrlEscapeStyle::Angle);
options.default_title = true;
append_markdown_link(&mut out, "text", "http://a\\", None, "http://a\\", &options, None);
assert_eq!(out, "[text](http://a\\ \"http://a\\\\\")");
}
#[test]
fn append_markdown_link_escapes_a_backslash_inside_the_angle_bracket_wrap_so_it_cannot_unescape_a_delimiter() {
let mut out = String::new();
let options = opts_with_style(UrlEscapeStyle::Angle);
append_markdown_link(&mut out, "text", "/my file\\>.pdf", None, "text", &options, None);
assert_eq!(out, "[text](</my file\\\\\\>.pdf>)");
}
}