use pulldown_cmark::{html, Event, Options, Parser, Tag, TagEnd, CodeBlockKind, HeadingLevel};
use std::path::Path;
#[derive(Debug, Clone)]
pub struct TocItem {
pub level: u8,
pub text: String,
pub id: String,
}
pub fn extract_headings(content: &str) -> Vec<TocItem> {
let content = fix_fullwidth_heading_spaces(content);
let mut options = Options::empty();
options.insert(Options::ENABLE_TABLES);
options.insert(Options::ENABLE_FOOTNOTES);
options.insert(Options::ENABLE_STRIKETHROUGH);
options.insert(Options::ENABLE_TASKLISTS);
options.insert(Options::ENABLE_HEADING_ATTRIBUTES);
let parser = Parser::new_ext(&content, options);
let mut headings = Vec::new();
let mut in_heading: Option<HeadingLevel> = None;
let mut heading_text = String::new();
for event in parser {
match &event {
Event::Start(Tag::Heading { level, .. }) => {
in_heading = Some(*level);
heading_text.clear();
}
Event::Text(text) if in_heading.is_some() => {
heading_text.push_str(text);
}
Event::End(TagEnd::Heading(level)) if in_heading.is_some() => {
let level_num = heading_level_to_num(*level);
if level_num >= 2 && level_num <= 4 {
let id = slugify(&heading_text);
headings.push(TocItem {
level: level_num,
text: heading_text.clone(),
id,
});
}
in_heading = None;
}
_ => {}
}
}
headings
}
pub fn render_markdown_with_path(content: &str, current_path: Option<&str>, hardbreaks: bool) -> String {
let html = render_markdown_internal(content, hardbreaks);
if let Some(path) = current_path {
convert_relative_links_to_absolute(&html, path)
} else {
html
}
}
pub fn render_markdown(content: &str) -> String {
render_markdown_internal(content, false)
}
pub fn render_markdown_with_hardbreaks(content: &str, hardbreaks: bool) -> String {
render_markdown_internal(content, hardbreaks)
}
fn render_markdown_internal(content: &str, hardbreaks: bool) -> String {
let content = fix_fullwidth_heading_spaces(content);
let content = fix_image_paths_with_spaces(&content);
let content = fix_multiline_footnotes(&content);
let content = fix_table_separator_columns(&content);
let content = convert_footnote_definitions_inline(&content, hardbreaks);
let content = convert_footnote_references_to_placeholder(&content);
let mut options = Options::empty();
options.insert(Options::ENABLE_TABLES);
options.insert(Options::ENABLE_STRIKETHROUGH);
options.insert(Options::ENABLE_TASKLISTS);
options.insert(Options::ENABLE_HEADING_ATTRIBUTES);
let parser = Parser::new_ext(&content, options);
let mut in_mermaid = false;
let mut mermaid_content = String::new();
let mut in_heading: Option<HeadingLevel> = None;
let mut heading_text = String::new();
let mut events: Vec<Event> = Vec::new();
for event in parser {
match &event {
Event::Start(Tag::CodeBlock(CodeBlockKind::Fenced(lang))) => {
let lang_str = lang.as_ref();
if lang_str == "mermaid" || lang_str.starts_with("mermaid") {
in_mermaid = true;
mermaid_content.clear();
continue;
}
}
Event::End(TagEnd::CodeBlock) if in_mermaid => {
let mermaid_html = format!(
r#"<div class="mermaid">{}</div>"#,
html_escape(&mermaid_content)
);
events.push(Event::Html(mermaid_html.into()));
in_mermaid = false;
continue;
}
Event::Text(text) if in_mermaid => {
mermaid_content.push_str(text);
continue;
}
Event::Start(Tag::Heading { level, .. }) => {
in_heading = Some(*level);
heading_text.clear();
events.push(event);
continue;
}
Event::Text(text) if in_heading.is_some() => {
heading_text.push_str(text);
events.push(event);
continue;
}
Event::End(TagEnd::Heading(level)) if in_heading.is_some() => {
let id = slugify(&heading_text);
let level_num = heading_level_to_num(*level);
let mut heading_events = Vec::new();
while let Some(ev) = events.pop() {
if matches!(ev, Event::Start(Tag::Heading { .. })) {
break;
}
heading_events.push(ev);
}
heading_events.reverse();
let open_tag = format!(r#"<h{} id="{}">"#, level_num, id);
events.push(Event::Html(open_tag.into()));
events.extend(heading_events);
events.push(Event::Html(format!("</h{}>", level_num).into()));
in_heading = None;
continue;
}
Event::SoftBreak if hardbreaks => {
events.push(Event::HardBreak);
continue;
}
_ => {}
}
events.push(event);
}
let mut html_output = String::new();
html::push_html(&mut html_output, events.into_iter());
html_output = fix_relative_links(&html_output);
html_output = autolink_urls(&html_output);
html_output = convert_remaining_markdown_images(&html_output);
html_output = convert_footnote_placeholders_to_html(&html_output);
html_output
}
fn heading_level_to_num(level: HeadingLevel) -> u8 {
match level {
HeadingLevel::H1 => 1,
HeadingLevel::H2 => 2,
HeadingLevel::H3 => 3,
HeadingLevel::H4 => 4,
HeadingLevel::H5 => 5,
HeadingLevel::H6 => 6,
}
}
fn slugify(text: &str) -> String {
text.to_lowercase()
.chars()
.filter_map(|c| {
if c.is_alphanumeric() || c == '-' || c == '_' {
Some(c)
} else if c.is_whitespace() {
Some('-')
} else if c > '\x7F' {
Some(c)
} else {
None
}
})
.collect::<String>()
.split('-')
.filter(|s| !s.is_empty())
.collect::<Vec<_>>()
.join("-")
}
fn html_escape(s: &str) -> String {
s.replace('&', "&")
.replace('<', "<")
.replace('>', ">")
.replace('"', """)
}
fn convert_footnote_definitions_inline(content: &str, hardbreaks: bool) -> String {
let mut result_lines = Vec::new();
let lines: Vec<&str> = content.lines().collect();
let mut i = 0;
while i < lines.len() {
let line = lines[i];
if let Some(captures) = parse_footnote_def_start(line) {
let (number, first_line_content) = captures;
let first_line_content = first_line_content.trim_end();
let mut continuation_lines: Vec<String> = Vec::new();
i += 1;
while i < lines.len() {
let next_line = lines[i];
let trimmed = next_line.trim_start();
if trimmed.is_empty() {
break;
}
if trimmed.starts_with("[^") && trimmed.contains("]:") {
break;
}
if trimmed.starts_with('#') {
break;
}
continuation_lines.push(next_line.to_string());
i += 1;
}
let return_link = format!(
"<a href=\"#reffn_{}\" title=\"Jump back to footnote [{}] in the text.\"> ↩</a>",
number, number
);
if continuation_lines.is_empty() {
result_lines.push(format!(
"<blockquote id=\"fn_{}\"><sup>{}</sup>. {}{}</blockquote>",
number, number, first_line_content, return_link
));
} else {
let continuation_content = continuation_lines.join("\n");
let continuation_html = render_footnote_continuation(&continuation_content, hardbreaks);
result_lines.push(format!(
"<blockquote id=\"fn_{}\"><sup>{}</sup>. {}{}\n{}</blockquote>",
number, number, first_line_content, return_link, continuation_html
));
}
} else {
result_lines.push(line.to_string());
i += 1;
}
}
result_lines.join("\n")
}
fn convert_footnote_references_to_placeholder(content: &str) -> String {
let mut result = String::new();
let mut chars = content.char_indices().peekable();
while let Some((i, c)) = chars.next() {
if c == '[' && content[i..].starts_with("[^") {
let rest = &content[i + 2..];
if let Some(end) = rest.find(']') {
let number = &rest[..end];
let after = &rest[end + 1..];
if !after.starts_with(':') && !number.is_empty() && number.chars().all(|c| c.is_alphanumeric()) {
result.push_str(&format!("%%FNREF_{}%%", number));
for _ in 0..(1 + end + 1) {
chars.next();
}
continue;
}
}
}
result.push(c);
}
result
}
fn convert_footnote_placeholders_to_html(html: &str) -> String {
let mut result = html.to_string();
let re_pattern = "%%FNREF_";
while let Some(start) = result.find(re_pattern) {
let after_prefix = &result[start + re_pattern.len()..];
if let Some(end) = after_prefix.find("%%") {
let number = &after_prefix[..end];
let replacement = format!(
"<sup><a href=\"#fn_{}\" id=\"reffn_{}\">{}</a></sup>",
number, number, number
);
let full_placeholder = format!("%%FNREF_{}%%", number);
result = result.replacen(&full_placeholder, &replacement, 1);
} else {
break;
}
}
result
}
fn parse_footnote_def_start(line: &str) -> Option<(&str, &str)> {
let trimmed = line.trim_start();
if !trimmed.starts_with("[^") {
return None;
}
let after_bracket = &trimmed[2..];
let end_bracket = after_bracket.find("]:")?;
let number = &after_bracket[..end_bracket];
let rest = &after_bracket[end_bracket + 2..].trim_start();
Some((number, rest))
}
fn render_footnote_continuation(content: &str, hardbreaks: bool) -> String {
let min_indent = content
.lines()
.filter(|line| !line.trim().is_empty())
.map(|line| line.len() - line.trim_start().len())
.min()
.unwrap_or(0);
let dedented: String = content
.lines()
.map(|line| {
if line.len() >= min_indent {
&line[min_indent..]
} else {
line.trim_start()
}
})
.collect::<Vec<_>>()
.join("\n");
let mut options = Options::empty();
options.insert(Options::ENABLE_TABLES);
options.insert(Options::ENABLE_STRIKETHROUGH);
let parser = Parser::new_ext(&dedented, options);
let events: Vec<Event> = parser.map(|event| {
if hardbreaks {
match event {
Event::SoftBreak => Event::HardBreak,
_ => event,
}
} else {
event
}
}).collect();
let mut html = String::new();
html::push_html(&mut html, events.into_iter());
html.trim().to_string()
}
fn fix_multiline_footnotes(content: &str) -> String {
let lines: Vec<&str> = content.lines().collect();
let mut result = Vec::new();
let mut in_footnote = false;
for line in lines {
if line.starts_with("[^") && line.contains("]:") {
in_footnote = true;
result.push(line.to_string());
} else if in_footnote {
let trimmed = line.trim_start();
if trimmed.is_empty() {
in_footnote = false;
result.push(line.to_string());
} else if trimmed.starts_with("[^") && trimmed.contains("]:") {
in_footnote = true;
result.push(line.to_string());
} else if trimmed.starts_with('#') {
in_footnote = false;
result.push(line.to_string());
} else {
result.push(format!(" {}", line));
}
} else {
result.push(line.to_string());
}
}
result.join("\n")
}
fn fix_table_separator_columns(content: &str) -> String {
let lines: Vec<&str> = content.lines().collect();
let mut result = Vec::new();
let mut i = 0;
while i < lines.len() {
let line = lines[i];
let trimmed = line.trim();
if trimmed.starts_with('|') {
if i + 1 < lines.len() {
let next_line = lines[i + 1];
if is_table_separator_row(next_line) {
let fixed_header = fix_table_row_trailing_pipe(line);
let header_cols = count_table_columns(&fixed_header);
let separator_cols = count_table_columns(next_line);
result.push(fixed_header);
i += 1;
if header_cols > 0 && separator_cols != header_cols {
let fixed_separator = generate_separator_row(header_cols, next_line);
result.push(fixed_separator);
} else {
result.push(next_line.to_string());
}
i += 1;
continue;
}
}
}
result.push(line.to_string());
i += 1;
}
result.join("\n")
}
fn fix_table_row_trailing_pipe(line: &str) -> String {
let trimmed = line.trim();
if trimmed.starts_with('|') && !trimmed.ends_with('|') {
format!("{}|", line)
} else {
line.to_string()
}
}
fn count_table_columns(line: &str) -> usize {
let trimmed = line.trim();
if !trimmed.starts_with('|') {
return 0;
}
let pipe_count = trimmed.chars().filter(|&c| c == '|').count();
if pipe_count > 1 {
pipe_count - 1
} else {
0
}
}
fn is_table_separator_row(line: &str) -> bool {
let trimmed = line.trim();
if !trimmed.starts_with('|') || !trimmed.ends_with('|') {
return false;
}
if !trimmed.contains('-') {
return false;
}
trimmed.chars().all(|c| c == '|' || c == '-' || c == ':' || c.is_whitespace())
}
fn generate_separator_row(col_count: usize, original: &str) -> String {
let alignment = if original.contains(":--:") || original.contains(":-:") {
":--:"
} else if original.contains(":--") || original.contains(":-") {
":--"
} else if original.contains("--:") || original.contains("-:") {
"--:"
} else {
"--"
};
let cols: Vec<&str> = std::iter::repeat(alignment).take(col_count).collect();
format!("|{}|", cols.join("|"))
}
fn fix_fullwidth_heading_spaces(content: &str) -> String {
content
.lines()
.map(|line| {
let trimmed = line.trim_start();
if trimmed.starts_with('#') {
let hash_count = trimmed.chars().take_while(|&c| c == '#').count();
if hash_count > 0 && hash_count <= 6 {
let after_hashes = &trimmed[hash_count..];
if after_hashes.starts_with('\u{3000}') {
let leading_whitespace = &line[..line.len() - trimmed.len()];
let rest = &after_hashes['\u{3000}'.len_utf8()..];
return format!("{}{} {}", leading_whitespace, "#".repeat(hash_count), rest);
}
}
}
line.to_string()
})
.collect::<Vec<_>>()
.join("\n")
}
fn fix_image_paths_with_spaces(content: &str) -> String {
let mut result = String::new();
let mut chars = content.chars().peekable();
while let Some(c) = chars.next() {
if c == '!' {
if chars.peek() == Some(&'[') {
let mut img_str = String::from("!");
img_str.push(chars.next().unwrap());
let mut bracket_depth = 1;
while let Some(&ch) = chars.peek() {
img_str.push(chars.next().unwrap());
if ch == '[' {
bracket_depth += 1;
} else if ch == ']' {
bracket_depth -= 1;
if bracket_depth == 0 {
break;
}
}
}
if chars.peek() == Some(&'(') {
img_str.push(chars.next().unwrap());
let mut url = String::new();
let mut paren_depth = 1;
while let Some(&ch) = chars.peek() {
if ch == '(' {
paren_depth += 1;
url.push(chars.next().unwrap());
} else if ch == ')' {
paren_depth -= 1;
if paren_depth == 0 {
chars.next(); break;
}
url.push(chars.next().unwrap());
} else {
url.push(chars.next().unwrap());
}
}
if url.contains(' ') && !url.starts_with('<') {
img_str.push('<');
img_str.push_str(&url);
img_str.push('>');
} else {
img_str.push_str(&url);
}
img_str.push(')');
}
result.push_str(&img_str);
} else {
result.push(c);
}
} else {
result.push(c);
}
}
result
}
fn fix_relative_links(html: &str) -> String {
let mut result = html.to_string();
let patterns = [
(r#".md""#, r#".html""#),
(r#".md#"#, r#".html#"#),
(r#".md'"#, r#".html'"#),
];
for (from, to) in patterns {
result = result.replace(from, to);
}
result
}
fn autolink_urls(html: &str) -> String {
let mut result = String::new();
let mut chars = html.char_indices().peekable();
let mut in_code = false;
while let Some((i, c)) = chars.next() {
if c == '<' {
result.push(c);
let mut tag_content = String::new();
while let Some((_, ch)) = chars.next() {
result.push(ch);
if ch == '>' {
break;
}
tag_content.push(ch);
}
let tag_lower = tag_content.to_lowercase();
if tag_lower.starts_with("code") || tag_lower.starts_with("pre") {
in_code = true;
} else if tag_lower.starts_with("/code") || tag_lower.starts_with("/pre") {
in_code = false;
}
continue;
}
if in_code {
result.push(c);
continue;
}
if c == 'h' && html[i..].starts_with("http://") || html[i..].starts_with("https://") {
if result.ends_with("href=\"") || result.ends_with("src=\"") {
result.push(c);
continue;
}
let url_start = i;
let mut url_end = i + 1;
while let Some(&(next_i, next_c)) = chars.peek() {
if next_c.is_whitespace() || next_c == '<' || next_c == '>'
|| next_c == '"' || next_c == '\'' {
break;
}
url_end = next_i + next_c.len_utf8();
chars.next();
}
let mut url = &html[url_start..url_end];
while url.ends_with('.') || url.ends_with(',') || url.ends_with(';')
|| url.ends_with(':') || url.ends_with(')') || url.ends_with('!') || url.ends_with('?') {
url = &url[..url.len() - 1];
}
result.push_str(&format!(
r#"<a href="{}" target="_blank">{}</a>"#,
url, url
));
let trimmed_len = url_end - url_start - url.len();
if trimmed_len > 0 {
result.push_str(&html[url_start + url.len()..url_end]);
}
} else {
result.push(c);
}
}
result
}
fn convert_remaining_markdown_images(html: &str) -> String {
let mut result = String::new();
let mut chars = html.char_indices().peekable();
let mut in_code = false;
while let Some((_, c)) = chars.next() {
if c == '<' {
result.push(c);
let mut tag_content = String::new();
while let Some((_, ch)) = chars.next() {
result.push(ch);
if ch == '>' {
break;
}
tag_content.push(ch);
}
let tag_lower = tag_content.to_lowercase();
if tag_lower.starts_with("code") || tag_lower.starts_with("pre") {
in_code = true;
} else if tag_lower.starts_with("/code") || tag_lower.starts_with("/pre") {
in_code = false;
}
continue;
}
if in_code {
result.push(c);
continue;
}
if c == '!' && chars.peek().map(|(_, ch)| *ch) == Some('[') {
chars.next();
let mut alt = String::new();
let mut bracket_depth = 1;
while let Some((_, ch)) = chars.next() {
if ch == '[' {
bracket_depth += 1;
alt.push(ch);
} else if ch == ']' {
bracket_depth -= 1;
if bracket_depth == 0 {
break;
}
alt.push(ch);
} else {
alt.push(ch);
}
}
if chars.peek().map(|(_, ch)| *ch) == Some('(') {
chars.next();
let mut url = String::new();
let mut paren_depth = 1;
while let Some((_, ch)) = chars.next() {
if ch == '(' {
paren_depth += 1;
url.push(ch);
} else if ch == ')' {
paren_depth -= 1;
if paren_depth == 0 {
break;
}
url.push(ch);
} else {
url.push(ch);
}
}
result.push_str(&format!(r#"<img src="{}" alt="{}">"#, url, alt));
} else {
result.push('!');
result.push('[');
result.push_str(&alt);
result.push(']');
}
} else {
result.push(c);
}
}
result
}
fn convert_relative_links_to_absolute(html: &str, current_path: &str) -> String {
let result = html.to_string();
let depth = Path::new(current_path)
.parent()
.map(|p| {
let dir = p.to_string_lossy();
if dir.is_empty() {
0
} else {
dir.matches('/').count() + 1
}
})
.unwrap_or(0);
let root_prefix: String = "../".repeat(depth);
let mut new_result = String::new();
let mut last_end = 0;
let href_pattern = r#"href=""#;
let mut search_start = 0;
while let Some(href_pos) = result[search_start..].find(href_pattern) {
let abs_href_pos = search_start + href_pos;
let url_start = abs_href_pos + href_pattern.len();
if let Some(url_end_offset) = result[url_start..].find('"') {
let url_end = url_start + url_end_offset;
let url = &result[url_start..url_end];
let needs_conversion = !url.is_empty()
&& !url.starts_with("http://")
&& !url.starts_with("https://")
&& !url.starts_with('#')
&& !url.starts_with("../")
&& !url.starts_with("./")
&& !url.starts_with('/')
&& !url.starts_with("mailto:")
&& !url.starts_with("javascript:")
&& depth > 0;
if needs_conversion {
new_result.push_str(&result[last_end..url_start]);
new_result.push_str(&root_prefix);
new_result.push_str(url);
last_end = url_end;
}
search_start = url_end + 1;
} else {
search_start = url_start + 1;
}
}
new_result.push_str(&result[last_end..]);
new_result
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_render_basic_markdown() {
let md = "# Hello\n\nThis is a **test**.";
let html = render_markdown(md);
assert!(html.contains("<h1 id=\"hello\">Hello</h1>"), "HTML: {}", html);
assert!(html.contains("<strong>test</strong>"));
}
#[test]
fn test_render_table() {
let md = r#"
| Header 1 | Header 2 |
|----------|----------|
| Cell 1 | Cell 2 |
"#;
let html = render_markdown(md);
assert!(html.contains("<table>"));
assert!(html.contains("<th>Header 1</th>"));
}
#[test]
fn test_render_mermaid() {
let md = r#"
```mermaid
sequenceDiagram
A->>B: Hello
```
"#;
let html = render_markdown(md);
assert!(html.contains(r#"<div class="mermaid">"#));
assert!(html.contains("sequenceDiagram"));
}
#[test]
fn test_fix_relative_links() {
let html = r#"<a href="chapter1.md">Link</a>"#;
let fixed = fix_relative_links(html);
assert!(fixed.contains(r#"href="chapter1.html""#));
}
#[test]
fn test_image_in_table() {
let md = r#"
| Col1 | Col2 |
|:--:|:--:|
||text|
"#;
let html = render_markdown(md);
println!("Generated HTML: {}", html);
assert!(html.contains("<img"), "Image tag should be generated: {}", html);
}
#[test]
fn test_image_in_table_japanese() {
let md = r#"## デザイン
|該当するタイムラインがある場合|該当するタイムラインがない場合|
|:--:|:--:|
|||
## 項目一覧"#;
let html = render_markdown(md);
println!("Generated HTML: {}", html);
assert!(html.contains("<img"), "Image tag should be generated: {}", html);
}
#[test]
fn test_image_with_space_in_filename() {
let md = r#"||"#;
let html = render_markdown(md);
println!("With space: {}", html);
let md2 = r#"||"#;
let html2 = render_markdown(md2);
println!("No space: {}", html2);
}
#[test]
fn test_autolink_urls() {
let md = "Guide Git:https://github.com/guide-inc-org/kcmsr-member-site-spec";
let html = render_markdown(md);
println!("Autolink result: {}", html);
assert!(html.contains(r#"<a href="https://github.com/guide-inc-org/kcmsr-member-site-spec" target="_blank">"#),
"URL should be auto-linked: {}", html);
}
#[test]
fn test_autolink_does_not_double_link() {
let md = "[Link](https://example.com)";
let html = render_markdown(md);
println!("Already linked result: {}", html);
let count = html.matches("https://example.com").count();
assert_eq!(count, 1, "URL should appear only once: {}", html);
}
#[test]
fn test_multiline_footnotes() {
let md = r#"Text with footnote[^1].
[^1]: First line
- Second line
- Third line
[^2]: Another footnote"#;
let html = render_markdown(md);
println!("Footnote HTML: {}", html);
assert!(html.contains("<li>"), "Footnote should contain list items: {}", html);
}
#[test]
fn test_fix_multiline_footnotes_preprocessing() {
let input = r#"[^1]: First line
- Second line
- Third line
[^2]: Another"#;
let output = fix_multiline_footnotes(input);
println!("Preprocessed:\n{}", output);
assert!(output.contains(" - Second line"), "Second line should be indented: {}", output);
assert!(output.contains(" - Third line"), "Third line should be indented: {}", output);
assert!(!output.contains(" [^2]"), "New footnote should not be indented: {}", output);
}
#[test]
fn test_slugify_matches_github_slugger() {
assert_eq!(slugify("/auth/verification-email/resend"), "authverification-emailresend");
assert_eq!(slugify("Hello World"), "hello-world");
assert_eq!(slugify("A.B.C"), "abc"); assert_eq!(slugify("日本語テスト"), "日本語テスト"); assert_eq!(slugify("test_underscore"), "test_underscore"); assert_eq!(slugify("a--b"), "a-b"); }
}
#[test]
fn test_footnote_in_table() {
let md = r#"| Col1 | Col2 | Col3 |
|------|------|------|
| [A][^1] | data | end |
[A]: #link
[^1]: Footnote one
"#;
let html = render_markdown(md);
println!("HTML: {}", html);
assert!(html.contains("<td>data</td>") || html.contains(">data<"), "data should be in its own cell: {}", html);
}
#[test]
fn test_footnote_with_list() {
let content = "- データソース項目の値\n- 上記以外の場合";
let html = render_footnote_continuation(content, false);
println!("Footnote continuation HTML: {}", html);
assert!(html.contains("<li>") && html.contains("<ul>"), "Should contain list: {}", html);
}