use std::path::Path;
use anyhow::{Context, Result};
pub const DOCUMENT_EXTENSIONS: &[&str] = &["html", "htm"];
pub fn is_document_file(path: &Path) -> bool {
match path.extension().and_then(|e| e.to_str()) {
Some(ext) => {
let ext = ext.to_ascii_lowercase();
DOCUMENT_EXTENSIONS.contains(&ext.as_str())
}
None => false,
}
}
pub fn extract_text(path: &Path) -> Result<String> {
let html = std::fs::read_to_string(path)
.with_context(|| format!("failed to read HTML document {}", path.display()))?;
Ok(html_to_markdownish(&html))
}
enum TagRole {
Heading(usize),
Block,
ListItem,
LineBreak,
Pre,
Cell,
Inline,
}
fn tag_role(name: &str) -> TagRole {
match name {
"h1" => TagRole::Heading(1),
"h2" => TagRole::Heading(2),
"h3" => TagRole::Heading(3),
"h4" => TagRole::Heading(4),
"h5" => TagRole::Heading(5),
"h6" => TagRole::Heading(6),
"p" | "div" | "section" | "article" | "header" | "footer" | "main" | "aside" | "nav"
| "blockquote" | "table" | "thead" | "tbody" | "tr" | "ul" | "ol" | "dl" | "dt" | "dd"
| "figure" | "figcaption" | "hr" | "form" | "fieldset" | "details" | "summary" => {
TagRole::Block
}
"li" => TagRole::ListItem,
"br" => TagRole::LineBreak,
"pre" => TagRole::Pre,
"td" | "th" => TagRole::Cell,
_ => TagRole::Inline,
}
}
pub fn html_to_markdownish(html: &str) -> String {
let bytes = html.as_bytes();
let mut out = String::with_capacity(html.len() / 2);
let mut i = 0;
let mut in_pre = false;
let mut pending_space = false;
while i < bytes.len() {
if bytes[i] == b'<' {
if html[i..].starts_with("<!--") {
i = match html[i + 4..].find("-->") {
Some(p) => i + 4 + p + 3,
None => bytes.len(), };
continue;
}
let (closing, name, tag_end) = match parse_tag(html, i) {
Some(t) => t,
None => {
flush_space(&mut out, &mut pending_space, in_pre);
out.push('<');
i += 1;
continue;
}
};
if !closing && matches!(name.as_str(), "script" | "style" | "head" | "title") {
i = skip_container(html, tag_end, &name);
continue;
}
match tag_role(&name) {
TagRole::Heading(level) => {
if closing {
push_newline(&mut out);
out.push('\n');
} else {
ensure_blank_line(&mut out);
out.push_str(&"#".repeat(level));
out.push(' ');
}
pending_space = false;
}
TagRole::Block => {
push_newline(&mut out);
pending_space = false;
}
TagRole::ListItem => {
if closing {
push_newline(&mut out);
} else {
push_newline(&mut out);
out.push_str("- ");
}
pending_space = false;
}
TagRole::LineBreak => {
if !closing {
out.push('\n');
}
pending_space = false;
}
TagRole::Pre => {
if closing {
push_newline(&mut out);
out.push_str("```\n");
in_pre = false;
} else {
ensure_blank_line(&mut out);
out.push_str("```\n");
in_pre = true;
}
pending_space = false;
}
TagRole::Cell => {
if !closing {
pending_space = true;
}
}
TagRole::Inline => {}
}
i = tag_end;
} else {
let (decoded, next) = if bytes[i] == b'&' {
decode_entity(html, i)
} else {
let ch = html[i..].chars().next().unwrap();
(ch, i + ch.len_utf8())
};
if in_pre {
out.push(decoded);
} else if decoded.is_whitespace() {
pending_space = true;
} else {
flush_space(&mut out, &mut pending_space, in_pre);
out.push(decoded);
}
i = next;
}
}
tidy(&out)
}
fn parse_tag(html: &str, start: usize) -> Option<(bool, String, usize)> {
let rest = &html[start + 1..];
let (closing, rest_off) = match rest.strip_prefix('/') {
Some(_) => (true, 1),
None => (false, 0),
};
let name_start = start + 1 + rest_off;
let name: String = html[name_start..]
.chars()
.take_while(|c| c.is_ascii_alphanumeric())
.collect();
if name.is_empty() {
if rest.starts_with('!') || rest.starts_with('?') {
let end = html[start..].find('>').map(|p| start + p + 1)?;
return Some((false, String::new(), end));
}
return None;
}
let end = html[start..].find('>').map(|p| start + p + 1)?;
Some((closing, name.to_ascii_lowercase(), end))
}
fn skip_container(html: &str, from: usize, name: &str) -> usize {
let lower = html.to_ascii_lowercase();
let close = format!("</{name}");
if let Some(p) = lower[from..].find(&close) {
let after = from + p;
return lower[after..]
.find('>')
.map(|q| after + q + 1)
.unwrap_or(html.len());
}
if name == "head" {
if let Some(p) = lower[from..].find("<body") {
return from + p;
}
}
html.len()
}
fn decode_entity(html: &str, start: usize) -> (char, usize) {
let rest = &html[start + 1..];
let Some(semi) = rest.find(';') else {
return ('&', start + 1);
};
if semi > 10 {
return ('&', start + 1);
}
let body = &rest[..semi];
let end = start + 1 + semi + 1;
let decoded = match body {
"amp" => Some('&'),
"lt" => Some('<'),
"gt" => Some('>'),
"quot" => Some('"'),
"apos" => Some('\''),
"nbsp" => Some(' '),
_ => body.strip_prefix("#").and_then(|num| {
let cp = if let Some(hex) = num.strip_prefix('x').or_else(|| num.strip_prefix('X')) {
u32::from_str_radix(hex, 16).ok()
} else {
num.parse::<u32>().ok()
};
cp.and_then(char::from_u32)
}),
};
match decoded {
Some(c) => (c, end),
None => ('&', start + 1),
}
}
fn flush_space(out: &mut String, pending: &mut bool, in_pre: bool) {
if *pending && !in_pre {
if !out.is_empty() && !out.ends_with('\n') {
out.push(' ');
}
*pending = false;
}
}
fn push_newline(out: &mut String) {
if !out.is_empty() && !out.ends_with('\n') {
out.push('\n');
}
}
fn ensure_blank_line(out: &mut String) {
while !out.is_empty() && !out.ends_with("\n\n") {
out.push('\n');
}
}
fn tidy(text: &str) -> String {
let mut lines: Vec<&str> = text.lines().map(str::trim_end).collect();
while lines.first().is_some_and(|l| l.is_empty()) {
lines.remove(0);
}
while lines.last().is_some_and(|l| l.is_empty()) {
lines.pop();
}
let mut out = String::with_capacity(text.len());
let mut blank_run = 0;
for line in lines {
if line.is_empty() {
blank_run += 1;
if blank_run > 1 {
continue;
}
} else {
blank_run = 0;
}
out.push_str(line);
out.push('\n');
}
out
}
#[cfg(test)]
#[path = "documents_tests.rs"]
mod tests;