use std::collections::HashMap;
use roxmltree::Document;
use crate::backend::ooxml::{self, Package};
use crate::backend::{
convert_html, maybe_prerender_html, DeclarativeBackend, MapImageResolver, NoFetch,
};
use crate::error::ConversionError;
use crate::source::SourceDocument;
use docling_core::{DoclingDocument, PictureImage};
#[derive(Default)]
pub struct EpubBackend {
pub fetch_images: bool,
pub use_web_browser: bool,
}
impl DeclarativeBackend for EpubBackend {
fn convert(&self, source: &SourceDocument) -> Result<DoclingDocument, ConversionError> {
let mut pkg = Package::open(&source.bytes)
.ok_or_else(|| ConversionError::Parse("epub: not a zip".into()))?;
let container = pkg
.read("META-INF/container.xml")
.ok_or_else(|| ConversionError::Parse("epub: no container.xml".into()))?;
let opf_path = rootfile_path(&container)
.ok_or_else(|| ConversionError::Parse("epub: no rootfile".into()))?;
let opf = pkg
.read(&opf_path)
.ok_or_else(|| ConversionError::Parse(format!("epub: missing {opf_path}")))?;
let opf_dir = opf_path.rsplit_once('/').map(|(d, _)| d).unwrap_or("");
let spine =
spine_files(&opf, opf_dir).map_err(|e| ConversionError::Parse(format!("epub: {e}")))?;
let mut combined =
String::from("<!DOCTYPE html><html><head><meta charset=\"utf-8\"/></head><body>");
let body_re = cached_regex!(r"(?is)<body[^>]*>(.*?)</body>");
let link_re = cached_regex!(r#"href="([^"]*\.(?:xhtml|xht|html?))(#[^"]*)""#);
let mut images: HashMap<String, PictureImage> = HashMap::new();
for file in &spine {
let Some(xhtml) = read_content(&mut pkg, file) else {
continue;
};
let body = body_re
.captures(&xhtml)
.map(|c| c[1].to_string())
.unwrap_or(xhtml);
let body = link_re.replace_all(&body, |caps: ®ex::Captures| {
if is_external_href(&caps[1]) {
caps[0].to_string()
} else {
format!("href=\"{}\"", &caps[2])
}
});
let body = if self.fetch_images {
let dir = file.rsplit_once('/').map(|(d, _)| d).unwrap_or("");
extract_images(&body, dir, &mut pkg, &mut images)
} else {
body.into_owned()
};
combined.push('\n');
combined.push_str(&body);
}
combined.push_str("\n</body></html>");
let combined = maybe_prerender_html(&combined, self.use_web_browser)?;
let doc = if self.fetch_images {
convert_html(&source.name, &combined, &MapImageResolver::new(images))
} else {
convert_html(&source.name, &combined, &NoFetch)
};
Ok(doc)
}
}
fn is_external_href(href: &str) -> bool {
if href.starts_with("//") {
return true;
}
let scheme_len = href
.bytes()
.take_while(|b| b.is_ascii_alphanumeric() || *b == b'_')
.count();
scheme_len > 0 && href[scheme_len..].starts_with(':')
}
fn read_content(pkg: &mut Package, path: &str) -> Option<String> {
let text = decode_content(pkg.read_bytes(path)?)?;
if let Err(e) = super::xml_depth::check(&text, path) {
eprintln!("docling: {e}; part skipped");
return None;
}
Some(text)
}
fn decode_content(bytes: Vec<u8>) -> Option<String> {
Some(match bytes.as_slice() {
[0xFF, 0xFE, rest @ ..] | [0xFE, 0xFF, rest @ ..] => {
let big_endian = bytes[0] == 0xFE;
let units: Vec<u16> = rest
.chunks_exact(2)
.map(|c| {
if big_endian {
u16::from_be_bytes([c[0], c[1]])
} else {
u16::from_le_bytes([c[0], c[1]])
}
})
.collect();
String::from_utf16_lossy(&units)
}
_ => String::from_utf8(bytes).ok()?,
})
}
fn extract_images(
body: &str,
dir: &str,
pkg: &mut Package,
images: &mut HashMap<String, PictureImage>,
) -> String {
let img_re = cached_regex!(r"(?is)<img\b[^>]*>");
let src_re = cached_regex!(r#"(?is)\bsrc\s*=\s*"([^"]*)""#);
img_re
.replace_all(body, |caps: ®ex::Captures| {
let tag = &caps[0];
let Some(raw) = src_re.captures(tag).map(|m| m[1].to_string()) else {
return tag.to_string();
};
if raw.is_empty()
|| raw.starts_with("data:")
|| raw.starts_with("http://")
|| raw.starts_with("https://")
{
return tag.to_string();
}
let archive_path = ooxml::resolve(dir, &raw);
if !images.contains_key(&archive_path) {
if let Some(pic) = pkg
.read_bytes(&archive_path)
.and_then(|bytes| ooxml::picture_image(&archive_path, bytes))
{
images.insert(archive_path.clone(), pic);
}
}
src_re
.replace(tag, format!(r#"src="{archive_path}""#).as_str())
.into_owned()
})
.into_owned()
}
fn rootfile_path(container: &str) -> Option<String> {
let dom = Document::parse(container).ok()?;
dom.descendants()
.find(|n| n.has_tag_name("rootfile"))
.and_then(|n| n.attribute("full-path"))
.map(str::to_string)
}
fn spine_files(opf: &str, opf_dir: &str) -> Result<Vec<String>, String> {
let dom = Document::parse(opf).map_err(|e| e.to_string())?;
let mut id_to_href = std::collections::HashMap::new();
for item in dom.descendants().filter(|n| n.has_tag_name("item")) {
if let (Some(id), Some(href)) = (item.attribute("id"), item.attribute("href")) {
id_to_href.insert(id.to_string(), percent_decode(href));
}
}
let mut files = Vec::new();
for itemref in dom.descendants().filter(|n| n.has_tag_name("itemref")) {
if let Some(href) = itemref.attribute("idref").and_then(|id| id_to_href.get(id)) {
files.push(normpath(&if opf_dir.is_empty() {
href.clone()
} else {
format!("{opf_dir}/{href}")
}));
}
}
Ok(files)
}
fn normpath(path: &str) -> String {
let absolute = path.starts_with('/');
let mut out: Vec<&str> = Vec::new();
for seg in path.split('/') {
match seg {
"" | "." => {}
".." => match out.last() {
Some(&last) if last != ".." => {
out.pop();
}
_ if absolute => {}
_ => out.push(".."),
},
s => out.push(s),
}
}
let joined = out.join("/");
match (absolute, joined.is_empty()) {
(true, _) => format!("/{joined}"),
(false, true) => ".".to_string(),
(false, false) => joined,
}
}
fn percent_decode(s: &str) -> String {
if !s.contains('%') {
return s.to_string();
}
let b = s.as_bytes();
let mut out = Vec::with_capacity(b.len());
let mut i = 0;
while i < b.len() {
if b[i] == b'%' && i + 2 < b.len() {
let hex = |c: u8| (c as char).to_digit(16);
if let (Some(h), Some(l)) = (hex(b[i + 1]), hex(b[i + 2])) {
out.push((h * 16 + l) as u8);
i += 3;
continue;
}
}
out.push(b[i]);
i += 1;
}
String::from_utf8_lossy(&out).into_owned()
}
#[cfg(test)]
mod tests {
#[test]
fn utf16_content_documents_decode_behind_their_bom() {
let le: Vec<u8> = [0xFF, 0xFE]
.into_iter()
.chain("<p>hé</p>".encode_utf16().flat_map(u16::to_le_bytes))
.collect();
assert_eq!(super::decode_content(le).as_deref(), Some("<p>hé</p>"));
let be: Vec<u8> = [0xFE, 0xFF]
.into_iter()
.chain("<p>x</p>".encode_utf16().flat_map(u16::to_be_bytes))
.collect();
assert_eq!(super::decode_content(be).as_deref(), Some("<p>x</p>"));
assert_eq!(
super::decode_content(b"<p>y</p>".to_vec()).as_deref(),
Some("<p>y</p>")
);
assert_eq!(super::decode_content(vec![0xC3, 0x28]), None);
}
use super::*;
#[test]
fn manifest_hrefs_stepping_out_of_the_opf_dir_normalize() {
assert_eq!(normpath("OEBPS/../Text/ch1.xhtml"), "Text/ch1.xhtml");
assert_eq!(normpath("OEBPS/./a//b.xhtml"), "OEBPS/a/b.xhtml");
assert_eq!(normpath("../x.xhtml"), "../x.xhtml");
assert_eq!(normpath("a/.."), ".");
let opf = r#"<package><manifest><item id="a" href="../Text/ch1.xhtml"/></manifest>
<spine><itemref idref="a"/></spine></package>"#;
assert_eq!(
spine_files(opf, "OEBPS").unwrap(),
vec!["Text/ch1.xhtml".to_string()]
);
}
#[test]
fn manifest_hrefs_are_percent_decoded() {
let opf = r#"<package xmlns="http://www.idpf.org/2007/opf">
<manifest>
<item id="a" href="chapter%201.xhtml"/>
<item id="b" href="%C3%A9pilogue.xhtml"/>
<item id="c" href="plain.xhtml"/>
<item id="d" href="odd%2.xhtml"/>
</manifest>
<spine><itemref idref="a"/><itemref idref="b"/><itemref idref="c"/><itemref idref="d"/></spine>
</package>"#;
assert_eq!(
spine_files(opf, "").unwrap(),
vec![
"chapter 1.xhtml",
"\u{e9}pilogue.xhtml",
"plain.xhtml",
"odd%2.xhtml"
]
);
}
#[test]
fn spine_order_resolves_manifest_hrefs() {
let opf = r#"<package xmlns="http://www.idpf.org/2007/opf">
<manifest>
<item id="c2" href="text/two.xhtml"/>
<item id="c1" href="text/one.xhtml"/>
<item id="css" href="x.css"/>
</manifest>
<spine><itemref idref="c1"/><itemref idref="c2"/></spine>
</package>"#;
assert_eq!(
spine_files(opf, "epub").unwrap(),
vec!["epub/text/one.xhtml", "epub/text/two.xhtml"]
);
}
#[test]
fn finds_opf_rootfile() {
let container = r#"<container xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
<rootfiles><rootfile full-path="epub/content.opf" media-type="application/oebps-package+xml"/></rootfiles>
</container>"#;
assert_eq!(
rootfile_path(container).as_deref(),
Some("epub/content.opf")
);
}
#[test]
fn internal_links_of_every_extension_become_anchors() {
let re = cached_regex!(r#"href="([^"]*\.(?:xhtml|xht|html?))(#[^"]*)""#);
let fix = |html: &str| -> String {
re.replace_all(html, |caps: ®ex::Captures| {
if is_external_href(&caps[1]) {
caps[0].to_string()
} else {
format!("href=\"{}\"", &caps[2])
}
})
.into_owned()
};
for ext in ["xhtml", "html", "htm", "xht"] {
assert_eq!(
fix(&format!("<a href=\"../text/ch2.{ext}#n1\">")),
"<a href=\"#n1\">"
);
}
assert_eq!(
fix("<a href=\"https://example.com/page.html#about\">"),
"<a href=\"https://example.com/page.html#about\">"
);
assert_eq!(
fix("<a href=\"//example.com/page.html#about\">"),
"<a href=\"//example.com/page.html#about\">"
);
assert_eq!(
fix("<a href=\"mailto:a@example.com\">"),
"<a href=\"mailto:a@example.com\">"
);
}
#[test]
fn extracts_archive_images_only_when_fetching() {
use crate::format::InputFormat;
use docling_core::{DoclingDocument, Node};
let path = concat!(
env!("CARGO_MANIFEST_DIR"),
"/../../tests/data/epub/sources/epub_purvis_poetry.epub"
);
let Ok(bytes) = std::fs::read(path) else {
return;
};
let src = SourceDocument::from_bytes("epub_purvis_poetry", InputFormat::Epub, bytes);
let embedded = |doc: &DoclingDocument| {
doc.nodes
.iter()
.filter_map(|n| match n {
Node::Picture {
image: Some(img), ..
} => Some(img),
_ => None,
})
.cloned()
.collect::<Vec<_>>()
};
let plain = EpubBackend {
fetch_images: false,
..Default::default()
}
.convert(&src)
.unwrap();
assert!(embedded(&plain).is_empty());
let fetched = EpubBackend {
fetch_images: true,
..Default::default()
}
.convert(&src)
.unwrap();
let imgs = embedded(&fetched);
assert!(!imgs.is_empty(), "expected extracted archive images");
assert!(imgs
.iter()
.all(|img| img.width > 0 && img.height > 0 && !img.data.is_empty()));
}
}