use super::DeclarativeBackend;
use crate::{InputFormat, SourceDocument};
use docling_core::tree::{ItemTree, TreeKind};
use docling_core::{DoclingDocument, Node, Table};
use std::collections::BTreeSet;
use std::io::Write;
type Words = BTreeSet<String>;
fn words_of(text: &str, into: &mut Words) {
let stripped = cached_regex!(r"!?\[([^\]]*)\]\([^)]*\)").replace_all(text, "$1");
let stripped = cached_regex!(r"<!--.*?-->").replace_all(&stripped, " ");
let stripped = stripped
.replace("<", "<")
.replace(">", ">")
.replace(""", "\"")
.replace("&", "&");
for w in cached_regex!(r"\p{L}{2,}").find_iter(&stripped) {
into.insert(w.as_str().to_lowercase());
}
}
fn table_words(t: &Table, into: &mut Words) {
for cell in t.rows.iter().flatten() {
words_of(cell, into);
}
if let Some(cells) = &t.cells {
for c in cells {
words_of(&c.text, into);
}
}
if let Some(blocks) = &t.cell_blocks {
for node in blocks.iter().flatten().flatten() {
node_words(node, into);
}
}
if let Some(c) = &t.caption {
words_of(c, into);
}
}
fn node_words(node: &Node, into: &mut Words) {
match node {
Node::Heading { text, .. }
| Node::Paragraph { text }
| Node::CheckboxItem { text, .. }
| Node::Caption { text, .. }
| Node::LabeledText { text, .. }
| Node::PageFurniture { text, .. }
| Node::FurnitureText { text, .. }
| Node::TextDump(text) => words_of(text, into),
Node::ListItem { text, dclx, .. } => match dclx {
Some(d) => words_of(&d.text, into),
None => words_of(text, into),
},
Node::Code { text, .. } => words_of(text, into),
Node::Table(t) => table_words(t, into),
Node::Picture { caption, .. } => {
if let Some(c) = caption {
words_of(c, into);
}
}
Node::Formula { latex, orig, .. } => {
words_of(latex, into);
words_of(orig, into);
}
Node::Chart { table, caption, .. } => {
table_words(table, into);
if let Some(c) = caption {
words_of(c, into);
}
}
Node::Group { children, .. } | Node::PictureChildren(children) => {
for c in children {
node_words(c, into);
}
}
Node::FieldRegion { items } => {
for f in items {
for s in [&f.marker, &f.key, &f.value].into_iter().flatten() {
words_of(s, into);
}
}
}
Node::KeyValueGraph { cells, .. } => {
for c in cells {
words_of(&c.text, into);
}
}
Node::InlineGroup { md_text, runs, .. } => {
words_of(md_text, into);
for r in runs {
words_of(&r.text, into);
}
}
Node::CommentSection { text, .. } => words_of(text, into),
Node::Furniture { inner, .. }
| Node::Commented { inner, .. }
| Node::Track { inner, .. }
| Node::Located { inner, .. }
| Node::Prov { inner, .. }
| Node::DoclangOnly(inner) => node_words(inner, into),
Node::PageBreak | Node::PageInfo { .. } => {}
}
}
fn flat_words(doc: &DoclingDocument) -> Words {
let mut out = Words::new();
for n in &doc.nodes {
node_words(n, &mut out);
}
out
}
fn tree_words(tree: &ItemTree) -> Words {
let mut out = Words::new();
for item in &tree.items {
if item.deleted {
continue;
}
match &item.kind {
TreeKind::Text { text, .. } => words_of(text, &mut out),
TreeKind::Code { text, .. } => words_of(text, &mut out),
TreeKind::Table { table, .. } => table_words(table, &mut out),
TreeKind::Picture { chart, .. } => {
if let Some(t) = chart {
table_words(t, &mut out);
}
}
TreeKind::FieldRegion { items } => {
for f in items {
for s in [&f.marker, &f.key, &f.value].into_iter().flatten() {
words_of(s, &mut out);
}
}
}
TreeKind::KeyValueGraph { cells, .. } => {
for c in cells {
words_of(&c.text, &mut out);
}
}
TreeKind::Group { .. } => {}
}
for note in &item.notes {
words_of(¬e.text, &mut out);
}
}
out
}
fn chain_divergence(bytes: Vec<u8>, name: &str) -> (Words, Words) {
let src = SourceDocument::from_bytes(name, InputFormat::Docx, bytes);
let doc = super::DocxBackend
.convert(&src)
.unwrap_or_else(|e| panic!("{name}: {e}"));
let flat = flat_words(&doc);
let tree = tree_words(
doc.tree
.as_ref()
.expect("the DOCX backend builds an item tree"),
);
(
flat.difference(&tree).cloned().collect(),
tree.difference(&flat).cloned().collect(),
)
}
const W: &str = "http://schemas.openxmlformats.org/wordprocessingml/2006/main";
const R: &str = "http://schemas.openxmlformats.org/officeDocument/2006/relationships";
const V: &str = "urn:schemas-microsoft-com:vml";
const WP: &str = "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing";
const A: &str = "http://schemas.openxmlformats.org/drawingml/2006/main";
const WPS: &str = "http://schemas.microsoft.com/office/word/2010/wordprocessingShape";
fn package(parts: &[(&str, String)]) -> Vec<u8> {
let mut zw = zip::ZipWriter::new(std::io::Cursor::new(Vec::new()));
let opts =
zip::write::SimpleFileOptions::default().compression_method(zip::CompressionMethod::Stored);
for (name, content) in parts {
zw.start_file(*name, opts).unwrap();
zw.write_all(content.as_bytes()).unwrap();
}
zw.finish().unwrap().into_inner()
}
fn docx_with_body(body: &str) -> Vec<u8> {
let document = format!(
r#"<w:document xmlns:w="{W}" xmlns:r="{R}" xmlns:v="{V}" xmlns:wp="{WP}" xmlns:a="{A}" xmlns:wps="{WPS}"><w:body>{body}<w:sectPr/></w:body></w:document>"#
);
let styles = format!(
r#"<w:styles xmlns:w="{W}"><w:style w:type="paragraph" w:styleId="Heading1"><w:name w:val="heading 1"/></w:style></w:styles>"#
);
let numbering = format!(
r#"<w:numbering xmlns:w="{W}"><w:abstractNum w:abstractNumId="0"><w:lvl w:ilvl="0"><w:start w:val="1"/><w:numFmt w:val="decimal"/><w:lvlText w:val="%1."/></w:lvl></w:abstractNum><w:num w:numId="1"><w:abstractNumId w:val="0"/></w:num></w:numbering>"#
);
let footnotes = format!(
r#"<w:footnotes xmlns:w="{W}"><w:footnote w:type="separator" w:id="-1"><w:p><w:r><w:separator/></w:r></w:p></w:footnote><w:footnote w:id="1"><w:p><w:r><w:t>footnote body quokka</w:t></w:r></w:p></w:footnote></w:footnotes>"#
);
let comments = format!(
r#"<w:comments xmlns:w="{W}"><w:comment w:id="0" w:author="Reviewer Name" w:date="2026-01-01T00:00:00Z"><w:p><w:r><w:t>comment body wombat</w:t></w:r></w:p></w:comment></w:comments>"#
);
let rels = r#"<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships"><Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/hyperlink" Target="https://example.com/path" TargetMode="External"/><Relationship Id="rId9" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/footnotes" Target="footnotes.xml"/></Relationships>"#.to_string();
package(&[
("word/document.xml", document),
("word/styles.xml", styles),
("word/numbering.xml", numbering),
("word/footnotes.xml", footnotes),
("word/comments.xml", comments),
("word/_rels/document.xml.rels", rels),
])
}
fn run(text: &str) -> String {
format!(r#"<w:r><w:t xml:space="preserve">{text}</w:t></w:r>"#)
}
fn para(inner: &str) -> String {
format!("<w:p>{inner}</w:p>")
}
fn table(payload: &str) -> String {
format!(
"<w:tbl><w:tblGrid><w:gridCol/><w:gridCol/></w:tblGrid><w:tr><w:tc>{payload}</w:tc><w:tc>{}</w:tc></w:tr></w:tbl>",
para(&run("sibling cell"))
)
}
fn vml_textbox(payload: &str) -> String {
para(&format!(
"<w:r><w:pict><v:shape><v:textbox><w:txbxContent>{payload}</w:txbxContent></v:textbox></v:shape></w:pict></w:r>"
))
}
fn dml_textbox(payload: &str) -> String {
para(&format!(
r#"<w:r><w:drawing><wp:inline><wp:extent cx="914400" cy="914400"/><wp:docPr id="1" name="Box"/><a:graphic><a:graphicData uri="{WPS}"><wps:wsp><wps:txbx><w:txbxContent>{payload}</w:txbxContent></wps:txbx></wps:wsp></a:graphicData></a:graphic></wp:inline></w:drawing></w:r>"#
))
}
fn block_sdt(payload: &str) -> String {
format!("<w:sdt><w:sdtPr/><w:sdtContent>{payload}</w:sdtContent></w:sdt>")
}
fn payloads() -> Vec<(&'static str, String)> {
vec![
("plain", para(&run("plain paragraph alpha"))),
(
"inline-sdt",
para(&format!(
"<w:sdt><w:sdtPr/><w:sdtContent>{}</w:sdtContent></w:sdt>",
run("inline control bravo")
)),
),
("block-sdt", block_sdt(¶(&run("block control charlie")))),
(
"ins",
para(&format!(
r#"<w:ins w:id="1" w:author="a" w:date="2026-01-01T00:00:00Z">{}</w:ins>"#,
run("inserted text delta")
)),
),
(
"moveTo",
para(&format!(
r#"<w:moveTo w:id="2" w:author="a" w:date="2026-01-01T00:00:00Z">{}</w:moveTo>"#,
run("moved text echo")
)),
),
(
"del",
para(&format!(
r#"{}<w:del w:id="3" w:author="a" w:date="2026-01-01T00:00:00Z"><w:r><w:delText>deleted text foxtrot</w:delText></w:r></w:del>"#,
run("kept before deletion")
)),
),
(
"customXml",
para(&format!(
r#"<w:customXml w:element="tag">{}</w:customXml>"#,
run("custom xml golf")
)),
),
(
"smartTag",
para(&format!(
r#"<w:smartTag w:uri="urn:x" w:element="place">{}</w:smartTag>"#,
run("smart tag hotel")
)),
),
(
"fldSimple",
para(&format!(
r#"<w:fldSimple w:instr=" REF x ">{}</w:fldSimple>"#,
run("field result india")
)),
),
(
"hyperlink",
para(&format!(
r#"{}<w:hyperlink r:id="rId1">{}</w:hyperlink>"#,
run("see "),
run("linked text juliet")
)),
),
(
"bold-runs",
para(&format!(
"{}<w:r><w:rPr><w:b/></w:rPr><w:t>bold kilo</w:t></w:r>{}",
run("plain before "),
run(" plain after lima")
)),
),
(
"heading",
para(&format!(
r#"<w:pPr><w:pStyle w:val="Heading1"/></w:pPr>{}"#,
run("heading text mike")
)),
),
(
"list-item",
para(&format!(
r#"<w:pPr><w:numPr><w:ilvl w:val="0"/><w:numId w:val="1"/></w:numPr></w:pPr>{}"#,
run("list item november")
)),
),
(
"footnote-ref",
para(&format!(
r#"{}<w:r><w:footnoteReference w:id="1"/></w:r>"#,
run("noted text oscar")
)),
),
(
"comment-range",
para(&format!(
r#"<w:commentRangeStart w:id="0"/>{}<w:commentRangeEnd w:id="0"/><w:r><w:commentReference w:id="0"/></w:r>"#,
run("commented text papa")
)),
),
(
"vml-textbox",
vml_textbox(¶(&run("vml box text quebec"))),
),
(
"dml-textbox",
dml_textbox(¶(&run("dml box text romeo"))),
),
("table", table(¶(&run("inner table sierra")))),
(
"line-break",
para(&format!(
"{}<w:r><w:br/><w:t>after break tango</w:t></w:r>",
run("before break")
)),
),
]
}
type Context = (&'static str, fn(&str) -> String);
fn contexts() -> Vec<Context> {
vec![
("body", |p| p.to_string()),
("table-cell", table),
("nested-table-cell", |p| table(&table(p))),
("vml-textbox", vml_textbox),
("dml-textbox", dml_textbox),
("textbox-in-cell", |p| table(&vml_textbox(p))),
("block-sdt-in-cell", |p| table(&block_sdt(p))),
("block-sdt", block_sdt),
]
}
#[test]
fn generated_matrix_both_chains_carry_the_same_words() {
let mut failures = Vec::new();
for (ctx_name, wrap) in contexts() {
for (payload_name, payload) in payloads() {
let body = format!("{}{}", para(&run("lead paragraph")), wrap(&payload));
let name = format!("{payload_name} @ {ctx_name}");
let (flat_only, tree_only) = chain_divergence(docx_with_body(&body), &name);
if !flat_only.is_empty() || !tree_only.is_empty() {
failures.push(format!(
"{name}: flat-only {flat_only:?}, tree-only {tree_only:?}"
));
}
}
}
assert!(
failures.is_empty(),
"the two DOCX walks disagree on:\n{}",
failures.join("\n")
);
}
#[test]
fn inline_custom_xml_is_in_both_chains() {
let body = para(&format!(
r#"{}<w:customXml w:element="tag">{}</w:customXml>{}"#,
run("Prefix text. "),
run("INSIDE_CUSTOM_XML"),
run(" Suffix text.")
));
let (flat_only, tree_only) = chain_divergence(docx_with_body(&body), "customXml");
assert!(
flat_only.is_empty() && tree_only.is_empty(),
"{flat_only:?} / {tree_only:?}"
);
}
#[test]
fn fixture_corpus_both_chains_carry_the_same_words() {
const KNOWN: &[(&str, &str)] = &[];
let root = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("../..");
let mut names = Vec::new();
for dir in [
root.join("tests/data/docx/sources"),
root.join("crates/docling/tests/data/docx/sources"),
] {
for entry in std::fs::read_dir(&dir).unwrap_or_else(|e| panic!("{}: {e}", dir.display())) {
let path = entry.unwrap().path();
if path.extension().is_some_and(|e| e == "docx") {
names.push(path);
}
}
}
names.sort();
assert!(names.len() > 50, "fixture dirs moved? {}", names.len());
let mut failures = Vec::new();
let mut stale: Vec<&str> = KNOWN.iter().map(|(n, _)| *n).collect();
for path in names {
let name = path.file_name().unwrap().to_string_lossy().into_owned();
let bytes = std::fs::read(&path).unwrap();
let (flat_only, tree_only) = chain_divergence(bytes, &name);
let diverges = !flat_only.is_empty() || !tree_only.is_empty();
if let Some((_, reason)) = KNOWN.iter().find(|(n, _)| *n == name) {
stale.retain(|n| *n != name);
if !diverges {
failures.push(format!(
"{name}: listed as known ({reason}) but the chains agree now"
));
}
} else if diverges {
failures.push(format!(
"{name}: flat-only {flat_only:?}, tree-only {tree_only:?}"
));
}
}
for name in stale {
failures.push(format!("{name}: in KNOWN but no such fixture"));
}
assert!(
failures.is_empty(),
"the two DOCX walks disagree on:\n{}",
failures.join("\n")
);
}