use docling_core::{DoclingDocument, Node, PictureImage, Table};
use crate::backend::cfb::CompoundFile;
use crate::backend::markdown::escape_text;
use crate::backend::officeart;
use crate::backend::symbol_fonts::{self, SymbolFont};
use crate::backend::DeclarativeBackend;
use crate::error::ConversionError;
use crate::source::SourceDocument;
pub struct DocBackend;
impl DeclarativeBackend for DocBackend {
fn convert(&self, source: &SourceDocument) -> Result<DoclingDocument, ConversionError> {
if is_word2(&source.bytes) {
return convert_word2(&source.name, &source.bytes);
}
let cfb = CompoundFile::open(&source.bytes).ok_or_else(|| {
ConversionError::Parse(CompoundFile::open_error("doc", &source.bytes))
})?;
let word = cfb
.stream("WordDocument")
.ok_or_else(|| ConversionError::Parse(cfb.stream_error("doc", "WordDocument")))?;
let w_ident = u16_at(&word, 0).ok_or_else(|| {
ConversionError::Parse(format!(
"doc: WordDocument stream too short for a FIB ({} bytes)",
word.len()
))
})?;
let n_fib = u16_at(&word, 2).unwrap_or(0);
let flags = u16_at(&word, 0x0A).unwrap_or(0);
if flags & 0x0100 != 0 {
return Err(ConversionError::Parse("doc: document is encrypted".into()));
}
if w_ident != 0xA5EC {
if (101..=105).contains(&n_fib) {
return convert_word6(&source.name, &word, flags);
}
return Err(ConversionError::Parse(format!(
"doc: unsupported Word binary format (wIdent {w_ident:#06X}, nFib {n_fib}); \
Word 6.0/95 and Word 97 or later are read"
)));
}
let fib = Fib::parse(&word)?;
let table_name = if flags & 0x0200 != 0 {
"1Table"
} else {
"0Table"
};
let table = cfb
.stream(table_name)
.ok_or_else(|| ConversionError::Parse(format!("doc: no {table_name} stream")))?;
let ccp_text = fib.ccp_text as u64;
let (fc_clx, lcb_clx) = fib.fc_lcb(Fib::CLX);
let clx = fc_clx
.checked_add(lcb_clx)
.and_then(|end| table.get(fc_clx..end))
.ok_or_else(|| {
ConversionError::Parse(format!(
"doc: piece table (fcClx {fc_clx}, lcbClx {lcb_clx}) lies outside the \
{table_name} stream ({} bytes)",
table.len()
))
})?;
let pieces = parse_piece_table(clx, None).ok_or_else(|| {
ConversionError::Parse(format!(
"doc: bad piece table (fcClx {fc_clx}, lcbClx {lcb_clx} in {table_name})"
))
})?;
let stis = parse_stsh(fib.part(&table, Fib::STSHF));
let bte = fib.part(&table, Fib::PLCF_BTE_PAPX);
let btec = fib.part(&table, Fib::PLCF_BTE_CHPX);
let mut chpx_cache = ChpxCache::default();
let (fc_lst, lcb_lst) = fib.fc_lcb(Fib::PLF_LST);
let lists = ListTables::parse(
table.get(fc_lst..).unwrap_or(&[]),
lcb_lst,
fib.part(&table, Fib::PLF_LFO),
);
let data = cfb.stream("Data").unwrap_or_default();
let spa = parse_plcf_spa(fib.part(&table, Fib::PLCSPA_MOM));
let drawings = Drawings::parse(fib.part(&table, Fib::DGG_INFO), &word);
let fonts = parse_sttbf_ffn(fib.part(&table, Fib::STTBF_FFN));
let ftn_base = ccp_text;
let hdd_base = ftn_base + fib.ccp_ftn as u64;
let edn_base = hdd_base + fib.ccp_hdd as u64 + fib.ccp_atn as u64;
let txbx_base = edn_base + fib.ccp_edn as u64;
let story = Story {
word: &word,
pieces: &pieces,
bte,
btec,
stis: &stis,
data: &data,
spa: &spa,
drawings: &drawings,
textboxes: parse_textboxes(
fib.part(&table, Fib::PLCF_TXBX_TXT),
txbx_base,
fib.ccp_txbx as u64,
),
placed: Default::default(),
fonts: &fonts,
lists: &lists,
};
let mut doc = DoclingDocument::new(&source.name);
story.walk(&mut chpx_cache, 0, ccp_text, true, &mut doc);
let mut unplaced: Vec<(u64, u64)> = story
.textboxes
.iter()
.filter(|(spid, _)| !story.placed.borrow().contains(spid))
.map(|(_, &range)| range)
.collect();
unplaced.sort_unstable();
for (a, b) in unplaced {
let mut inner = DoclingDocument::new("");
story.walk(&mut chpx_cache, a, b, false, &mut inner);
if !inner.nodes.is_empty() {
doc.push(Node::Group {
label: "section".into(),
name: Some("textbox".into()),
layer: None,
children: inner.nodes,
});
}
}
let header_boxes = HeaderBoxes {
anchors: parse_plcf_spa(fib.part(&table, Fib::PLCSPA_HDR)),
stories: parse_textboxes(
fib.part(&table, Fib::PLCF_HDR_TXBX_TXT),
txbx_base + fib.ccp_txbx as u64,
fib.ccp_hdr_txbx as u64,
),
};
for (footer, text) in header_footer_texts(
&story,
&mut chpx_cache,
fib.part(&table, Fib::PLCF_HDD),
hdd_base,
fib.ccp_hdd as u64,
&header_boxes,
) {
doc.push(Node::FurnitureText {
label: if footer { "page_footer" } else { "page_header" }.into(),
text: escape_text(&text),
});
}
for (refs, txt, base, len) in [
(Fib::PLCF_FND_REF, Fib::PLCF_FND_TXT, ftn_base, fib.ccp_ftn),
(Fib::PLCF_END_REF, Fib::PLCF_END_TXT, edn_base, fib.ccp_edn),
] {
for text in note_texts(
&story,
&mut chpx_cache,
fib.part(&table, refs),
fib.part(&table, txt),
base,
len as u64,
) {
doc.push(Node::FurnitureText {
label: "footnote".into(),
text: escape_text(&text),
});
}
}
Ok(doc)
}
}
struct Story<'a> {
word: &'a [u8],
pieces: &'a [Piece],
bte: &'a [u8],
btec: &'a [u8],
stis: &'a [StyleDef],
data: &'a [u8],
spa: &'a [(u64, u32)],
drawings: &'a Drawings,
textboxes: std::collections::HashMap<u32, (u64, u64)>,
placed: std::cell::RefCell<std::collections::HashSet<u32>>,
fonts: &'a [String],
lists: &'a ListTables,
}
impl Story<'_> {
fn char_at(&self, cp: u64) -> Option<(char, u64)> {
let k = self.pieces.partition_point(|p| p.cp_end <= cp);
let piece = self.pieces.get(k).filter(|p| p.cp_start <= cp)?;
let i = cp - piece.cp_start;
Some((piece_char(self.word, piece, i), piece_fc(piece, i)))
}
fn push_char(&self, para: &mut ParaAccum, ch: char, fmt: CharFmt) {
let font = |ftc: u16| {
self.fonts
.get(ftc as usize)
.and_then(|name| SymbolFont::from_name(name))
};
let ch = match fmt.symbol {
Some((ftc, code)) if ch == '(' => font(ftc)
.and_then(|f| f.glyph(code as u32))
.or_else(|| char::from_u32(code as u32).filter(|c| !c.is_control()))
.unwrap_or(ch),
_ if symbol_fonts::is_glyph_code(ch) => {
let ftc = if (ch as u32) < 0x80 {
fmt.ftc_ascii.or(fmt.ftc_other)
} else {
fmt.ftc_other.or(fmt.ftc_ascii)
};
ftc.and_then(font)
.and_then(|f| f.glyph(ch as u32))
.unwrap_or(ch)
}
_ => ch,
};
para.push(ch, fmt);
}
fn walk(
&self,
cache: &mut ChpxCache,
from: u64,
to: u64,
textboxes: bool,
doc: &mut DoclingDocument,
) {
let mut builder = NodeBuilder::new(self.lists.clone());
let mut para = ParaAccum::default();
for cp in from..to {
let Some((ch, fc)) = self.char_at(cp) else {
break;
};
match ch {
'\r' | '\u{0007}' | '\u{000C}' => {
let props = paragraph_props(self.word, self.bte, fc);
para.finish(ch, props, self.stis, &mut builder, doc);
}
'\u{0001}' => {
if let Some(pic_fc) = cache.props(self.word, self.btec, fc).pic_fc {
para.add_picture(inline_picture(self.data, pic_fc));
}
}
'\u{0008}' => {
if let Some(spid) = spa_shape_at(self.spa, cp) {
for image in self.drawings.shape_pictures(spid, 0) {
para.add_picture(image);
}
if let Some(&(a, b)) = self.textboxes.get(&spid).filter(|_| textboxes) {
self.placed.borrow_mut().insert(spid);
let mut inner = DoclingDocument::new("");
self.walk(cache, a, b, false, &mut inner);
if !inner.nodes.is_empty() {
para.textboxes.push(inner.nodes);
}
}
}
}
_ => self.push_char(&mut para, ch, cache.props(self.word, self.btec, fc)),
}
}
para.finish('\r', ParaProps::default(), self.stis, &mut builder, doc);
builder.flush(doc);
}
fn paragraphs(&self, cache: &mut ChpxCache, from: u64, to: u64) -> Vec<String> {
let mut out = Vec::new();
let mut para = ParaAccum::default();
for cp in from..to {
let Some((ch, fc)) = self.char_at(cp) else {
break;
};
match ch {
'\r' | '\u{0007}' | '\u{000C}' => {
let text = para.plain().trim().to_string();
para.segments.clear();
para.field_stack.clear();
if !text.is_empty() {
out.push(text);
}
}
_ => self.push_char(&mut para, ch, cache.props(self.word, self.btec, fc)),
}
}
let text = para.plain().trim().to_string();
if !text.is_empty() {
out.push(text);
}
out
}
}
fn parse_textboxes(plc: &[u8], base: u64, len: u64) -> std::collections::HashMap<u32, (u64, u64)> {
let mut out = std::collections::HashMap::new();
if plc.len() < 4 {
return out;
}
let n = (plc.len() - 4) / (4 + 22);
for i in 0..n {
let (Some(a), Some(b)) = (u32_at(plc, i * 4), u32_at(plc, i * 4 + 4)) else {
break;
};
let Some(lid) = u32_at(plc, (n + 1) * 4 + i * 22 + 14) else {
break;
};
let (a, b) = (a as u64, (b as u64).min(len));
if lid != 0 && lid != u32::MAX && b > a {
out.insert(lid, (base + a, base + b));
}
}
out
}
struct HeaderBoxes {
anchors: Vec<(u64, u32)>,
stories: std::collections::HashMap<u32, (u64, u64)>,
}
fn header_footer_texts(
story: &Story,
cache: &mut ChpxCache,
plc: &[u8],
base: u64,
len: u64,
boxes: &HeaderBoxes,
) -> Vec<(bool, String)> {
let cps: Vec<u64> = plc
.chunks_exact(4)
.filter_map(|c| u32_at(c, 0).map(u64::from))
.collect();
let mut out: Vec<(bool, String)> = Vec::new();
let push = |out: &mut Vec<(bool, String)>, footer: bool, text: String| {
if !out.iter().any(|(f, t)| *f == footer && *t == text) {
out.push((footer, text));
}
};
let mut placed: std::collections::HashSet<u32> = Default::default();
let mut section = 6;
while section + 6 < cps.len() {
for (k, footer) in [
(1, false),
(4, false),
(0, false),
(3, true),
(5, true),
(2, true),
] {
let (a, b) = (cps[section + k], cps[section + k + 1].min(len));
if a >= b {
continue;
}
for &(cp, spid) in &boxes.anchors {
let Some(&(ta, tb)) = boxes.stories.get(&spid) else {
continue;
};
if (a..b).contains(&cp) && placed.insert(spid) {
for text in story.paragraphs(cache, ta, tb) {
push(&mut out, footer, text);
}
}
}
for text in story.paragraphs(cache, base + a, base + b) {
push(&mut out, footer, text);
}
}
section += 6;
}
let mut rest: Vec<(u64, u64)> = boxes
.stories
.iter()
.filter(|(spid, _)| !placed.contains(spid))
.map(|(_, &range)| range)
.collect();
rest.sort_unstable();
for (ta, tb) in rest {
for text in story.paragraphs(cache, ta, tb) {
push(&mut out, false, text);
}
}
out
}
fn note_texts(
story: &Story,
cache: &mut ChpxCache,
refs: &[u8],
txt: &[u8],
base: u64,
len: u64,
) -> Vec<String> {
if refs.len() < 4 || len == 0 {
return Vec::new();
}
let n = (refs.len() - 4) / 6;
(0..n)
.filter_map(|i| {
let a = u32_at(txt, i * 4)? as u64;
let b = (u32_at(txt, i * 4 + 4)? as u64).min(len);
(a < b).then(|| story.paragraphs(cache, base + a, base + b).join(" "))
})
.filter(|t| !t.is_empty())
.collect()
}
struct Fib<'a> {
ccp_text: u32,
ccp_ftn: u32,
ccp_hdd: u32,
ccp_atn: u32,
ccp_edn: u32,
ccp_txbx: u32,
ccp_hdr_txbx: u32,
fc_lcb: &'a [u8],
}
impl<'a> Fib<'a> {
const STSHF: usize = 1;
const PLCF_FND_REF: usize = 2;
const PLCF_FND_TXT: usize = 3;
const PLCF_HDD: usize = 11;
const PLCF_BTE_CHPX: usize = 12;
const PLCF_BTE_PAPX: usize = 13;
const STTBF_FFN: usize = 15;
const CLX: usize = 33;
const PLCSPA_MOM: usize = 40;
const PLCSPA_HDR: usize = 41;
const PLCF_END_REF: usize = 46;
const PLCF_END_TXT: usize = 47;
const DGG_INFO: usize = 50;
const PLCF_TXBX_TXT: usize = 56;
const PLCF_HDR_TXBX_TXT: usize = 58;
const PLF_LST: usize = 73;
const PLF_LFO: usize = 74;
fn parse(word: &'a [u8]) -> Result<Self, ConversionError> {
let short = |what: &str, need: usize| {
ConversionError::Parse(format!(
"doc: FIB truncated — {what} needs {need} bytes, WordDocument has {}",
word.len()
))
};
let csw = u16_at(word, 32).ok_or_else(|| short("csw", 34))? as usize;
let cslw_at = 34 + csw * 2;
let cslw = u16_at(word, cslw_at).ok_or_else(|| short("cslw", cslw_at + 2))? as usize;
let lw_at = cslw_at + 2;
if cslw < 4 {
return Err(ConversionError::Parse(format!(
"doc: FIB's fibRgLw has {cslw} entries, too few for ccpText"
)));
}
let ccp_text = u32_at(word, lw_at + 12).ok_or_else(|| short("ccpText", lw_at + 16))?;
let lw = |i: usize| {
if i < cslw {
u32_at(word, lw_at + i * 4).unwrap_or(0)
} else {
0
}
};
let cb_at = lw_at + cslw * 4;
let cb = u16_at(word, cb_at).ok_or_else(|| short("cbRgFcLcb", cb_at + 2))? as usize;
let blob_at = cb_at + 2;
let fc_lcb = word
.get(blob_at..blob_at + cb * 8)
.ok_or_else(|| short("fibRgFcLcbBlob", blob_at + cb * 8))?;
Ok(Self {
ccp_text,
ccp_ftn: lw(4),
ccp_hdd: lw(5),
ccp_atn: lw(7),
ccp_edn: lw(8),
ccp_txbx: lw(9),
ccp_hdr_txbx: lw(10),
fc_lcb,
})
}
fn fc_lcb(&self, index: usize) -> (usize, usize) {
let fc = u32_at(self.fc_lcb, index * 8).unwrap_or(0) as usize;
let lcb = u32_at(self.fc_lcb, index * 8 + 4).unwrap_or(0) as usize;
(fc, lcb)
}
fn part<'s>(&self, stream: &'s [u8], index: usize) -> &'s [u8] {
let (fc, lcb) = self.fc_lcb(index);
fc.checked_add(lcb)
.and_then(|end| stream.get(fc..end))
.unwrap_or(&[])
}
}
fn convert_word6(name: &str, word: &[u8], flags: u16) -> Result<DoclingDocument, ConversionError> {
let field = |o: usize, what: &str| {
u32_at(word, o).map(|v| v as usize).ok_or_else(|| {
ConversionError::Parse(format!(
"doc: Word 6/95 FIB truncated — no {what} at {o:#x} ({} bytes)",
word.len()
))
})
};
let fc_min = field(0x18, "fcMin")?;
let ccp_text = field(0x34, "ccpText")? as u64;
let unicode = flags & 0x1000 != 0;
let pieces = if flags & 0x0004 != 0 {
let fc_clx = field(0x160, "fcClx")?;
let lcb_clx = field(0x164, "lcbClx")?;
let clx = fc_clx
.checked_add(lcb_clx)
.and_then(|end| word.get(fc_clx..end))
.ok_or_else(|| {
ConversionError::Parse(format!(
"doc: piece table (fcClx {fc_clx}, lcbClx {lcb_clx}) lies outside the \
WordDocument stream ({} bytes)",
word.len()
))
})?;
parse_piece_table(clx, Some(unicode)).ok_or_else(|| {
ConversionError::Parse(format!(
"doc: bad piece table (fcClx {fc_clx}, lcbClx {lcb_clx} in WordDocument)"
))
})?
} else {
vec![Piece {
cp_start: 0,
cp_end: ccp_text,
fc: fc_min as u64,
compressed: !unicode,
}]
};
let mut doc = DoclingDocument::new(name);
let mut builder = NodeBuilder::new(ListTables::default());
let mut para = ParaAccum::default();
let mut cp: u64 = 0;
'pieces: for piece in &pieces {
for i in 0..piece.cp_end.saturating_sub(piece.cp_start) {
if cp >= ccp_text {
break 'pieces;
}
cp += 1;
match piece_char(word, piece, i) {
'\r' | '\u{0007}' | '\u{000C}' => {
para.finish('\r', ParaProps::default(), &[], &mut builder, &mut doc);
}
ch => para.push(ch, CharFmt::default()),
}
}
}
para.finish('\r', ParaProps::default(), &[], &mut builder, &mut doc);
builder.flush(&mut doc);
Ok(doc)
}
pub(crate) fn is_word2(data: &[u8]) -> bool {
matches!(u16_at(data, 0), Some(0xA59B | 0xA59C | 0xA5DB))
&& u16_at(data, 2).is_some_and(|n| (1..=63).contains(&n))
&& u32_at(data, 0x18).is_some_and(|fc_min| (fc_min as usize) < data.len())
}
fn convert_word2(name: &str, data: &[u8]) -> Result<DoclingDocument, ConversionError> {
let flags = u16_at(data, 0x0A).unwrap_or(0);
if flags & 0x0100 != 0 {
return Err(ConversionError::Parse("doc: document is encrypted".into()));
}
if flags & 0x0004 != 0 {
return Err(ConversionError::Parse(
"doc: fast-saved (complex) Word 1.x/2.0 file — its text is in pieces this reader \
does not follow; open it in Word and save without Fast Save"
.into(),
));
}
let field = |o: usize, what: &str| {
u32_at(data, o).map(|v| v as usize).ok_or_else(|| {
ConversionError::Parse(format!(
"doc: Word 2.0 FIB truncated — no {what} at {o:#x} ({} bytes)",
data.len()
))
})
};
let fc_min = field(0x18, "fcMin")?;
let fc_mac = field(0x1C, "fcMac")?;
let ccp_text = field(0x34, "ccpText")?;
let end = fc_min.saturating_add(ccp_text).min(fc_mac).min(data.len());
let text = data.get(fc_min..end).unwrap_or(&[]);
let plc = |o: usize| -> &[u8] {
let (Some(fc), Some(cb)) = (u32_at(data, o), u16_at(data, o + 4)) else {
return &[];
};
let (fc, cb) = (fc as usize, cb as usize);
fc.checked_add(cb)
.and_then(|e| data.get(fc..e))
.unwrap_or(&[])
};
let chpx_plc = plc(0xA0);
let papx_plc = plc(0xA6);
let styles = word2_styles(data);
let mut doc = DoclingDocument::new(name);
let mut builder = NodeBuilder::new(ListTables::default());
let mut para = ParaAccum::default();
let mut i = 0usize;
while i < text.len() {
let fc = fc_min + i;
let b = text[i];
if b == 0x0D {
let next = text.get(i + 1).copied();
i += if matches!(next, Some(0x0A | 0x07)) {
2
} else {
1
};
let mut props = ParaProps::default();
let papx = word2_fkp(data, papx_plc, fc as u64);
props.istd = papx.and_then(|p| p.first()).copied().unwrap_or(0) as u16;
if next == Some(0x07) {
let (in_table, ttp) = papx.map(word2_table_flags).unwrap_or((true, false));
props.in_table = in_table || !ttp;
props.ttp = ttp;
}
let mark = if next == Some(0x07) { '\u{0007}' } else { '\r' };
para.finish(mark, props, &styles, &mut builder, &mut doc);
continue;
}
let fmt = word2_fkp(data, chpx_plc, fc as u64)
.and_then(|chpx| chpx.first())
.map_or(CharFmt::default(), |&flags| CharFmt {
bold: flags & 0x01 != 0,
italic: flags & 0x02 != 0,
..CharFmt::default()
});
para.push(cp1252(b), fmt);
i += 1;
}
para.finish('\r', ParaProps::default(), &styles, &mut builder, &mut doc);
builder.flush(&mut doc);
Ok(doc)
}
const WORD2_STC_HEADING1: u8 = 254;
const WORD2_STC_HEADING9: u8 = 246;
fn word2_styles(data: &[u8]) -> Vec<StyleDef> {
let none = StyleDef {
sti: 0x0FFF,
outline: None,
};
let heading_sti = |stc: u8| {
(WORD2_STC_HEADING9..=WORD2_STC_HEADING1)
.contains(&stc)
.then(|| (255 - stc) as u16)
};
let mut styles = vec![none; 256];
for (stc, def) in styles.iter_mut().enumerate() {
if let Some(sti) = heading_sti(stc as u8) {
def.sti = sti;
}
}
let (Some(fc), Some(cb)) = (u32_at(data, 0x5E), u16_at(data, 0x62)) else {
return styles;
};
let Some(stsh) = data.get(fc as usize..(fc as usize).saturating_add(cb as usize)) else {
return styles;
};
let Some(cstc_std) = u16_at(stsh, 0) else {
return styles;
};
let mut pos = 2usize;
for _ in 0..3 {
let Some(cb_block) = u16_at(stsh, pos) else {
return styles;
};
pos = pos.saturating_add(cb_block.max(2) as usize);
}
let Some(imac) = u16_at(stsh, pos) else {
return styles;
};
pos += 2;
for stcp in 0..imac as usize {
let stc = stcp.wrapping_sub(cstc_std as usize) & 255;
let (Some(&_next), Some(&base)) = (stsh.get(pos), stsh.get(pos + 1)) else {
break;
};
pos += 2;
if styles[stc].sti == 0x0FFF && base as usize != stc {
if let Some(sti) = heading_sti(base) {
styles[stc].sti = sti;
}
}
}
styles
}
fn word2_fkp<'a>(data: &'a [u8], plc: &[u8], fc: u64) -> Option<&'a [u8]> {
let n = plc.len().checked_sub(4)? / 6;
let pn = (0..n).find_map(|k| {
let start = u32_at(plc, k * 4)? as u64;
let end = u32_at(plc, (k + 1) * 4)? as u64;
(start <= fc && fc < end).then(|| u16_at(plc, (n + 1) * 4 + k * 2))?
})?;
let page = data.get(pn as usize * 512..pn as usize * 512 + 512)?;
let crun = *page.last()? as usize;
let fcs_end = (crun + 1) * 4;
let run = (0..crun).find(|&r| {
let start = u32_at(page, r * 4).unwrap_or(u32::MAX) as u64;
let end = u32_at(page, (r + 1) * 4).unwrap_or(0) as u64;
start <= fc && fc < end
})?;
let bx = *page.get(fcs_end + run)? as usize;
if bx == 0 {
return None;
}
let cb = *page.get(bx * 2)? as usize;
page.get(bx * 2 + 1..bx * 2 + 1 + cb)
}
fn word2_table_flags(papx: &[u8]) -> (bool, bool) {
let (mut in_table, mut ttp) = (false, false);
let mut i = 7usize;
while i < papx.len() {
match papx[i] {
0x11 => {
in_table = true;
i += 1;
}
0x18 => {
in_table |= papx.get(i + 1).is_some_and(|&v| v != 0);
i += 2;
}
0x19 => {
ttp |= papx.get(i + 1).is_some_and(|&v| v != 0);
i += 2;
}
_ => break,
}
}
(in_table, ttp)
}
struct Piece {
cp_start: u64,
cp_end: u64,
fc: u64,
compressed: bool,
}
fn piece_char(word: &[u8], piece: &Piece, i: u64) -> char {
if piece.compressed {
let b = word.get((piece.fc + i) as usize).copied().unwrap_or(0);
cp1252(b)
} else {
let o = (piece.fc + 2 * i) as usize;
let u = u16_at(word, o).unwrap_or(0xFFFD);
char::from_u32(u as u32).unwrap_or('\u{FFFD}')
}
}
fn piece_fc(piece: &Piece, i: u64) -> u64 {
if piece.compressed {
piece.fc + i
} else {
piece.fc + 2 * i
}
}
fn parse_piece_table(clx: &[u8], word6: Option<bool>) -> Option<Vec<Piece>> {
let mut pos = 0usize;
loop {
match clx.get(pos)? {
0x01 => {
let cb = u16_at(clx, pos + 1)? as usize;
pos += 3 + cb;
}
0x02 => {
let lcb = u32_at(clx, pos + 1)? as usize;
let plc = clx.get(pos + 5..(pos + 5).checked_add(lcb)?)?;
let n = (lcb.checked_sub(4)?) / 12;
let mut pieces = Vec::with_capacity(n);
for i in 0..n {
let cp_start = u32_at(plc, i * 4)? as u64;
let cp_end = u32_at(plc, (i + 1) * 4)? as u64;
let pcd = (n + 1) * 4 + i * 8;
let fc_raw = u32_at(plc, pcd + 2)?;
let compressed = match word6 {
Some(unicode) => !unicode,
None => fc_raw & 0x4000_0000 != 0,
};
let fc = if word6.is_some() {
fc_raw as u64
} else if compressed {
((fc_raw & 0x3FFF_FFFF) / 2) as u64
} else {
fc_raw as u64
};
pieces.push(Piece {
cp_start,
cp_end,
fc,
compressed,
});
}
return Some(pieces);
}
_ => return None,
}
}
}
#[derive(Default, Clone, Copy)]
struct ParaProps {
istd: u16,
in_table: bool,
ttp: bool,
ilfo: u16,
ilvl: u8,
outline: Option<u8>,
}
fn paragraph_props(word: &[u8], bte: &[u8], fc: u64) -> ParaProps {
let mut props = ParaProps::default();
let Some(n) = bte.len().checked_sub(4).map(|l| l / 8) else {
return props;
};
if n == 0 {
return props;
}
let mut pn = None;
for i in 0..n {
let lo = u32_at(bte, i * 4).unwrap_or(u32::MAX) as u64;
let hi = u32_at(bte, (i + 1) * 4).unwrap_or(0) as u64;
if fc >= lo && fc < hi {
pn = u32_at(bte, (n + 1) * 4 + i * 4);
break;
}
}
let Some(pn) = pn else { return props };
let page_off = (pn & 0x003F_FFFF) as usize * 512;
let Some(page) = word.get(page_off..page_off + 512) else {
return props;
};
let crun = page[511] as usize;
if crun == 0 || (crun + 1) * 4 + crun * 13 > 511 {
return props;
}
let mut run = None;
for j in 0..crun {
let lo = u32_at(page, j * 4).unwrap_or(u32::MAX) as u64;
let hi = u32_at(page, (j + 1) * 4).unwrap_or(0) as u64;
if fc >= lo && fc < hi {
run = Some(j);
break;
}
}
let Some(j) = run else { return props };
let b_offset = page[(crun + 1) * 4 + j * 13] as usize;
if b_offset == 0 {
return props; }
let mut o = b_offset * 2;
let Some(&cb) = page.get(o) else { return props };
let grpprl_len = if cb == 0 {
o += 2;
page.get(b_offset * 2 + 1).map(|&c| c as usize * 2)
} else {
o += 1;
Some(cb as usize * 2 - 1)
};
let Some(len) = grpprl_len else { return props };
let Some(grpprl) = page.get(o..(o + len).min(512)) else {
return props;
};
if grpprl.len() < 2 {
return props;
}
props.istd = u16::from_le_bytes([grpprl[0], grpprl[1]]);
apply_pap_sprms(&grpprl[2..], &mut props);
props
}
fn apply_pap_sprms(mut sprms: &[u8], props: &mut ParaProps) {
while sprms.len() >= 2 {
let sprm = u16::from_le_bytes([sprms[0], sprms[1]]);
sprms = &sprms[2..];
let spra = sprm >> 13;
let operand_len = match spra {
0 | 1 => 1,
2 | 4 | 5 => 2,
3 => 4,
7 => 3,
_ => {
match sprms.first() {
Some(&cb) => 1 + cb as usize,
None => return,
}
}
};
if sprms.len() < operand_len {
return;
}
match sprm {
0x2416 => props.in_table = sprms[0] != 0, 0x2417 => props.ttp = sprms[0] != 0, 0x460B => props.ilfo = u16::from_le_bytes([sprms[0], sprms[1]]), 0x260A => props.ilvl = sprms[0], 0x2640 => props.outline = Some(sprms[0]), _ => {}
}
sprms = &sprms[operand_len..];
}
}
#[derive(Clone, Copy)]
struct StyleDef {
sti: u16,
outline: Option<u8>,
}
fn parse_stsh(stsh: &[u8]) -> Vec<StyleDef> {
let none = StyleDef {
sti: 0x0FFF,
outline: None,
};
let Some(cb_stshi) = u16_at(stsh, 0) else {
return Vec::new();
};
let Some(cstd) = u16_at(stsh, 2) else {
return Vec::new();
};
let cb_std_base = u16_at(stsh, 4).unwrap_or(10) as usize;
let mut styles = Vec::with_capacity(cstd as usize);
let mut pos = 2 + cb_stshi as usize;
for _ in 0..cstd {
let Some(cb_std) = u16_at(stsh, pos) else {
break;
};
pos += 2;
let mut def = none;
if cb_std >= 2 {
let std = stsh.get(pos..pos + cb_std as usize).unwrap_or(&[]);
def.sti = u16_at(std, 0).map(|w| w & 0x0FFF).unwrap_or(0x0FFF);
let sgc = u16_at(std, 2).map(|w| w & 0x0F).unwrap_or(0);
if sgc == 1 {
if let Some(cch) = u16_at(std, cb_std_base) {
let mut upx = cb_std_base + 2 + (cch as usize + 1) * 2;
upx += upx & 1;
if let Some(cb_upx) = u16_at(std, upx) {
let g = std.get(upx + 2..upx + 2 + cb_upx as usize).unwrap_or(&[]);
if g.len() >= 2 {
let mut props = ParaProps::default();
apply_pap_sprms(&g[2..], &mut props);
def.outline = props.outline;
}
}
}
}
}
styles.push(def);
pos += cb_std as usize;
pos += pos & 1;
}
styles
}
#[derive(Default, Clone, Copy, PartialEq, Eq)]
struct CharFmt {
bold: bool,
italic: bool,
pic_fc: Option<u32>,
ftc_ascii: Option<u16>,
ftc_other: Option<u16>,
symbol: Option<(u16, u16)>,
}
#[derive(Default)]
struct ChpxCache {
lo: u64,
hi: u64,
fmt: CharFmt,
}
impl ChpxCache {
fn props(&mut self, word: &[u8], btec: &[u8], fc: u64) -> CharFmt {
if fc >= self.lo && fc < self.hi {
return self.fmt;
}
let (fmt, lo, hi) = char_props(word, btec, fc);
self.lo = lo;
self.hi = hi;
self.fmt = fmt;
fmt
}
}
fn char_props(word: &[u8], btec: &[u8], fc: u64) -> (CharFmt, u64, u64) {
let fmt = CharFmt::default();
let Some(n) = btec.len().checked_sub(4).map(|l| l / 8) else {
return (fmt, fc, fc + 1);
};
if n == 0 {
return (fmt, fc, fc + 1);
}
let mut pn = None;
for i in 0..n {
let lo = u32_at(btec, i * 4).unwrap_or(u32::MAX) as u64;
let hi = u32_at(btec, (i + 1) * 4).unwrap_or(0) as u64;
if fc >= lo && fc < hi {
pn = u32_at(btec, (n + 1) * 4 + i * 4);
break;
}
}
let Some(pn) = pn else {
return (fmt, fc, fc + 1);
};
let page_off = (pn & 0x003F_FFFF) as usize * 512;
let Some(page) = word.get(page_off..page_off + 512) else {
return (fmt, fc, fc + 1);
};
let crun = page[511] as usize;
if crun == 0 || (crun + 1) * 4 + crun > 511 {
return (fmt, fc, fc + 1);
}
for j in 0..crun {
let lo = u32_at(page, j * 4).unwrap_or(u32::MAX) as u64;
let hi = u32_at(page, (j + 1) * 4).unwrap_or(0) as u64;
if fc < lo || fc >= hi {
continue;
}
let b = page[(crun + 1) * 4 + j] as usize;
let mut out = CharFmt::default();
if b != 0 {
if let Some(&cb) = page.get(b * 2) {
if let Some(grpprl) = page.get(b * 2 + 1..(b * 2 + 1 + cb as usize).min(512)) {
apply_chp_sprms(grpprl, &mut out);
}
}
}
return (out, lo, hi);
}
(fmt, fc, fc + 1)
}
fn apply_chp_sprms(mut sprms: &[u8], fmt: &mut CharFmt) {
while sprms.len() >= 2 {
let sprm = u16::from_le_bytes([sprms[0], sprms[1]]);
sprms = &sprms[2..];
let operand_len = match sprm >> 13 {
0 | 1 => 1,
2 | 4 | 5 => 2,
3 => 4,
7 => 3,
_ => match sprms.first() {
Some(&cb) => 1 + cb as usize,
None => return,
},
};
if sprms.len() < operand_len {
return;
}
match sprm {
0x0835 => fmt.bold = sprms[0] == 1 || sprms[0] == 0x81, 0x0836 => fmt.italic = sprms[0] == 1 || sprms[0] == 0x81, 0x4A4F => fmt.ftc_ascii = Some(u16::from_le_bytes([sprms[0], sprms[1]])), 0x4A51 => fmt.ftc_other = Some(u16::from_le_bytes([sprms[0], sprms[1]])), 0x6A09 => {
fmt.symbol = Some((
u16::from_le_bytes([sprms[0], sprms[1]]),
u16::from_le_bytes([sprms[2], sprms[3]]),
))
}
0x6A03 => {
fmt.pic_fc = Some(u32::from_le_bytes([sprms[0], sprms[1], sprms[2], sprms[3]]))
}
_ => {}
}
sprms = &sprms[operand_len..];
}
}
fn inline_picture(data: &[u8], pic_fc: u32) -> Option<PictureImage> {
let base = pic_fc as usize;
let lcb = u32::from_le_bytes(data.get(base..base + 4)?.try_into().ok()?) as usize;
let cb_header = u16::from_le_bytes(data.get(base + 4..base + 6)?.try_into().ok()?) as usize;
let mm = u16::from_le_bytes(data.get(base + 6..base + 8)?.try_into().ok()?);
let mut start = base + cb_header;
if mm == 0x0066 {
let cch = *data.get(start)? as usize;
start += 1 + cch;
}
let body = data.get(start..base + lcb.max(cb_header))?;
officeart::first_blip(body, 0)
}
fn parse_sttbf_ffn(sttb: &[u8]) -> Vec<String> {
let mut names = Vec::new();
let Some(&c_data) = u16_at(sttb, 0).as_ref() else {
return names;
};
let (c_data, mut pos) = if c_data == 0xFFFF {
(u16_at(sttb, 2).unwrap_or(0) as usize, 6)
} else {
(c_data as usize, 4)
};
let cb_extra = u16_at(sttb, pos - 2).unwrap_or(0) as usize;
for _ in 0..c_data {
let Some(&cch) = sttb.get(pos) else { break };
let Some(ffn) = sttb.get(pos + 1..pos + 1 + cch as usize) else {
break;
};
let name: String = ffn
.get(39..)
.unwrap_or(&[])
.chunks_exact(2)
.map(|c| u16::from_le_bytes([c[0], c[1]]))
.take_while(|&u| u != 0)
.map(|u| char::from_u32(u as u32).unwrap_or('\u{FFFD}'))
.collect();
names.push(name);
pos += 1 + cch as usize + cb_extra;
}
names
}
fn parse_plcf_spa(plc: &[u8]) -> Vec<(u64, u32)> {
let Some(n) = plc.len().checked_sub(4).map(|l| l / 30) else {
return Vec::new();
};
(0..n)
.filter_map(|i| {
let cp = u32_at(plc, i * 4)? as u64;
let spid = u32_at(plc, (n + 1) * 4 + i * 26)?;
Some((cp, spid))
})
.collect()
}
fn spa_shape_at(spa: &[(u64, u32)], cp: u64) -> Option<u32> {
spa.iter()
.find(|(acp, _)| *acp == cp)
.map(|(_, spid)| *spid)
}
struct Drawings {
shape_pib: std::collections::HashMap<u32, u32>,
groups: std::collections::HashMap<u32, Vec<u32>>,
blips: Vec<Option<PictureImage>>,
}
impl Drawings {
fn parse(dgg: &[u8], delay_stream: &[u8]) -> Self {
let mut out = Self {
shape_pib: std::collections::HashMap::new(),
groups: std::collections::HashMap::new(),
blips: Vec::new(),
};
let mut pos = 0usize;
while pos + 8 <= dgg.len() {
let rec_type = u16::from_le_bytes([dgg[pos + 2], dgg[pos + 3]]);
let len = u32::from_le_bytes([dgg[pos + 4], dgg[pos + 5], dgg[pos + 6], dgg[pos + 7]])
as usize;
if (rec_type & 0xFF00) == 0xF000 && pos + 8 + len <= dgg.len() {
out.walk(&dgg[pos..pos + 8 + len], delay_stream, 0);
pos += 8 + len;
} else {
pos += 1;
}
}
out
}
fn sp_spid(sp_body: &[u8]) -> Option<u32> {
officeart::Records::new(sp_body)
.find(|(h, _)| h.rec_type == 0xF00A)
.and_then(|(_, b)| {
b.get(..4)
.map(|x| u32::from_le_bytes([x[0], x[1], x[2], x[3]]))
})
}
fn walk(&mut self, body: &[u8], delay: &[u8], depth: usize) {
if depth > 16 {
return;
}
for (h, b) in officeart::Records::new(body) {
match h.rec_type {
0xF007 => {
let embedded = b.get(36..).and_then(|tail| officeart::first_blip(tail, 0));
let img = embedded.or_else(|| {
let fo = u32::from_le_bytes(b.get(28..32)?.try_into().ok()?) as usize;
let rec = delay.get(fo..)?;
officeart::Records::new(rec)
.next()
.filter(|(rh, _)| officeart::is_blip(rh.rec_type))
.and_then(|(rh, rb)| officeart::decode_blip(&rh, rb))
});
self.blips.push(img);
}
0xF003 => {
let mut frame = None;
let mut members = Vec::new();
for (h2, b2) in officeart::Records::new(b) {
match h2.rec_type {
0xF004 => {
let spid = Self::sp_spid(b2);
let is_frame = officeart::Records::new(b2)
.any(|(h3, _)| h3.rec_type == 0xF009);
match (is_frame, frame) {
(true, None) => frame = spid,
_ => members.extend(spid),
}
}
0xF003 => {
let nested_frame = officeart::Records::new(b2)
.find(|(h3, _)| h3.rec_type == 0xF004)
.and_then(|(_, b3)| Self::sp_spid(b3));
members.extend(nested_frame);
}
_ => {}
}
}
if let Some(frame) = frame {
self.groups.insert(frame, members);
}
self.walk(b, delay, depth + 1);
}
0xF004 => {
let mut spid = None;
let mut pib = None;
for (h2, b2) in officeart::Records::new(b) {
match h2.rec_type {
0xF00A => {
spid = b2
.get(..4)
.map(|x| u32::from_le_bytes([x[0], x[1], x[2], x[3]]));
}
0xF00B => {
for e in 0..h2.instance as usize {
let o = e * 6;
let Some(entry) = b2.get(o..o + 6) else { break };
let id = u16::from_le_bytes([entry[0], entry[1]]) & 0x3FFF;
if id == 260 {
pib = Some(u32::from_le_bytes([
entry[2], entry[3], entry[4], entry[5],
]));
}
}
}
_ => {}
}
}
if let (Some(spid), Some(pib)) = (spid, pib) {
self.shape_pib.insert(spid, pib);
}
}
_ if h.version == 0xF => self.walk(b, delay, depth + 1),
_ => {}
}
}
}
fn shape_pictures(&self, spid: u32, depth: usize) -> Vec<Option<PictureImage>> {
if depth > 8 {
return Vec::new();
}
if let Some(pib) = self.shape_pib.get(&spid) {
let img = pib
.checked_sub(1)
.and_then(|ix| self.blips.get(ix as usize))
.cloned()
.flatten();
return vec![img];
}
match self.groups.get(&spid) {
Some(members) => members
.iter()
.flat_map(|&m| self.shape_pictures(m, depth + 1))
.collect(),
None => Vec::new(),
}
}
}
#[derive(Clone)]
enum LvlPart {
Level(u8),
Text(String),
}
#[derive(Clone, Default)]
struct LvlInfo {
nfc: u8,
start: u32,
template: Vec<LvlPart>,
}
#[derive(Default, Clone)]
struct ListTables {
lfo_lsids: Vec<u32>,
lists: std::collections::HashMap<u32, Vec<LvlInfo>>,
}
impl ListTables {
fn parse(lst_tail: &[u8], lcb_lst: usize, plflfo: &[u8]) -> Self {
let plflst = lst_tail;
let mut out = Self::default();
let c_lst = u16_at(plflst, 0).unwrap_or(0) as usize;
let mut lstfs = Vec::with_capacity(c_lst);
for i in 0..c_lst {
let base = 2 + i * 28;
let Some(lsid) = u32_at(plflst, base) else {
break;
};
let simple = plflst.get(base + 26).is_some_and(|&f| f & 0x01 != 0);
lstfs.push((lsid, if simple { 1usize } else { 9usize }));
}
let mut pos = (2 + c_lst * 28).max(lcb_lst);
'lists: for (lsid, nlvl) in lstfs {
let mut lvls = Vec::with_capacity(nlvl);
for _ in 0..nlvl {
let Some(start) = u32_at(plflst, pos) else {
break 'lists;
};
let Some(&nfc) = plflst.get(pos + 4) else {
break 'lists;
};
let cb_chpx = plflst.get(pos + 24).copied().unwrap_or(0) as usize;
let cb_papx = plflst.get(pos + 25).copied().unwrap_or(0) as usize;
pos += 28 + cb_papx + cb_chpx;
let cch = u16_at(plflst, pos).unwrap_or(0) as usize;
pos += 2;
let mut template: Vec<LvlPart> = Vec::new();
for i in 0..cch {
let Some(u) = u16_at(plflst, pos + i * 2) else {
break;
};
if u <= 8 {
template.push(LvlPart::Level(u as u8));
} else if let Some(c) = char::from_u32(u as u32) {
match template.last_mut() {
Some(LvlPart::Text(t)) => t.push(c),
_ => template.push(LvlPart::Text(c.to_string())),
}
}
}
pos += cch * 2;
lvls.push(LvlInfo {
nfc,
start,
template,
});
}
out.lists.insert(lsid, lvls);
}
let lfo_mac = u32_at(plflfo, 0).unwrap_or(0) as usize;
for i in 0..lfo_mac {
match u32_at(plflfo, 4 + i * 16) {
Some(lsid) => out.lfo_lsids.push(lsid),
None => break,
}
}
out
}
fn level(&self, ilfo: u16, ilvl: u8) -> Option<LvlInfo> {
let lsid = *self.lfo_lsids.get(ilfo.checked_sub(1)? as usize)?;
let lvls = self.lists.get(&lsid)?;
lvls.get(ilvl as usize).or_else(|| lvls.first()).cloned()
}
fn level_start(&self, ilfo: u16, ilvl: u8) -> u64 {
self.level(ilfo, ilvl).map(|l| l.start as u64).unwrap_or(1)
}
}
#[derive(Default)]
struct ParaAccum {
segments: Vec<(String, CharFmt)>,
pictures: Vec<(bool, Option<PictureImage>)>,
field_stack: Vec<bool>, textboxes: Vec<Vec<Node>>,
}
impl ParaAccum {
fn add_picture(&mut self, image: Option<PictureImage>) {
let before_text = self.segments.iter().all(|(t, _)| t.trim().is_empty());
self.pictures.push((before_text, image));
}
fn push(&mut self, ch: char, fmt: CharFmt) {
match ch {
'\u{0013}' => self.field_stack.push(false),
'\u{0014}' => {
if let Some(top) = self.field_stack.last_mut() {
*top = true;
}
}
'\u{0015}' => {
self.field_stack.pop();
}
'\u{0001}' | '\u{0002}' | '\u{0005}' | '\u{0008}' => {}
'\u{000B}' => self.keep('\n', fmt), '\u{001E}' => self.keep('-', fmt), '\u{001F}' => {} _ => self.keep(ch, fmt),
}
}
fn keep(&mut self, ch: char, fmt: CharFmt) {
if !self.field_stack.iter().all(|&r| r) {
return;
}
match self.segments.last_mut() {
Some((text, last)) if *last == fmt => text.push(ch),
_ => self.segments.push((ch.to_string(), fmt)),
}
}
fn plain(&self) -> String {
self.segments.iter().map(|(t, _)| t.as_str()).collect()
}
fn markdown(&self) -> String {
let mut out = String::new();
for (text, fmt) in &self.segments {
if !fmt.bold && !fmt.italic {
out.push_str(text);
continue;
}
let core = text.trim();
if core.is_empty() {
out.push_str(text);
continue;
}
let lead = &text[..text.len() - text.trim_start().len()];
let trail = &text[text.trim_end().len()..];
let marker = match (fmt.bold, fmt.italic) {
(true, true) => "***",
(true, false) => "**",
(false, true) => "*",
_ => unreachable!(),
};
out.push_str(lead);
out.push_str(marker);
out.push_str(core);
out.push_str(marker);
out.push_str(trail);
}
out
}
fn finish(
&mut self,
mark: char,
props: ParaProps,
stis: &[StyleDef],
builder: &mut NodeBuilder,
doc: &mut DoclingDocument,
) {
let plain = self.plain();
let markdown = self.markdown();
let pictures = std::mem::take(&mut self.pictures);
let textboxes = std::mem::take(&mut self.textboxes);
self.segments.clear();
self.field_stack.clear();
builder.paragraph(plain, markdown, pictures, textboxes, mark, props, stis, doc);
}
}
#[derive(Default)]
struct NodeBuilder {
rows: Vec<Vec<String>>,
cells: Vec<String>,
cell_text: String,
last_ilfo: Option<u16>,
run_base: Option<u8>,
lists: ListTables,
counters: std::collections::HashMap<(u16, u8), u64>,
}
impl NodeBuilder {
fn new(lists: ListTables) -> Self {
Self {
lists,
..Self::default()
}
}
#[allow(clippy::too_many_arguments)]
fn paragraph(
&mut self,
plain: String,
markdown: String,
pictures: Vec<(bool, Option<PictureImage>)>,
textboxes: Vec<Vec<Node>>,
mark: char,
props: ParaProps,
stis: &[StyleDef],
doc: &mut DoclingDocument,
) {
if props.ttp {
if !self.cell_text.is_empty() {
self.cells.push(std::mem::take(&mut self.cell_text));
}
if !self.cells.is_empty() {
self.rows.push(std::mem::take(&mut self.cells));
}
return;
}
if props.in_table {
if !self.cell_text.is_empty() {
self.cell_text.push_str("\n\n");
}
self.cell_text
.push_str(markdown.trim_end_matches('\u{0007}'));
for text in textboxes.iter().flatten().filter_map(node_text) {
if !self.cell_text.is_empty() {
self.cell_text.push_str("\n\n");
}
self.cell_text.push_str(&text);
}
if mark == '\u{0007}' {
self.cells.push(std::mem::take(&mut self.cell_text));
}
return;
}
self.flush(doc);
for children in textboxes {
doc.push(Node::Group {
label: "section".into(),
name: Some("textbox".into()),
layer: None,
children,
});
self.last_ilfo = None;
self.run_base = None;
}
let picture_node = |image: Option<PictureImage>| Node::Picture {
caption: None,
caption_href: None,
image,
classification: None,
caption_parent: Default::default(),
caption_location: None,
};
let plain = plain.trim().to_string();
let text = escape_text(markdown.trim());
if plain.is_empty() {
for (_, image) in pictures {
doc.push(picture_node(image));
}
return;
}
for (_, image) in pictures.iter().filter(|(before, _)| *before) {
doc.push(picture_node(image.clone()));
}
let after: Vec<_> = pictures
.into_iter()
.filter(|(before, _)| !before)
.map(|(_, image)| image)
.collect();
let style = stis.get(props.istd as usize).copied().unwrap_or(StyleDef {
sti: 0x0FFF,
outline: None,
});
let sti = style.sti;
let outline_heading = (!(1..=9).contains(&sti) && sti != 62)
.then_some(style.outline)
.flatten()
.filter(|l| *l <= 8);
if (1..=9).contains(&sti) || sti == 62 || outline_heading.is_some() {
let level = if sti == 62 {
1
} else if let Some(l) = outline_heading {
l + 2
} else {
sti as u8 + 1
};
doc.push(Node::Heading {
level,
text: escape_text(&plain),
});
self.last_ilfo = None;
self.run_base = None;
} else if props.ilfo != 0 {
let lvl = self.lists.level(props.ilfo, props.ilvl);
let numbered = lvl.as_ref().is_some_and(|l| l.nfc != 0x17 && l.nfc != 0xFF);
let first_in_list = self.last_ilfo != Some(props.ilfo);
let base = *self.run_base.get_or_insert(props.ilvl);
let level = props.ilvl.saturating_sub(base);
if numbered {
let start = self.lists.level_start(props.ilfo, props.ilvl);
let c = self
.counters
.entry((props.ilfo, props.ilvl))
.or_insert(start.saturating_sub(1));
*c += 1;
for ((f, l), v) in self.counters.iter_mut() {
if *f == props.ilfo && *l > props.ilvl {
*v = 0;
}
}
let marker = self.build_marker(props.ilfo, props.ilvl, lvl.as_ref());
let number = marker
.trim_end_matches(['.', ')'])
.rsplit(['.', ')'])
.next()
.and_then(|s| s.trim().parse::<u64>().ok())
.unwrap_or(1);
if cached_regex!(r"^\d+[.)]$").is_match(&marker) {
doc.push(Node::ListItem {
ordered: true,
number,
first_in_list,
text,
level,
marker: Some(marker),
location: None,
dclx: None,
href: None,
layer: None,
});
} else {
let dclx = Some(docling_core::ListItemDclx {
ordered: true,
marker: Some(marker.clone()),
text: text.clone(),
runs: Vec::new(),
});
doc.push(Node::ListItem {
ordered: false,
number,
first_in_list,
text: format!("{marker} {text}"),
level,
marker: None,
location: None,
dclx,
href: None,
layer: None,
});
}
} else {
doc.push(Node::ListItem {
ordered: false,
number: 0,
first_in_list,
text,
level,
marker: None,
location: None,
dclx: None,
href: None,
layer: None,
});
}
self.last_ilfo = Some(props.ilfo);
} else {
doc.push(Node::Paragraph { text });
self.last_ilfo = None;
self.run_base = None;
}
for image in after {
doc.push(picture_node(image));
}
}
fn flush(&mut self, doc: &mut DoclingDocument) {
if !self.cell_text.is_empty() {
self.cells.push(std::mem::take(&mut self.cell_text));
}
if !self.cells.is_empty() {
self.rows.push(std::mem::take(&mut self.cells));
}
if self.rows.is_empty() {
return;
}
let rows = std::mem::take(&mut self.rows);
let width = rows.iter().map(|r| r.len()).max().unwrap_or(0);
let rows: Vec<Vec<String>> = rows
.into_iter()
.map(|mut r| {
r.resize(width, String::new());
r
})
.collect();
doc.push(Node::Table(Table {
rows,
location: None,
structure: None,
cell_blocks: None,
cells: None,
caption: None,
caption_parent: Default::default(),
caption_location: None,
}));
self.last_ilfo = None;
self.run_base = None;
}
fn build_marker(&self, ilfo: u16, ilvl: u8, lvl: Option<&LvlInfo>) -> String {
let counter_at = |l: u8| -> u64 {
self.counters
.get(&(ilfo, l))
.copied()
.unwrap_or_else(|| self.lists.level_start(ilfo, l))
};
if let Some(lvl) = lvl {
let literal: String = lvl
.template
.iter()
.filter_map(|p| match p {
LvlPart::Text(t) => Some(t.as_str()),
LvlPart::Level(_) => None,
})
.collect();
let has_levels = lvl.template.iter().any(|p| matches!(p, LvlPart::Level(_)));
if has_levels
&& !literal
.trim_matches(|c: char| " .)(:[]".contains(c))
.is_empty()
{
return lvl
.template
.iter()
.map(|p| match p {
LvlPart::Level(l) => counter_at(*l).to_string(),
LvlPart::Text(t) => t.clone(),
})
.collect();
}
}
let parts: Vec<String> = (0..=ilvl).map(|l| counter_at(l).to_string()).collect();
parts.join(".") + "."
}
}
fn node_text(node: &Node) -> Option<String> {
let text = match node {
Node::Paragraph { text } | Node::Heading { text, .. } | Node::ListItem { text, .. } => {
text.clone()
}
Node::Table(t) => t
.rows
.iter()
.flatten()
.filter(|c| !c.trim().is_empty())
.cloned()
.collect::<Vec<_>>()
.join(" "),
Node::Group { children, .. } => children
.iter()
.filter_map(node_text)
.collect::<Vec<_>>()
.join(" "),
_ => return None,
};
(!text.trim().is_empty()).then_some(text)
}
fn u16_at(d: &[u8], o: usize) -> Option<u16> {
Some(u16::from_le_bytes(d.get(o..o + 2)?.try_into().ok()?))
}
fn u32_at(d: &[u8], o: usize) -> Option<u32> {
Some(u32::from_le_bytes(d.get(o..o + 4)?.try_into().ok()?))
}
pub(crate) fn cp1252(b: u8) -> char {
match b {
0x80 => '€',
0x82 => '‚',
0x83 => 'ƒ',
0x84 => '„',
0x85 => '…',
0x86 => '†',
0x87 => '‡',
0x88 => 'ˆ',
0x89 => '‰',
0x8A => 'Š',
0x8B => '‹',
0x8C => 'Œ',
0x8E => 'Ž',
0x91 => '\u{2018}',
0x92 => '\u{2019}',
0x93 => '\u{201C}',
0x94 => '\u{201D}',
0x95 => '•',
0x96 => '–',
0x97 => '—',
0x98 => '˜',
0x99 => '™',
0x9A => 'š',
0x9B => '›',
0x9C => 'œ',
0x9E => 'ž',
0x9F => 'Ÿ',
other => other as char,
}
}
#[cfg(test)]
mod tests {
use super::*;
fn fib_bytes(cb: u16) -> Vec<u8> {
let mut w = vec![0u8; 32];
w[0..2].copy_from_slice(&0xA5ECu16.to_le_bytes());
w.extend_from_slice(&14u16.to_le_bytes());
w.extend(std::iter::repeat_n(0, 28));
w.extend_from_slice(&22u16.to_le_bytes());
let mut lw = vec![0u8; 88];
lw[12..16].copy_from_slice(&1234u32.to_le_bytes());
w.extend(lw);
w.extend_from_slice(&cb.to_le_bytes());
let mut blob = vec![0u8; cb as usize * 8];
if cb > 33 {
blob[264..268].copy_from_slice(&7u32.to_le_bytes());
blob[268..272].copy_from_slice(&9u32.to_le_bytes());
}
w.extend(blob);
w
}
#[test]
fn fib_reads_through_the_declared_counts() {
let w = fib_bytes(0x5D);
let fib = Fib::parse(&w).expect("well-formed FIB");
assert_eq!(fib.ccp_text, 1234);
assert_eq!(fib.fc_lcb(Fib::CLX), (7, 9));
assert_eq!(u32_at(&w, 418), Some(7));
assert_eq!(u32_at(&w, 422), Some(9));
let short = fib_bytes(20);
let fib = Fib::parse(&short).unwrap();
assert_eq!(fib.fc_lcb(Fib::CLX), (0, 0));
assert!(fib.part(&[1, 2, 3], Fib::CLX).is_empty());
}
#[test]
fn truncated_fib_is_an_error_naming_the_field() {
let w = fib_bytes(0x5D);
for (cut, field) in [
(20, "csw"),
(40, "cslw"),
(70, "ccpText"),
(153, "cbRgFcLcb"),
(300, "fibRgFcLcbBlob"),
] {
let err = Fib::parse(&w[..cut])
.err()
.expect("truncated FIB must fail");
assert!(err.to_string().contains(field), "cut {cut}: {err}");
}
}
#[test]
fn part_survives_an_overflowing_range() {
let mut w = fib_bytes(0x5D);
w[154 + 264..154 + 272].copy_from_slice(&[0xFF; 8]);
let fib = Fib::parse(&w).unwrap();
assert!(fib.part(&[0u8; 16], Fib::CLX).is_empty());
}
#[test]
fn unsupported_word_version_is_named() {
let mut data = std::fs::read(concat!(
env!("CARGO_MANIFEST_DIR"),
"/../../tests/data/doc/sources/docx_lists.doc"
))
.unwrap();
let at = (512..data.len())
.step_by(512)
.find(|&o| data[o..o + 2] == [0xEC, 0xA5])
.expect("FIB sector");
data[at..at + 4].copy_from_slice(&[0x34, 0x12, 0x40, 0x00]);
let src = SourceDocument::from_bytes("x.doc", InputFormat::Doc, data);
let Err(err) = DocBackend.convert(&src) else {
panic!("an unknown Word version must be rejected");
};
let err = err.to_string();
assert!(
err.contains("wIdent 0x1234") && err.contains("nFib 64"),
"{err}"
);
}
use crate::InputFormat;
fn fixture(name: &str) -> SourceDocument {
let path = format!(
"{}/../../tests/data/doc/sources/{name}",
env!("CARGO_MANIFEST_DIR")
);
let bytes = std::fs::read(&path).expect("fixture exists");
SourceDocument::from_bytes(name, InputFormat::Doc, bytes)
}
#[test]
fn text_boxes_headers_and_footnotes_are_read() {
let path = format!(
"{}/tests/data/doc/sources/doc_textbox_header_footnote.doc",
env!("CARGO_MANIFEST_DIR")
);
let bytes = std::fs::read(&path).expect("fixture exists");
let doc = DocBackend
.convert(&SourceDocument::from_bytes(
"doc_textbox_header_footnote.doc",
InputFormat::Doc,
bytes,
))
.expect("converts");
let md = doc.export_to_markdown();
assert!(md.contains("MARKTB1"), "{md}");
assert!(md.contains("| MARKTB2 | b2"), "{md}");
assert!(!md.contains("MARKHDR") && !md.contains("MARKFN"), "{md}");
let json: serde_json::Value = serde_json::from_str(&doc.export_to_json()).unwrap();
let furniture: Vec<(String, String)> = json["texts"]
.as_array()
.unwrap()
.iter()
.filter(|t| t["content_layer"] == "furniture")
.map(|t| {
(
t["label"].as_str().unwrap().to_string(),
t["text"].as_str().unwrap().to_string(),
)
})
.collect();
assert_eq!(
furniture,
vec![
("page_header".to_string(), "MARKHDR".to_string()),
("footnote".to_string(), "MARKFN".to_string()),
]
);
let groups = json["groups"].as_array().unwrap();
assert_eq!(groups[0]["name"], "textbox");
assert_eq!(json["body"]["children"][1]["$ref"], "#/groups/0");
}
#[test]
fn header_text_boxes_are_header_furniture() {
let path = format!(
"{}/tests/data/doc/sources/doc_header_textbox.doc",
env!("CARGO_MANIFEST_DIR")
);
let bytes = std::fs::read(&path).expect("fixture exists");
let doc = DocBackend
.convert(&SourceDocument::from_bytes(
"h.doc",
InputFormat::Doc,
bytes,
))
.expect("converts");
let md = doc.export_to_markdown();
assert_eq!(md.trim(), "Body paragraph before header test.");
let json: serde_json::Value = serde_json::from_str(&doc.export_to_json()).unwrap();
let furniture: Vec<(&str, &str)> = json["texts"]
.as_array()
.unwrap()
.iter()
.filter(|t| t["content_layer"] == "furniture")
.map(|t| (t["label"].as_str().unwrap(), t["text"].as_str().unwrap()))
.collect();
assert_eq!(furniture, [("page_header", "MARK_HEADER_TEXTBOX_TEXT")]);
}
#[test]
fn extracts_headings_lists_and_paragraphs() {
let doc = DocBackend
.convert(&fixture("docx_lists.doc"))
.expect("converts");
let headings = doc
.nodes
.iter()
.filter(|n| matches!(n, Node::Heading { .. }))
.count();
let lists = doc
.nodes
.iter()
.filter(|n| matches!(n, Node::ListItem { .. }))
.count();
assert!(headings > 0, "expected headings: {:?}", doc.nodes);
assert!(lists > 0, "expected list items: {:?}", doc.nodes);
}
#[test]
fn extracts_tables_with_cells() {
let doc = DocBackend
.convert(&fixture("docx_rich_tables_01.doc"))
.expect("converts");
let tables: Vec<&Table> = doc
.nodes
.iter()
.filter_map(|n| match n {
Node::Table(t) => Some(t),
_ => None,
})
.collect();
assert!(!tables.is_empty(), "expected tables: {:?}", doc.nodes);
assert!(tables[0].rows.len() > 1 && tables[0].rows[0].len() > 1);
}
#[test]
fn underscores_are_escaped_in_prose() {
let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR"))
.join("tests/data/doc/sources/doc_underscores.doc");
let bytes = std::fs::read(path).expect("fixture");
let src = SourceDocument::from_bytes("u", InputFormat::Doc, bytes);
let doc = DocBackend.convert(&src).expect("converts");
assert_eq!(
doc.export_to_markdown(),
"Identifiers: OBJ\\_DIR, MER\\_BAX, LO\\_SNO02, plain text.\n"
);
let json: serde_json::Value = serde_json::from_str(&doc.export_to_json()).unwrap();
assert_eq!(
json["texts"][0]["text"],
"Identifiers: OBJ_DIR, MER_BAX, LO_SNO02, plain text."
);
}
#[test]
fn garbage_is_an_error_not_a_panic() {
let src = SourceDocument::from_bytes("x.doc", InputFormat::Doc, vec![0u8; 128]);
assert!(DocBackend.convert(&src).is_err());
}
#[test]
fn extracts_inline_and_floating_images_with_bytes() {
let doc = DocBackend
.convert(&fixture("docx_grouped_images.doc"))
.expect("converts");
let images: Vec<_> = doc
.nodes
.iter()
.filter_map(|n| match n {
Node::Picture { image, .. } => Some(image),
_ => None,
})
.collect();
assert_eq!(images.len(), 6, "expected 6 pictures: {:?}", doc.nodes);
assert!(
images.iter().all(|i| i.is_some()),
"every picture should carry decoded bytes"
);
let img = images[0].as_ref().unwrap();
assert!(img.width > 0 && img.height > 0 && !img.data.is_empty());
}
#[test]
fn word2_flat_file_converts_like_its_docx_resave() {
let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR"))
.join("tests/data/doc/sources/word2_pcjs.doc");
let bytes = std::fs::read(path).expect("fixture");
assert!(is_word2(&bytes));
assert_eq!(crate::sniff::detect(&bytes), Some(InputFormat::Doc));
let src = SourceDocument::from_bytes("word2", InputFormat::Doc, bytes.clone());
let doc = DocBackend.convert(&src).expect("converts");
let md = doc.export_to_markdown();
assert_eq!(
md.trim_end(),
"This IS a dummy word document\n\n1.\tsdfsdf\n\n2.\tsdfsdf\n\n3.\tlorem\n\n4.\tipsum\n\n\
**BOLD TEXT**\n\n***Italic text***\n\n***Underligned***\n\n\
| Animals | testa | testb |\n\
|-------------|-----------|-----------|\n\
| cat | loremtab | ipsumtab |\n\
| dog | testacell | testbcell |\n\
| empty cells | | |"
);
let mut complex = bytes;
complex[0x0A] |= 0x04;
let err = DocBackend
.convert(&SourceDocument::from_bytes("c", InputFormat::Doc, complex))
.unwrap_err()
.to_string();
assert!(err.contains("fast-saved"), "{err}");
}
#[test]
fn word2_heading_styles_are_headings() {
let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR"))
.join("tests/data/doc/sources/word2_heading_styles.doc");
let bytes = std::fs::read(path).expect("fixture");
let styles = word2_styles(&bytes);
assert_eq!(styles[254].sti, 1);
assert_eq!(styles[253].sti, 2);
assert_eq!(styles[246].sti, 9);
for stc in [0u8, 1, 2, 222, 255] {
assert_eq!(styles[stc as usize].sti, 0x0FFF, "stc {stc}");
}
let src = SourceDocument::from_bytes("w2", InputFormat::Doc, bytes);
let md = DocBackend
.convert(&src)
.expect("converts")
.export_to_markdown();
let headings: Vec<&str> = md.lines().filter(|l| l.starts_with('#')).collect();
assert_eq!(
headings,
["## HEADING NUMBER ONE TITLE", "## HEADING NUMBER"],
"{md}"
);
let mut fib = vec![0u8; 0x70];
fib[0..2].copy_from_slice(&0xA5DBu16.to_le_bytes());
let stsh_at = fib.len() as u32;
let mut stsh = Vec::new();
stsh.extend_from_slice(&0u16.to_le_bytes()); for block in [vec![0xFFu8; 9], vec![0xFF; 9], vec![0xFF; 9]] {
stsh.extend_from_slice(&((block.len() + 2) as u16).to_le_bytes());
stsh.extend_from_slice(&block);
}
stsh.extend_from_slice(&9u16.to_le_bytes()); for stc in 0u8..9 {
let base = match stc {
7 => 252,
8 => 0,
_ => stc,
};
stsh.extend_from_slice(&[0, base]); }
fib[0x5E..0x62].copy_from_slice(&stsh_at.to_le_bytes());
fib[0x62..0x64].copy_from_slice(&(stsh.len() as u16).to_le_bytes());
fib.extend_from_slice(&stsh);
let styles = word2_styles(&fib);
assert_eq!(styles[7].sti, 3);
assert_eq!(styles[8].sti, 0x0FFF);
assert_eq!(styles[0].sti, 0x0FFF);
assert_eq!(styles[252].sti, 3);
assert_eq!(word2_styles(&[0xDB, 0xA5])[254].sti, 1);
}
#[test]
fn word2_papx_table_flags() {
let cell = [0, 0, 0, 0, 0, 0, 0, 0x11];
assert_eq!(word2_table_flags(&cell), (true, false));
let row_end = [0, 0, 0, 0, 0, 0, 0, 0x18, 1, 0x19, 1, 0x94, 0x6c];
assert_eq!(word2_table_flags(&row_end), (true, true));
let plain = [0, 0, 0, 0, 0, 0, 0];
assert_eq!(word2_table_flags(&plain), (false, false));
assert!(!is_word2(b"\xd0\xcf\x11\xe0 not word 2"));
}
}