Skip to main content

strypt_core/formats/
ooxml.rs

1//! Office Open XML — `.docx`, `.xlsx`, `.pptx`.
2//!
3//! Structurally unlike anything in Phase 1. A JPEG or a PNG is one file with metadata segments
4//! in it; an OOXML document is a ZIP archive of XML parts, described by `[Content_Types].xml`
5//! and wired together by relationship parts under `_rels/`. Its metadata is spread across at
6//! least five places, and one of them — the pictures the author pasted in — is a set of
7//! complete image files carrying whatever their cameras wrote.
8//!
9//! The ZIP layer lives in [`crate::container::zip`] and is written rather than imported
10//! (ADR-0028). The descent into embedded images is bounded at exactly one level and to images
11//! only (ADR-0029). What this handler removes, keeps, and merely reports is ADR-0030.
12//!
13//! # Where the metadata is
14//!
15//! - `docProps/core.xml` — Dublin Core: `dc:creator`, `cp:lastModifiedBy`, `dcterms:created`,
16//!   `dcterms:modified`, `cp:revision`.
17//! - `docProps/app.xml` — the producing application and version, `Company`, `Manager`, and
18//!   `TotalTime`, which is cumulative editing minutes and therefore a record of how long
19//!   somebody worked on a document and, across saves, when.
20//! - `docProps/custom.xml` — arbitrary named properties. Document management systems write
21//!   internal matter numbers and usernames here.
22//! - `docProps/thumbnail.*` — a rendered preview of the first page. It survives every kind of
23//!   redaction applied to the text, in the same way an Exif thumbnail survives cropping.
24//! - Revision-save identifiers, spread through the document body: `w:rsid*` attributes and the
25//!   `w:rsids` table in `settings.xml`. Each one marks an editing session, and two documents
26//!   sharing an rsid were edited in the same session on the same machine.
27//! - `w14:paraId` and `w14:textId`, which are per-paragraph identifiers stable across saves and
28//!   across copies of a document.
29//! - Author names and timestamps on tracked changes and comments, sitting inline in the body.
30//!
31//! # Removing a part is not enough
32//!
33//! `[Content_Types].xml` still declares a removed part's type and `_rels/.rels` still points at
34//! it, and a document referencing parts that are not there is invalid. Word offers to repair
35//! it, which for a user trying not to draw attention to a document is a worse outcome than a
36//! slightly larger file. So both are rewritten, by deleting the byte ranges of the entries that
37//! referred to what went — never by re-serialising, which would change bytes that had no reason
38//! to change.
39//!
40//! # What is reported rather than removed
41//!
42//! The *content* of comments and tracked changes stays. Removing a tracked insertion means
43//! deciding whether the document accepts or rejects it, and that changes the document's words —
44//! `docs/PRD.md` §8.1 says the payload wins, and a document's visible text is its payload in
45//! the most direct sense there is. A tool that silently accepted every pending revision would
46//! hand a journalist a document that says something different from the one they reviewed.
47//!
48//! Their **author names, initials, and timestamps** are a different matter: those are metadata
49//! sitting on content, and removing them changes no words at all. They go.
50//!
51//! This is a recorded limitation, and for a document whose comments themselves must not be
52//! published, mat2 is the better recommendation — ADR-0012 requires saying so where it is true.
53
54use std::collections::{BTreeMap, BTreeSet};
55
56use crate::container::package::{self, Action, Decision, Embedded, Part, as_u64};
57use crate::container::zip::{self, Output};
58use crate::detect::Format;
59use crate::error::{MalformedDetail, Result, StryptError};
60use crate::formats::xml;
61use crate::formats::{MetadataHandler, ParseLimits, StripOptions, Stripped};
62use crate::report::{
63    Finding, InspectOptions, MetadataKind, MetadataReport, MetadataValue, Note, StripReport,
64};
65
66mod rules;
67
68/// The part that describes every other part's type. Required in every OOXML package.
69const CONTENT_TYPES: &str = "[Content_Types].xml";
70/// The package-level relationship part, which is where the properties parts are referenced from.
71const ROOT_RELS: &str = "_rels/.rels";
72
73/// Content types whose parts are metadata in their entirety and are removed whole.
74///
75/// Matched on the **content type, not the path**: the `docProps/` convention is what producers
76/// happen to do, and the content type is what the format actually guarantees. A producer that
77/// puts its core properties at `custom/props.xml` is unusual, not exempt.
78const PROPERTY_CONTENT_TYPES: [(&str, MetadataKind); 3] = [
79    (
80        "application/vnd.openxmlformats-package.core-properties+xml",
81        MetadataKind::PersonalIdentity,
82    ),
83    (
84        "application/vnd.openxmlformats-officedocument.extended-properties+xml",
85        MetadataKind::SoftwareFingerprint,
86    ),
87    (
88        "application/vnd.openxmlformats-officedocument.custom-properties+xml",
89        MetadataKind::PersonalIdentity,
90    ),
91];
92
93/// The relationship type of the package thumbnail (ECMA-376 Part 2, §10.1.4).
94///
95/// The thumbnail has no content-type override of its own — it is covered by the `Default` for
96/// its extension — so it is found through the relationship instead.
97const THUMBNAIL_RELATIONSHIP: &str =
98    "http://schemas.openxmlformats.org/package/2006/relationships/metadata/thumbnail";
99
100/// The main-document content type for each format this handler serves.
101///
102/// The macro-enabled variants are deliberately absent: they are refused at detection, because a
103/// `vbaProject.bin` is an OLE compound file that strypt cannot read, and reporting a document
104/// clean while a container inside it went unexamined is the failure this project exists to
105/// avoid.
106const MAIN_PART_TYPES: [(&str, Format); 3] = [
107    (
108        "application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml",
109        Format::Docx,
110    ),
111    (
112        "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml",
113        Format::Xlsx,
114    ),
115    (
116        "application/vnd.openxmlformats-officedocument.presentationml.presentation.main+xml",
117        Format::Pptx,
118    ),
119];
120
121/// Removal of metadata from Office Open XML documents.
122///
123/// One handler type serving three formats, instantiated once per format rather than branching
124/// internally, so that [`MetadataHandler::format`] keeps returning the format the registry
125/// dispatched on.
126#[derive(Debug, Clone, Copy)]
127pub struct OoxmlHandler {
128    format: Format,
129}
130
131impl OoxmlHandler {
132    /// The handler for `WordprocessingML` documents.
133    pub const DOCX: Self = Self {
134        format: Format::Docx,
135    };
136    /// The handler for `SpreadsheetML` workbooks.
137    pub const XLSX: Self = Self {
138        format: Format::Xlsx,
139    };
140    /// The handler for `PresentationML` presentations.
141    pub const PPTX: Self = Self {
142        format: Format::Pptx,
143    };
144}
145
146impl MetadataHandler for OoxmlHandler {
147    fn name(&self) -> &'static str {
148        self.format.id()
149    }
150
151    fn format(&self) -> Format {
152        self.format
153    }
154
155    fn inspect(&self, input: &[u8], options: &InspectOptions) -> Result<MetadataReport> {
156        // The same pass stripping uses, with the output discarded — so "everything `strip`
157        // removes is something `inspect` can see" holds by construction rather than by two code
158        // paths agreeing to stay in step. The pipeline's verification pass depends on it, and
159        // for this format it depends on it twice over: a document with a photograph in it would
160        // fail verification if the descent happened only on the strip side (ADR-0029).
161        let processed = process(input, self.format, options, &ParseLimits::default())?;
162        Ok(MetadataReport {
163            format: self.format,
164            findings: processed.findings,
165            notes: processed.notes,
166        })
167    }
168
169    fn strip(&self, input: &[u8], options: &StripOptions) -> Result<Stripped> {
170        let processed = process(input, self.format, &options.inspect, &options.limits)?;
171        Ok(Stripped {
172            report: StripReport {
173                format: self.format,
174                removed: processed.findings,
175                retained: Vec::new(),
176                notes: processed.notes,
177                input_bytes: as_u64(input.len()),
178                output_bytes: as_u64(processed.output.len()),
179            },
180            bytes: processed.output,
181        })
182    }
183}
184
185/// The result of one pass over a document.
186struct Processed {
187    findings: Vec<Finding>,
188    notes: Vec<Note>,
189    output: Vec<u8>,
190}
191
192/// Walk the package once and produce both the report and the sanitised archive.
193fn process(
194    input: &[u8],
195    format: Format,
196    options: &InspectOptions,
197    limits: &ParseLimits,
198) -> Result<Processed> {
199    let parts = package::read_parts(input, format, limits)?;
200
201    let types = ContentTypes::parse(&parts, format)?;
202    types.confirm_format(format)?;
203
204    let mut findings = Vec::new();
205    let mut notes = Vec::new();
206
207    let rels = Relationships::parse(&parts);
208    let dropped = parts_to_drop(&parts, &types, &rels);
209    let dead_rels = dead_relationships(&parts);
210
211    package::refuse_nested_containers(&parts, format, &mut notes)?;
212
213    let mut outputs: Vec<Output<'_>> = Vec::with_capacity(parts.len());
214    for part in &parts {
215        let decision = decide(part, &types, &dropped, &dead_rels, options, limits)?;
216        findings.extend(decision.findings);
217        notes.extend(decision.notes);
218        match decision.action {
219            Action::Copy => outputs.push(Output::Copied(part.entry.clone())),
220            Action::Drop => {}
221            Action::Rewrite(data) => outputs.push(Output::Rewritten {
222                name: part.entry.name.to_vec(),
223                data,
224                flags: part.entry.flags,
225            }),
226        }
227    }
228
229    findings.extend(package::container_findings(&parts));
230
231    let output = zip::write(&outputs).map_err(|e| e.into_strypt(format))?;
232    Ok(Processed {
233        findings,
234        notes,
235        output,
236    })
237}
238
239/// `[Content_Types].xml`, parsed into the two lookups the rest of this module needs.
240struct ContentTypes {
241    /// Part path (without its leading slash) to content type, from `Override` elements.
242    overrides: Vec<(String, String)>,
243    /// Lowercase extension to content type, from `Default` elements.
244    defaults: Vec<(String, String)>,
245}
246
247impl ContentTypes {
248    fn parse(parts: &[Part<'_>], format: Format) -> Result<Self> {
249        let part = parts
250            .iter()
251            .find(|p| p.name() == Some(CONTENT_TYPES))
252            .ok_or_else(|| malformed(format, MalformedDetail::MissingMarker))?;
253        let text = part
254            .text()
255            .ok_or_else(|| malformed(format, MalformedDetail::BrokenIndex))?;
256
257        let mut overrides = Vec::new();
258        let mut defaults = Vec::new();
259        for tag in xml::tags(text) {
260            match tag.name {
261                "Override" => {
262                    if let (Some(name), Some(kind)) =
263                        (tag.attribute("PartName"), tag.attribute("ContentType"))
264                    {
265                        overrides.push((normalise_part_name(name), kind.to_owned()));
266                    }
267                }
268                "Default" => {
269                    if let (Some(ext), Some(kind)) =
270                        (tag.attribute("Extension"), tag.attribute("ContentType"))
271                    {
272                        defaults.push((ext.to_ascii_lowercase(), kind.to_owned()));
273                    }
274                }
275                _ => {}
276            }
277        }
278        Ok(Self {
279            overrides,
280            defaults,
281        })
282    }
283
284    /// The content type declared for `part`, by override first and by extension second.
285    fn type_of(&self, part: &str) -> Option<&str> {
286        if let Some((_, kind)) = self.overrides.iter().find(|(name, _)| name == part) {
287            return Some(kind);
288        }
289        let ext = part.rsplit_once('.')?.1.to_ascii_lowercase();
290        self.defaults
291            .iter()
292            .find(|(candidate, _)| *candidate == ext)
293            .map(|(_, kind)| kind.as_str())
294    }
295
296    /// Refuse a package whose main part is not the one this handler was dispatched for.
297    ///
298    /// Detection decides the format by reading this same declaration, so a mismatch here means
299    /// the package changed underneath us or the two disagree — either way, guessing is how a
300    /// handler ends up confidently reporting on a file it does not understand.
301    fn confirm_format(&self, format: Format) -> Result<()> {
302        let declared = self.overrides.iter().find_map(|(_, kind)| {
303            MAIN_PART_TYPES
304                .iter()
305                .find(|(candidate, _)| candidate == kind)
306                .map(|(_, f)| *f)
307        });
308        if declared == Some(format) {
309            Ok(())
310        } else {
311            Err(malformed(format, MalformedDetail::MissingMarker))
312        }
313    }
314}
315
316/// The package-level relationships, which is where the thumbnail is named.
317#[derive(Default)]
318struct Relationships {
319    /// Targets of relationships whose type marks them as metadata rather than content.
320    metadata_targets: Vec<String>,
321    /// Relationships pointing outside the package, which frequently carry a local filesystem
322    /// path — an attached template on a user's desktop names that user.
323    external_targets: Vec<String>,
324}
325
326impl Relationships {
327    fn parse(parts: &[Part<'_>]) -> Self {
328        let Some(text) = parts
329            .iter()
330            .find(|p| p.name() == Some(ROOT_RELS))
331            .and_then(Part::text)
332        else {
333            return Self::default();
334        };
335        let mut rels = Self::default();
336        for tag in xml::tags(text) {
337            if tag.name != "Relationship" {
338                continue;
339            }
340            let Some(target) = tag.attribute("Target") else {
341                continue;
342            };
343            if tag.attribute("Type") == Some(THUMBNAIL_RELATIONSHIP) {
344                rels.metadata_targets.push(normalise_part_name(target));
345            }
346            if tag.attribute("TargetMode") == Some("External") {
347                rels.external_targets.push(target.to_owned());
348            }
349        }
350        rels
351    }
352}
353
354/// Every part path this pass will remove.
355fn parts_to_drop(
356    parts: &[Part<'_>],
357    types: &ContentTypes,
358    rels: &Relationships,
359) -> BTreeSet<String> {
360    let mut dropped: BTreeSet<String> = rels.metadata_targets.iter().cloned().collect();
361    for part in parts {
362        let Some(name) = part.name() else { continue };
363        let Some(kind) = types.type_of(name) else {
364            continue;
365        };
366        if PROPERTY_CONTENT_TYPES
367            .iter()
368            .any(|(candidate, _)| *candidate == kind)
369        {
370            dropped.insert(name.to_owned());
371        }
372    }
373    dropped
374}
375
376/// Decide about one part.
377fn decide(
378    part: &Part<'_>,
379    types: &ContentTypes,
380    dropped: &BTreeSet<String>,
381    dead_rels: &DeadRelationships,
382    options: &InspectOptions,
383    limits: &ParseLimits,
384) -> Result<Decision> {
385    let Some(name) = part.name() else {
386        // A part whose name is not UTF-8 cannot be one this handler knows, and cannot be
387        // referenced by any relationship, whose targets are text. Copied, and declared.
388        return Ok(Decision::unexamined(
389            "an entry whose name is not valid UTF-8",
390            part.entry.compressed.len(),
391        ));
392    };
393    if part.entry.is_directory() {
394        return Ok(Decision::copy());
395    }
396    // The two index parts, which have to stop referring to whatever went. They are handled
397    // here rather than in a pass of their own so that they keep their position in the archive —
398    // and, more to the point, so that they are written exactly once. Writing them in a second
399    // pass produced a package containing two `[Content_Types].xml` entries, which readers
400    // tolerate and which quietly broke byte-identical idempotence.
401    if name == CONTENT_TYPES || name == ROOT_RELS {
402        let Some(text) = part.text() else {
403            return Ok(Decision::copy());
404        };
405        let dereferenced = rules::drop_references(text, dropped);
406        let base = dereferenced.as_deref().unwrap_or(text);
407        let scrubbed = rules::scrub(base, name, dead_rels.for_part(name), options);
408        let action = match scrubbed.output.or(dereferenced) {
409            Some(rewritten) => Action::Rewrite(rewritten.into_bytes()),
410            None => Action::Copy,
411        };
412        return Ok(Decision {
413            action,
414            findings: scrubbed.findings,
415            notes: scrubbed.notes,
416        });
417    }
418
419    if dropped.contains(name) {
420        return Ok(Decision {
421            action: Action::Drop,
422            findings: property_findings(part, name, types, options),
423            notes: Vec::new(),
424        });
425    }
426
427    let Some(data) = part.data.as_deref() else {
428        return Ok(Decision::copy());
429    };
430
431    // An embedded image goes through the *same* handler the CLI uses on a loose file, one level
432    // deep and images only (ADR-0029). The descent itself lives in `container::package` so that
433    // there is exactly one of it.
434    if let Some(embedded) = package::embedded_image_format(data) {
435        return match package::strip_embedded_image(embedded, data, name, options, limits)? {
436            package::Embedded::Unchanged => Ok(Decision::copy()),
437            Embedded::Stripped {
438                bytes,
439                findings,
440                notes,
441            } => Ok(Decision {
442                action: Action::Rewrite(bytes),
443                findings,
444                notes,
445            }),
446        };
447    }
448
449    match part.text() {
450        Some(text) => Ok(scrub_part(text, name, dead_rels.for_part(name), options)),
451        // Not text, not an image strypt handles: a font, an audio clip, a binary blob.
452        None => Ok(Decision::unexamined(name, data.len())),
453    }
454}
455
456/// Report what a properties part held, before it is dropped.
457///
458/// The part goes whole either way. Naming its fields is what makes `strypt show` useful — "this
459/// document names an author and records 340 minutes of editing" is actionable, where "a
460/// properties part was removed" is not.
461fn property_findings(
462    part: &Part<'_>,
463    name: &str,
464    types: &ContentTypes,
465    options: &InspectOptions,
466) -> Vec<Finding> {
467    let kind = types
468        .type_of(name)
469        .and_then(|declared| {
470            PROPERTY_CONTENT_TYPES
471                .iter()
472                .find(|(candidate, _)| *candidate == declared)
473                .map(|(_, kind)| *kind)
474        })
475        .unwrap_or(MetadataKind::Other);
476
477    let Some(text) = part.text() else {
478        // The thumbnail, which is an image rather than XML, and anything else non-textual.
479        return vec![Finding::new(
480            MetadataKind::Thumbnail,
481            name.to_owned(),
482            as_u64(part.data.as_ref().map_or(0, Vec::len)),
483        )];
484    };
485
486    let mut findings = Vec::new();
487    for element in xml::elements_with_text(text) {
488        if element.text.trim().is_empty() {
489            continue;
490        }
491        findings.push(
492            Finding::new(
493                kind_of_property(element.name, kind),
494                name.to_owned(),
495                as_u64(element.text.len()),
496            )
497            .with_field(element.name.to_owned())
498            .with_value(options, || MetadataValue::Text(element.text.to_owned())),
499        );
500    }
501    if findings.is_empty() {
502        // An empty properties part is still a part that should not be published, and a report
503        // that said nothing about it would be a report claiming there was nothing there.
504        findings.push(Finding::new(kind, name.to_owned(), as_u64(text.len())));
505    }
506    findings
507}
508
509/// Classify a property element more precisely than its part's default.
510fn kind_of_property(element: &str, fallback: MetadataKind) -> MetadataKind {
511    match element {
512        "dc:creator" | "cp:lastModifiedBy" | "Manager" | "Company" => {
513            MetadataKind::PersonalIdentity
514        }
515        "dcterms:created" | "dcterms:modified" | "cp:lastPrinted" => MetadataKind::Timestamp,
516        "Application" | "AppVersion" | "Template" => MetadataKind::SoftwareFingerprint,
517        // Cumulative editing minutes, and the revision counter beside it. Neither names anyone
518        // and both narrow the field considerably — `docs/THREAT_MODEL.md` §4.7 on correlation.
519        "TotalTime" | "cp:revision" => MetadataKind::EditingHistory,
520        "cp:contentStatus" | "dc:description" | "cp:keywords" | "dc:subject" => {
521            MetadataKind::Comment
522        }
523        _ => fallback,
524    }
525}
526
527/// Scrub identifying attributes out of one XML part.
528fn scrub_part(
529    text: &str,
530    name: &str,
531    dead_rel_ids: &BTreeSet<String>,
532    options: &InspectOptions,
533) -> Decision {
534    let scrubbed = rules::scrub(text, name, dead_rel_ids, options);
535    let action = match scrubbed.output {
536        // Unchanged parts keep their original compressed bytes, so a document with nothing to
537        // remove differs from its input only in its entry headers.
538        None => Action::Copy,
539        Some(text) => Action::Rewrite(text.into_bytes()),
540    };
541    Decision {
542        action,
543        findings: scrubbed.findings,
544        notes: scrubbed.notes,
545    }
546}
547
548/// The relationships this package will remove, indexed by the part that has to change.
549///
550/// A relationship removal has two ends: the `Relationship` element in the `.rels` part, and
551/// whatever carried the `r:id` in the part that `.rels` belongs to. Both are collected here,
552/// before any part is rewritten, because a part cannot see the other end from inside its own
553/// scrub pass — and removing one end without the other leaves a reference pointing at nothing,
554/// which is the repair prompt this handler exists to avoid.
555#[derive(Default)]
556struct DeadRelationships {
557    by_part: BTreeMap<String, BTreeSet<String>>,
558}
559
560impl DeadRelationships {
561    /// The ids this part must stop referring to. Empty for a part with nothing to change.
562    fn for_part(&self, name: &str) -> &BTreeSet<String> {
563        static NONE: BTreeSet<String> = BTreeSet::new();
564        self.by_part.get(name).unwrap_or(&NONE)
565    }
566}
567
568/// Find every external relationship whose target is a local or network path.
569fn dead_relationships(parts: &[Part<'_>]) -> DeadRelationships {
570    let mut dead = DeadRelationships::default();
571    for part in parts {
572        let (Some(name), Some(text)) = (part.name(), part.text()) else {
573            continue;
574        };
575        if !name.to_ascii_lowercase().ends_with(".rels") {
576            continue;
577        }
578        let ids = rules::external_local_relationships(text);
579        if ids.is_empty() {
580            continue;
581        }
582        // The part a `.rels` belongs to: `word/_rels/settings.xml.rels` describes
583        // `word/settings.xml` (ECMA-376 Part 2, §9.3.2). The package-level `_rels/.rels` has no
584        // owning part, and `owner_part` returns `None` for it.
585        if let Some(owner) = owner_part(name) {
586            dead.by_part.entry(owner).or_default().extend(ids.clone());
587        }
588        dead.by_part.entry(name.to_owned()).or_default().extend(ids);
589    }
590    dead
591}
592
593/// The part a relationship part describes, or [`None`] for the package-level one.
594fn owner_part(rels_name: &str) -> Option<String> {
595    let stem = rels_name.strip_suffix(".rels")?;
596    let (directory, file) = match stem.rsplit_once('/') {
597        Some((directory, file)) => (directory, file),
598        None => ("", stem),
599    };
600    let base = directory.strip_suffix("_rels")?;
601    if file.is_empty() {
602        // `_rels/.rels`, which describes the package rather than a part.
603        return None;
604    }
605    Some(format!("{base}{file}"))
606}
607
608/// Strip a leading slash from a part path so that the content-types, relationship, and entry
609/// spellings of the same part compare equal.
610fn normalise_part_name(name: &str) -> String {
611    name.strip_prefix('/').unwrap_or(name).to_owned()
612}
613
614/// A malformed-structure error for this format.
615const fn malformed(format: Format, detail: MalformedDetail) -> StryptError {
616    StryptError::Malformed {
617        format,
618        offset: None,
619        detail,
620    }
621}