Skip to main content

strypt_core/formats/
pdf.rs

1//! PDF.
2//!
3//! The hardest of the Phase 1 formats, and the one where hidden data is least likely to be
4//! where you look for it. The Document Information Dictionary is the easy part and the part
5//! every tutorial covers; the leaks that matter live in XMP packets, per-object metadata,
6//! annotation authorship, and — above all — in objects left behind by incremental updates,
7//! which are physically present in the file and reachable with a hex editor long after the
8//! "current" version of the document stopped referring to them.
9//!
10//! # Full rewrite, not incremental patching (ADR-0020)
11//!
12//! A PDF can be edited by appending: the original bytes stay, and a new cross-reference
13//! section at the end says which objects supersede which. Nulling the Info dictionary with
14//! another such append is easy, fast, and preserves the file almost perfectly — and it leaves
15//! every previous author name exactly where it was, four kilobytes up the file. For this
16//! tool that is not a lesser fix, it is a silent failure: the user is told the document is
17//! clean and publishes it.
18//!
19//! So the document is parsed into its object graph, scrubbed, pruned to what the catalogue
20//! can actually reach, renumbered, and written out fresh. Everything unreachable — every
21//! superseded revision — is gone because it is never written, not because it was overwritten.
22//!
23//! The cost is honest and worth stating: the output is not byte-comparable with the input,
24//! object numbering changes, and files using features the rewrite cannot faithfully reproduce
25//! are refused rather than mangled. Refusing is the correct half of that trade.
26
27use lopdf::{Dictionary, Document, Object, ObjectId};
28
29use crate::container::package::{self, Embedded};
30use crate::detect::Format;
31use crate::error::{MalformedDetail, ResourceLimit, Result, StryptError};
32use crate::formats::xmp::name_of;
33use crate::formats::{MetadataHandler, ParseLimits, StripOptions, Stripped, xmp};
34use crate::report::{
35    Finding, InspectOptions, MetadataKind, MetadataReport, MetadataValue, Note, StripReport,
36};
37
38/// Removal of metadata from PDF documents.
39#[derive(Debug, Clone, Copy, Default)]
40pub struct PdfHandler;
41
42impl MetadataHandler for PdfHandler {
43    fn name(&self) -> &'static str {
44        Format::Pdf.id()
45    }
46
47    fn format(&self) -> Format {
48        Format::Pdf
49    }
50
51    fn inspect(&self, input: &[u8], options: &InspectOptions) -> Result<MetadataReport> {
52        // Inspection loads its own copy of the document and runs the *same* scrub that
53        // stripping does, then throws the result away. That is deliberate: it makes
54        // "everything `strip` removes is something `inspect` can see" true by construction
55        // rather than by two code paths agreeing to stay in step. The verification pass in
56        // `crate::pipeline` is only meaningful if that holds (`docs/ARCHITECTURE.md` §3).
57        let limits = ParseLimits::default();
58        let mut doc = load(input, &limits)?;
59        let scrubbed = scrub(&mut doc, input, options, &limits)?;
60        Ok(MetadataReport {
61            format: Format::Pdf,
62            findings: scrubbed.findings,
63            notes: scrubbed.notes,
64        })
65    }
66
67    fn strip(&self, input: &[u8], options: &StripOptions) -> Result<Stripped> {
68        let mut doc = load(input, &options.limits)?;
69        let mut scrubbed = scrub(&mut doc, input, &options.inspect, &options.limits)?;
70
71        // Drop every object the catalogue can no longer reach. This is the step that removes
72        // superseded revisions, and it has to come after scrubbing so that objects orphaned
73        // *by* the scrub — the Info dictionary, XMP streams — go with them.
74        let pruned = doc.prune_objects();
75        if !pruned.is_empty() {
76            scrubbed.notes.push(Note::OrphanedObjectsRemoved {
77                objects: pruned.len(),
78            });
79        }
80
81        // Renumber so that output depends only on the object graph, not on the numbering the
82        // input happened to use. Without this, two documents that scrub to the same content
83        // serialise differently, and the determinism invariant
84        // (`docs/TESTING_STRATEGY.md` §1) fails for no good reason.
85        renumber_stably(&mut doc)?;
86
87        // Collapse negative zero to zero for the same reason renumbering exists above: without
88        // it the output is not stable under a second strip. lopdf writes `Real(-0.0)` as `-0`,
89        // dropping the decimal point; re-parsing `-0` yields `Integer(0)`, which writes as `0`.
90        // So one strip of a file containing `-0.` differs from two, breaking the idempotence
91        // invariant (`docs/TESTING_STRATEGY.md` §1, invariant 3).
92        //
93        // Rewriting a number in the user's document needs justifying, since this handler
94        // otherwise refuses rather than repairs (ADR-0018). It is sound here because ISO
95        // 32000-1 §7.3.3 gives PDF numbers no signed zero: `-0` and `0` denote the same value,
96        // there is no operator that can distinguish them, and no renderer can. The alternative
97        // — refusing a valid file over a lost minus sign that changes nothing — costs the user
98        // their document to protect a distinction the format does not make.
99        normalise_negative_zero(&mut doc, &options.limits)?;
100
101        // Guarded for the same reason as `load`: serialisation walks a document graph built
102        // from hostile input, so it is dependency code on untrusted data just as parsing is.
103        // Failing here means nothing is written, which is the correct half of the trade.
104        let mut bytes = Vec::new();
105        crate::panic_guard::guard(
106            || {
107                doc.save_to(&mut bytes).map_err(|source| StryptError::Io {
108                    action: crate::error::IoAction::WritingOutput,
109                    source,
110                })
111            },
112            || StryptError::Malformed {
113                format: Format::Pdf,
114                offset: None,
115                detail: MalformedDetail::DependencyPanic,
116            },
117        )?;
118
119        // Nothing is handed back until the bytes just written have been read again and shown
120        // to be the document that was written. See `verify_round_trip`.
121        verify_round_trip(&doc, &bytes)?;
122
123        Ok(Stripped {
124            report: StripReport {
125                format: Format::Pdf,
126                removed: scrubbed.findings,
127                retained: Vec::new(),
128                notes: scrubbed.notes,
129                input_bytes: as_u64(input.len()),
130                output_bytes: as_u64(bytes.len()),
131            },
132            bytes,
133        })
134    }
135}
136
137/// Refuse output that does not read back as the document that was written.
138///
139/// The rewrite (ADR-0020) assumes serialising a parsed document and re-parsing it returns the
140/// same document. For a lenient parser on hostile input that assumption does not hold, and
141/// when it breaks it breaks silently: `lopdf` will accept a dictionary whose keys came out of
142/// mangled bytes — `/Annotst 1 /[^@018064665...` in the file that found this — and then write
143/// it back in a form it cannot itself read. The object is written, and disappears when the
144/// file is next opened.
145///
146/// That is the failure this guard exists for, and it is the dangerous kind. A document whose
147/// only `/Page` is lost on reload has a `/Pages` node still claiming `/Count 1` and a `/Kids`
148/// array pointing at an object that is no longer there. strypt wrote that file and reported
149/// success; a second strip then pruned what had become unreachable and wrote a 230-byte
150/// document with a dangling page reference, reporting success again. No metadata survived
151/// either pass, so the verification pass — which searches output for residual metadata — saw
152/// nothing wrong. It cannot: it is looking for what should be absent, not for what should
153/// still be present.
154///
155/// Checking a full structural equivalence would be a second implementation of the rewrite, so
156/// this checks the two properties whose failure means the output is not the document:
157///
158/// 1. **Every object written is present on reload.** This is what catches an object that
159///    serialised into something unparseable. Comparing ids alone is enough — an object that
160///    round-trips to a different id is caught by the same comparison.
161/// 2. **The page tree survives.** A document that had pages before writing and none after has
162///    lost the thing it exists to carry, even when every object id happens to match.
163///
164/// A failure refuses the file. That is the fail-closed answer (`CLAUDE.md` §3 rule 6): strypt
165/// cannot faithfully rewrite this document, so it declines to rather than handing back a
166/// broken one with a success report. Refusing costs the user a file that was already too
167/// damaged to survive a rewrite; the alternative cost them a document they believed was clean.
168fn verify_round_trip(written: &Document, bytes: &[u8]) -> Result<()> {
169    let reloaded = crate::panic_guard::guard(
170        || Document::load_mem(bytes).map_err(|e| map_parse_error(&e)),
171        || StryptError::Malformed {
172            format: Format::Pdf,
173            offset: None,
174            detail: MalformedDetail::DependencyPanic,
175        },
176    )
177    .map_err(|_| StryptError::Malformed {
178        format: Format::Pdf,
179        offset: None,
180        detail: MalformedDetail::NotRoundTrippable,
181    })?;
182
183    let written_ids: Vec<ObjectId> = written.objects.keys().copied().collect();
184    let reloaded_ids: Vec<ObjectId> = reloaded.objects.keys().copied().collect();
185    if written_ids != reloaded_ids {
186        return Err(StryptError::Malformed {
187            format: Format::Pdf,
188            offset: None,
189            detail: MalformedDetail::NotRoundTrippable,
190        });
191    }
192
193    // Only an emptied page tree is a failure, not an empty one: a document that had no pages
194    // to begin with is degenerate but not something this rewrite broke.
195    if written.page_iter().next().is_some() && reloaded.page_iter().next().is_none() {
196        return Err(StryptError::Malformed {
197            format: Format::Pdf,
198            offset: None,
199            detail: MalformedDetail::NotRoundTrippable,
200        });
201    }
202
203    Ok(())
204}
205
206/// How many times `renumber_stably` will renumber before refusing the document.
207///
208/// A well-formed file reaches its fixed point on the second call — the first assigns
209/// `1..=n`, the second confirms nothing moved. The degenerate case below needs a third.
210/// Four is that plus margin; a document still moving after four is not converging, and
211/// looping harder would only delay the refusal.
212const MAX_RENUMBER_ROUNDS: usize = 4;
213
214/// Renumber until the numbering stops changing, or refuse the document.
215///
216/// `lopdf::renumber_objects` is **not idempotent**, which matters because strypt's idempotence
217/// invariant (`docs/TESTING_STRATEGY.md` §1, invariant 3) is stated byte-for-byte on the *first*
218/// re-strip. Before renumbering sequentially, `renumber_objects_with` checks whether the page
219/// order matches ascending object ids and, if it does not, permutes the page objects so that it
220/// does (`lopdf` 0.44.0 `src/processor.rs`). That check reads the numbering the previous step
221/// produced, so one pass can leave a document that a second pass would reorder again.
222///
223/// A sustained fuzz run found the case where that is reachable: a document whose page tree is
224/// self-referential — object 2 is a `/Page` whose own `/Kids` array lists object 2 — so pruning
225/// and renumbering changed which objects `page_iter` yields and in which order. The first strip
226/// produced pages ordered `[3, 2]`, descending; the second saw the mismatch and swapped objects
227/// 2 and 3. Same length, same content, 145 bytes different. It settled from the third strip on,
228/// so this was never an endless flip — but "stable eventually" is not the invariant, and a user
229/// who strips a file twice must not get two different files.
230///
231/// Iterating to a fixed point fixes it by construction rather than by reasoning about a
232/// dependency's internals: the document is only serialised once renumbering has been shown to be
233/// a no-op on it, so re-loading and renumbering that output cannot move anything either.
234///
235/// This does not change the output of any document that was already stable — for those the
236/// second round is the confirmation that would have been skipped, not a second permutation.
237///
238/// Non-convergence is refused rather than accepted at whatever state the last round left, which
239/// is the fail-closed half of the trade (ADR-0018, and `CLAUDE.md` §3 rule 6). Emitting a file
240/// whose numbering strypt could not settle would mean handing the user output it cannot promise
241/// is reproducible.
242fn renumber_stably(doc: &mut Document) -> Result<()> {
243    for _ in 0..MAX_RENUMBER_ROUNDS {
244        // Both the id set and the page order have to be compared. The ids alone are not enough:
245        // the reordering step permutes which object holds which page while leaving the set of
246        // ids exactly as it was, so comparing ids only would report a fixed point on the very
247        // pass that moved something.
248        let ids_before: Vec<ObjectId> = doc.objects.keys().copied().collect();
249        let pages_before: Vec<ObjectId> = doc.page_iter().collect();
250
251        doc.renumber_objects();
252
253        // A page tree naming one object twice makes lopdf's page permutation map two ids onto
254        // one, dropping an object — in the file that found this, the document's only /Page.
255        if doc.objects.len() < ids_before.len() {
256            return Err(StryptError::Malformed {
257                format: Format::Pdf,
258                offset: None,
259                detail: MalformedDetail::NotRoundTrippable,
260            });
261        }
262
263        let ids_after: Vec<ObjectId> = doc.objects.keys().copied().collect();
264        let pages_after: Vec<ObjectId> = doc.page_iter().collect();
265
266        if ids_before == ids_after && pages_before == pages_after {
267            return Ok(());
268        }
269    }
270
271    Err(StryptError::Malformed {
272        format: Format::Pdf,
273        offset: None,
274        detail: MalformedDetail::CyclicReference,
275    })
276}
277
278/// Findings and caveats from one pass over a document.
279struct Scrubbed {
280    findings: Vec<Finding>,
281    notes: Vec<Note>,
282}
283
284/// Parse `input`, refusing documents this handler must not rewrite.
285fn load(input: &[u8], limits: &ParseLimits) -> Result<Document> {
286    // `lopdf` is third-party code parsing attacker-controlled bytes, and ADR-0006's no-panic
287    // rule does not reach inside it (ADR-0018). A sustained fuzz run found an integer overflow
288    // in its cross-reference parser, which with `overflow-checks` on in release meant the
289    // shipped binary aborted with a stack trace instead of refusing the file. Contained here
290    // so it reaches the user as an ordinary refusal; see `crate::panic_guard` for what that
291    // does and does not cover.
292    let doc = crate::panic_guard::guard(
293        || Document::load_mem(input).map_err(|e| map_parse_error(&e)),
294        || StryptError::Malformed {
295            format: Format::Pdf,
296            offset: None,
297            detail: MalformedDetail::DependencyPanic,
298        },
299    )?;
300
301    // Encrypted documents are refused rather than rewritten. lopdf can open one protected by
302    // an empty owner password, and it would be technically easy to emit a decrypted copy —
303    // but that hands the user a file with its protection quietly removed, which is a change
304    // to their document's security they did not ask for and would not necessarily notice.
305    // Fail closed and say so.
306    if doc.is_encrypted() || doc.was_encrypted() {
307        return Err(StryptError::Malformed {
308            format: Format::Pdf,
309            offset: None,
310            detail: MalformedDetail::UnsupportedFeature,
311        });
312    }
313
314    // A trailer with no /Root is refused rather than processed.
315    //
316    // ISO 32000-1 §7.5.5 makes /Root a required trailer entry: it names the document catalogue,
317    // which is the single root every other object hangs off. Without it the file has no defined
318    // entry point, and no viewer will open it.
319    //
320    // strypt used to accept such a file, and the result was worse than a refusal. The rewrite in
321    // `strip` walks reachable objects from the root and drops the rest (ADR-0020); with no root
322    // to walk from, which objects survive is not stable across runs. A fuzz run found a document
323    // whose second strip differed from its first — 609 bytes, then 485 — because the second pass
324    // dropped an annotation object that the page still referenced through /Annots. Renumbering
325    // then filled that slot with the catalogue, so the page's annotation array pointed at the
326    // document catalogue. strypt had introduced that corruption itself, while returning success
327    // both times.
328    //
329    // Refusing is the fail-closed answer and costs nothing real: a PDF this broken is not one
330    // the user can publish anyway.
331    if !doc.trailer.has(b"Root") {
332        return Err(StryptError::Malformed {
333            format: Format::Pdf,
334            offset: None,
335            detail: MalformedDetail::MissingMarker,
336        });
337    }
338
339    let too_many = u32::try_from(doc.objects.len()).map_or(true, |count| count > limits.max_items);
340    if too_many {
341        return Err(StryptError::LimitExceeded {
342            format: Format::Pdf,
343            limit: ResourceLimit::ItemCount,
344        });
345    }
346
347    // Every stream's declared /Length must be an integer that matches the content actually
348    // parsed out of it.
349    //
350    // ISO 32000-1 §7.3.8.2 requires /Length to be an integer giving the exact byte count
351    // between "stream" and "endstream". When it is not — a fuzz case reached here by writing
352    // "/Length 45." instead of "/Length 45" — lopdf 0.44 parses the object, keeps the
353    // malformed value in the dictionary, and stores *empty* content, because it cannot locate
354    // the stream's end. It reports no error while doing so.
355    //
356    // Left unchecked, the document reaches the rewriter with the stream's bytes already gone.
357    // strypt then writes a file whose /Length still claims 45 bytes over an empty stream, and
358    // reports "nothing to remove; wrote a clean copy" — a structurally invalid PDF, missing
359    // the user's page content, presented as a success. That is the §5.4 failure the whole
360    // design is arranged to avoid, and the reason it went unnoticed is that the verification
361    // pass looks for residual *metadata*, which an emptied stream has none of.
362    //
363    // Refusing costs nothing on real documents: across the synthetic corpus and all 23
364    // parseable PDFs of the real-producer corpus — pdfLaTeX, LibreOffice, Google Docs,
365    // Acrobat and ImageMagick output, compressed streams included — not one stream disagrees
366    // with its declared length.
367    for object in doc.objects.values() {
368        let Ok(stream) = object.as_stream() else {
369            continue;
370        };
371        let declared = stream
372            .dict
373            .get(b"Length")
374            .ok()
375            .and_then(|length| length.as_i64().ok())
376            .and_then(|length| usize::try_from(length).ok());
377        if declared != Some(stream.content.len()) {
378            return Err(StryptError::Malformed {
379                format: Format::Pdf,
380                offset: None,
381                detail: MalformedDetail::LengthOutOfRange,
382            });
383        }
384    }
385
386    Ok(doc)
387}
388
389/// Translate a parser failure into something a user can act on.
390///
391/// The mapping is coarse on purpose. "Your file is truncated" and "your file is not really a
392/// PDF" lead to different actions; which of thirty-four internal variants fired does not.
393fn map_parse_error(error: &lopdf::Error) -> StryptError {
394    use lopdf::Error as E;
395    let detail = match *error {
396        E::Parse(_) | E::Syntax(_) | E::IndirectObject { .. } | E::ObjectIdMismatch => {
397            MalformedDetail::UnexpectedMarker
398        }
399        E::Xref(_) | E::MissingXrefEntry | E::InvalidObjectStream(_) => {
400            MalformedDetail::BrokenIndex
401        }
402        E::InvalidOffset(_) | E::ObjectNotFound(_) | E::NumericCast(_) | E::TryFromInt(_) => {
403            MalformedDetail::LengthOutOfRange
404        }
405        E::ReferenceCycle(_) | E::ReferenceLimit => MalformedDetail::CyclicReference,
406        E::IO(_) => MalformedDetail::Truncated,
407        E::Decryption(_)
408        | E::InvalidPassword
409        | E::AlreadyEncrypted
410        | E::UnsupportedSecurityHandler(_)
411        | E::Unimplemented(_) => MalformedDetail::UnsupportedFeature,
412        _ => MalformedDetail::MissingMarker,
413    };
414    let offset = match *error {
415        E::InvalidOffset(at) | E::IndirectObject { offset: at } => u64::try_from(at).ok(),
416        _ => None,
417    };
418    StryptError::Malformed {
419        format: Format::Pdf,
420        offset,
421        detail,
422    }
423}
424
425/// Keys in the Document Information Dictionary, and what each one exposes.
426///
427/// `/Creator` names the application the document was *authored* in and `/Producer` the one
428/// that wrote the PDF — so a LaTeX paper typically confesses both its editor and its
429/// toolchain version here. Neither identifies a person alone; together with a timestamp and
430/// a font list they narrow the field a great deal (`docs/THREAT_MODEL.md` §4.7).
431const INFO_KEYS: &[(&[u8], MetadataKind)] = &[
432    (b"Author", MetadataKind::PersonalIdentity),
433    (b"Creator", MetadataKind::SoftwareFingerprint),
434    (b"Producer", MetadataKind::SoftwareFingerprint),
435    (b"CreationDate", MetadataKind::Timestamp),
436    (b"ModDate", MetadataKind::Timestamp),
437    (b"Title", MetadataKind::Comment),
438    (b"Subject", MetadataKind::Comment),
439    (b"Keywords", MetadataKind::Comment),
440    (b"Trapped", MetadataKind::Other),
441];
442
443/// Keys removed from *any* dictionary in the document, wherever they appear.
444///
445/// These are safe to remove anywhere because the PDF specification gives them one meaning
446/// each and nothing renders differently without them. `/PieceInfo` is the interesting one:
447/// it is a scratch area where an application may store whatever private state it likes
448/// between editing sessions, and what ends up in it is entirely up to that application.
449const GLOBAL_KEYS: &[(&[u8], MetadataKind)] = &[
450    (b"Metadata", MetadataKind::Other),
451    (b"PieceInfo", MetadataKind::EditingHistory),
452    (b"LastModified", MetadataKind::Timestamp),
453];
454
455/// Annotation subtypes whose `/T` entry is the annotating person's name.
456///
457/// This distinction matters. On a markup annotation `/T` is the author — exactly what we are
458/// here to remove. On a `/Widget`, which is how every interactive form field is drawn, `/T`
459/// is the *field name* that the form's logic and its saved data refer to. Stripping it would
460/// silently break the document, and breaking a user's file to protect them is not a trade
461/// this tool gets to make on their behalf without saying so.
462const MARKUP_ANNOTATION_SUBTYPES: &[&[u8]] = &[
463    b"Text",
464    b"FreeText",
465    b"Line",
466    b"Square",
467    b"Circle",
468    b"Polygon",
469    b"PolyLine",
470    b"Highlight",
471    b"Underline",
472    b"Squiggly",
473    b"StrikeOut",
474    b"Stamp",
475    b"Caret",
476    b"Ink",
477    b"FileAttachment",
478    b"Sound",
479    b"Movie",
480    b"Redact",
481];
482
483/// Walk the document, recording what is there and removing it.
484fn scrub(
485    doc: &mut Document,
486    raw: &[u8],
487    options: &InspectOptions,
488    limits: &ParseLimits,
489) -> Result<Scrubbed> {
490    let mut findings = Vec::new();
491    let mut notes = Vec::new();
492
493    let info_id = trailer_reference(doc, b"Info");
494
495    // Phase one reads. Every object is examined, not merely the ones the catalogue can reach:
496    // an orphan from a superseded revision is exactly the thing worth telling the user about,
497    // and it is invisible to a walk that starts at the root.
498    let object_ids: Vec<ObjectId> = doc.objects.keys().copied().collect();
499    for id in &object_ids {
500        let Some(object) = doc.objects.get(id) else {
501            continue;
502        };
503        examine_object(object, *id, info_id, options, limits, 0, &mut findings)?;
504    }
505    if doc.trailer.has(b"ID") {
506        // The file identifier is a pair of strings that stays stable across saves of the same
507        // document. It identifies nobody by itself and links every copy and every revision of
508        // the document to each other, which for a leaked draft is the whole question.
509        findings.push(Finding::new(
510            MetadataKind::DocumentIdentifier,
511            "trailer /ID",
512            0,
513        ));
514    }
515
516    // A PDF that has been saved more than once ends with more than one %%EOF. Counting them
517    // is cruder than walking the cross-reference chain and tells the user the thing that
518    // actually matters: earlier versions of this document were sitting inside it.
519    let revisions = count_revisions(raw);
520    if revisions > 1 {
521        notes.push(Note::IncrementalHistory {
522            revisions: revisions.saturating_sub(1),
523        });
524    }
525
526    // Phase two writes.
527    doc.trailer.remove(b"Info");
528    doc.trailer.remove(b"ID");
529    for id in &object_ids {
530        if let Some(object) = doc.objects.get_mut(id) {
531            remove_from_object(object, *id, info_id, limits, 0)?;
532        }
533    }
534
535    if object_ids
536        .iter()
537        .filter_map(|id| doc.objects.get(id))
538        .any(is_embedded_file_holder)
539    {
540        // strypt does not open embedded files. Recursing into them means recursing into
541        // arbitrary nested content, which is a zip-bomb-shaped problem that Phase 2 has to
542        // decide about explicitly. Until then the user is told, because an attachment
543        // carrying its own metadata inside a document reported as clean is precisely the
544        // over-trust this tool must not create (`docs/THREAT_MODEL.md` §5.6).
545        notes.push(Note::OutOfScopeContent {
546            location: "embedded file attachment".into(),
547        });
548    }
549
550    strip_embedded_images(doc, &object_ids, options, limits, &mut findings, &mut notes)?;
551
552    Ok(Scrubbed { findings, notes })
553}
554
555/// Strip the JPEGs a PDF carries, through the JPEG handler (ADR-0056, extending ADR-0029).
556///
557/// A `DCTDecode`-only stream is a JPEG file byte for byte (ISO 32000-1 §7.4.8), Exif included, so
558/// it needs no decoding to reach. Keyed on the filter rather than `/Subtype /Image` so that page
559/// thumbnails (`/Thumb`, §12.3.4) are covered too. JPEG 2000 and filter chains ending in a JPEG
560/// are copied with a note: reaching them means a JPX parser or inflating first.
561fn strip_embedded_images(
562    doc: &mut Document,
563    object_ids: &[ObjectId],
564    options: &InspectOptions,
565    limits: &ParseLimits,
566    findings: &mut Vec<Finding>,
567    notes: &mut Vec<Note>,
568) -> Result<()> {
569    for id in object_ids {
570        let Some(Object::Stream(stream)) = doc.objects.get_mut(id) else {
571            continue;
572        };
573        let Ok(filters) = stream.filters() else {
574            continue;
575        };
576        let name = format!("image object {}", id.0);
577        let is_jpeg = matches!(filters.as_slice(), [b"DCTDecode"])
578            && package::embedded_image_format(&stream.content) == Some(Format::Jpeg);
579        if !is_jpeg {
580            if filters
581                .iter()
582                .any(|f| *f == b"DCTDecode" || *f == b"JPXDecode")
583            {
584                notes.push(Note::UnparsedRegion {
585                    location: name,
586                    bytes: as_u64(stream.content.len()),
587                });
588            }
589            continue;
590        }
591        match package::strip_embedded_image(Format::Jpeg, &stream.content, &name, options, limits)?
592        {
593            Embedded::Unchanged => {}
594            Embedded::Stripped {
595                bytes,
596                findings: image_findings,
597                notes: image_notes,
598            } => {
599                findings.extend(image_findings);
600                notes.extend(image_notes);
601                stream.set_content(bytes);
602            }
603        }
604    }
605    Ok(())
606}
607
608/// Resolve a reference held in the trailer, if it is one.
609fn trailer_reference(doc: &Document, key: &[u8]) -> Option<ObjectId> {
610    doc.trailer
611        .get(key)
612        .ok()
613        .and_then(|o| o.as_reference().ok())
614}
615
616/// Count `%%EOF` markers, each of which terminates one revision of the document.
617fn count_revisions(raw: &[u8]) -> usize {
618    const EOF: &[u8] = b"%%EOF";
619    raw.windows(EOF.len()).filter(|w| *w == EOF).count()
620}
621
622/// Record everything identifying inside one object.
623fn examine_object(
624    object: &Object,
625    id: ObjectId,
626    info_id: Option<ObjectId>,
627    options: &InspectOptions,
628    limits: &ParseLimits,
629    depth: u32,
630    out: &mut Vec<Finding>,
631) -> Result<()> {
632    if depth > limits.max_depth {
633        return Err(StryptError::LimitExceeded {
634            format: Format::Pdf,
635            limit: ResourceLimit::Depth,
636        });
637    }
638    match object {
639        Object::Dictionary(dict) => {
640            if Some(id) == info_id {
641                examine_info(dict, options, out);
642            }
643            examine_dictionary(dict, id, info_id, options, limits, depth, out)?;
644        }
645        Object::Stream(stream) => {
646            if is_metadata_stream(&stream.dict) {
647                examine_xmp(stream, options, out);
648            }
649            examine_dictionary(&stream.dict, id, info_id, options, limits, depth, out)?;
650        }
651        Object::Array(items) => {
652            for item in items {
653                examine_object(
654                    item,
655                    id,
656                    info_id,
657                    options,
658                    limits,
659                    depth.saturating_add(1),
660                    out,
661                )?;
662            }
663        }
664        _ => {}
665    }
666    Ok(())
667}
668
669/// Record the Document Information Dictionary, including keys the specification never
670/// defined — applications add their own freely, and a custom key is no less identifying for
671/// being non-standard.
672fn examine_info(dict: &Dictionary, options: &InspectOptions, out: &mut Vec<Finding>) {
673    for (key, value) in dict {
674        let kind = INFO_KEYS
675            .iter()
676            .find(|(name, _)| *name == key.as_slice())
677            .map_or(MetadataKind::Other, |(_, kind)| *kind);
678        out.push(
679            Finding::new(kind, "/Info", value_size(value))
680                .with_field(name_of(key))
681                .with_value(options, || describe(value)),
682        );
683    }
684}
685
686/// Record the globally-removable keys, and annotation authorship.
687fn examine_dictionary(
688    dict: &Dictionary,
689    id: ObjectId,
690    info_id: Option<ObjectId>,
691    options: &InspectOptions,
692    limits: &ParseLimits,
693    depth: u32,
694    out: &mut Vec<Finding>,
695) -> Result<()> {
696    for (name, kind) in GLOBAL_KEYS {
697        if let Ok(value) = dict.get(name) {
698            out.push(
699                Finding::new(*kind, format!("/{}", name_of(name)), value_size(value))
700                    .with_field(name_of(name)),
701            );
702        }
703    }
704    if is_markup_annotation(dict) {
705        for (name, kind) in [
706            (&b"T"[..], MetadataKind::PersonalIdentity),
707            (&b"M"[..], MetadataKind::Timestamp),
708            (&b"CreationDate"[..], MetadataKind::Timestamp),
709            (&b"NM"[..], MetadataKind::DocumentIdentifier),
710        ] {
711            if let Ok(value) = dict.get(name) {
712                out.push(
713                    Finding::new(kind, "annotation", value_size(value))
714                        .with_field(name_of(name))
715                        .with_value(options, || describe(value)),
716                );
717            }
718        }
719    }
720    // Embedded-file parameters carry their own creation and modification dates, which survive
721    // every scrub aimed only at the containing document.
722    if let Ok(Object::Dictionary(params)) = dict.get(b"Params") {
723        for name in [&b"CreationDate"[..], &b"ModDate"[..], &b"CheckSum"[..]] {
724            if let Ok(value) = params.get(name) {
725                out.push(
726                    Finding::new(MetadataKind::Timestamp, "/Params", value_size(value))
727                        .with_field(name_of(name)),
728                );
729            }
730        }
731    }
732    for (_, value) in dict {
733        examine_object(
734            value,
735            id,
736            info_id,
737            options,
738            limits,
739            depth.saturating_add(1),
740            out,
741        )?;
742    }
743    Ok(())
744}
745
746/// Scan an XMP packet for the properties worth naming.
747///
748/// Only unfiltered packets are scanned. ISO 32000-1 §14.3.2 recommends that a metadata stream
749/// be left uncompressed precisely so it can be read without parsing the whole document, and
750/// in practice they almost always are. Refusing to inflate the rare compressed one avoids
751/// handing an attacker a decompression bomb in exchange for a slightly more detailed report —
752/// the packet is still found, still reported, and still removed either way.
753fn examine_xmp(stream: &lopdf::Stream, options: &InspectOptions, out: &mut Vec<Finding>) {
754    if stream.dict.has(b"Filter") {
755        out.push(
756            Finding::new(
757                MetadataKind::Other,
758                "XMP packet",
759                as_u64(stream.content.len()),
760            )
761            .with_field("Metadata (encoded)"),
762        );
763        return;
764    }
765    out.extend(xmp::scan(&stream.content, "XMP packet", options));
766}
767
768/// Remove everything [`examine_object`] reports, from one object.
769fn remove_from_object(
770    object: &mut Object,
771    id: ObjectId,
772    info_id: Option<ObjectId>,
773    limits: &ParseLimits,
774    depth: u32,
775) -> Result<()> {
776    if depth > limits.max_depth {
777        return Err(StryptError::LimitExceeded {
778            format: Format::Pdf,
779            limit: ResourceLimit::Depth,
780        });
781    }
782    match object {
783        Object::Dictionary(dict) => {
784            if Some(id) == info_id {
785                // The Info dictionary is emptied as well as unlinked. Unlinking alone would
786                // be enough for a correct pruner, and relying on that would make this
787                // handler's correctness depend on the pruner's — a dependency worth not
788                // having in the one place where being wrong means a name survives.
789                *dict = Dictionary::new();
790                return Ok(());
791            }
792            remove_from_dictionary(dict, id, info_id, limits, depth)?;
793        }
794        Object::Stream(stream) => {
795            remove_from_dictionary(&mut stream.dict, id, info_id, limits, depth)?;
796        }
797        Object::Array(items) => {
798            for item in items {
799                remove_from_object(item, id, info_id, limits, depth.saturating_add(1))?;
800            }
801        }
802        _ => {}
803    }
804    Ok(())
805}
806
807/// Rewrite every `Real(-0.0)` in the document as `Real(0.0)`.
808///
809/// See the call site for why this is done at all. Note the deliberate use of `is_sign_negative`
810/// rather than `== -0.0`: in IEEE 754 `-0.0 == 0.0` is true, so the obvious comparison matches
811/// positive zero as well and would rewrite values that were never a problem.
812fn normalise_negative_zero(doc: &mut Document, limits: &ParseLimits) -> Result<()> {
813    for object in doc.objects.values_mut() {
814        normalise_object(object, limits, 0)?;
815    }
816    // The trailer is not in `objects` and is reached only by walking it explicitly. Missing it
817    // is how the first version of this fix passed every local test and still failed: CI's fuzz
818    // run moved a negative zero into the trailer within minutes, and the assertion fired again
819    // on a document whose object graph was entirely clean.
820    for (_, value) in &mut doc.trailer {
821        normalise_object(value, limits, 0)?;
822    }
823    Ok(())
824}
825
826/// Walk one object, collapsing negative zeros wherever they nest.
827fn normalise_object(object: &mut Object, limits: &ParseLimits, depth: u32) -> Result<()> {
828    if depth > limits.max_depth {
829        return Err(StryptError::LimitExceeded {
830            format: Format::Pdf,
831            limit: ResourceLimit::Depth,
832        });
833    }
834    match object {
835        Object::Real(value) if value.is_sign_negative() && *value == 0.0 => {
836            *value = 0.0;
837        }
838        Object::Dictionary(dict) => {
839            for (_, value) in dict.iter_mut() {
840                normalise_object(value, limits, depth.saturating_add(1))?;
841            }
842        }
843        Object::Stream(stream) => {
844            for (_, value) in &mut stream.dict {
845                normalise_object(value, limits, depth.saturating_add(1))?;
846            }
847        }
848        Object::Array(items) => {
849            for item in items {
850                normalise_object(item, limits, depth.saturating_add(1))?;
851            }
852        }
853        _ => {}
854    }
855    Ok(())
856}
857
858/// Remove identifying keys from one dictionary and everything nested inside it.
859fn remove_from_dictionary(
860    dict: &mut Dictionary,
861    id: ObjectId,
862    info_id: Option<ObjectId>,
863    limits: &ParseLimits,
864    depth: u32,
865) -> Result<()> {
866    for (name, _) in GLOBAL_KEYS {
867        dict.remove(name);
868    }
869    if is_markup_annotation(dict) {
870        dict.remove(b"T");
871        dict.remove(b"M");
872        dict.remove(b"CreationDate");
873        dict.remove(b"NM");
874    }
875    if let Ok(Object::Dictionary(params)) = dict.get_mut(b"Params") {
876        params.remove(b"CreationDate");
877        params.remove(b"ModDate");
878        params.remove(b"CheckSum");
879    }
880    for (_, value) in &mut *dict {
881        remove_from_object(value, id, info_id, limits, depth.saturating_add(1))?;
882    }
883    Ok(())
884}
885
886/// True for a stream that is an XMP metadata packet.
887fn is_metadata_stream(dict: &Dictionary) -> bool {
888    dict.get_type().is_ok_and(|t| t == b"Metadata")
889        || dict
890            .get(b"Subtype")
891            .and_then(Object::as_name)
892            .is_ok_and(|s| s == b"XML")
893}
894
895/// True for an annotation whose `/T` names a person rather than a form field.
896fn is_markup_annotation(dict: &Dictionary) -> bool {
897    let Ok(subtype) = dict.get(b"Subtype").and_then(Object::as_name) else {
898        return false;
899    };
900    MARKUP_ANNOTATION_SUBTYPES.contains(&subtype)
901}
902
903/// True for a file-specification dictionary, which is how a PDF carries an attachment.
904fn is_embedded_file_holder(object: &Object) -> bool {
905    let dict = match object {
906        Object::Dictionary(dict) => dict,
907        Object::Stream(stream) => &stream.dict,
908        _ => return false,
909    };
910    dict.has_type(b"Filespec") || dict.has(b"EmbeddedFiles")
911}
912
913/// The size of a value in bytes, where it has a meaningful one.
914fn value_size(object: &Object) -> u64 {
915    match object {
916        Object::String(bytes, _) | Object::Name(bytes) => as_u64(bytes.len()),
917        Object::Stream(stream) => as_u64(stream.content.len()),
918        _ => 0,
919    }
920}
921
922/// Render a value, for the callers that opted into seeing values.
923fn describe(object: &Object) -> MetadataValue {
924    match object {
925        Object::String(bytes, _) | Object::Name(bytes) => MetadataValue::Text(name_of(bytes)),
926        Object::Integer(n) => MetadataValue::Text(n.to_string()),
927        Object::Boolean(b) => MetadataValue::Text(b.to_string()),
928        other => MetadataValue::Opaque {
929            bytes: value_size(other),
930        },
931    }
932}
933
934/// Widen a length for reporting. Saturating rather than fallible: a report field is not worth
935/// failing an otherwise-successful strip over.
936fn as_u64(value: usize) -> u64 {
937    u64::try_from(value).unwrap_or(u64::MAX)
938}