Skip to main content

strypt_core/formats/
pdf.rs

1//! PDF.
2//!
3//! The hardest of the Phase 1 formats, and the one where hidden data is least likely to be
4//! where you look for it. The Document Information Dictionary is the easy part and the part
5//! every tutorial covers; the leaks that matter live in XMP packets, per-object metadata,
6//! annotation authorship, and — above all — in objects left behind by incremental updates,
7//! which are physically present in the file and reachable with a hex editor long after the
8//! "current" version of the document stopped referring to them.
9//!
10//! # Full rewrite, not incremental patching (ADR-0020)
11//!
12//! A PDF can be edited by appending: the original bytes stay, and a new cross-reference
13//! section at the end says which objects supersede which. Nulling the Info dictionary with
14//! another such append is easy, fast, and preserves the file almost perfectly — and it leaves
15//! every previous author name exactly where it was, four kilobytes up the file. For this
16//! tool that is not a lesser fix, it is a silent failure: the user is told the document is
17//! clean and publishes it.
18//!
19//! So the document is parsed into its object graph, scrubbed, pruned to what the catalogue
20//! can actually reach, renumbered, and written out fresh. Everything unreachable — every
21//! superseded revision — is gone because it is never written, not because it was overwritten.
22//!
23//! The cost is honest and worth stating: the output is not byte-comparable with the input,
24//! object numbering changes, and files using features the rewrite cannot faithfully reproduce
25//! are refused rather than mangled. Refusing is the correct half of that trade.
26
27use lopdf::{Dictionary, Document, Object, ObjectId};
28
29use crate::detect::Format;
30use crate::error::{MalformedDetail, ResourceLimit, Result, StryptError};
31use crate::formats::xmp::name_of;
32use crate::formats::{MetadataHandler, ParseLimits, StripOptions, Stripped, xmp};
33use crate::report::{
34    Finding, InspectOptions, MetadataKind, MetadataReport, MetadataValue, Note, StripReport,
35};
36
37/// Removal of metadata from PDF documents.
38#[derive(Debug, Clone, Copy, Default)]
39pub struct PdfHandler;
40
41impl MetadataHandler for PdfHandler {
42    fn name(&self) -> &'static str {
43        Format::Pdf.id()
44    }
45
46    fn format(&self) -> Format {
47        Format::Pdf
48    }
49
50    fn inspect(&self, input: &[u8], options: &InspectOptions) -> Result<MetadataReport> {
51        // Inspection loads its own copy of the document and runs the *same* scrub that
52        // stripping does, then throws the result away. That is deliberate: it makes
53        // "everything `strip` removes is something `inspect` can see" true by construction
54        // rather than by two code paths agreeing to stay in step. The verification pass in
55        // `crate::pipeline` is only meaningful if that holds (`docs/ARCHITECTURE.md` §3).
56        let limits = ParseLimits::default();
57        let mut doc = load(input, &limits)?;
58        let scrubbed = scrub(&mut doc, input, options, &limits)?;
59        Ok(MetadataReport {
60            format: Format::Pdf,
61            findings: scrubbed.findings,
62            notes: scrubbed.notes,
63        })
64    }
65
66    fn strip(&self, input: &[u8], options: &StripOptions) -> Result<Stripped> {
67        let mut doc = load(input, &options.limits)?;
68        let mut scrubbed = scrub(&mut doc, input, &options.inspect, &options.limits)?;
69
70        // Drop every object the catalogue can no longer reach. This is the step that removes
71        // superseded revisions, and it has to come after scrubbing so that objects orphaned
72        // *by* the scrub — the Info dictionary, XMP streams — go with them.
73        let pruned = doc.prune_objects();
74        if !pruned.is_empty() {
75            scrubbed.notes.push(Note::OrphanedObjectsRemoved {
76                objects: pruned.len(),
77            });
78        }
79
80        // Renumber so that output depends only on the object graph, not on the numbering the
81        // input happened to use. Without this, two documents that scrub to the same content
82        // serialise differently, and the determinism invariant
83        // (`docs/TESTING_STRATEGY.md` §1) fails for no good reason.
84        renumber_stably(&mut doc)?;
85
86        // Collapse negative zero to zero for the same reason renumbering exists above: without
87        // it the output is not stable under a second strip. lopdf writes `Real(-0.0)` as `-0`,
88        // dropping the decimal point; re-parsing `-0` yields `Integer(0)`, which writes as `0`.
89        // So one strip of a file containing `-0.` differs from two, breaking the idempotence
90        // invariant (`docs/TESTING_STRATEGY.md` §1, invariant 3).
91        //
92        // Rewriting a number in the user's document needs justifying, since this handler
93        // otherwise refuses rather than repairs (ADR-0018). It is sound here because ISO
94        // 32000-1 §7.3.3 gives PDF numbers no signed zero: `-0` and `0` denote the same value,
95        // there is no operator that can distinguish them, and no renderer can. The alternative
96        // — refusing a valid file over a lost minus sign that changes nothing — costs the user
97        // their document to protect a distinction the format does not make.
98        normalise_negative_zero(&mut doc, &options.limits)?;
99
100        // Guarded for the same reason as `load`: serialisation walks a document graph built
101        // from hostile input, so it is dependency code on untrusted data just as parsing is.
102        // Failing here means nothing is written, which is the correct half of the trade.
103        let mut bytes = Vec::new();
104        crate::panic_guard::guard(
105            || {
106                doc.save_to(&mut bytes).map_err(|source| StryptError::Io {
107                    action: crate::error::IoAction::WritingOutput,
108                    source,
109                })
110            },
111            || StryptError::Malformed {
112                format: Format::Pdf,
113                offset: None,
114                detail: MalformedDetail::DependencyPanic,
115            },
116        )?;
117
118        // Nothing is handed back until the bytes just written have been read again and shown
119        // to be the document that was written. See `verify_round_trip`.
120        verify_round_trip(&doc, &bytes)?;
121
122        Ok(Stripped {
123            report: StripReport {
124                format: Format::Pdf,
125                removed: scrubbed.findings,
126                retained: Vec::new(),
127                notes: scrubbed.notes,
128                input_bytes: as_u64(input.len()),
129                output_bytes: as_u64(bytes.len()),
130            },
131            bytes,
132        })
133    }
134}
135
136/// Refuse output that does not read back as the document that was written.
137///
138/// The rewrite (ADR-0020) assumes serialising a parsed document and re-parsing it returns the
139/// same document. For a lenient parser on hostile input that assumption does not hold, and
140/// when it breaks it breaks silently: `lopdf` will accept a dictionary whose keys came out of
141/// mangled bytes — `/Annotst 1 /[^@018064665...` in the file that found this — and then write
142/// it back in a form it cannot itself read. The object is written, and disappears when the
143/// file is next opened.
144///
145/// That is the failure this guard exists for, and it is the dangerous kind. A document whose
146/// only `/Page` is lost on reload has a `/Pages` node still claiming `/Count 1` and a `/Kids`
147/// array pointing at an object that is no longer there. strypt wrote that file and reported
148/// success; a second strip then pruned what had become unreachable and wrote a 230-byte
149/// document with a dangling page reference, reporting success again. No metadata survived
150/// either pass, so the verification pass — which searches output for residual metadata — saw
151/// nothing wrong. It cannot: it is looking for what should be absent, not for what should
152/// still be present.
153///
154/// Checking a full structural equivalence would be a second implementation of the rewrite, so
155/// this checks the two properties whose failure means the output is not the document:
156///
157/// 1. **Every object written is present on reload.** This is what catches an object that
158///    serialised into something unparseable. Comparing ids alone is enough — an object that
159///    round-trips to a different id is caught by the same comparison.
160/// 2. **The page tree survives.** A document that had pages before writing and none after has
161///    lost the thing it exists to carry, even when every object id happens to match.
162///
163/// A failure refuses the file. That is the fail-closed answer (`CLAUDE.md` §3 rule 6): strypt
164/// cannot faithfully rewrite this document, so it declines to rather than handing back a
165/// broken one with a success report. Refusing costs the user a file that was already too
166/// damaged to survive a rewrite; the alternative cost them a document they believed was clean.
167fn verify_round_trip(written: &Document, bytes: &[u8]) -> Result<()> {
168    let reloaded = crate::panic_guard::guard(
169        || Document::load_mem(bytes).map_err(|e| map_parse_error(&e)),
170        || StryptError::Malformed {
171            format: Format::Pdf,
172            offset: None,
173            detail: MalformedDetail::DependencyPanic,
174        },
175    )
176    .map_err(|_| StryptError::Malformed {
177        format: Format::Pdf,
178        offset: None,
179        detail: MalformedDetail::NotRoundTrippable,
180    })?;
181
182    let written_ids: Vec<ObjectId> = written.objects.keys().copied().collect();
183    let reloaded_ids: Vec<ObjectId> = reloaded.objects.keys().copied().collect();
184    if written_ids != reloaded_ids {
185        return Err(StryptError::Malformed {
186            format: Format::Pdf,
187            offset: None,
188            detail: MalformedDetail::NotRoundTrippable,
189        });
190    }
191
192    // Only an emptied page tree is a failure, not an empty one: a document that had no pages
193    // to begin with is degenerate but not something this rewrite broke.
194    if written.page_iter().next().is_some() && reloaded.page_iter().next().is_none() {
195        return Err(StryptError::Malformed {
196            format: Format::Pdf,
197            offset: None,
198            detail: MalformedDetail::NotRoundTrippable,
199        });
200    }
201
202    Ok(())
203}
204
205/// How many times `renumber_stably` will renumber before refusing the document.
206///
207/// A well-formed file reaches its fixed point on the second call — the first assigns
208/// `1..=n`, the second confirms nothing moved. The degenerate case below needs a third.
209/// Four is that plus margin; a document still moving after four is not converging, and
210/// looping harder would only delay the refusal.
211const MAX_RENUMBER_ROUNDS: usize = 4;
212
213/// Renumber until the numbering stops changing, or refuse the document.
214///
215/// `lopdf::renumber_objects` is **not idempotent**, which matters because strypt's idempotence
216/// invariant (`docs/TESTING_STRATEGY.md` §1, invariant 3) is stated byte-for-byte on the *first*
217/// re-strip. Before renumbering sequentially, `renumber_objects_with` checks whether the page
218/// order matches ascending object ids and, if it does not, permutes the page objects so that it
219/// does (`lopdf` 0.44.0 `src/processor.rs`). That check reads the numbering the previous step
220/// produced, so one pass can leave a document that a second pass would reorder again.
221///
222/// A sustained fuzz run found the case where that is reachable: a document whose page tree is
223/// self-referential — object 2 is a `/Page` whose own `/Kids` array lists object 2 — so pruning
224/// and renumbering changed which objects `page_iter` yields and in which order. The first strip
225/// produced pages ordered `[3, 2]`, descending; the second saw the mismatch and swapped objects
226/// 2 and 3. Same length, same content, 145 bytes different. It settled from the third strip on,
227/// so this was never an endless flip — but "stable eventually" is not the invariant, and a user
228/// who strips a file twice must not get two different files.
229///
230/// Iterating to a fixed point fixes it by construction rather than by reasoning about a
231/// dependency's internals: the document is only serialised once renumbering has been shown to be
232/// a no-op on it, so re-loading and renumbering that output cannot move anything either.
233///
234/// This does not change the output of any document that was already stable — for those the
235/// second round is the confirmation that would have been skipped, not a second permutation.
236///
237/// Non-convergence is refused rather than accepted at whatever state the last round left, which
238/// is the fail-closed half of the trade (ADR-0018, and `CLAUDE.md` §3 rule 6). Emitting a file
239/// whose numbering strypt could not settle would mean handing the user output it cannot promise
240/// is reproducible.
241fn renumber_stably(doc: &mut Document) -> Result<()> {
242    for _ in 0..MAX_RENUMBER_ROUNDS {
243        // Both the id set and the page order have to be compared. The ids alone are not enough:
244        // the reordering step permutes which object holds which page while leaving the set of
245        // ids exactly as it was, so comparing ids only would report a fixed point on the very
246        // pass that moved something.
247        let ids_before: Vec<ObjectId> = doc.objects.keys().copied().collect();
248        let pages_before: Vec<ObjectId> = doc.page_iter().collect();
249
250        doc.renumber_objects();
251
252        let ids_after: Vec<ObjectId> = doc.objects.keys().copied().collect();
253        let pages_after: Vec<ObjectId> = doc.page_iter().collect();
254
255        if ids_before == ids_after && pages_before == pages_after {
256            return Ok(());
257        }
258    }
259
260    Err(StryptError::Malformed {
261        format: Format::Pdf,
262        offset: None,
263        detail: MalformedDetail::CyclicReference,
264    })
265}
266
267/// Findings and caveats from one pass over a document.
268struct Scrubbed {
269    findings: Vec<Finding>,
270    notes: Vec<Note>,
271}
272
273/// Parse `input`, refusing documents this handler must not rewrite.
274fn load(input: &[u8], limits: &ParseLimits) -> Result<Document> {
275    // `lopdf` is third-party code parsing attacker-controlled bytes, and ADR-0006's no-panic
276    // rule does not reach inside it (ADR-0018). A sustained fuzz run found an integer overflow
277    // in its cross-reference parser, which with `overflow-checks` on in release meant the
278    // shipped binary aborted with a stack trace instead of refusing the file. Contained here
279    // so it reaches the user as an ordinary refusal; see `crate::panic_guard` for what that
280    // does and does not cover.
281    let doc = crate::panic_guard::guard(
282        || Document::load_mem(input).map_err(|e| map_parse_error(&e)),
283        || StryptError::Malformed {
284            format: Format::Pdf,
285            offset: None,
286            detail: MalformedDetail::DependencyPanic,
287        },
288    )?;
289
290    // Encrypted documents are refused rather than rewritten. lopdf can open one protected by
291    // an empty owner password, and it would be technically easy to emit a decrypted copy —
292    // but that hands the user a file with its protection quietly removed, which is a change
293    // to their document's security they did not ask for and would not necessarily notice.
294    // Fail closed and say so.
295    if doc.is_encrypted() || doc.was_encrypted() {
296        return Err(StryptError::Malformed {
297            format: Format::Pdf,
298            offset: None,
299            detail: MalformedDetail::UnsupportedFeature,
300        });
301    }
302
303    // A trailer with no /Root is refused rather than processed.
304    //
305    // ISO 32000-1 §7.5.5 makes /Root a required trailer entry: it names the document catalogue,
306    // which is the single root every other object hangs off. Without it the file has no defined
307    // entry point, and no viewer will open it.
308    //
309    // strypt used to accept such a file, and the result was worse than a refusal. The rewrite in
310    // `strip` walks reachable objects from the root and drops the rest (ADR-0020); with no root
311    // to walk from, which objects survive is not stable across runs. A fuzz run found a document
312    // whose second strip differed from its first — 609 bytes, then 485 — because the second pass
313    // dropped an annotation object that the page still referenced through /Annots. Renumbering
314    // then filled that slot with the catalogue, so the page's annotation array pointed at the
315    // document catalogue. strypt had introduced that corruption itself, while returning success
316    // both times.
317    //
318    // Refusing is the fail-closed answer and costs nothing real: a PDF this broken is not one
319    // the user can publish anyway.
320    if !doc.trailer.has(b"Root") {
321        return Err(StryptError::Malformed {
322            format: Format::Pdf,
323            offset: None,
324            detail: MalformedDetail::MissingMarker,
325        });
326    }
327
328    let too_many = u32::try_from(doc.objects.len()).map_or(true, |count| count > limits.max_items);
329    if too_many {
330        return Err(StryptError::LimitExceeded {
331            format: Format::Pdf,
332            limit: ResourceLimit::ItemCount,
333        });
334    }
335
336    // Every stream's declared /Length must be an integer that matches the content actually
337    // parsed out of it.
338    //
339    // ISO 32000-1 §7.3.8.2 requires /Length to be an integer giving the exact byte count
340    // between "stream" and "endstream". When it is not — a fuzz case reached here by writing
341    // "/Length 45." instead of "/Length 45" — lopdf 0.44 parses the object, keeps the
342    // malformed value in the dictionary, and stores *empty* content, because it cannot locate
343    // the stream's end. It reports no error while doing so.
344    //
345    // Left unchecked, the document reaches the rewriter with the stream's bytes already gone.
346    // strypt then writes a file whose /Length still claims 45 bytes over an empty stream, and
347    // reports "nothing to remove; wrote a clean copy" — a structurally invalid PDF, missing
348    // the user's page content, presented as a success. That is the §5.4 failure the whole
349    // design is arranged to avoid, and the reason it went unnoticed is that the verification
350    // pass looks for residual *metadata*, which an emptied stream has none of.
351    //
352    // Refusing costs nothing on real documents: across the synthetic corpus and all 23
353    // parseable PDFs of the real-producer corpus — pdfLaTeX, LibreOffice, Google Docs,
354    // Acrobat and ImageMagick output, compressed streams included — not one stream disagrees
355    // with its declared length.
356    for object in doc.objects.values() {
357        let Ok(stream) = object.as_stream() else {
358            continue;
359        };
360        let declared = stream
361            .dict
362            .get(b"Length")
363            .ok()
364            .and_then(|length| length.as_i64().ok())
365            .and_then(|length| usize::try_from(length).ok());
366        if declared != Some(stream.content.len()) {
367            return Err(StryptError::Malformed {
368                format: Format::Pdf,
369                offset: None,
370                detail: MalformedDetail::LengthOutOfRange,
371            });
372        }
373    }
374
375    Ok(doc)
376}
377
378/// Translate a parser failure into something a user can act on.
379///
380/// The mapping is coarse on purpose. "Your file is truncated" and "your file is not really a
381/// PDF" lead to different actions; which of thirty-four internal variants fired does not.
382fn map_parse_error(error: &lopdf::Error) -> StryptError {
383    use lopdf::Error as E;
384    let detail = match *error {
385        E::Parse(_) | E::Syntax(_) | E::IndirectObject { .. } | E::ObjectIdMismatch => {
386            MalformedDetail::UnexpectedMarker
387        }
388        E::Xref(_) | E::MissingXrefEntry | E::InvalidObjectStream(_) => {
389            MalformedDetail::BrokenIndex
390        }
391        E::InvalidOffset(_) | E::ObjectNotFound(_) | E::NumericCast(_) | E::TryFromInt(_) => {
392            MalformedDetail::LengthOutOfRange
393        }
394        E::ReferenceCycle(_) | E::ReferenceLimit => MalformedDetail::CyclicReference,
395        E::IO(_) => MalformedDetail::Truncated,
396        E::Decryption(_)
397        | E::InvalidPassword
398        | E::AlreadyEncrypted
399        | E::UnsupportedSecurityHandler(_)
400        | E::Unimplemented(_) => MalformedDetail::UnsupportedFeature,
401        _ => MalformedDetail::MissingMarker,
402    };
403    let offset = match *error {
404        E::InvalidOffset(at) | E::IndirectObject { offset: at } => u64::try_from(at).ok(),
405        _ => None,
406    };
407    StryptError::Malformed {
408        format: Format::Pdf,
409        offset,
410        detail,
411    }
412}
413
414/// Keys in the Document Information Dictionary, and what each one exposes.
415///
416/// `/Creator` names the application the document was *authored* in and `/Producer` the one
417/// that wrote the PDF — so a LaTeX paper typically confesses both its editor and its
418/// toolchain version here. Neither identifies a person alone; together with a timestamp and
419/// a font list they narrow the field a great deal (`docs/THREAT_MODEL.md` §4.7).
420const INFO_KEYS: &[(&[u8], MetadataKind)] = &[
421    (b"Author", MetadataKind::PersonalIdentity),
422    (b"Creator", MetadataKind::SoftwareFingerprint),
423    (b"Producer", MetadataKind::SoftwareFingerprint),
424    (b"CreationDate", MetadataKind::Timestamp),
425    (b"ModDate", MetadataKind::Timestamp),
426    (b"Title", MetadataKind::Comment),
427    (b"Subject", MetadataKind::Comment),
428    (b"Keywords", MetadataKind::Comment),
429    (b"Trapped", MetadataKind::Other),
430];
431
432/// Keys removed from *any* dictionary in the document, wherever they appear.
433///
434/// These are safe to remove anywhere because the PDF specification gives them one meaning
435/// each and nothing renders differently without them. `/PieceInfo` is the interesting one:
436/// it is a scratch area where an application may store whatever private state it likes
437/// between editing sessions, and what ends up in it is entirely up to that application.
438const GLOBAL_KEYS: &[(&[u8], MetadataKind)] = &[
439    (b"Metadata", MetadataKind::Other),
440    (b"PieceInfo", MetadataKind::EditingHistory),
441    (b"LastModified", MetadataKind::Timestamp),
442];
443
444/// Annotation subtypes whose `/T` entry is the annotating person's name.
445///
446/// This distinction matters. On a markup annotation `/T` is the author — exactly what we are
447/// here to remove. On a `/Widget`, which is how every interactive form field is drawn, `/T`
448/// is the *field name* that the form's logic and its saved data refer to. Stripping it would
449/// silently break the document, and breaking a user's file to protect them is not a trade
450/// this tool gets to make on their behalf without saying so.
451const MARKUP_ANNOTATION_SUBTYPES: &[&[u8]] = &[
452    b"Text",
453    b"FreeText",
454    b"Line",
455    b"Square",
456    b"Circle",
457    b"Polygon",
458    b"PolyLine",
459    b"Highlight",
460    b"Underline",
461    b"Squiggly",
462    b"StrikeOut",
463    b"Stamp",
464    b"Caret",
465    b"Ink",
466    b"FileAttachment",
467    b"Sound",
468    b"Movie",
469    b"Redact",
470];
471
472/// Walk the document, recording what is there and removing it.
473fn scrub(
474    doc: &mut Document,
475    raw: &[u8],
476    options: &InspectOptions,
477    limits: &ParseLimits,
478) -> Result<Scrubbed> {
479    let mut findings = Vec::new();
480    let mut notes = Vec::new();
481
482    let info_id = trailer_reference(doc, b"Info");
483
484    // Phase one reads. Every object is examined, not merely the ones the catalogue can reach:
485    // an orphan from a superseded revision is exactly the thing worth telling the user about,
486    // and it is invisible to a walk that starts at the root.
487    let object_ids: Vec<ObjectId> = doc.objects.keys().copied().collect();
488    for id in &object_ids {
489        let Some(object) = doc.objects.get(id) else {
490            continue;
491        };
492        examine_object(object, *id, info_id, options, limits, 0, &mut findings)?;
493    }
494    if doc.trailer.has(b"ID") {
495        // The file identifier is a pair of strings that stays stable across saves of the same
496        // document. It identifies nobody by itself and links every copy and every revision of
497        // the document to each other, which for a leaked draft is the whole question.
498        findings.push(Finding::new(
499            MetadataKind::DocumentIdentifier,
500            "trailer /ID",
501            0,
502        ));
503    }
504
505    // A PDF that has been saved more than once ends with more than one %%EOF. Counting them
506    // is cruder than walking the cross-reference chain and tells the user the thing that
507    // actually matters: earlier versions of this document were sitting inside it.
508    let revisions = count_revisions(raw);
509    if revisions > 1 {
510        notes.push(Note::IncrementalHistory {
511            revisions: revisions.saturating_sub(1),
512        });
513    }
514
515    // Phase two writes.
516    doc.trailer.remove(b"Info");
517    doc.trailer.remove(b"ID");
518    for id in &object_ids {
519        if let Some(object) = doc.objects.get_mut(id) {
520            remove_from_object(object, *id, info_id, limits, 0)?;
521        }
522    }
523
524    if object_ids
525        .iter()
526        .filter_map(|id| doc.objects.get(id))
527        .any(is_embedded_file_holder)
528    {
529        // strypt does not open embedded files. Recursing into them means recursing into
530        // arbitrary nested content, which is a zip-bomb-shaped problem that Phase 2 has to
531        // decide about explicitly. Until then the user is told, because an attachment
532        // carrying its own metadata inside a document reported as clean is precisely the
533        // over-trust this tool must not create (`docs/THREAT_MODEL.md` §5.6).
534        notes.push(Note::OutOfScopeContent {
535            location: "embedded file attachment".into(),
536        });
537    }
538
539    Ok(Scrubbed { findings, notes })
540}
541
542/// Resolve a reference held in the trailer, if it is one.
543fn trailer_reference(doc: &Document, key: &[u8]) -> Option<ObjectId> {
544    doc.trailer
545        .get(key)
546        .ok()
547        .and_then(|o| o.as_reference().ok())
548}
549
550/// Count `%%EOF` markers, each of which terminates one revision of the document.
551fn count_revisions(raw: &[u8]) -> usize {
552    const EOF: &[u8] = b"%%EOF";
553    raw.windows(EOF.len()).filter(|w| *w == EOF).count()
554}
555
556/// Record everything identifying inside one object.
557fn examine_object(
558    object: &Object,
559    id: ObjectId,
560    info_id: Option<ObjectId>,
561    options: &InspectOptions,
562    limits: &ParseLimits,
563    depth: u32,
564    out: &mut Vec<Finding>,
565) -> Result<()> {
566    if depth > limits.max_depth {
567        return Err(StryptError::LimitExceeded {
568            format: Format::Pdf,
569            limit: ResourceLimit::Depth,
570        });
571    }
572    match object {
573        Object::Dictionary(dict) => {
574            if Some(id) == info_id {
575                examine_info(dict, options, out);
576            }
577            examine_dictionary(dict, id, info_id, options, limits, depth, out)?;
578        }
579        Object::Stream(stream) => {
580            if is_metadata_stream(&stream.dict) {
581                examine_xmp(stream, options, out);
582            }
583            examine_dictionary(&stream.dict, id, info_id, options, limits, depth, out)?;
584        }
585        Object::Array(items) => {
586            for item in items {
587                examine_object(
588                    item,
589                    id,
590                    info_id,
591                    options,
592                    limits,
593                    depth.saturating_add(1),
594                    out,
595                )?;
596            }
597        }
598        _ => {}
599    }
600    Ok(())
601}
602
603/// Record the Document Information Dictionary, including keys the specification never
604/// defined — applications add their own freely, and a custom key is no less identifying for
605/// being non-standard.
606fn examine_info(dict: &Dictionary, options: &InspectOptions, out: &mut Vec<Finding>) {
607    for (key, value) in dict {
608        let kind = INFO_KEYS
609            .iter()
610            .find(|(name, _)| *name == key.as_slice())
611            .map_or(MetadataKind::Other, |(_, kind)| *kind);
612        out.push(
613            Finding::new(kind, "/Info", value_size(value))
614                .with_field(name_of(key))
615                .with_value(options, || describe(value)),
616        );
617    }
618}
619
620/// Record the globally-removable keys, and annotation authorship.
621fn examine_dictionary(
622    dict: &Dictionary,
623    id: ObjectId,
624    info_id: Option<ObjectId>,
625    options: &InspectOptions,
626    limits: &ParseLimits,
627    depth: u32,
628    out: &mut Vec<Finding>,
629) -> Result<()> {
630    for (name, kind) in GLOBAL_KEYS {
631        if let Ok(value) = dict.get(name) {
632            out.push(
633                Finding::new(*kind, format!("/{}", name_of(name)), value_size(value))
634                    .with_field(name_of(name)),
635            );
636        }
637    }
638    if is_markup_annotation(dict) {
639        for (name, kind) in [
640            (&b"T"[..], MetadataKind::PersonalIdentity),
641            (&b"M"[..], MetadataKind::Timestamp),
642            (&b"CreationDate"[..], MetadataKind::Timestamp),
643            (&b"NM"[..], MetadataKind::DocumentIdentifier),
644        ] {
645            if let Ok(value) = dict.get(name) {
646                out.push(
647                    Finding::new(kind, "annotation", value_size(value))
648                        .with_field(name_of(name))
649                        .with_value(options, || describe(value)),
650                );
651            }
652        }
653    }
654    // Embedded-file parameters carry their own creation and modification dates, which survive
655    // every scrub aimed only at the containing document.
656    if let Ok(Object::Dictionary(params)) = dict.get(b"Params") {
657        for name in [&b"CreationDate"[..], &b"ModDate"[..], &b"CheckSum"[..]] {
658            if let Ok(value) = params.get(name) {
659                out.push(
660                    Finding::new(MetadataKind::Timestamp, "/Params", value_size(value))
661                        .with_field(name_of(name)),
662                );
663            }
664        }
665    }
666    for (_, value) in dict {
667        examine_object(
668            value,
669            id,
670            info_id,
671            options,
672            limits,
673            depth.saturating_add(1),
674            out,
675        )?;
676    }
677    Ok(())
678}
679
680/// Scan an XMP packet for the properties worth naming.
681///
682/// Only unfiltered packets are scanned. ISO 32000-1 §14.3.2 recommends that a metadata stream
683/// be left uncompressed precisely so it can be read without parsing the whole document, and
684/// in practice they almost always are. Refusing to inflate the rare compressed one avoids
685/// handing an attacker a decompression bomb in exchange for a slightly more detailed report —
686/// the packet is still found, still reported, and still removed either way.
687fn examine_xmp(stream: &lopdf::Stream, options: &InspectOptions, out: &mut Vec<Finding>) {
688    if stream.dict.has(b"Filter") {
689        out.push(
690            Finding::new(
691                MetadataKind::Other,
692                "XMP packet",
693                as_u64(stream.content.len()),
694            )
695            .with_field("Metadata (encoded)"),
696        );
697        return;
698    }
699    out.extend(xmp::scan(&stream.content, "XMP packet", options));
700}
701
702/// Remove everything [`examine_object`] reports, from one object.
703fn remove_from_object(
704    object: &mut Object,
705    id: ObjectId,
706    info_id: Option<ObjectId>,
707    limits: &ParseLimits,
708    depth: u32,
709) -> Result<()> {
710    if depth > limits.max_depth {
711        return Err(StryptError::LimitExceeded {
712            format: Format::Pdf,
713            limit: ResourceLimit::Depth,
714        });
715    }
716    match object {
717        Object::Dictionary(dict) => {
718            if Some(id) == info_id {
719                // The Info dictionary is emptied as well as unlinked. Unlinking alone would
720                // be enough for a correct pruner, and relying on that would make this
721                // handler's correctness depend on the pruner's — a dependency worth not
722                // having in the one place where being wrong means a name survives.
723                *dict = Dictionary::new();
724                return Ok(());
725            }
726            remove_from_dictionary(dict, id, info_id, limits, depth)?;
727        }
728        Object::Stream(stream) => {
729            remove_from_dictionary(&mut stream.dict, id, info_id, limits, depth)?;
730        }
731        Object::Array(items) => {
732            for item in items {
733                remove_from_object(item, id, info_id, limits, depth.saturating_add(1))?;
734            }
735        }
736        _ => {}
737    }
738    Ok(())
739}
740
741/// Rewrite every `Real(-0.0)` in the document as `Real(0.0)`.
742///
743/// See the call site for why this is done at all. Note the deliberate use of `is_sign_negative`
744/// rather than `== -0.0`: in IEEE 754 `-0.0 == 0.0` is true, so the obvious comparison matches
745/// positive zero as well and would rewrite values that were never a problem.
746fn normalise_negative_zero(doc: &mut Document, limits: &ParseLimits) -> Result<()> {
747    for object in doc.objects.values_mut() {
748        normalise_object(object, limits, 0)?;
749    }
750    // The trailer is not in `objects` and is reached only by walking it explicitly. Missing it
751    // is how the first version of this fix passed every local test and still failed: CI's fuzz
752    // run moved a negative zero into the trailer within minutes, and the assertion fired again
753    // on a document whose object graph was entirely clean.
754    for (_, value) in &mut doc.trailer {
755        normalise_object(value, limits, 0)?;
756    }
757    Ok(())
758}
759
760/// Walk one object, collapsing negative zeros wherever they nest.
761fn normalise_object(object: &mut Object, limits: &ParseLimits, depth: u32) -> Result<()> {
762    if depth > limits.max_depth {
763        return Err(StryptError::LimitExceeded {
764            format: Format::Pdf,
765            limit: ResourceLimit::Depth,
766        });
767    }
768    match object {
769        Object::Real(value) if value.is_sign_negative() && *value == 0.0 => {
770            *value = 0.0;
771        }
772        Object::Dictionary(dict) => {
773            for (_, value) in dict.iter_mut() {
774                normalise_object(value, limits, depth.saturating_add(1))?;
775            }
776        }
777        Object::Stream(stream) => {
778            for (_, value) in &mut stream.dict {
779                normalise_object(value, limits, depth.saturating_add(1))?;
780            }
781        }
782        Object::Array(items) => {
783            for item in items {
784                normalise_object(item, limits, depth.saturating_add(1))?;
785            }
786        }
787        _ => {}
788    }
789    Ok(())
790}
791
792/// Remove identifying keys from one dictionary and everything nested inside it.
793fn remove_from_dictionary(
794    dict: &mut Dictionary,
795    id: ObjectId,
796    info_id: Option<ObjectId>,
797    limits: &ParseLimits,
798    depth: u32,
799) -> Result<()> {
800    for (name, _) in GLOBAL_KEYS {
801        dict.remove(name);
802    }
803    if is_markup_annotation(dict) {
804        dict.remove(b"T");
805        dict.remove(b"M");
806        dict.remove(b"CreationDate");
807        dict.remove(b"NM");
808    }
809    if let Ok(Object::Dictionary(params)) = dict.get_mut(b"Params") {
810        params.remove(b"CreationDate");
811        params.remove(b"ModDate");
812        params.remove(b"CheckSum");
813    }
814    for (_, value) in &mut *dict {
815        remove_from_object(value, id, info_id, limits, depth.saturating_add(1))?;
816    }
817    Ok(())
818}
819
820/// True for a stream that is an XMP metadata packet.
821fn is_metadata_stream(dict: &Dictionary) -> bool {
822    dict.get_type().is_ok_and(|t| t == b"Metadata")
823        || dict
824            .get(b"Subtype")
825            .and_then(Object::as_name)
826            .is_ok_and(|s| s == b"XML")
827}
828
829/// True for an annotation whose `/T` names a person rather than a form field.
830fn is_markup_annotation(dict: &Dictionary) -> bool {
831    let Ok(subtype) = dict.get(b"Subtype").and_then(Object::as_name) else {
832        return false;
833    };
834    MARKUP_ANNOTATION_SUBTYPES.contains(&subtype)
835}
836
837/// True for a file-specification dictionary, which is how a PDF carries an attachment.
838fn is_embedded_file_holder(object: &Object) -> bool {
839    let dict = match object {
840        Object::Dictionary(dict) => dict,
841        Object::Stream(stream) => &stream.dict,
842        _ => return false,
843    };
844    dict.has_type(b"Filespec") || dict.has(b"EmbeddedFiles")
845}
846
847/// The size of a value in bytes, where it has a meaningful one.
848fn value_size(object: &Object) -> u64 {
849    match object {
850        Object::String(bytes, _) | Object::Name(bytes) => as_u64(bytes.len()),
851        Object::Stream(stream) => as_u64(stream.content.len()),
852        _ => 0,
853    }
854}
855
856/// Render a value, for the callers that opted into seeing values.
857fn describe(object: &Object) -> MetadataValue {
858    match object {
859        Object::String(bytes, _) | Object::Name(bytes) => MetadataValue::Text(name_of(bytes)),
860        Object::Integer(n) => MetadataValue::Text(n.to_string()),
861        Object::Boolean(b) => MetadataValue::Text(b.to_string()),
862        other => MetadataValue::Opaque {
863            bytes: value_size(other),
864        },
865    }
866}
867
868/// Widen a length for reporting. Saturating rather than fallible: a report field is not worth
869/// failing an otherwise-successful strip over.
870fn as_u64(value: usize) -> u64 {
871    u64::try_from(value).unwrap_or(u64::MAX)
872}