Skip to main content

strypt_core/formats/
pdf.rs

1//! PDF.
2//!
3//! The hardest of the Phase 1 formats, and the one where hidden data is least likely to be
4//! where you look for it. The Document Information Dictionary is the easy part and the part
5//! every tutorial covers; the leaks that matter live in XMP packets, per-object metadata,
6//! annotation authorship, and — above all — in objects left behind by incremental updates,
7//! which are physically present in the file and reachable with a hex editor long after the
8//! "current" version of the document stopped referring to them.
9//!
10//! # Full rewrite, not incremental patching (ADR-0020)
11//!
12//! A PDF can be edited by appending: the original bytes stay, and a new cross-reference
13//! section at the end says which objects supersede which. Nulling the Info dictionary with
14//! another such append is easy, fast, and preserves the file almost perfectly — and it leaves
15//! every previous author name exactly where it was, four kilobytes up the file. For this
16//! tool that is not a lesser fix, it is a silent failure: the user is told the document is
17//! clean and publishes it.
18//!
19//! So the document is parsed into its object graph, scrubbed, pruned to what the catalogue
20//! can actually reach, renumbered, and written out fresh. Everything unreachable — every
21//! superseded revision — is gone because it is never written, not because it was overwritten.
22//!
23//! The cost is honest and worth stating: the output is not byte-comparable with the input,
24//! object numbering changes, and files using features the rewrite cannot faithfully reproduce
25//! are refused rather than mangled. Refusing is the correct half of that trade.
26
27use lopdf::{Dictionary, Document, Object, ObjectId};
28
29use crate::detect::Format;
30use crate::error::{MalformedDetail, ResourceLimit, Result, StryptError};
31use crate::formats::xmp::name_of;
32use crate::formats::{MetadataHandler, ParseLimits, StripOptions, Stripped, xmp};
33use crate::report::{
34    Finding, InspectOptions, MetadataKind, MetadataReport, MetadataValue, Note, StripReport,
35};
36
37/// Removal of metadata from PDF documents.
38#[derive(Debug, Clone, Copy, Default)]
39pub struct PdfHandler;
40
41impl MetadataHandler for PdfHandler {
42    fn name(&self) -> &'static str {
43        Format::Pdf.id()
44    }
45
46    fn format(&self) -> Format {
47        Format::Pdf
48    }
49
50    fn inspect(&self, input: &[u8], options: &InspectOptions) -> Result<MetadataReport> {
51        // Inspection loads its own copy of the document and runs the *same* scrub that
52        // stripping does, then throws the result away. That is deliberate: it makes
53        // "everything `strip` removes is something `inspect` can see" true by construction
54        // rather than by two code paths agreeing to stay in step. The verification pass in
55        // `crate::pipeline` is only meaningful if that holds (`docs/ARCHITECTURE.md` §3).
56        let limits = ParseLimits::default();
57        let mut doc = load(input, &limits)?;
58        let scrubbed = scrub(&mut doc, input, options, &limits)?;
59        Ok(MetadataReport {
60            format: Format::Pdf,
61            findings: scrubbed.findings,
62            notes: scrubbed.notes,
63        })
64    }
65
66    fn strip(&self, input: &[u8], options: &StripOptions) -> Result<Stripped> {
67        let mut doc = load(input, &options.limits)?;
68        let mut scrubbed = scrub(&mut doc, input, &options.inspect, &options.limits)?;
69
70        // Drop every object the catalogue can no longer reach. This is the step that removes
71        // superseded revisions, and it has to come after scrubbing so that objects orphaned
72        // *by* the scrub — the Info dictionary, XMP streams — go with them.
73        let pruned = doc.prune_objects();
74        if !pruned.is_empty() {
75            scrubbed.notes.push(Note::OrphanedObjectsRemoved {
76                objects: pruned.len(),
77            });
78        }
79
80        // Renumber so that output depends only on the object graph, not on the numbering the
81        // input happened to use. Without this, two documents that scrub to the same content
82        // serialise differently, and the determinism invariant
83        // (`docs/TESTING_STRATEGY.md` §1) fails for no good reason.
84        doc.renumber_objects();
85
86        // Collapse negative zero to zero for the same reason renumbering exists above: without
87        // it the output is not stable under a second strip. lopdf writes `Real(-0.0)` as `-0`,
88        // dropping the decimal point; re-parsing `-0` yields `Integer(0)`, which writes as `0`.
89        // So one strip of a file containing `-0.` differs from two, breaking the idempotence
90        // invariant (`docs/TESTING_STRATEGY.md` §1, invariant 3).
91        //
92        // Rewriting a number in the user's document needs justifying, since this handler
93        // otherwise refuses rather than repairs (ADR-0018). It is sound here because ISO
94        // 32000-1 §7.3.3 gives PDF numbers no signed zero: `-0` and `0` denote the same value,
95        // there is no operator that can distinguish them, and no renderer can. The alternative
96        // — refusing a valid file over a lost minus sign that changes nothing — costs the user
97        // their document to protect a distinction the format does not make.
98        normalise_negative_zero(&mut doc, &options.limits)?;
99
100        // Guarded for the same reason as `load`: serialisation walks a document graph built
101        // from hostile input, so it is dependency code on untrusted data just as parsing is.
102        // Failing here means nothing is written, which is the correct half of the trade.
103        let mut bytes = Vec::new();
104        crate::panic_guard::guard(
105            || {
106                doc.save_to(&mut bytes).map_err(|source| StryptError::Io {
107                    action: crate::error::IoAction::WritingOutput,
108                    source,
109                })
110            },
111            || StryptError::Malformed {
112                format: Format::Pdf,
113                offset: None,
114                detail: MalformedDetail::DependencyPanic,
115            },
116        )?;
117
118        Ok(Stripped {
119            report: StripReport {
120                format: Format::Pdf,
121                removed: scrubbed.findings,
122                retained: Vec::new(),
123                notes: scrubbed.notes,
124                input_bytes: as_u64(input.len()),
125                output_bytes: as_u64(bytes.len()),
126            },
127            bytes,
128        })
129    }
130}
131
132/// Findings and caveats from one pass over a document.
133struct Scrubbed {
134    findings: Vec<Finding>,
135    notes: Vec<Note>,
136}
137
138/// Parse `input`, refusing documents this handler must not rewrite.
139fn load(input: &[u8], limits: &ParseLimits) -> Result<Document> {
140    // `lopdf` is third-party code parsing attacker-controlled bytes, and ADR-0006's no-panic
141    // rule does not reach inside it (ADR-0018). A sustained fuzz run found an integer overflow
142    // in its cross-reference parser, which with `overflow-checks` on in release meant the
143    // shipped binary aborted with a stack trace instead of refusing the file. Contained here
144    // so it reaches the user as an ordinary refusal; see `crate::panic_guard` for what that
145    // does and does not cover.
146    let doc = crate::panic_guard::guard(
147        || Document::load_mem(input).map_err(|e| map_parse_error(&e)),
148        || StryptError::Malformed {
149            format: Format::Pdf,
150            offset: None,
151            detail: MalformedDetail::DependencyPanic,
152        },
153    )?;
154
155    // Encrypted documents are refused rather than rewritten. lopdf can open one protected by
156    // an empty owner password, and it would be technically easy to emit a decrypted copy —
157    // but that hands the user a file with its protection quietly removed, which is a change
158    // to their document's security they did not ask for and would not necessarily notice.
159    // Fail closed and say so.
160    if doc.is_encrypted() || doc.was_encrypted() {
161        return Err(StryptError::Malformed {
162            format: Format::Pdf,
163            offset: None,
164            detail: MalformedDetail::UnsupportedFeature,
165        });
166    }
167
168    // A trailer with no /Root is refused rather than processed.
169    //
170    // ISO 32000-1 §7.5.5 makes /Root a required trailer entry: it names the document catalogue,
171    // which is the single root every other object hangs off. Without it the file has no defined
172    // entry point, and no viewer will open it.
173    //
174    // strypt used to accept such a file, and the result was worse than a refusal. The rewrite in
175    // `strip` walks reachable objects from the root and drops the rest (ADR-0020); with no root
176    // to walk from, which objects survive is not stable across runs. A fuzz run found a document
177    // whose second strip differed from its first — 609 bytes, then 485 — because the second pass
178    // dropped an annotation object that the page still referenced through /Annots. Renumbering
179    // then filled that slot with the catalogue, so the page's annotation array pointed at the
180    // document catalogue. strypt had introduced that corruption itself, while returning success
181    // both times.
182    //
183    // Refusing is the fail-closed answer and costs nothing real: a PDF this broken is not one
184    // the user can publish anyway.
185    if !doc.trailer.has(b"Root") {
186        return Err(StryptError::Malformed {
187            format: Format::Pdf,
188            offset: None,
189            detail: MalformedDetail::MissingMarker,
190        });
191    }
192
193    let too_many = u32::try_from(doc.objects.len()).map_or(true, |count| count > limits.max_items);
194    if too_many {
195        return Err(StryptError::LimitExceeded {
196            format: Format::Pdf,
197            limit: ResourceLimit::ItemCount,
198        });
199    }
200
201    // Every stream's declared /Length must be an integer that matches the content actually
202    // parsed out of it.
203    //
204    // ISO 32000-1 §7.3.8.2 requires /Length to be an integer giving the exact byte count
205    // between "stream" and "endstream". When it is not — a fuzz case reached here by writing
206    // "/Length 45." instead of "/Length 45" — lopdf 0.44 parses the object, keeps the
207    // malformed value in the dictionary, and stores *empty* content, because it cannot locate
208    // the stream's end. It reports no error while doing so.
209    //
210    // Left unchecked, the document reaches the rewriter with the stream's bytes already gone.
211    // strypt then writes a file whose /Length still claims 45 bytes over an empty stream, and
212    // reports "nothing to remove; wrote a clean copy" — a structurally invalid PDF, missing
213    // the user's page content, presented as a success. That is the §5.4 failure the whole
214    // design is arranged to avoid, and the reason it went unnoticed is that the verification
215    // pass looks for residual *metadata*, which an emptied stream has none of.
216    //
217    // Refusing costs nothing on real documents: across the synthetic corpus and all 23
218    // parseable PDFs of the real-producer corpus — pdfLaTeX, LibreOffice, Google Docs,
219    // Acrobat and ImageMagick output, compressed streams included — not one stream disagrees
220    // with its declared length.
221    for object in doc.objects.values() {
222        let Ok(stream) = object.as_stream() else {
223            continue;
224        };
225        let declared = stream
226            .dict
227            .get(b"Length")
228            .ok()
229            .and_then(|length| length.as_i64().ok())
230            .and_then(|length| usize::try_from(length).ok());
231        if declared != Some(stream.content.len()) {
232            return Err(StryptError::Malformed {
233                format: Format::Pdf,
234                offset: None,
235                detail: MalformedDetail::LengthOutOfRange,
236            });
237        }
238    }
239
240    Ok(doc)
241}
242
243/// Translate a parser failure into something a user can act on.
244///
245/// The mapping is coarse on purpose. "Your file is truncated" and "your file is not really a
246/// PDF" lead to different actions; which of thirty-four internal variants fired does not.
247fn map_parse_error(error: &lopdf::Error) -> StryptError {
248    use lopdf::Error as E;
249    let detail = match *error {
250        E::Parse(_) | E::Syntax(_) | E::IndirectObject { .. } | E::ObjectIdMismatch => {
251            MalformedDetail::UnexpectedMarker
252        }
253        E::Xref(_) | E::MissingXrefEntry | E::InvalidObjectStream(_) => {
254            MalformedDetail::BrokenIndex
255        }
256        E::InvalidOffset(_) | E::ObjectNotFound(_) | E::NumericCast(_) | E::TryFromInt(_) => {
257            MalformedDetail::LengthOutOfRange
258        }
259        E::ReferenceCycle(_) | E::ReferenceLimit => MalformedDetail::CyclicReference,
260        E::IO(_) => MalformedDetail::Truncated,
261        E::Decryption(_)
262        | E::InvalidPassword
263        | E::AlreadyEncrypted
264        | E::UnsupportedSecurityHandler(_)
265        | E::Unimplemented(_) => MalformedDetail::UnsupportedFeature,
266        _ => MalformedDetail::MissingMarker,
267    };
268    let offset = match *error {
269        E::InvalidOffset(at) | E::IndirectObject { offset: at } => u64::try_from(at).ok(),
270        _ => None,
271    };
272    StryptError::Malformed {
273        format: Format::Pdf,
274        offset,
275        detail,
276    }
277}
278
279/// Keys in the Document Information Dictionary, and what each one exposes.
280///
281/// `/Creator` names the application the document was *authored* in and `/Producer` the one
282/// that wrote the PDF — so a LaTeX paper typically confesses both its editor and its
283/// toolchain version here. Neither identifies a person alone; together with a timestamp and
284/// a font list they narrow the field a great deal (`docs/THREAT_MODEL.md` §4.7).
285const INFO_KEYS: &[(&[u8], MetadataKind)] = &[
286    (b"Author", MetadataKind::PersonalIdentity),
287    (b"Creator", MetadataKind::SoftwareFingerprint),
288    (b"Producer", MetadataKind::SoftwareFingerprint),
289    (b"CreationDate", MetadataKind::Timestamp),
290    (b"ModDate", MetadataKind::Timestamp),
291    (b"Title", MetadataKind::Comment),
292    (b"Subject", MetadataKind::Comment),
293    (b"Keywords", MetadataKind::Comment),
294    (b"Trapped", MetadataKind::Other),
295];
296
297/// Keys removed from *any* dictionary in the document, wherever they appear.
298///
299/// These are safe to remove anywhere because the PDF specification gives them one meaning
300/// each and nothing renders differently without them. `/PieceInfo` is the interesting one:
301/// it is a scratch area where an application may store whatever private state it likes
302/// between editing sessions, and what ends up in it is entirely up to that application.
303const GLOBAL_KEYS: &[(&[u8], MetadataKind)] = &[
304    (b"Metadata", MetadataKind::Other),
305    (b"PieceInfo", MetadataKind::EditingHistory),
306    (b"LastModified", MetadataKind::Timestamp),
307];
308
309/// Annotation subtypes whose `/T` entry is the annotating person's name.
310///
311/// This distinction matters. On a markup annotation `/T` is the author — exactly what we are
312/// here to remove. On a `/Widget`, which is how every interactive form field is drawn, `/T`
313/// is the *field name* that the form's logic and its saved data refer to. Stripping it would
314/// silently break the document, and breaking a user's file to protect them is not a trade
315/// this tool gets to make on their behalf without saying so.
316const MARKUP_ANNOTATION_SUBTYPES: &[&[u8]] = &[
317    b"Text",
318    b"FreeText",
319    b"Line",
320    b"Square",
321    b"Circle",
322    b"Polygon",
323    b"PolyLine",
324    b"Highlight",
325    b"Underline",
326    b"Squiggly",
327    b"StrikeOut",
328    b"Stamp",
329    b"Caret",
330    b"Ink",
331    b"FileAttachment",
332    b"Sound",
333    b"Movie",
334    b"Redact",
335];
336
337/// Walk the document, recording what is there and removing it.
338fn scrub(
339    doc: &mut Document,
340    raw: &[u8],
341    options: &InspectOptions,
342    limits: &ParseLimits,
343) -> Result<Scrubbed> {
344    let mut findings = Vec::new();
345    let mut notes = Vec::new();
346
347    let info_id = trailer_reference(doc, b"Info");
348
349    // Phase one reads. Every object is examined, not merely the ones the catalogue can reach:
350    // an orphan from a superseded revision is exactly the thing worth telling the user about,
351    // and it is invisible to a walk that starts at the root.
352    let object_ids: Vec<ObjectId> = doc.objects.keys().copied().collect();
353    for id in &object_ids {
354        let Some(object) = doc.objects.get(id) else {
355            continue;
356        };
357        examine_object(object, *id, info_id, options, limits, 0, &mut findings)?;
358    }
359    if doc.trailer.has(b"ID") {
360        // The file identifier is a pair of strings that stays stable across saves of the same
361        // document. It identifies nobody by itself and links every copy and every revision of
362        // the document to each other, which for a leaked draft is the whole question.
363        findings.push(Finding::new(
364            MetadataKind::DocumentIdentifier,
365            "trailer /ID",
366            0,
367        ));
368    }
369
370    // A PDF that has been saved more than once ends with more than one %%EOF. Counting them
371    // is cruder than walking the cross-reference chain and tells the user the thing that
372    // actually matters: earlier versions of this document were sitting inside it.
373    let revisions = count_revisions(raw);
374    if revisions > 1 {
375        notes.push(Note::IncrementalHistory {
376            revisions: revisions.saturating_sub(1),
377        });
378    }
379
380    // Phase two writes.
381    doc.trailer.remove(b"Info");
382    doc.trailer.remove(b"ID");
383    for id in &object_ids {
384        if let Some(object) = doc.objects.get_mut(id) {
385            remove_from_object(object, *id, info_id, limits, 0)?;
386        }
387    }
388
389    if object_ids
390        .iter()
391        .filter_map(|id| doc.objects.get(id))
392        .any(is_embedded_file_holder)
393    {
394        // strypt does not open embedded files. Recursing into them means recursing into
395        // arbitrary nested content, which is a zip-bomb-shaped problem that Phase 2 has to
396        // decide about explicitly. Until then the user is told, because an attachment
397        // carrying its own metadata inside a document reported as clean is precisely the
398        // over-trust this tool must not create (`docs/THREAT_MODEL.md` §5.6).
399        notes.push(Note::OutOfScopeContent {
400            location: "embedded file attachment".into(),
401        });
402    }
403
404    Ok(Scrubbed { findings, notes })
405}
406
407/// Resolve a reference held in the trailer, if it is one.
408fn trailer_reference(doc: &Document, key: &[u8]) -> Option<ObjectId> {
409    doc.trailer
410        .get(key)
411        .ok()
412        .and_then(|o| o.as_reference().ok())
413}
414
415/// Count `%%EOF` markers, each of which terminates one revision of the document.
416fn count_revisions(raw: &[u8]) -> usize {
417    const EOF: &[u8] = b"%%EOF";
418    raw.windows(EOF.len()).filter(|w| *w == EOF).count()
419}
420
421/// Record everything identifying inside one object.
422fn examine_object(
423    object: &Object,
424    id: ObjectId,
425    info_id: Option<ObjectId>,
426    options: &InspectOptions,
427    limits: &ParseLimits,
428    depth: u32,
429    out: &mut Vec<Finding>,
430) -> Result<()> {
431    if depth > limits.max_depth {
432        return Err(StryptError::LimitExceeded {
433            format: Format::Pdf,
434            limit: ResourceLimit::Depth,
435        });
436    }
437    match object {
438        Object::Dictionary(dict) => {
439            if Some(id) == info_id {
440                examine_info(dict, options, out);
441            }
442            examine_dictionary(dict, id, info_id, options, limits, depth, out)?;
443        }
444        Object::Stream(stream) => {
445            if is_metadata_stream(&stream.dict) {
446                examine_xmp(stream, options, out);
447            }
448            examine_dictionary(&stream.dict, id, info_id, options, limits, depth, out)?;
449        }
450        Object::Array(items) => {
451            for item in items {
452                examine_object(
453                    item,
454                    id,
455                    info_id,
456                    options,
457                    limits,
458                    depth.saturating_add(1),
459                    out,
460                )?;
461            }
462        }
463        _ => {}
464    }
465    Ok(())
466}
467
468/// Record the Document Information Dictionary, including keys the specification never
469/// defined — applications add their own freely, and a custom key is no less identifying for
470/// being non-standard.
471fn examine_info(dict: &Dictionary, options: &InspectOptions, out: &mut Vec<Finding>) {
472    for (key, value) in dict {
473        let kind = INFO_KEYS
474            .iter()
475            .find(|(name, _)| *name == key.as_slice())
476            .map_or(MetadataKind::Other, |(_, kind)| *kind);
477        out.push(
478            Finding::new(kind, "/Info", value_size(value))
479                .with_field(name_of(key))
480                .with_value(options, || describe(value)),
481        );
482    }
483}
484
485/// Record the globally-removable keys, and annotation authorship.
486fn examine_dictionary(
487    dict: &Dictionary,
488    id: ObjectId,
489    info_id: Option<ObjectId>,
490    options: &InspectOptions,
491    limits: &ParseLimits,
492    depth: u32,
493    out: &mut Vec<Finding>,
494) -> Result<()> {
495    for (name, kind) in GLOBAL_KEYS {
496        if let Ok(value) = dict.get(name) {
497            out.push(
498                Finding::new(*kind, format!("/{}", name_of(name)), value_size(value))
499                    .with_field(name_of(name)),
500            );
501        }
502    }
503    if is_markup_annotation(dict) {
504        for (name, kind) in [
505            (&b"T"[..], MetadataKind::PersonalIdentity),
506            (&b"M"[..], MetadataKind::Timestamp),
507            (&b"CreationDate"[..], MetadataKind::Timestamp),
508            (&b"NM"[..], MetadataKind::DocumentIdentifier),
509        ] {
510            if let Ok(value) = dict.get(name) {
511                out.push(
512                    Finding::new(kind, "annotation", value_size(value))
513                        .with_field(name_of(name))
514                        .with_value(options, || describe(value)),
515                );
516            }
517        }
518    }
519    // Embedded-file parameters carry their own creation and modification dates, which survive
520    // every scrub aimed only at the containing document.
521    if let Ok(Object::Dictionary(params)) = dict.get(b"Params") {
522        for name in [&b"CreationDate"[..], &b"ModDate"[..], &b"CheckSum"[..]] {
523            if let Ok(value) = params.get(name) {
524                out.push(
525                    Finding::new(MetadataKind::Timestamp, "/Params", value_size(value))
526                        .with_field(name_of(name)),
527                );
528            }
529        }
530    }
531    for (_, value) in dict {
532        examine_object(
533            value,
534            id,
535            info_id,
536            options,
537            limits,
538            depth.saturating_add(1),
539            out,
540        )?;
541    }
542    Ok(())
543}
544
545/// Scan an XMP packet for the properties worth naming.
546///
547/// Only unfiltered packets are scanned. ISO 32000-1 §14.3.2 recommends that a metadata stream
548/// be left uncompressed precisely so it can be read without parsing the whole document, and
549/// in practice they almost always are. Refusing to inflate the rare compressed one avoids
550/// handing an attacker a decompression bomb in exchange for a slightly more detailed report —
551/// the packet is still found, still reported, and still removed either way.
552fn examine_xmp(stream: &lopdf::Stream, options: &InspectOptions, out: &mut Vec<Finding>) {
553    if stream.dict.has(b"Filter") {
554        out.push(
555            Finding::new(
556                MetadataKind::Other,
557                "XMP packet",
558                as_u64(stream.content.len()),
559            )
560            .with_field("Metadata (encoded)"),
561        );
562        return;
563    }
564    out.extend(xmp::scan(&stream.content, "XMP packet", options));
565}
566
567/// Remove everything [`examine_object`] reports, from one object.
568fn remove_from_object(
569    object: &mut Object,
570    id: ObjectId,
571    info_id: Option<ObjectId>,
572    limits: &ParseLimits,
573    depth: u32,
574) -> Result<()> {
575    if depth > limits.max_depth {
576        return Err(StryptError::LimitExceeded {
577            format: Format::Pdf,
578            limit: ResourceLimit::Depth,
579        });
580    }
581    match object {
582        Object::Dictionary(dict) => {
583            if Some(id) == info_id {
584                // The Info dictionary is emptied as well as unlinked. Unlinking alone would
585                // be enough for a correct pruner, and relying on that would make this
586                // handler's correctness depend on the pruner's — a dependency worth not
587                // having in the one place where being wrong means a name survives.
588                *dict = Dictionary::new();
589                return Ok(());
590            }
591            remove_from_dictionary(dict, id, info_id, limits, depth)?;
592        }
593        Object::Stream(stream) => {
594            remove_from_dictionary(&mut stream.dict, id, info_id, limits, depth)?;
595        }
596        Object::Array(items) => {
597            for item in items {
598                remove_from_object(item, id, info_id, limits, depth.saturating_add(1))?;
599            }
600        }
601        _ => {}
602    }
603    Ok(())
604}
605
606/// Rewrite every `Real(-0.0)` in the document as `Real(0.0)`.
607///
608/// See the call site for why this is done at all. Note the deliberate use of `is_sign_negative`
609/// rather than `== -0.0`: in IEEE 754 `-0.0 == 0.0` is true, so the obvious comparison matches
610/// positive zero as well and would rewrite values that were never a problem.
611fn normalise_negative_zero(doc: &mut Document, limits: &ParseLimits) -> Result<()> {
612    for object in doc.objects.values_mut() {
613        normalise_object(object, limits, 0)?;
614    }
615    // The trailer is not in `objects` and is reached only by walking it explicitly. Missing it
616    // is how the first version of this fix passed every local test and still failed: CI's fuzz
617    // run moved a negative zero into the trailer within minutes, and the assertion fired again
618    // on a document whose object graph was entirely clean.
619    for (_, value) in &mut doc.trailer {
620        normalise_object(value, limits, 0)?;
621    }
622    Ok(())
623}
624
625/// Walk one object, collapsing negative zeros wherever they nest.
626fn normalise_object(object: &mut Object, limits: &ParseLimits, depth: u32) -> Result<()> {
627    if depth > limits.max_depth {
628        return Err(StryptError::LimitExceeded {
629            format: Format::Pdf,
630            limit: ResourceLimit::Depth,
631        });
632    }
633    match object {
634        Object::Real(value) if value.is_sign_negative() && *value == 0.0 => {
635            *value = 0.0;
636        }
637        Object::Dictionary(dict) => {
638            for (_, value) in dict.iter_mut() {
639                normalise_object(value, limits, depth.saturating_add(1))?;
640            }
641        }
642        Object::Stream(stream) => {
643            for (_, value) in &mut stream.dict {
644                normalise_object(value, limits, depth.saturating_add(1))?;
645            }
646        }
647        Object::Array(items) => {
648            for item in items {
649                normalise_object(item, limits, depth.saturating_add(1))?;
650            }
651        }
652        _ => {}
653    }
654    Ok(())
655}
656
657/// Remove identifying keys from one dictionary and everything nested inside it.
658fn remove_from_dictionary(
659    dict: &mut Dictionary,
660    id: ObjectId,
661    info_id: Option<ObjectId>,
662    limits: &ParseLimits,
663    depth: u32,
664) -> Result<()> {
665    for (name, _) in GLOBAL_KEYS {
666        dict.remove(name);
667    }
668    if is_markup_annotation(dict) {
669        dict.remove(b"T");
670        dict.remove(b"M");
671        dict.remove(b"CreationDate");
672        dict.remove(b"NM");
673    }
674    if let Ok(Object::Dictionary(params)) = dict.get_mut(b"Params") {
675        params.remove(b"CreationDate");
676        params.remove(b"ModDate");
677        params.remove(b"CheckSum");
678    }
679    for (_, value) in &mut *dict {
680        remove_from_object(value, id, info_id, limits, depth.saturating_add(1))?;
681    }
682    Ok(())
683}
684
685/// True for a stream that is an XMP metadata packet.
686fn is_metadata_stream(dict: &Dictionary) -> bool {
687    dict.get_type().is_ok_and(|t| t == b"Metadata")
688        || dict
689            .get(b"Subtype")
690            .and_then(Object::as_name)
691            .is_ok_and(|s| s == b"XML")
692}
693
694/// True for an annotation whose `/T` names a person rather than a form field.
695fn is_markup_annotation(dict: &Dictionary) -> bool {
696    let Ok(subtype) = dict.get(b"Subtype").and_then(Object::as_name) else {
697        return false;
698    };
699    MARKUP_ANNOTATION_SUBTYPES.contains(&subtype)
700}
701
702/// True for a file-specification dictionary, which is how a PDF carries an attachment.
703fn is_embedded_file_holder(object: &Object) -> bool {
704    let dict = match object {
705        Object::Dictionary(dict) => dict,
706        Object::Stream(stream) => &stream.dict,
707        _ => return false,
708    };
709    dict.has_type(b"Filespec") || dict.has(b"EmbeddedFiles")
710}
711
712/// The size of a value in bytes, where it has a meaningful one.
713fn value_size(object: &Object) -> u64 {
714    match object {
715        Object::String(bytes, _) | Object::Name(bytes) => as_u64(bytes.len()),
716        Object::Stream(stream) => as_u64(stream.content.len()),
717        _ => 0,
718    }
719}
720
721/// Render a value, for the callers that opted into seeing values.
722fn describe(object: &Object) -> MetadataValue {
723    match object {
724        Object::String(bytes, _) | Object::Name(bytes) => MetadataValue::Text(name_of(bytes)),
725        Object::Integer(n) => MetadataValue::Text(n.to_string()),
726        Object::Boolean(b) => MetadataValue::Text(b.to_string()),
727        other => MetadataValue::Opaque {
728            bytes: value_size(other),
729        },
730    }
731}
732
733/// Widen a length for reporting. Saturating rather than fallible: a report field is not worth
734/// failing an otherwise-successful strip over.
735fn as_u64(value: usize) -> u64 {
736    u64::try_from(value).unwrap_or(u64::MAX)
737}