strypt_core/formats/pdf.rs
1//! PDF.
2//!
3//! The hardest of the Phase 1 formats, and the one where hidden data is least likely to be
4//! where you look for it. The Document Information Dictionary is the easy part and the part
5//! every tutorial covers; the leaks that matter live in XMP packets, per-object metadata,
6//! annotation authorship, and — above all — in objects left behind by incremental updates,
7//! which are physically present in the file and reachable with a hex editor long after the
8//! "current" version of the document stopped referring to them.
9//!
10//! # Full rewrite, not incremental patching (ADR-0020)
11//!
12//! A PDF can be edited by appending: the original bytes stay, and a new cross-reference
13//! section at the end says which objects supersede which. Nulling the Info dictionary with
14//! another such append is easy, fast, and preserves the file almost perfectly — and it leaves
15//! every previous author name exactly where it was, four kilobytes up the file. For this
16//! tool that is not a lesser fix, it is a silent failure: the user is told the document is
17//! clean and publishes it.
18//!
19//! So the document is parsed into its object graph, scrubbed, pruned to what the catalogue
20//! can actually reach, renumbered, and written out fresh. Everything unreachable — every
21//! superseded revision — is gone because it is never written, not because it was overwritten.
22//!
23//! The cost is honest and worth stating: the output is not byte-comparable with the input,
24//! object numbering changes, and files using features the rewrite cannot faithfully reproduce
25//! are refused rather than mangled. Refusing is the correct half of that trade.
26
27use lopdf::{Dictionary, Document, Object, ObjectId};
28
29use crate::detect::Format;
30use crate::error::{MalformedDetail, ResourceLimit, Result, StryptError};
31use crate::formats::xmp::name_of;
32use crate::formats::{MetadataHandler, ParseLimits, StripOptions, Stripped, xmp};
33use crate::report::{
34 Finding, InspectOptions, MetadataKind, MetadataReport, MetadataValue, Note, StripReport,
35};
36
37/// Removal of metadata from PDF documents.
38#[derive(Debug, Clone, Copy, Default)]
39pub struct PdfHandler;
40
41impl MetadataHandler for PdfHandler {
42 fn name(&self) -> &'static str {
43 Format::Pdf.id()
44 }
45
46 fn format(&self) -> Format {
47 Format::Pdf
48 }
49
50 fn inspect(&self, input: &[u8], options: &InspectOptions) -> Result<MetadataReport> {
51 // Inspection loads its own copy of the document and runs the *same* scrub that
52 // stripping does, then throws the result away. That is deliberate: it makes
53 // "everything `strip` removes is something `inspect` can see" true by construction
54 // rather than by two code paths agreeing to stay in step. The verification pass in
55 // `crate::pipeline` is only meaningful if that holds (`docs/ARCHITECTURE.md` §3).
56 let limits = ParseLimits::default();
57 let mut doc = load(input, &limits)?;
58 let scrubbed = scrub(&mut doc, input, options, &limits)?;
59 Ok(MetadataReport {
60 format: Format::Pdf,
61 findings: scrubbed.findings,
62 notes: scrubbed.notes,
63 })
64 }
65
66 fn strip(&self, input: &[u8], options: &StripOptions) -> Result<Stripped> {
67 let mut doc = load(input, &options.limits)?;
68 let mut scrubbed = scrub(&mut doc, input, &options.inspect, &options.limits)?;
69
70 // Drop every object the catalogue can no longer reach. This is the step that removes
71 // superseded revisions, and it has to come after scrubbing so that objects orphaned
72 // *by* the scrub — the Info dictionary, XMP streams — go with them.
73 let pruned = doc.prune_objects();
74 if !pruned.is_empty() {
75 scrubbed.notes.push(Note::OrphanedObjectsRemoved {
76 objects: pruned.len(),
77 });
78 }
79
80 // Renumber so that output depends only on the object graph, not on the numbering the
81 // input happened to use. Without this, two documents that scrub to the same content
82 // serialise differently, and the determinism invariant
83 // (`docs/TESTING_STRATEGY.md` §1) fails for no good reason.
84 renumber_stably(&mut doc)?;
85
86 // Collapse negative zero to zero for the same reason renumbering exists above: without
87 // it the output is not stable under a second strip. lopdf writes `Real(-0.0)` as `-0`,
88 // dropping the decimal point; re-parsing `-0` yields `Integer(0)`, which writes as `0`.
89 // So one strip of a file containing `-0.` differs from two, breaking the idempotence
90 // invariant (`docs/TESTING_STRATEGY.md` §1, invariant 3).
91 //
92 // Rewriting a number in the user's document needs justifying, since this handler
93 // otherwise refuses rather than repairs (ADR-0018). It is sound here because ISO
94 // 32000-1 §7.3.3 gives PDF numbers no signed zero: `-0` and `0` denote the same value,
95 // there is no operator that can distinguish them, and no renderer can. The alternative
96 // — refusing a valid file over a lost minus sign that changes nothing — costs the user
97 // their document to protect a distinction the format does not make.
98 normalise_negative_zero(&mut doc, &options.limits)?;
99
100 // Guarded for the same reason as `load`: serialisation walks a document graph built
101 // from hostile input, so it is dependency code on untrusted data just as parsing is.
102 // Failing here means nothing is written, which is the correct half of the trade.
103 let mut bytes = Vec::new();
104 crate::panic_guard::guard(
105 || {
106 doc.save_to(&mut bytes).map_err(|source| StryptError::Io {
107 action: crate::error::IoAction::WritingOutput,
108 source,
109 })
110 },
111 || StryptError::Malformed {
112 format: Format::Pdf,
113 offset: None,
114 detail: MalformedDetail::DependencyPanic,
115 },
116 )?;
117
118 // Nothing is handed back until the bytes just written have been read again and shown
119 // to be the document that was written. See `verify_round_trip`.
120 verify_round_trip(&doc, &bytes)?;
121
122 Ok(Stripped {
123 report: StripReport {
124 format: Format::Pdf,
125 removed: scrubbed.findings,
126 retained: Vec::new(),
127 notes: scrubbed.notes,
128 input_bytes: as_u64(input.len()),
129 output_bytes: as_u64(bytes.len()),
130 },
131 bytes,
132 })
133 }
134}
135
136/// Refuse output that does not read back as the document that was written.
137///
138/// The rewrite (ADR-0020) assumes serialising a parsed document and re-parsing it returns the
139/// same document. For a lenient parser on hostile input that assumption does not hold, and
140/// when it breaks it breaks silently: `lopdf` will accept a dictionary whose keys came out of
141/// mangled bytes — `/Annotst 1 /[^@018064665...` in the file that found this — and then write
142/// it back in a form it cannot itself read. The object is written, and disappears when the
143/// file is next opened.
144///
145/// That is the failure this guard exists for, and it is the dangerous kind. A document whose
146/// only `/Page` is lost on reload has a `/Pages` node still claiming `/Count 1` and a `/Kids`
147/// array pointing at an object that is no longer there. strypt wrote that file and reported
148/// success; a second strip then pruned what had become unreachable and wrote a 230-byte
149/// document with a dangling page reference, reporting success again. No metadata survived
150/// either pass, so the verification pass — which searches output for residual metadata — saw
151/// nothing wrong. It cannot: it is looking for what should be absent, not for what should
152/// still be present.
153///
154/// Checking a full structural equivalence would be a second implementation of the rewrite, so
155/// this checks the two properties whose failure means the output is not the document:
156///
157/// 1. **Every object written is present on reload.** This is what catches an object that
158/// serialised into something unparseable. Comparing ids alone is enough — an object that
159/// round-trips to a different id is caught by the same comparison.
160/// 2. **The page tree survives.** A document that had pages before writing and none after has
161/// lost the thing it exists to carry, even when every object id happens to match.
162///
163/// A failure refuses the file. That is the fail-closed answer (`CLAUDE.md` §3 rule 6): strypt
164/// cannot faithfully rewrite this document, so it declines to rather than handing back a
165/// broken one with a success report. Refusing costs the user a file that was already too
166/// damaged to survive a rewrite; the alternative cost them a document they believed was clean.
167fn verify_round_trip(written: &Document, bytes: &[u8]) -> Result<()> {
168 let reloaded = crate::panic_guard::guard(
169 || Document::load_mem(bytes).map_err(|e| map_parse_error(&e)),
170 || StryptError::Malformed {
171 format: Format::Pdf,
172 offset: None,
173 detail: MalformedDetail::DependencyPanic,
174 },
175 )
176 .map_err(|_| StryptError::Malformed {
177 format: Format::Pdf,
178 offset: None,
179 detail: MalformedDetail::NotRoundTrippable,
180 })?;
181
182 let written_ids: Vec<ObjectId> = written.objects.keys().copied().collect();
183 let reloaded_ids: Vec<ObjectId> = reloaded.objects.keys().copied().collect();
184 if written_ids != reloaded_ids {
185 return Err(StryptError::Malformed {
186 format: Format::Pdf,
187 offset: None,
188 detail: MalformedDetail::NotRoundTrippable,
189 });
190 }
191
192 // Only an emptied page tree is a failure, not an empty one: a document that had no pages
193 // to begin with is degenerate but not something this rewrite broke.
194 if written.page_iter().next().is_some() && reloaded.page_iter().next().is_none() {
195 return Err(StryptError::Malformed {
196 format: Format::Pdf,
197 offset: None,
198 detail: MalformedDetail::NotRoundTrippable,
199 });
200 }
201
202 Ok(())
203}
204
205/// How many times `renumber_stably` will renumber before refusing the document.
206///
207/// A well-formed file reaches its fixed point on the second call — the first assigns
208/// `1..=n`, the second confirms nothing moved. The degenerate case below needs a third.
209/// Four is that plus margin; a document still moving after four is not converging, and
210/// looping harder would only delay the refusal.
211const MAX_RENUMBER_ROUNDS: usize = 4;
212
213/// Renumber until the numbering stops changing, or refuse the document.
214///
215/// `lopdf::renumber_objects` is **not idempotent**, which matters because strypt's idempotence
216/// invariant (`docs/TESTING_STRATEGY.md` §1, invariant 3) is stated byte-for-byte on the *first*
217/// re-strip. Before renumbering sequentially, `renumber_objects_with` checks whether the page
218/// order matches ascending object ids and, if it does not, permutes the page objects so that it
219/// does (`lopdf` 0.44.0 `src/processor.rs`). That check reads the numbering the previous step
220/// produced, so one pass can leave a document that a second pass would reorder again.
221///
222/// A sustained fuzz run found the case where that is reachable: a document whose page tree is
223/// self-referential — object 2 is a `/Page` whose own `/Kids` array lists object 2 — so pruning
224/// and renumbering changed which objects `page_iter` yields and in which order. The first strip
225/// produced pages ordered `[3, 2]`, descending; the second saw the mismatch and swapped objects
226/// 2 and 3. Same length, same content, 145 bytes different. It settled from the third strip on,
227/// so this was never an endless flip — but "stable eventually" is not the invariant, and a user
228/// who strips a file twice must not get two different files.
229///
230/// Iterating to a fixed point fixes it by construction rather than by reasoning about a
231/// dependency's internals: the document is only serialised once renumbering has been shown to be
232/// a no-op on it, so re-loading and renumbering that output cannot move anything either.
233///
234/// This does not change the output of any document that was already stable — for those the
235/// second round is the confirmation that would have been skipped, not a second permutation.
236///
237/// Non-convergence is refused rather than accepted at whatever state the last round left, which
238/// is the fail-closed half of the trade (ADR-0018, and `CLAUDE.md` §3 rule 6). Emitting a file
239/// whose numbering strypt could not settle would mean handing the user output it cannot promise
240/// is reproducible.
241fn renumber_stably(doc: &mut Document) -> Result<()> {
242 for _ in 0..MAX_RENUMBER_ROUNDS {
243 // Both the id set and the page order have to be compared. The ids alone are not enough:
244 // the reordering step permutes which object holds which page while leaving the set of
245 // ids exactly as it was, so comparing ids only would report a fixed point on the very
246 // pass that moved something.
247 let ids_before: Vec<ObjectId> = doc.objects.keys().copied().collect();
248 let pages_before: Vec<ObjectId> = doc.page_iter().collect();
249
250 doc.renumber_objects();
251
252 let ids_after: Vec<ObjectId> = doc.objects.keys().copied().collect();
253 let pages_after: Vec<ObjectId> = doc.page_iter().collect();
254
255 if ids_before == ids_after && pages_before == pages_after {
256 return Ok(());
257 }
258 }
259
260 Err(StryptError::Malformed {
261 format: Format::Pdf,
262 offset: None,
263 detail: MalformedDetail::CyclicReference,
264 })
265}
266
267/// Findings and caveats from one pass over a document.
268struct Scrubbed {
269 findings: Vec<Finding>,
270 notes: Vec<Note>,
271}
272
273/// Parse `input`, refusing documents this handler must not rewrite.
274fn load(input: &[u8], limits: &ParseLimits) -> Result<Document> {
275 // `lopdf` is third-party code parsing attacker-controlled bytes, and ADR-0006's no-panic
276 // rule does not reach inside it (ADR-0018). A sustained fuzz run found an integer overflow
277 // in its cross-reference parser, which with `overflow-checks` on in release meant the
278 // shipped binary aborted with a stack trace instead of refusing the file. Contained here
279 // so it reaches the user as an ordinary refusal; see `crate::panic_guard` for what that
280 // does and does not cover.
281 let doc = crate::panic_guard::guard(
282 || Document::load_mem(input).map_err(|e| map_parse_error(&e)),
283 || StryptError::Malformed {
284 format: Format::Pdf,
285 offset: None,
286 detail: MalformedDetail::DependencyPanic,
287 },
288 )?;
289
290 // Encrypted documents are refused rather than rewritten. lopdf can open one protected by
291 // an empty owner password, and it would be technically easy to emit a decrypted copy —
292 // but that hands the user a file with its protection quietly removed, which is a change
293 // to their document's security they did not ask for and would not necessarily notice.
294 // Fail closed and say so.
295 if doc.is_encrypted() || doc.was_encrypted() {
296 return Err(StryptError::Malformed {
297 format: Format::Pdf,
298 offset: None,
299 detail: MalformedDetail::UnsupportedFeature,
300 });
301 }
302
303 // A trailer with no /Root is refused rather than processed.
304 //
305 // ISO 32000-1 §7.5.5 makes /Root a required trailer entry: it names the document catalogue,
306 // which is the single root every other object hangs off. Without it the file has no defined
307 // entry point, and no viewer will open it.
308 //
309 // strypt used to accept such a file, and the result was worse than a refusal. The rewrite in
310 // `strip` walks reachable objects from the root and drops the rest (ADR-0020); with no root
311 // to walk from, which objects survive is not stable across runs. A fuzz run found a document
312 // whose second strip differed from its first — 609 bytes, then 485 — because the second pass
313 // dropped an annotation object that the page still referenced through /Annots. Renumbering
314 // then filled that slot with the catalogue, so the page's annotation array pointed at the
315 // document catalogue. strypt had introduced that corruption itself, while returning success
316 // both times.
317 //
318 // Refusing is the fail-closed answer and costs nothing real: a PDF this broken is not one
319 // the user can publish anyway.
320 if !doc.trailer.has(b"Root") {
321 return Err(StryptError::Malformed {
322 format: Format::Pdf,
323 offset: None,
324 detail: MalformedDetail::MissingMarker,
325 });
326 }
327
328 let too_many = u32::try_from(doc.objects.len()).map_or(true, |count| count > limits.max_items);
329 if too_many {
330 return Err(StryptError::LimitExceeded {
331 format: Format::Pdf,
332 limit: ResourceLimit::ItemCount,
333 });
334 }
335
336 // Every stream's declared /Length must be an integer that matches the content actually
337 // parsed out of it.
338 //
339 // ISO 32000-1 §7.3.8.2 requires /Length to be an integer giving the exact byte count
340 // between "stream" and "endstream". When it is not — a fuzz case reached here by writing
341 // "/Length 45." instead of "/Length 45" — lopdf 0.44 parses the object, keeps the
342 // malformed value in the dictionary, and stores *empty* content, because it cannot locate
343 // the stream's end. It reports no error while doing so.
344 //
345 // Left unchecked, the document reaches the rewriter with the stream's bytes already gone.
346 // strypt then writes a file whose /Length still claims 45 bytes over an empty stream, and
347 // reports "nothing to remove; wrote a clean copy" — a structurally invalid PDF, missing
348 // the user's page content, presented as a success. That is the §5.4 failure the whole
349 // design is arranged to avoid, and the reason it went unnoticed is that the verification
350 // pass looks for residual *metadata*, which an emptied stream has none of.
351 //
352 // Refusing costs nothing on real documents: across the synthetic corpus and all 23
353 // parseable PDFs of the real-producer corpus — pdfLaTeX, LibreOffice, Google Docs,
354 // Acrobat and ImageMagick output, compressed streams included — not one stream disagrees
355 // with its declared length.
356 for object in doc.objects.values() {
357 let Ok(stream) = object.as_stream() else {
358 continue;
359 };
360 let declared = stream
361 .dict
362 .get(b"Length")
363 .ok()
364 .and_then(|length| length.as_i64().ok())
365 .and_then(|length| usize::try_from(length).ok());
366 if declared != Some(stream.content.len()) {
367 return Err(StryptError::Malformed {
368 format: Format::Pdf,
369 offset: None,
370 detail: MalformedDetail::LengthOutOfRange,
371 });
372 }
373 }
374
375 Ok(doc)
376}
377
378/// Translate a parser failure into something a user can act on.
379///
380/// The mapping is coarse on purpose. "Your file is truncated" and "your file is not really a
381/// PDF" lead to different actions; which of thirty-four internal variants fired does not.
382fn map_parse_error(error: &lopdf::Error) -> StryptError {
383 use lopdf::Error as E;
384 let detail = match *error {
385 E::Parse(_) | E::Syntax(_) | E::IndirectObject { .. } | E::ObjectIdMismatch => {
386 MalformedDetail::UnexpectedMarker
387 }
388 E::Xref(_) | E::MissingXrefEntry | E::InvalidObjectStream(_) => {
389 MalformedDetail::BrokenIndex
390 }
391 E::InvalidOffset(_) | E::ObjectNotFound(_) | E::NumericCast(_) | E::TryFromInt(_) => {
392 MalformedDetail::LengthOutOfRange
393 }
394 E::ReferenceCycle(_) | E::ReferenceLimit => MalformedDetail::CyclicReference,
395 E::IO(_) => MalformedDetail::Truncated,
396 E::Decryption(_)
397 | E::InvalidPassword
398 | E::AlreadyEncrypted
399 | E::UnsupportedSecurityHandler(_)
400 | E::Unimplemented(_) => MalformedDetail::UnsupportedFeature,
401 _ => MalformedDetail::MissingMarker,
402 };
403 let offset = match *error {
404 E::InvalidOffset(at) | E::IndirectObject { offset: at } => u64::try_from(at).ok(),
405 _ => None,
406 };
407 StryptError::Malformed {
408 format: Format::Pdf,
409 offset,
410 detail,
411 }
412}
413
414/// Keys in the Document Information Dictionary, and what each one exposes.
415///
416/// `/Creator` names the application the document was *authored* in and `/Producer` the one
417/// that wrote the PDF — so a LaTeX paper typically confesses both its editor and its
418/// toolchain version here. Neither identifies a person alone; together with a timestamp and
419/// a font list they narrow the field a great deal (`docs/THREAT_MODEL.md` §4.7).
420const INFO_KEYS: &[(&[u8], MetadataKind)] = &[
421 (b"Author", MetadataKind::PersonalIdentity),
422 (b"Creator", MetadataKind::SoftwareFingerprint),
423 (b"Producer", MetadataKind::SoftwareFingerprint),
424 (b"CreationDate", MetadataKind::Timestamp),
425 (b"ModDate", MetadataKind::Timestamp),
426 (b"Title", MetadataKind::Comment),
427 (b"Subject", MetadataKind::Comment),
428 (b"Keywords", MetadataKind::Comment),
429 (b"Trapped", MetadataKind::Other),
430];
431
432/// Keys removed from *any* dictionary in the document, wherever they appear.
433///
434/// These are safe to remove anywhere because the PDF specification gives them one meaning
435/// each and nothing renders differently without them. `/PieceInfo` is the interesting one:
436/// it is a scratch area where an application may store whatever private state it likes
437/// between editing sessions, and what ends up in it is entirely up to that application.
438const GLOBAL_KEYS: &[(&[u8], MetadataKind)] = &[
439 (b"Metadata", MetadataKind::Other),
440 (b"PieceInfo", MetadataKind::EditingHistory),
441 (b"LastModified", MetadataKind::Timestamp),
442];
443
444/// Annotation subtypes whose `/T` entry is the annotating person's name.
445///
446/// This distinction matters. On a markup annotation `/T` is the author — exactly what we are
447/// here to remove. On a `/Widget`, which is how every interactive form field is drawn, `/T`
448/// is the *field name* that the form's logic and its saved data refer to. Stripping it would
449/// silently break the document, and breaking a user's file to protect them is not a trade
450/// this tool gets to make on their behalf without saying so.
451const MARKUP_ANNOTATION_SUBTYPES: &[&[u8]] = &[
452 b"Text",
453 b"FreeText",
454 b"Line",
455 b"Square",
456 b"Circle",
457 b"Polygon",
458 b"PolyLine",
459 b"Highlight",
460 b"Underline",
461 b"Squiggly",
462 b"StrikeOut",
463 b"Stamp",
464 b"Caret",
465 b"Ink",
466 b"FileAttachment",
467 b"Sound",
468 b"Movie",
469 b"Redact",
470];
471
472/// Walk the document, recording what is there and removing it.
473fn scrub(
474 doc: &mut Document,
475 raw: &[u8],
476 options: &InspectOptions,
477 limits: &ParseLimits,
478) -> Result<Scrubbed> {
479 let mut findings = Vec::new();
480 let mut notes = Vec::new();
481
482 let info_id = trailer_reference(doc, b"Info");
483
484 // Phase one reads. Every object is examined, not merely the ones the catalogue can reach:
485 // an orphan from a superseded revision is exactly the thing worth telling the user about,
486 // and it is invisible to a walk that starts at the root.
487 let object_ids: Vec<ObjectId> = doc.objects.keys().copied().collect();
488 for id in &object_ids {
489 let Some(object) = doc.objects.get(id) else {
490 continue;
491 };
492 examine_object(object, *id, info_id, options, limits, 0, &mut findings)?;
493 }
494 if doc.trailer.has(b"ID") {
495 // The file identifier is a pair of strings that stays stable across saves of the same
496 // document. It identifies nobody by itself and links every copy and every revision of
497 // the document to each other, which for a leaked draft is the whole question.
498 findings.push(Finding::new(
499 MetadataKind::DocumentIdentifier,
500 "trailer /ID",
501 0,
502 ));
503 }
504
505 // A PDF that has been saved more than once ends with more than one %%EOF. Counting them
506 // is cruder than walking the cross-reference chain and tells the user the thing that
507 // actually matters: earlier versions of this document were sitting inside it.
508 let revisions = count_revisions(raw);
509 if revisions > 1 {
510 notes.push(Note::IncrementalHistory {
511 revisions: revisions.saturating_sub(1),
512 });
513 }
514
515 // Phase two writes.
516 doc.trailer.remove(b"Info");
517 doc.trailer.remove(b"ID");
518 for id in &object_ids {
519 if let Some(object) = doc.objects.get_mut(id) {
520 remove_from_object(object, *id, info_id, limits, 0)?;
521 }
522 }
523
524 if object_ids
525 .iter()
526 .filter_map(|id| doc.objects.get(id))
527 .any(is_embedded_file_holder)
528 {
529 // strypt does not open embedded files. Recursing into them means recursing into
530 // arbitrary nested content, which is a zip-bomb-shaped problem that Phase 2 has to
531 // decide about explicitly. Until then the user is told, because an attachment
532 // carrying its own metadata inside a document reported as clean is precisely the
533 // over-trust this tool must not create (`docs/THREAT_MODEL.md` §5.6).
534 notes.push(Note::OutOfScopeContent {
535 location: "embedded file attachment".into(),
536 });
537 }
538
539 Ok(Scrubbed { findings, notes })
540}
541
542/// Resolve a reference held in the trailer, if it is one.
543fn trailer_reference(doc: &Document, key: &[u8]) -> Option<ObjectId> {
544 doc.trailer
545 .get(key)
546 .ok()
547 .and_then(|o| o.as_reference().ok())
548}
549
550/// Count `%%EOF` markers, each of which terminates one revision of the document.
551fn count_revisions(raw: &[u8]) -> usize {
552 const EOF: &[u8] = b"%%EOF";
553 raw.windows(EOF.len()).filter(|w| *w == EOF).count()
554}
555
556/// Record everything identifying inside one object.
557fn examine_object(
558 object: &Object,
559 id: ObjectId,
560 info_id: Option<ObjectId>,
561 options: &InspectOptions,
562 limits: &ParseLimits,
563 depth: u32,
564 out: &mut Vec<Finding>,
565) -> Result<()> {
566 if depth > limits.max_depth {
567 return Err(StryptError::LimitExceeded {
568 format: Format::Pdf,
569 limit: ResourceLimit::Depth,
570 });
571 }
572 match object {
573 Object::Dictionary(dict) => {
574 if Some(id) == info_id {
575 examine_info(dict, options, out);
576 }
577 examine_dictionary(dict, id, info_id, options, limits, depth, out)?;
578 }
579 Object::Stream(stream) => {
580 if is_metadata_stream(&stream.dict) {
581 examine_xmp(stream, options, out);
582 }
583 examine_dictionary(&stream.dict, id, info_id, options, limits, depth, out)?;
584 }
585 Object::Array(items) => {
586 for item in items {
587 examine_object(
588 item,
589 id,
590 info_id,
591 options,
592 limits,
593 depth.saturating_add(1),
594 out,
595 )?;
596 }
597 }
598 _ => {}
599 }
600 Ok(())
601}
602
603/// Record the Document Information Dictionary, including keys the specification never
604/// defined — applications add their own freely, and a custom key is no less identifying for
605/// being non-standard.
606fn examine_info(dict: &Dictionary, options: &InspectOptions, out: &mut Vec<Finding>) {
607 for (key, value) in dict {
608 let kind = INFO_KEYS
609 .iter()
610 .find(|(name, _)| *name == key.as_slice())
611 .map_or(MetadataKind::Other, |(_, kind)| *kind);
612 out.push(
613 Finding::new(kind, "/Info", value_size(value))
614 .with_field(name_of(key))
615 .with_value(options, || describe(value)),
616 );
617 }
618}
619
620/// Record the globally-removable keys, and annotation authorship.
621fn examine_dictionary(
622 dict: &Dictionary,
623 id: ObjectId,
624 info_id: Option<ObjectId>,
625 options: &InspectOptions,
626 limits: &ParseLimits,
627 depth: u32,
628 out: &mut Vec<Finding>,
629) -> Result<()> {
630 for (name, kind) in GLOBAL_KEYS {
631 if let Ok(value) = dict.get(name) {
632 out.push(
633 Finding::new(*kind, format!("/{}", name_of(name)), value_size(value))
634 .with_field(name_of(name)),
635 );
636 }
637 }
638 if is_markup_annotation(dict) {
639 for (name, kind) in [
640 (&b"T"[..], MetadataKind::PersonalIdentity),
641 (&b"M"[..], MetadataKind::Timestamp),
642 (&b"CreationDate"[..], MetadataKind::Timestamp),
643 (&b"NM"[..], MetadataKind::DocumentIdentifier),
644 ] {
645 if let Ok(value) = dict.get(name) {
646 out.push(
647 Finding::new(kind, "annotation", value_size(value))
648 .with_field(name_of(name))
649 .with_value(options, || describe(value)),
650 );
651 }
652 }
653 }
654 // Embedded-file parameters carry their own creation and modification dates, which survive
655 // every scrub aimed only at the containing document.
656 if let Ok(Object::Dictionary(params)) = dict.get(b"Params") {
657 for name in [&b"CreationDate"[..], &b"ModDate"[..], &b"CheckSum"[..]] {
658 if let Ok(value) = params.get(name) {
659 out.push(
660 Finding::new(MetadataKind::Timestamp, "/Params", value_size(value))
661 .with_field(name_of(name)),
662 );
663 }
664 }
665 }
666 for (_, value) in dict {
667 examine_object(
668 value,
669 id,
670 info_id,
671 options,
672 limits,
673 depth.saturating_add(1),
674 out,
675 )?;
676 }
677 Ok(())
678}
679
680/// Scan an XMP packet for the properties worth naming.
681///
682/// Only unfiltered packets are scanned. ISO 32000-1 §14.3.2 recommends that a metadata stream
683/// be left uncompressed precisely so it can be read without parsing the whole document, and
684/// in practice they almost always are. Refusing to inflate the rare compressed one avoids
685/// handing an attacker a decompression bomb in exchange for a slightly more detailed report —
686/// the packet is still found, still reported, and still removed either way.
687fn examine_xmp(stream: &lopdf::Stream, options: &InspectOptions, out: &mut Vec<Finding>) {
688 if stream.dict.has(b"Filter") {
689 out.push(
690 Finding::new(
691 MetadataKind::Other,
692 "XMP packet",
693 as_u64(stream.content.len()),
694 )
695 .with_field("Metadata (encoded)"),
696 );
697 return;
698 }
699 out.extend(xmp::scan(&stream.content, "XMP packet", options));
700}
701
702/// Remove everything [`examine_object`] reports, from one object.
703fn remove_from_object(
704 object: &mut Object,
705 id: ObjectId,
706 info_id: Option<ObjectId>,
707 limits: &ParseLimits,
708 depth: u32,
709) -> Result<()> {
710 if depth > limits.max_depth {
711 return Err(StryptError::LimitExceeded {
712 format: Format::Pdf,
713 limit: ResourceLimit::Depth,
714 });
715 }
716 match object {
717 Object::Dictionary(dict) => {
718 if Some(id) == info_id {
719 // The Info dictionary is emptied as well as unlinked. Unlinking alone would
720 // be enough for a correct pruner, and relying on that would make this
721 // handler's correctness depend on the pruner's — a dependency worth not
722 // having in the one place where being wrong means a name survives.
723 *dict = Dictionary::new();
724 return Ok(());
725 }
726 remove_from_dictionary(dict, id, info_id, limits, depth)?;
727 }
728 Object::Stream(stream) => {
729 remove_from_dictionary(&mut stream.dict, id, info_id, limits, depth)?;
730 }
731 Object::Array(items) => {
732 for item in items {
733 remove_from_object(item, id, info_id, limits, depth.saturating_add(1))?;
734 }
735 }
736 _ => {}
737 }
738 Ok(())
739}
740
741/// Rewrite every `Real(-0.0)` in the document as `Real(0.0)`.
742///
743/// See the call site for why this is done at all. Note the deliberate use of `is_sign_negative`
744/// rather than `== -0.0`: in IEEE 754 `-0.0 == 0.0` is true, so the obvious comparison matches
745/// positive zero as well and would rewrite values that were never a problem.
746fn normalise_negative_zero(doc: &mut Document, limits: &ParseLimits) -> Result<()> {
747 for object in doc.objects.values_mut() {
748 normalise_object(object, limits, 0)?;
749 }
750 // The trailer is not in `objects` and is reached only by walking it explicitly. Missing it
751 // is how the first version of this fix passed every local test and still failed: CI's fuzz
752 // run moved a negative zero into the trailer within minutes, and the assertion fired again
753 // on a document whose object graph was entirely clean.
754 for (_, value) in &mut doc.trailer {
755 normalise_object(value, limits, 0)?;
756 }
757 Ok(())
758}
759
760/// Walk one object, collapsing negative zeros wherever they nest.
761fn normalise_object(object: &mut Object, limits: &ParseLimits, depth: u32) -> Result<()> {
762 if depth > limits.max_depth {
763 return Err(StryptError::LimitExceeded {
764 format: Format::Pdf,
765 limit: ResourceLimit::Depth,
766 });
767 }
768 match object {
769 Object::Real(value) if value.is_sign_negative() && *value == 0.0 => {
770 *value = 0.0;
771 }
772 Object::Dictionary(dict) => {
773 for (_, value) in dict.iter_mut() {
774 normalise_object(value, limits, depth.saturating_add(1))?;
775 }
776 }
777 Object::Stream(stream) => {
778 for (_, value) in &mut stream.dict {
779 normalise_object(value, limits, depth.saturating_add(1))?;
780 }
781 }
782 Object::Array(items) => {
783 for item in items {
784 normalise_object(item, limits, depth.saturating_add(1))?;
785 }
786 }
787 _ => {}
788 }
789 Ok(())
790}
791
792/// Remove identifying keys from one dictionary and everything nested inside it.
793fn remove_from_dictionary(
794 dict: &mut Dictionary,
795 id: ObjectId,
796 info_id: Option<ObjectId>,
797 limits: &ParseLimits,
798 depth: u32,
799) -> Result<()> {
800 for (name, _) in GLOBAL_KEYS {
801 dict.remove(name);
802 }
803 if is_markup_annotation(dict) {
804 dict.remove(b"T");
805 dict.remove(b"M");
806 dict.remove(b"CreationDate");
807 dict.remove(b"NM");
808 }
809 if let Ok(Object::Dictionary(params)) = dict.get_mut(b"Params") {
810 params.remove(b"CreationDate");
811 params.remove(b"ModDate");
812 params.remove(b"CheckSum");
813 }
814 for (_, value) in &mut *dict {
815 remove_from_object(value, id, info_id, limits, depth.saturating_add(1))?;
816 }
817 Ok(())
818}
819
820/// True for a stream that is an XMP metadata packet.
821fn is_metadata_stream(dict: &Dictionary) -> bool {
822 dict.get_type().is_ok_and(|t| t == b"Metadata")
823 || dict
824 .get(b"Subtype")
825 .and_then(Object::as_name)
826 .is_ok_and(|s| s == b"XML")
827}
828
829/// True for an annotation whose `/T` names a person rather than a form field.
830fn is_markup_annotation(dict: &Dictionary) -> bool {
831 let Ok(subtype) = dict.get(b"Subtype").and_then(Object::as_name) else {
832 return false;
833 };
834 MARKUP_ANNOTATION_SUBTYPES.contains(&subtype)
835}
836
837/// True for a file-specification dictionary, which is how a PDF carries an attachment.
838fn is_embedded_file_holder(object: &Object) -> bool {
839 let dict = match object {
840 Object::Dictionary(dict) => dict,
841 Object::Stream(stream) => &stream.dict,
842 _ => return false,
843 };
844 dict.has_type(b"Filespec") || dict.has(b"EmbeddedFiles")
845}
846
847/// The size of a value in bytes, where it has a meaningful one.
848fn value_size(object: &Object) -> u64 {
849 match object {
850 Object::String(bytes, _) | Object::Name(bytes) => as_u64(bytes.len()),
851 Object::Stream(stream) => as_u64(stream.content.len()),
852 _ => 0,
853 }
854}
855
856/// Render a value, for the callers that opted into seeing values.
857fn describe(object: &Object) -> MetadataValue {
858 match object {
859 Object::String(bytes, _) | Object::Name(bytes) => MetadataValue::Text(name_of(bytes)),
860 Object::Integer(n) => MetadataValue::Text(n.to_string()),
861 Object::Boolean(b) => MetadataValue::Text(b.to_string()),
862 other => MetadataValue::Opaque {
863 bytes: value_size(other),
864 },
865 }
866}
867
868/// Widen a length for reporting. Saturating rather than fallible: a report field is not worth
869/// failing an otherwise-successful strip over.
870fn as_u64(value: usize) -> u64 {
871 u64::try_from(value).unwrap_or(u64::MAX)
872}