strypt_core/formats/pdf.rs
1//! PDF.
2//!
3//! The hardest of the Phase 1 formats, and the one where hidden data is least likely to be
4//! where you look for it. The Document Information Dictionary is the easy part and the part
5//! every tutorial covers; the leaks that matter live in XMP packets, per-object metadata,
6//! annotation authorship, and — above all — in objects left behind by incremental updates,
7//! which are physically present in the file and reachable with a hex editor long after the
8//! "current" version of the document stopped referring to them.
9//!
10//! # Full rewrite, not incremental patching (ADR-0020)
11//!
12//! A PDF can be edited by appending: the original bytes stay, and a new cross-reference
13//! section at the end says which objects supersede which. Nulling the Info dictionary with
14//! another such append is easy, fast, and preserves the file almost perfectly — and it leaves
15//! every previous author name exactly where it was, four kilobytes up the file. For this
16//! tool that is not a lesser fix, it is a silent failure: the user is told the document is
17//! clean and publishes it.
18//!
19//! So the document is parsed into its object graph, scrubbed, pruned to what the catalogue
20//! can actually reach, renumbered, and written out fresh. Everything unreachable — every
21//! superseded revision — is gone because it is never written, not because it was overwritten.
22//!
23//! The cost is honest and worth stating: the output is not byte-comparable with the input,
24//! object numbering changes, and files using features the rewrite cannot faithfully reproduce
25//! are refused rather than mangled. Refusing is the correct half of that trade.
26
27use lopdf::{Dictionary, Document, Object, ObjectId};
28
29use crate::container::package::{self, Embedded};
30use crate::detect::Format;
31use crate::error::{MalformedDetail, ResourceLimit, Result, StryptError};
32use crate::formats::xmp::name_of;
33use crate::formats::{MetadataHandler, ParseLimits, StripOptions, Stripped, xmp};
34use crate::report::{
35 Finding, InspectOptions, MetadataKind, MetadataReport, MetadataValue, Note, StripReport,
36};
37
38/// Removal of metadata from PDF documents.
39#[derive(Debug, Clone, Copy, Default)]
40pub struct PdfHandler;
41
42impl MetadataHandler for PdfHandler {
43 fn name(&self) -> &'static str {
44 Format::Pdf.id()
45 }
46
47 fn format(&self) -> Format {
48 Format::Pdf
49 }
50
51 fn inspect(&self, input: &[u8], options: &InspectOptions) -> Result<MetadataReport> {
52 // Inspection loads its own copy of the document and runs the *same* scrub that
53 // stripping does, then throws the result away. That is deliberate: it makes
54 // "everything `strip` removes is something `inspect` can see" true by construction
55 // rather than by two code paths agreeing to stay in step. The verification pass in
56 // `crate::pipeline` is only meaningful if that holds (`docs/ARCHITECTURE.md` §3).
57 let limits = ParseLimits::default();
58 let mut doc = load(input, &limits)?;
59 let scrubbed = scrub(&mut doc, input, options, &limits)?;
60 Ok(MetadataReport {
61 format: Format::Pdf,
62 findings: scrubbed.findings,
63 notes: scrubbed.notes,
64 })
65 }
66
67 fn strip(&self, input: &[u8], options: &StripOptions) -> Result<Stripped> {
68 let mut doc = load(input, &options.limits)?;
69 let mut scrubbed = scrub(&mut doc, input, &options.inspect, &options.limits)?;
70
71 // Drop every object the catalogue can no longer reach. This is the step that removes
72 // superseded revisions, and it has to come after scrubbing so that objects orphaned
73 // *by* the scrub — the Info dictionary, XMP streams — go with them.
74 let pruned = doc.prune_objects();
75 if !pruned.is_empty() {
76 scrubbed.notes.push(Note::OrphanedObjectsRemoved {
77 objects: pruned.len(),
78 });
79 }
80
81 // Renumber so that output depends only on the object graph, not on the numbering the
82 // input happened to use. Without this, two documents that scrub to the same content
83 // serialise differently, and the determinism invariant
84 // (`docs/TESTING_STRATEGY.md` §1) fails for no good reason.
85 renumber_stably(&mut doc)?;
86
87 // Collapse negative zero to zero for the same reason renumbering exists above: without
88 // it the output is not stable under a second strip. lopdf writes `Real(-0.0)` as `-0`,
89 // dropping the decimal point; re-parsing `-0` yields `Integer(0)`, which writes as `0`.
90 // So one strip of a file containing `-0.` differs from two, breaking the idempotence
91 // invariant (`docs/TESTING_STRATEGY.md` §1, invariant 3).
92 //
93 // Rewriting a number in the user's document needs justifying, since this handler
94 // otherwise refuses rather than repairs (ADR-0018). It is sound here because ISO
95 // 32000-1 §7.3.3 gives PDF numbers no signed zero: `-0` and `0` denote the same value,
96 // there is no operator that can distinguish them, and no renderer can. The alternative
97 // — refusing a valid file over a lost minus sign that changes nothing — costs the user
98 // their document to protect a distinction the format does not make.
99 normalise_negative_zero(&mut doc, &options.limits)?;
100
101 // Guarded for the same reason as `load`: serialisation walks a document graph built
102 // from hostile input, so it is dependency code on untrusted data just as parsing is.
103 // Failing here means nothing is written, which is the correct half of the trade.
104 let mut bytes = Vec::new();
105 crate::panic_guard::guard(
106 || {
107 doc.save_to(&mut bytes).map_err(|source| StryptError::Io {
108 action: crate::error::IoAction::WritingOutput,
109 source,
110 })
111 },
112 || StryptError::Malformed {
113 format: Format::Pdf,
114 offset: None,
115 detail: MalformedDetail::DependencyPanic,
116 },
117 )?;
118
119 // Nothing is handed back until the bytes just written have been read again and shown
120 // to be the document that was written. See `verify_round_trip`.
121 verify_round_trip(&doc, &bytes)?;
122
123 Ok(Stripped {
124 report: StripReport {
125 format: Format::Pdf,
126 removed: scrubbed.findings,
127 retained: Vec::new(),
128 notes: scrubbed.notes,
129 input_bytes: as_u64(input.len()),
130 output_bytes: as_u64(bytes.len()),
131 },
132 bytes,
133 })
134 }
135}
136
137/// Refuse output that does not read back as the document that was written.
138///
139/// The rewrite (ADR-0020) assumes serialising a parsed document and re-parsing it returns the
140/// same document. For a lenient parser on hostile input that assumption does not hold, and
141/// when it breaks it breaks silently: `lopdf` will accept a dictionary whose keys came out of
142/// mangled bytes — `/Annotst 1 /[^@018064665...` in the file that found this — and then write
143/// it back in a form it cannot itself read. The object is written, and disappears when the
144/// file is next opened.
145///
146/// That is the failure this guard exists for, and it is the dangerous kind. A document whose
147/// only `/Page` is lost on reload has a `/Pages` node still claiming `/Count 1` and a `/Kids`
148/// array pointing at an object that is no longer there. strypt wrote that file and reported
149/// success; a second strip then pruned what had become unreachable and wrote a 230-byte
150/// document with a dangling page reference, reporting success again. No metadata survived
151/// either pass, so the verification pass — which searches output for residual metadata — saw
152/// nothing wrong. It cannot: it is looking for what should be absent, not for what should
153/// still be present.
154///
155/// Checking a full structural equivalence would be a second implementation of the rewrite, so
156/// this checks the two properties whose failure means the output is not the document:
157///
158/// 1. **Every object written is present on reload.** This is what catches an object that
159/// serialised into something unparseable. Comparing ids alone is enough — an object that
160/// round-trips to a different id is caught by the same comparison.
161/// 2. **The page tree survives.** A document that had pages before writing and none after has
162/// lost the thing it exists to carry, even when every object id happens to match.
163///
164/// A failure refuses the file. That is the fail-closed answer (`CLAUDE.md` §3 rule 6): strypt
165/// cannot faithfully rewrite this document, so it declines to rather than handing back a
166/// broken one with a success report. Refusing costs the user a file that was already too
167/// damaged to survive a rewrite; the alternative cost them a document they believed was clean.
168fn verify_round_trip(written: &Document, bytes: &[u8]) -> Result<()> {
169 let reloaded = crate::panic_guard::guard(
170 || Document::load_mem(bytes).map_err(|e| map_parse_error(&e)),
171 || StryptError::Malformed {
172 format: Format::Pdf,
173 offset: None,
174 detail: MalformedDetail::DependencyPanic,
175 },
176 )
177 .map_err(|_| StryptError::Malformed {
178 format: Format::Pdf,
179 offset: None,
180 detail: MalformedDetail::NotRoundTrippable,
181 })?;
182
183 let written_ids: Vec<ObjectId> = written.objects.keys().copied().collect();
184 let reloaded_ids: Vec<ObjectId> = reloaded.objects.keys().copied().collect();
185 if written_ids != reloaded_ids {
186 return Err(StryptError::Malformed {
187 format: Format::Pdf,
188 offset: None,
189 detail: MalformedDetail::NotRoundTrippable,
190 });
191 }
192
193 // Only an emptied page tree is a failure, not an empty one: a document that had no pages
194 // to begin with is degenerate but not something this rewrite broke.
195 if written.page_iter().next().is_some() && reloaded.page_iter().next().is_none() {
196 return Err(StryptError::Malformed {
197 format: Format::Pdf,
198 offset: None,
199 detail: MalformedDetail::NotRoundTrippable,
200 });
201 }
202
203 Ok(())
204}
205
206/// How many times `renumber_stably` will renumber before refusing the document.
207///
208/// A well-formed file reaches its fixed point on the second call — the first assigns
209/// `1..=n`, the second confirms nothing moved. The degenerate case below needs a third.
210/// Four is that plus margin; a document still moving after four is not converging, and
211/// looping harder would only delay the refusal.
212const MAX_RENUMBER_ROUNDS: usize = 4;
213
214/// Renumber until the numbering stops changing, or refuse the document.
215///
216/// `lopdf::renumber_objects` is **not idempotent**, which matters because strypt's idempotence
217/// invariant (`docs/TESTING_STRATEGY.md` §1, invariant 3) is stated byte-for-byte on the *first*
218/// re-strip. Before renumbering sequentially, `renumber_objects_with` checks whether the page
219/// order matches ascending object ids and, if it does not, permutes the page objects so that it
220/// does (`lopdf` 0.44.0 `src/processor.rs`). That check reads the numbering the previous step
221/// produced, so one pass can leave a document that a second pass would reorder again.
222///
223/// A sustained fuzz run found the case where that is reachable: a document whose page tree is
224/// self-referential — object 2 is a `/Page` whose own `/Kids` array lists object 2 — so pruning
225/// and renumbering changed which objects `page_iter` yields and in which order. The first strip
226/// produced pages ordered `[3, 2]`, descending; the second saw the mismatch and swapped objects
227/// 2 and 3. Same length, same content, 145 bytes different. It settled from the third strip on,
228/// so this was never an endless flip — but "stable eventually" is not the invariant, and a user
229/// who strips a file twice must not get two different files.
230///
231/// Iterating to a fixed point fixes it by construction rather than by reasoning about a
232/// dependency's internals: the document is only serialised once renumbering has been shown to be
233/// a no-op on it, so re-loading and renumbering that output cannot move anything either.
234///
235/// This does not change the output of any document that was already stable — for those the
236/// second round is the confirmation that would have been skipped, not a second permutation.
237///
238/// Non-convergence is refused rather than accepted at whatever state the last round left, which
239/// is the fail-closed half of the trade (ADR-0018, and `CLAUDE.md` §3 rule 6). Emitting a file
240/// whose numbering strypt could not settle would mean handing the user output it cannot promise
241/// is reproducible.
242fn renumber_stably(doc: &mut Document) -> Result<()> {
243 for _ in 0..MAX_RENUMBER_ROUNDS {
244 // Both the id set and the page order have to be compared. The ids alone are not enough:
245 // the reordering step permutes which object holds which page while leaving the set of
246 // ids exactly as it was, so comparing ids only would report a fixed point on the very
247 // pass that moved something.
248 let ids_before: Vec<ObjectId> = doc.objects.keys().copied().collect();
249 let pages_before: Vec<ObjectId> = doc.page_iter().collect();
250
251 doc.renumber_objects();
252
253 let ids_after: Vec<ObjectId> = doc.objects.keys().copied().collect();
254 let pages_after: Vec<ObjectId> = doc.page_iter().collect();
255
256 if ids_before == ids_after && pages_before == pages_after {
257 return Ok(());
258 }
259 }
260
261 Err(StryptError::Malformed {
262 format: Format::Pdf,
263 offset: None,
264 detail: MalformedDetail::CyclicReference,
265 })
266}
267
268/// Findings and caveats from one pass over a document.
269struct Scrubbed {
270 findings: Vec<Finding>,
271 notes: Vec<Note>,
272}
273
274/// Parse `input`, refusing documents this handler must not rewrite.
275fn load(input: &[u8], limits: &ParseLimits) -> Result<Document> {
276 // `lopdf` is third-party code parsing attacker-controlled bytes, and ADR-0006's no-panic
277 // rule does not reach inside it (ADR-0018). A sustained fuzz run found an integer overflow
278 // in its cross-reference parser, which with `overflow-checks` on in release meant the
279 // shipped binary aborted with a stack trace instead of refusing the file. Contained here
280 // so it reaches the user as an ordinary refusal; see `crate::panic_guard` for what that
281 // does and does not cover.
282 let doc = crate::panic_guard::guard(
283 || Document::load_mem(input).map_err(|e| map_parse_error(&e)),
284 || StryptError::Malformed {
285 format: Format::Pdf,
286 offset: None,
287 detail: MalformedDetail::DependencyPanic,
288 },
289 )?;
290
291 // Encrypted documents are refused rather than rewritten. lopdf can open one protected by
292 // an empty owner password, and it would be technically easy to emit a decrypted copy —
293 // but that hands the user a file with its protection quietly removed, which is a change
294 // to their document's security they did not ask for and would not necessarily notice.
295 // Fail closed and say so.
296 if doc.is_encrypted() || doc.was_encrypted() {
297 return Err(StryptError::Malformed {
298 format: Format::Pdf,
299 offset: None,
300 detail: MalformedDetail::UnsupportedFeature,
301 });
302 }
303
304 // A trailer with no /Root is refused rather than processed.
305 //
306 // ISO 32000-1 §7.5.5 makes /Root a required trailer entry: it names the document catalogue,
307 // which is the single root every other object hangs off. Without it the file has no defined
308 // entry point, and no viewer will open it.
309 //
310 // strypt used to accept such a file, and the result was worse than a refusal. The rewrite in
311 // `strip` walks reachable objects from the root and drops the rest (ADR-0020); with no root
312 // to walk from, which objects survive is not stable across runs. A fuzz run found a document
313 // whose second strip differed from its first — 609 bytes, then 485 — because the second pass
314 // dropped an annotation object that the page still referenced through /Annots. Renumbering
315 // then filled that slot with the catalogue, so the page's annotation array pointed at the
316 // document catalogue. strypt had introduced that corruption itself, while returning success
317 // both times.
318 //
319 // Refusing is the fail-closed answer and costs nothing real: a PDF this broken is not one
320 // the user can publish anyway.
321 if !doc.trailer.has(b"Root") {
322 return Err(StryptError::Malformed {
323 format: Format::Pdf,
324 offset: None,
325 detail: MalformedDetail::MissingMarker,
326 });
327 }
328
329 let too_many = u32::try_from(doc.objects.len()).map_or(true, |count| count > limits.max_items);
330 if too_many {
331 return Err(StryptError::LimitExceeded {
332 format: Format::Pdf,
333 limit: ResourceLimit::ItemCount,
334 });
335 }
336
337 // Every stream's declared /Length must be an integer that matches the content actually
338 // parsed out of it.
339 //
340 // ISO 32000-1 §7.3.8.2 requires /Length to be an integer giving the exact byte count
341 // between "stream" and "endstream". When it is not — a fuzz case reached here by writing
342 // "/Length 45." instead of "/Length 45" — lopdf 0.44 parses the object, keeps the
343 // malformed value in the dictionary, and stores *empty* content, because it cannot locate
344 // the stream's end. It reports no error while doing so.
345 //
346 // Left unchecked, the document reaches the rewriter with the stream's bytes already gone.
347 // strypt then writes a file whose /Length still claims 45 bytes over an empty stream, and
348 // reports "nothing to remove; wrote a clean copy" — a structurally invalid PDF, missing
349 // the user's page content, presented as a success. That is the §5.4 failure the whole
350 // design is arranged to avoid, and the reason it went unnoticed is that the verification
351 // pass looks for residual *metadata*, which an emptied stream has none of.
352 //
353 // Refusing costs nothing on real documents: across the synthetic corpus and all 23
354 // parseable PDFs of the real-producer corpus — pdfLaTeX, LibreOffice, Google Docs,
355 // Acrobat and ImageMagick output, compressed streams included — not one stream disagrees
356 // with its declared length.
357 for object in doc.objects.values() {
358 let Ok(stream) = object.as_stream() else {
359 continue;
360 };
361 let declared = stream
362 .dict
363 .get(b"Length")
364 .ok()
365 .and_then(|length| length.as_i64().ok())
366 .and_then(|length| usize::try_from(length).ok());
367 if declared != Some(stream.content.len()) {
368 return Err(StryptError::Malformed {
369 format: Format::Pdf,
370 offset: None,
371 detail: MalformedDetail::LengthOutOfRange,
372 });
373 }
374 }
375
376 Ok(doc)
377}
378
379/// Translate a parser failure into something a user can act on.
380///
381/// The mapping is coarse on purpose. "Your file is truncated" and "your file is not really a
382/// PDF" lead to different actions; which of thirty-four internal variants fired does not.
383fn map_parse_error(error: &lopdf::Error) -> StryptError {
384 use lopdf::Error as E;
385 let detail = match *error {
386 E::Parse(_) | E::Syntax(_) | E::IndirectObject { .. } | E::ObjectIdMismatch => {
387 MalformedDetail::UnexpectedMarker
388 }
389 E::Xref(_) | E::MissingXrefEntry | E::InvalidObjectStream(_) => {
390 MalformedDetail::BrokenIndex
391 }
392 E::InvalidOffset(_) | E::ObjectNotFound(_) | E::NumericCast(_) | E::TryFromInt(_) => {
393 MalformedDetail::LengthOutOfRange
394 }
395 E::ReferenceCycle(_) | E::ReferenceLimit => MalformedDetail::CyclicReference,
396 E::IO(_) => MalformedDetail::Truncated,
397 E::Decryption(_)
398 | E::InvalidPassword
399 | E::AlreadyEncrypted
400 | E::UnsupportedSecurityHandler(_)
401 | E::Unimplemented(_) => MalformedDetail::UnsupportedFeature,
402 _ => MalformedDetail::MissingMarker,
403 };
404 let offset = match *error {
405 E::InvalidOffset(at) | E::IndirectObject { offset: at } => u64::try_from(at).ok(),
406 _ => None,
407 };
408 StryptError::Malformed {
409 format: Format::Pdf,
410 offset,
411 detail,
412 }
413}
414
415/// Keys in the Document Information Dictionary, and what each one exposes.
416///
417/// `/Creator` names the application the document was *authored* in and `/Producer` the one
418/// that wrote the PDF — so a LaTeX paper typically confesses both its editor and its
419/// toolchain version here. Neither identifies a person alone; together with a timestamp and
420/// a font list they narrow the field a great deal (`docs/THREAT_MODEL.md` §4.7).
421const INFO_KEYS: &[(&[u8], MetadataKind)] = &[
422 (b"Author", MetadataKind::PersonalIdentity),
423 (b"Creator", MetadataKind::SoftwareFingerprint),
424 (b"Producer", MetadataKind::SoftwareFingerprint),
425 (b"CreationDate", MetadataKind::Timestamp),
426 (b"ModDate", MetadataKind::Timestamp),
427 (b"Title", MetadataKind::Comment),
428 (b"Subject", MetadataKind::Comment),
429 (b"Keywords", MetadataKind::Comment),
430 (b"Trapped", MetadataKind::Other),
431];
432
433/// Keys removed from *any* dictionary in the document, wherever they appear.
434///
435/// These are safe to remove anywhere because the PDF specification gives them one meaning
436/// each and nothing renders differently without them. `/PieceInfo` is the interesting one:
437/// it is a scratch area where an application may store whatever private state it likes
438/// between editing sessions, and what ends up in it is entirely up to that application.
439const GLOBAL_KEYS: &[(&[u8], MetadataKind)] = &[
440 (b"Metadata", MetadataKind::Other),
441 (b"PieceInfo", MetadataKind::EditingHistory),
442 (b"LastModified", MetadataKind::Timestamp),
443];
444
445/// Annotation subtypes whose `/T` entry is the annotating person's name.
446///
447/// This distinction matters. On a markup annotation `/T` is the author — exactly what we are
448/// here to remove. On a `/Widget`, which is how every interactive form field is drawn, `/T`
449/// is the *field name* that the form's logic and its saved data refer to. Stripping it would
450/// silently break the document, and breaking a user's file to protect them is not a trade
451/// this tool gets to make on their behalf without saying so.
452const MARKUP_ANNOTATION_SUBTYPES: &[&[u8]] = &[
453 b"Text",
454 b"FreeText",
455 b"Line",
456 b"Square",
457 b"Circle",
458 b"Polygon",
459 b"PolyLine",
460 b"Highlight",
461 b"Underline",
462 b"Squiggly",
463 b"StrikeOut",
464 b"Stamp",
465 b"Caret",
466 b"Ink",
467 b"FileAttachment",
468 b"Sound",
469 b"Movie",
470 b"Redact",
471];
472
473/// Walk the document, recording what is there and removing it.
474fn scrub(
475 doc: &mut Document,
476 raw: &[u8],
477 options: &InspectOptions,
478 limits: &ParseLimits,
479) -> Result<Scrubbed> {
480 let mut findings = Vec::new();
481 let mut notes = Vec::new();
482
483 let info_id = trailer_reference(doc, b"Info");
484
485 // Phase one reads. Every object is examined, not merely the ones the catalogue can reach:
486 // an orphan from a superseded revision is exactly the thing worth telling the user about,
487 // and it is invisible to a walk that starts at the root.
488 let object_ids: Vec<ObjectId> = doc.objects.keys().copied().collect();
489 for id in &object_ids {
490 let Some(object) = doc.objects.get(id) else {
491 continue;
492 };
493 examine_object(object, *id, info_id, options, limits, 0, &mut findings)?;
494 }
495 if doc.trailer.has(b"ID") {
496 // The file identifier is a pair of strings that stays stable across saves of the same
497 // document. It identifies nobody by itself and links every copy and every revision of
498 // the document to each other, which for a leaked draft is the whole question.
499 findings.push(Finding::new(
500 MetadataKind::DocumentIdentifier,
501 "trailer /ID",
502 0,
503 ));
504 }
505
506 // A PDF that has been saved more than once ends with more than one %%EOF. Counting them
507 // is cruder than walking the cross-reference chain and tells the user the thing that
508 // actually matters: earlier versions of this document were sitting inside it.
509 let revisions = count_revisions(raw);
510 if revisions > 1 {
511 notes.push(Note::IncrementalHistory {
512 revisions: revisions.saturating_sub(1),
513 });
514 }
515
516 // Phase two writes.
517 doc.trailer.remove(b"Info");
518 doc.trailer.remove(b"ID");
519 for id in &object_ids {
520 if let Some(object) = doc.objects.get_mut(id) {
521 remove_from_object(object, *id, info_id, limits, 0)?;
522 }
523 }
524
525 if object_ids
526 .iter()
527 .filter_map(|id| doc.objects.get(id))
528 .any(is_embedded_file_holder)
529 {
530 // strypt does not open embedded files. Recursing into them means recursing into
531 // arbitrary nested content, which is a zip-bomb-shaped problem that Phase 2 has to
532 // decide about explicitly. Until then the user is told, because an attachment
533 // carrying its own metadata inside a document reported as clean is precisely the
534 // over-trust this tool must not create (`docs/THREAT_MODEL.md` §5.6).
535 notes.push(Note::OutOfScopeContent {
536 location: "embedded file attachment".into(),
537 });
538 }
539
540 strip_embedded_images(doc, &object_ids, options, limits, &mut findings, &mut notes)?;
541
542 Ok(Scrubbed { findings, notes })
543}
544
545/// Strip the JPEGs a PDF carries, through the JPEG handler (ADR-0056, extending ADR-0029).
546///
547/// A `DCTDecode`-only stream is a JPEG file byte for byte (ISO 32000-1 §7.4.8), Exif included, so
548/// it needs no decoding to reach. Keyed on the filter rather than `/Subtype /Image` so that page
549/// thumbnails (`/Thumb`, §12.3.4) are covered too. JPEG 2000 and filter chains ending in a JPEG
550/// are copied with a note: reaching them means a JPX parser or inflating first.
551fn strip_embedded_images(
552 doc: &mut Document,
553 object_ids: &[ObjectId],
554 options: &InspectOptions,
555 limits: &ParseLimits,
556 findings: &mut Vec<Finding>,
557 notes: &mut Vec<Note>,
558) -> Result<()> {
559 for id in object_ids {
560 let Some(Object::Stream(stream)) = doc.objects.get_mut(id) else {
561 continue;
562 };
563 let Ok(filters) = stream.filters() else {
564 continue;
565 };
566 let name = format!("image object {}", id.0);
567 let is_jpeg = matches!(filters.as_slice(), [b"DCTDecode"])
568 && package::embedded_image_format(&stream.content) == Some(Format::Jpeg);
569 if !is_jpeg {
570 if filters
571 .iter()
572 .any(|f| *f == b"DCTDecode" || *f == b"JPXDecode")
573 {
574 notes.push(Note::UnparsedRegion {
575 location: name,
576 bytes: as_u64(stream.content.len()),
577 });
578 }
579 continue;
580 }
581 match package::strip_embedded_image(Format::Jpeg, &stream.content, &name, options, limits)?
582 {
583 Embedded::Unchanged => {}
584 Embedded::Stripped {
585 bytes,
586 findings: image_findings,
587 notes: image_notes,
588 } => {
589 findings.extend(image_findings);
590 notes.extend(image_notes);
591 stream.set_content(bytes);
592 }
593 }
594 }
595 Ok(())
596}
597
598/// Resolve a reference held in the trailer, if it is one.
599fn trailer_reference(doc: &Document, key: &[u8]) -> Option<ObjectId> {
600 doc.trailer
601 .get(key)
602 .ok()
603 .and_then(|o| o.as_reference().ok())
604}
605
606/// Count `%%EOF` markers, each of which terminates one revision of the document.
607fn count_revisions(raw: &[u8]) -> usize {
608 const EOF: &[u8] = b"%%EOF";
609 raw.windows(EOF.len()).filter(|w| *w == EOF).count()
610}
611
612/// Record everything identifying inside one object.
613fn examine_object(
614 object: &Object,
615 id: ObjectId,
616 info_id: Option<ObjectId>,
617 options: &InspectOptions,
618 limits: &ParseLimits,
619 depth: u32,
620 out: &mut Vec<Finding>,
621) -> Result<()> {
622 if depth > limits.max_depth {
623 return Err(StryptError::LimitExceeded {
624 format: Format::Pdf,
625 limit: ResourceLimit::Depth,
626 });
627 }
628 match object {
629 Object::Dictionary(dict) => {
630 if Some(id) == info_id {
631 examine_info(dict, options, out);
632 }
633 examine_dictionary(dict, id, info_id, options, limits, depth, out)?;
634 }
635 Object::Stream(stream) => {
636 if is_metadata_stream(&stream.dict) {
637 examine_xmp(stream, options, out);
638 }
639 examine_dictionary(&stream.dict, id, info_id, options, limits, depth, out)?;
640 }
641 Object::Array(items) => {
642 for item in items {
643 examine_object(
644 item,
645 id,
646 info_id,
647 options,
648 limits,
649 depth.saturating_add(1),
650 out,
651 )?;
652 }
653 }
654 _ => {}
655 }
656 Ok(())
657}
658
659/// Record the Document Information Dictionary, including keys the specification never
660/// defined — applications add their own freely, and a custom key is no less identifying for
661/// being non-standard.
662fn examine_info(dict: &Dictionary, options: &InspectOptions, out: &mut Vec<Finding>) {
663 for (key, value) in dict {
664 let kind = INFO_KEYS
665 .iter()
666 .find(|(name, _)| *name == key.as_slice())
667 .map_or(MetadataKind::Other, |(_, kind)| *kind);
668 out.push(
669 Finding::new(kind, "/Info", value_size(value))
670 .with_field(name_of(key))
671 .with_value(options, || describe(value)),
672 );
673 }
674}
675
676/// Record the globally-removable keys, and annotation authorship.
677fn examine_dictionary(
678 dict: &Dictionary,
679 id: ObjectId,
680 info_id: Option<ObjectId>,
681 options: &InspectOptions,
682 limits: &ParseLimits,
683 depth: u32,
684 out: &mut Vec<Finding>,
685) -> Result<()> {
686 for (name, kind) in GLOBAL_KEYS {
687 if let Ok(value) = dict.get(name) {
688 out.push(
689 Finding::new(*kind, format!("/{}", name_of(name)), value_size(value))
690 .with_field(name_of(name)),
691 );
692 }
693 }
694 if is_markup_annotation(dict) {
695 for (name, kind) in [
696 (&b"T"[..], MetadataKind::PersonalIdentity),
697 (&b"M"[..], MetadataKind::Timestamp),
698 (&b"CreationDate"[..], MetadataKind::Timestamp),
699 (&b"NM"[..], MetadataKind::DocumentIdentifier),
700 ] {
701 if let Ok(value) = dict.get(name) {
702 out.push(
703 Finding::new(kind, "annotation", value_size(value))
704 .with_field(name_of(name))
705 .with_value(options, || describe(value)),
706 );
707 }
708 }
709 }
710 // Embedded-file parameters carry their own creation and modification dates, which survive
711 // every scrub aimed only at the containing document.
712 if let Ok(Object::Dictionary(params)) = dict.get(b"Params") {
713 for name in [&b"CreationDate"[..], &b"ModDate"[..], &b"CheckSum"[..]] {
714 if let Ok(value) = params.get(name) {
715 out.push(
716 Finding::new(MetadataKind::Timestamp, "/Params", value_size(value))
717 .with_field(name_of(name)),
718 );
719 }
720 }
721 }
722 for (_, value) in dict {
723 examine_object(
724 value,
725 id,
726 info_id,
727 options,
728 limits,
729 depth.saturating_add(1),
730 out,
731 )?;
732 }
733 Ok(())
734}
735
736/// Scan an XMP packet for the properties worth naming.
737///
738/// Only unfiltered packets are scanned. ISO 32000-1 §14.3.2 recommends that a metadata stream
739/// be left uncompressed precisely so it can be read without parsing the whole document, and
740/// in practice they almost always are. Refusing to inflate the rare compressed one avoids
741/// handing an attacker a decompression bomb in exchange for a slightly more detailed report —
742/// the packet is still found, still reported, and still removed either way.
743fn examine_xmp(stream: &lopdf::Stream, options: &InspectOptions, out: &mut Vec<Finding>) {
744 if stream.dict.has(b"Filter") {
745 out.push(
746 Finding::new(
747 MetadataKind::Other,
748 "XMP packet",
749 as_u64(stream.content.len()),
750 )
751 .with_field("Metadata (encoded)"),
752 );
753 return;
754 }
755 out.extend(xmp::scan(&stream.content, "XMP packet", options));
756}
757
758/// Remove everything [`examine_object`] reports, from one object.
759fn remove_from_object(
760 object: &mut Object,
761 id: ObjectId,
762 info_id: Option<ObjectId>,
763 limits: &ParseLimits,
764 depth: u32,
765) -> Result<()> {
766 if depth > limits.max_depth {
767 return Err(StryptError::LimitExceeded {
768 format: Format::Pdf,
769 limit: ResourceLimit::Depth,
770 });
771 }
772 match object {
773 Object::Dictionary(dict) => {
774 if Some(id) == info_id {
775 // The Info dictionary is emptied as well as unlinked. Unlinking alone would
776 // be enough for a correct pruner, and relying on that would make this
777 // handler's correctness depend on the pruner's — a dependency worth not
778 // having in the one place where being wrong means a name survives.
779 *dict = Dictionary::new();
780 return Ok(());
781 }
782 remove_from_dictionary(dict, id, info_id, limits, depth)?;
783 }
784 Object::Stream(stream) => {
785 remove_from_dictionary(&mut stream.dict, id, info_id, limits, depth)?;
786 }
787 Object::Array(items) => {
788 for item in items {
789 remove_from_object(item, id, info_id, limits, depth.saturating_add(1))?;
790 }
791 }
792 _ => {}
793 }
794 Ok(())
795}
796
797/// Rewrite every `Real(-0.0)` in the document as `Real(0.0)`.
798///
799/// See the call site for why this is done at all. Note the deliberate use of `is_sign_negative`
800/// rather than `== -0.0`: in IEEE 754 `-0.0 == 0.0` is true, so the obvious comparison matches
801/// positive zero as well and would rewrite values that were never a problem.
802fn normalise_negative_zero(doc: &mut Document, limits: &ParseLimits) -> Result<()> {
803 for object in doc.objects.values_mut() {
804 normalise_object(object, limits, 0)?;
805 }
806 // The trailer is not in `objects` and is reached only by walking it explicitly. Missing it
807 // is how the first version of this fix passed every local test and still failed: CI's fuzz
808 // run moved a negative zero into the trailer within minutes, and the assertion fired again
809 // on a document whose object graph was entirely clean.
810 for (_, value) in &mut doc.trailer {
811 normalise_object(value, limits, 0)?;
812 }
813 Ok(())
814}
815
816/// Walk one object, collapsing negative zeros wherever they nest.
817fn normalise_object(object: &mut Object, limits: &ParseLimits, depth: u32) -> Result<()> {
818 if depth > limits.max_depth {
819 return Err(StryptError::LimitExceeded {
820 format: Format::Pdf,
821 limit: ResourceLimit::Depth,
822 });
823 }
824 match object {
825 Object::Real(value) if value.is_sign_negative() && *value == 0.0 => {
826 *value = 0.0;
827 }
828 Object::Dictionary(dict) => {
829 for (_, value) in dict.iter_mut() {
830 normalise_object(value, limits, depth.saturating_add(1))?;
831 }
832 }
833 Object::Stream(stream) => {
834 for (_, value) in &mut stream.dict {
835 normalise_object(value, limits, depth.saturating_add(1))?;
836 }
837 }
838 Object::Array(items) => {
839 for item in items {
840 normalise_object(item, limits, depth.saturating_add(1))?;
841 }
842 }
843 _ => {}
844 }
845 Ok(())
846}
847
848/// Remove identifying keys from one dictionary and everything nested inside it.
849fn remove_from_dictionary(
850 dict: &mut Dictionary,
851 id: ObjectId,
852 info_id: Option<ObjectId>,
853 limits: &ParseLimits,
854 depth: u32,
855) -> Result<()> {
856 for (name, _) in GLOBAL_KEYS {
857 dict.remove(name);
858 }
859 if is_markup_annotation(dict) {
860 dict.remove(b"T");
861 dict.remove(b"M");
862 dict.remove(b"CreationDate");
863 dict.remove(b"NM");
864 }
865 if let Ok(Object::Dictionary(params)) = dict.get_mut(b"Params") {
866 params.remove(b"CreationDate");
867 params.remove(b"ModDate");
868 params.remove(b"CheckSum");
869 }
870 for (_, value) in &mut *dict {
871 remove_from_object(value, id, info_id, limits, depth.saturating_add(1))?;
872 }
873 Ok(())
874}
875
876/// True for a stream that is an XMP metadata packet.
877fn is_metadata_stream(dict: &Dictionary) -> bool {
878 dict.get_type().is_ok_and(|t| t == b"Metadata")
879 || dict
880 .get(b"Subtype")
881 .and_then(Object::as_name)
882 .is_ok_and(|s| s == b"XML")
883}
884
885/// True for an annotation whose `/T` names a person rather than a form field.
886fn is_markup_annotation(dict: &Dictionary) -> bool {
887 let Ok(subtype) = dict.get(b"Subtype").and_then(Object::as_name) else {
888 return false;
889 };
890 MARKUP_ANNOTATION_SUBTYPES.contains(&subtype)
891}
892
893/// True for a file-specification dictionary, which is how a PDF carries an attachment.
894fn is_embedded_file_holder(object: &Object) -> bool {
895 let dict = match object {
896 Object::Dictionary(dict) => dict,
897 Object::Stream(stream) => &stream.dict,
898 _ => return false,
899 };
900 dict.has_type(b"Filespec") || dict.has(b"EmbeddedFiles")
901}
902
903/// The size of a value in bytes, where it has a meaningful one.
904fn value_size(object: &Object) -> u64 {
905 match object {
906 Object::String(bytes, _) | Object::Name(bytes) => as_u64(bytes.len()),
907 Object::Stream(stream) => as_u64(stream.content.len()),
908 _ => 0,
909 }
910}
911
912/// Render a value, for the callers that opted into seeing values.
913fn describe(object: &Object) -> MetadataValue {
914 match object {
915 Object::String(bytes, _) | Object::Name(bytes) => MetadataValue::Text(name_of(bytes)),
916 Object::Integer(n) => MetadataValue::Text(n.to_string()),
917 Object::Boolean(b) => MetadataValue::Text(b.to_string()),
918 other => MetadataValue::Opaque {
919 bytes: value_size(other),
920 },
921 }
922}
923
924/// Widen a length for reporting. Saturating rather than fallible: a report field is not worth
925/// failing an otherwise-successful strip over.
926fn as_u64(value: usize) -> u64 {
927 u64::try_from(value).unwrap_or(u64::MAX)
928}