Skip to main content

pdfrum_parser/
doc.rs

1//! Opening a document: the header, the load ladder, and the page tree.
2//!
3//! # The ladder
4//!
5//! [`load`] is a sequence of attempts, each one falling back to a cruder
6//! repair. Find the header; read the cross-reference information, rebuilding
7//! it from a full-file scan if the structured paths fail; set up decryption;
8//! then check that the trailer names a catalog and that the catalog names at
9//! least one page. If that last check fails the reader *throws away the
10//! table it just built* and rebuilds anyway, because a table that yields no
11//! pages is more likely stale than the file is empty.
12//!
13//! That retry is why the ladder is written as a ladder rather than a straight
14//! line: the same steps run twice with different inputs, and which rung a
15//! file lands on decides whether it opens at all.
16//!
17//! # `/Root` must be a reference
18//!
19//! A trailer whose `/Root` is a dictionary written inline is treated as
20//! having no catalog, even though the dictionary is right there. It reads as
21//! damage, and damage triggers the rebuild — which on real files finds a
22//! better catalog. Honoring the inline dictionary would skip that.
23//!
24//! # Counting pages without walking them
25//!
26//! A `/Pages` node's `/Count` is believed whenever it is positive and below
27//! the cap, without checking it against the tree. Files whose counts are
28//! wrong therefore report the wrong number — and lookups past the real end
29//! fail individually, which is exactly what a reader that trusted the walk
30//! instead would not reproduce.
31
32use std::sync::{Arc, Mutex};
33
34use pdfrum_common::{
35    DiagKind, Diagnostics, LimitExceeded, Limits, Operation, PageIndex, PdfVersion, Severity,
36};
37use pdfrum_crypt::{Permissions, SecurityHandler};
38use pdfrum_object::{ByteSpan, Dict, NoResolve, ObjRef, Object, Resolve, names};
39
40use crate::error::Error;
41use crate::store::ObjectStore;
42use crate::xref::{Trailer, Xref, XrefShape};
43
44/// How many bytes a `%PDF-1.7\n` header occupies, and the least a file can be.
45const HEADER_SIZE: usize = 9;
46
47/// Why a document could not be opened.
48#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
49#[non_exhaustive]
50pub enum LoadError {
51    /// No `%PDF` header within the first kilobyte, or a file too short to
52    /// hold one.
53    #[error("not a PDF file")]
54    NotPdf,
55
56    /// The document is encrypted and the password given does not open it.
57    /// Distinct from the others because callers ask again rather than give
58    /// up.
59    #[error("wrong password")]
60    WrongPassword,
61
62    /// The document uses a security handler this reader does not implement.
63    #[error("unsupported encryption: {0}")]
64    UnsupportedEncryption(String),
65
66    /// The file is damaged past what recovery could repair: no usable
67    /// cross-reference information, or no catalog with pages in it.
68    #[error("damaged beyond recovery: {0}")]
69    Broken(String),
70
71    /// The caller's `Limits::deadline` had passed when the open began, or
72    /// passed during the cross-reference rebuild scan.
73    #[error(transparent)]
74    Limit(LimitExceeded),
75}
76
77/// How to open a document.
78#[derive(Debug, Clone, Default)]
79pub struct LoadOptions {
80    /// The password to try, as raw bytes. Not capped in length.
81    pub password: Option<Vec<u8>>,
82    /// Caps to enforce while reading.
83    pub limits: Limits,
84}
85
86impl LoadOptions {
87    /// Try this password — a `&str`, a `String`, a `&[u8]` or a `Vec<u8>`.
88    ///
89    /// PDF passwords are byte strings and need not be UTF-8.
90    #[must_use]
91    pub fn with_password(mut self, password: impl AsRef<[u8]>) -> Self {
92        self.password = Some(password.as_ref().to_vec());
93        self
94    }
95}
96
97/// One page's dictionary, with the attributes it inherits already resolved.
98#[derive(Debug, Clone, PartialEq)]
99pub struct PageDict {
100    /// The page's own dictionary.
101    pub dict: Dict,
102    /// The reference it was reached through, when it had one. A page written
103    /// inline in its parent's `/Kids` has none.
104    pub reference: Option<ObjRef>,
105}
106
107impl PageDict {
108    /// An attribute of this page, looked up through its `/Parent` chain
109    /// (ISO 32000-1 §7.7.3.4).
110    ///
111    /// `/Resources`, `/MediaBox`, `/CropBox` and `/Rotate` are inheritable:
112    /// a page that does not state one takes its parent's, or its
113    /// grandparent's. The walk stops at the first node that states the key
114    /// *directly* — a value that is itself a reference is resolved, but a
115    /// `/Parent` that is not a dictionary ends the chain.
116    #[must_use]
117    pub fn inherited(&self, key: &pdfrum_object::Name, r: &impl Resolve) -> Option<Object> {
118        let mut node = self.dict.clone();
119        let mut seen: Vec<Dict> = Vec::new();
120        for _ in 0..64 {
121            if let Some(value) = node.get(key, r) {
122                return Some(value.get().clone());
123            }
124            // A cycle in the parent chain would otherwise spin forever.
125            if seen.contains(&node) {
126                return None;
127            }
128            seen.push(node.clone());
129            node = node.dict(names::PARENT, r)?;
130        }
131        None
132    }
133}
134
135/// An opened document.
136///
137/// Holds the file, everything the reader learned about where its objects are,
138/// and the lazy store that turns references into objects. `Send + Sync`, so
139/// pages can be rendered in parallel.
140#[derive(Debug)]
141pub struct Document {
142    /// The file from its header onwards; every offset indexes into this.
143    bytes: ByteSpan,
144    /// The trailer, merged across every section that contributed one.
145    trailer: Trailer,
146    /// The object store.
147    store: Arc<ObjectStore>,
148    /// The version the header declared, or `None` when it declared none.
149    version: Option<PdfVersion>,
150    /// Where the header was found in the original file.
151    header_offset: u64,
152    /// The shape of the cross-reference the load used, for the writer.
153    xref_shape: XrefShape,
154    /// The `/Encrypt` dictionary as the file wrote it, and whether the
155    /// trailer held it directly rather than by reference.
156    encrypt: Option<(Dict, bool)>,
157    /// How many pages the catalog says there are.
158    page_count: u32,
159    /// Page dictionaries found so far, by index.
160    pages: Mutex<PageCache>,
161    /// Everything repaired while opening the file.
162    pub diags: Diagnostics,
163}
164
165/// The page lookup's memory. The cache, not the index.
166#[derive(Debug, Default)]
167struct PageCache {
168    /// One slot per page, filled as pages are found.
169    slots: Vec<Option<PageDict>>,
170    /// Whether the tree turned out to be deeper than the cap, which stops
171    /// every later lookup as well.
172    poisoned: bool,
173}
174
175/// Open a document.
176///
177/// `bytes` is the whole file. Every repair the reader performed is in
178/// [`Document::diags`] afterwards, and a document that opened with a rebuilt
179/// table reports so through [`Document::xref_was_rebuilt`].
180///
181/// # Errors
182///
183/// [`LoadError::NotPdf`] for a file with no header, [`LoadError::WrongPassword`]
184/// and [`LoadError::UnsupportedEncryption`] for encryption the password or
185/// the reader cannot handle, and [`LoadError::Broken`] for damage recovery
186/// could not repair.
187///
188/// ```
189/// use std::sync::Arc;
190/// use pdfrum_parser::{LoadError, LoadOptions, load};
191///
192/// let not_a_pdf: Arc<[u8]> = Arc::from(&b"just some bytes"[..]);
193/// assert_eq!(load(not_a_pdf, &LoadOptions::default()).err(), Some(LoadError::NotPdf));
194/// ```
195pub fn load(bytes: impl Into<ByteSpan>, opts: &LoadOptions) -> Result<Document, LoadError> {
196    opts.limits
197        .check_deadline(Operation::Open)
198        .map_err(LoadError::Limit)?;
199    let mut diags = Diagnostics::default();
200    let bytes: ByteSpan = bytes.into();
201    let header_offset = find_header(&bytes, &opts.limits).ok_or(LoadError::NotPdf)?;
202    if bytes.len() < header_offset.saturating_add(HEADER_SIZE) {
203        return Err(LoadError::NotPdf);
204    }
205    if header_offset > 0 {
206        diags.record(
207            Severity::Recovered,
208            DiagKind::HeaderOffset,
209            Some(header_offset as u64),
210        );
211    }
212
213    // Everything before the header is invisible: offsets in the file are
214    // relative to it, so the reader works on the slice from there on.
215    //
216    // A window, not a copy: junk before `%PDF` costs a refcount bump like
217    // any other offset. Every stream is then a window into this one.
218    let body = bytes
219        .subspan(header_offset..bytes.len())
220        .unwrap_or_else(|_| ByteSpan::empty());
221    let version = read_version(&body);
222
223    let (xref, mut trailer, mut xref_shape) =
224        crate::xref::read_xref_full(&body, &opts.limits, &mut diags).map_err(|e| match e {
225            crate::Error::Limit(limit) => LoadError::Limit(limit),
226            other => LoadError::Broken(other.to_string()),
227        })?;
228
229    // First attempt: the catalog has to be reachable *and* have pages in it.
230    // Shared from here on: the stores below only read the table, and a slot
231    // vector is expensive to copy — see `ObjectStore`'s `xref` field. The
232    // rebuild path below re-shares after it mutates.
233    let mut shared_xref = Arc::new(xref);
234    let mut security = build_security(&body, &shared_xref, &trailer.dict, opts, &mut diags)?;
235    let mut store = build_store(&body, &shared_xref, opts, &trailer.dict, security);
236    let mut page_count = catalog_page_count(&store, &trailer.dict, &opts.limits);
237
238    if page_count.is_none() {
239        if xref_shape.rebuilt {
240            return Err(LoadError::Broken("no document catalog".into()));
241        }
242        // The table is the suspect, not the file: scan it and try again.
243        diags.record(Severity::Recovered, DiagKind::RootRecovered, None);
244        let mut fresh = Xref::new();
245        let mut fresh_trailer = Trailer::default();
246        let rebuilt = crate::xref::rebuild(
247            &body,
248            &mut fresh,
249            &mut fresh_trailer,
250            &opts.limits,
251            &mut diags,
252            &NoResolve,
253        )
254        .map_err(LoadError::Limit)?;
255        if !rebuilt {
256            return Err(LoadError::Broken("no document catalog".into()));
257        }
258        // The recovery path, so the copy this makes is not the common case:
259        // the first attempt's store still holds a reference, and the merged
260        // table has to be a fresh value for the second attempt's stores to
261        // share in turn.
262        let mut merged = (*shared_xref).clone();
263        merged.merge_up(&fresh);
264        crate::xref::merge_trailers(&mut trailer, &fresh_trailer);
265        xref_shape = XrefShape::rebuilt();
266
267        shared_xref = Arc::new(merged);
268        security = build_security(&body, &shared_xref, &trailer.dict, opts, &mut diags)?;
269        store = build_store(&body, &shared_xref, opts, &trailer.dict, security);
270        // Second attempt asks only for a catalog. A rebuilt document whose
271        // catalog is reachable but describes no pages still opens — it is
272        // then a document of zero pages, which is a thing a file can be.
273        if catalog(&store, &trailer.dict).is_none() {
274            return Err(LoadError::Broken("no document catalog".into()));
275        }
276        page_count = catalog_page_count(&store, &trailer.dict, &opts.limits);
277    }
278
279    let page_count = page_count.unwrap_or(0);
280    // Read last, from the trailer as it finally stands, so a rebuild that
281    // replaced the trailer is reflected.
282    let encrypt = encrypt_dict_located(&body, &shared_xref, &trailer.dict, &opts.limits);
283
284    diags.extend(&store.drain_diags());
285
286    Ok(Document {
287        bytes: body,
288        trailer,
289        store,
290        version,
291        header_offset: header_offset as u64,
292        xref_shape,
293        encrypt,
294        page_count,
295        pages: Mutex::new(PageCache {
296            slots: vec![None; usize::try_from(page_count).unwrap_or(0)],
297            poisoned: false,
298        }),
299        diags,
300    })
301}
302
303/// Find `%PDF` within the first `limits.header_scan` bytes.
304fn find_header(bytes: &[u8], limits: &Limits) -> Option<usize> {
305    let window = usize::try_from(limits.header_scan).unwrap_or(usize::MAX);
306    let last = bytes.len().checked_sub(4)?.min(window);
307    (0..=last).find(|&i| bytes.get(i..i + 4) == Some(b"%PDF"))
308}
309
310/// Read the version digits out of `%PDF-M.N`.
311///
312/// Never validated: a header claiming version 9.9 opens like any other, and a
313/// non-digit contributes nothing.
314///
315/// This is the one place in the workspace that knows the `major × 10 + minor`
316/// packing the old `Document::version() -> u8` published: the digits are read,
317/// packed, and immediately unpacked into a [`PdfVersion`]. The round trip is
318/// kept rather than removed because `0` — the answer when neither digit is
319/// readable — is what distinguishes "no version declared" from version 0.0,
320/// and collapsing the two would change which documents the writer gives its
321/// 1.7 fallback to.
322fn read_version(body: &[u8]) -> Option<PdfVersion> {
323    let digit = |i: usize| -> u8 {
324        body.get(i)
325            .filter(|b| b.is_ascii_digit())
326            .map_or(0, |b| b - b'0')
327    };
328    match digit(5).saturating_mul(10).saturating_add(digit(7)) {
329        0 => None,
330        packed => Some(PdfVersion::new(packed / 10, packed % 10)),
331    }
332}
333
334/// Build the security handler the trailer's `/Encrypt` calls for.
335fn build_security(
336    body: &ByteSpan,
337    xref: &Arc<Xref>,
338    trailer: &Dict,
339    opts: &LoadOptions,
340    diags: &mut Diagnostics,
341) -> Result<SecurityHandler, LoadError> {
342    let Some(encrypt) = encrypt_dict(body, xref, trailer, &opts.limits) else {
343        return Ok(SecurityHandler::Identity);
344    };
345    // The handler name is type-checked before being resolved, so a `/Filter`
346    // written as a string is not the standard handler however it spells it.
347    if encrypt.name(names::FILTER) != Some(names::STANDARD) {
348        return Err(LoadError::UnsupportedEncryption(
349            encrypt
350                .name(names::FILTER)
351                .map_or_else(|| "unnamed".to_owned(), |n| n.as_text().into_owned()),
352        ));
353    }
354
355    let file_id = trailer
356        .array(names::ID, &NoResolve)
357        .and_then(|a| a.string_at(0).map(|s| s.as_bytes().to_vec()))
358        .unwrap_or_default();
359    let password = opts.password.clone().unwrap_or_default();
360
361    match SecurityHandler::from_encrypt_dict(&encrypt, &file_id, &password, &NoResolve) {
362        Ok(handler) => {
363            if handler.password_encoding() != pdfrum_crypt::PasswordEncoding::AsGiven {
364                diags.record(Severity::Recovered, DiagKind::PasswordReencoded, None);
365            }
366            Ok(handler)
367        }
368        Err(pdfrum_crypt::Error::WrongPassword) => Err(LoadError::WrongPassword),
369        Err(pdfrum_crypt::Error::UnsupportedHandler(name)) => Err(
370            LoadError::UnsupportedEncryption(String::from_utf8_lossy(&name).into_owned()),
371        ),
372        Err(e) => Err(LoadError::UnsupportedEncryption(e.to_string())),
373    }
374}
375
376/// The `/Encrypt` dictionary, written inline or reached through one
377/// reference.
378///
379/// Most files write it indirectly, so the reference has to be chased — but
380/// the real store does not exist yet, and could not read this dictionary if
381/// it did, since it would try to decrypt it with the key this dictionary
382/// defines. So the lookup goes through a throwaway store that decrypts
383/// nothing. That is not a shortcut: the encryption dictionary is the one
384/// object in a document that is always plaintext.
385fn encrypt_dict(
386    body: &ByteSpan,
387    xref: &Arc<Xref>,
388    trailer: &Dict,
389    limits: &Limits,
390) -> Option<Dict> {
391    encrypt_dict_located(body, xref, trailer, limits).map(|(d, _)| d)
392}
393
394/// The same lookup, additionally reporting whether the trailer held the
395/// dictionary **directly** rather than by reference.
396///
397/// A writer needs that second fact: an inline encryption dictionary has no
398/// object number of its own, so it must be promoted to a fresh indirect
399/// object before the trailer's `/Encrypt` can name it (ISO 32000-1 §7.6.1
400/// requires `/Encrypt` be indirect).
401fn encrypt_dict_located(
402    body: &ByteSpan,
403    xref: &Arc<Xref>,
404    trailer: &Dict,
405    limits: &Limits,
406) -> Option<(Dict, bool)> {
407    match trailer.raw(names::ENCRYPT)? {
408        Object::Dict(d) => Some((d.clone(), true)),
409        Object::Ref(r) => {
410            let plain = ObjectStore::new(
411                body.clone(),
412                Arc::clone(xref),
413                limits.clone(),
414                SecurityHandler::Identity,
415            );
416            Some((plain.get(r.num).ok()?.as_dict().cloned()?, false))
417        }
418        _ => None,
419    }
420}
421
422/// Record the metadata object as exempt from decryption when the document
423/// says its metadata is not encrypted.
424fn exempt_metadata(store: &mut ObjectStore, trailer: &Dict) {
425    if store.security().encrypt_metadata() {
426        return;
427    }
428    let Some(root) = trailer.reference(names::ROOT) else {
429        return;
430    };
431    let Ok(catalog) = store.get(root.num) else {
432        return;
433    };
434    if let Some(metadata) = catalog.as_dict().and_then(|d| d.reference(names::METADATA)) {
435        store.exempt_from_decryption(metadata.num);
436    }
437}
438
439/// Build a store over the table, with the metadata exemption applied.
440fn build_store(
441    body: &ByteSpan,
442    xref: &Arc<Xref>,
443    opts: &LoadOptions,
444    trailer: &Dict,
445    security: SecurityHandler,
446) -> Arc<ObjectStore> {
447    let mut store = ObjectStore::new(
448        body.clone(),
449        Arc::clone(xref),
450        opts.limits.clone(),
451        security,
452    );
453    exempt_metadata(&mut store, trailer);
454    Arc::new(store)
455}
456
457/// The document catalog, if the trailer names one reachably.
458///
459/// A `/Root` written as anything but a reference does not name a catalog,
460/// even when it is a perfectly good dictionary written inline: that reads as
461/// damage, and damage is what triggers the rebuild that finds a better one.
462fn catalog(store: &ObjectStore, trailer: &Dict) -> Option<Dict> {
463    let root = trailer.reference(names::ROOT)?;
464    store.get(root.num).ok()?.as_dict().cloned()
465}
466
467/// How many pages the catalog claims, or `None` when there is no usable
468/// catalog or it describes no pages at all.
469fn catalog_page_count(store: &ObjectStore, trailer: &Dict, limits: &Limits) -> Option<u32> {
470    let catalog = catalog(store, trailer)?;
471    let count = page_count_of(store, &catalog, limits);
472    (count > 0).then_some(count)
473}
474
475/// The number of pages under a catalog.
476///
477/// A catalog without `/Pages` has none. A `/Pages` node without `/Kids` is
478/// itself the single page — that is not a repair, it is what a file with one
479/// page and no tree means.
480fn page_count_of(store: &ObjectStore, catalog: &Dict, limits: &Limits) -> u32 {
481    let Some(pages) = catalog.dict(names::PAGES, store) else {
482        return 0;
483    };
484    if pages.raw(names::KIDS).is_none() {
485        return 1;
486    }
487    // The root counts as an ancestor from the start, so a kid pointing back
488    // at it is a loop rather than a subtree.
489    let mut ancestors = vec![pages.clone()];
490    count_subtree(store, &pages, limits, &mut ancestors).unwrap_or(0)
491}
492
493/// Count the leaves under a node, or `None` when the tree claims more pages
494/// than a document may have.
495///
496/// `/Count` is believed whenever it is positive and under the cap, without
497/// checking it against the tree — so a file that lies about its length
498/// reports the lie, and the individual lookups past its real end are what
499/// fail.
500///
501/// The `None` propagates all the way out rather than being absorbed as a
502/// zero: a subtree that overflows makes the *whole* document uncountable,
503/// which is why this returns an `Option` instead of saturating.
504///
505/// `ancestors` holds the nodes currently being descended through, and is the
506/// cycle guard — the only one, since there is no depth cap here. Note what it
507/// is *not*: a record of every node already seen. A node listed twice among
508/// one parent's `/Kids` is counted twice, because the second listing is a
509/// sibling rather than a loop, and a file whose tree shares subtrees that way
510/// really does have that many pages.
511fn count_subtree(
512    store: &ObjectStore,
513    node: &Dict,
514    limits: &Limits,
515    ancestors: &mut Vec<Dict>,
516) -> Option<u32> {
517    if let Some(count) = node.int(names::COUNT, store)
518        && count > 0
519        && count < i64::from(limits.max_page_count)
520        && let Ok(count) = u32::try_from(count)
521    {
522        return Some(count);
523    }
524
525    let Some(kids) = node.array(names::KIDS, store) else {
526        return Some(0);
527    };
528    let mut total: u32 = 0;
529    for kid in kids.iter() {
530        let Some(kid) = kid.resolve(store).ok().and_then(|k| k.as_dict().cloned()) else {
531            continue;
532        };
533        // Only a kid that is already an ancestor would loop.
534        if ancestors.contains(&kid) {
535            continue;
536        }
537        total = total.saturating_add(match node_kind(&kid) {
538            NodeKind::Branch => {
539                ancestors.push(kid.clone());
540                let under = count_subtree(store, &kid, limits, ancestors);
541                ancestors.pop();
542                under?
543            }
544            NodeKind::Leaf => 1,
545        });
546        if total >= limits.max_page_count {
547            return None;
548        }
549    }
550    Some(total)
551}
552
553/// What a page-tree node is.
554enum NodeKind {
555    /// An interior node whose `/Kids` hold more nodes.
556    Branch,
557    /// A page.
558    Leaf,
559}
560
561/// Classify a node, guessing when `/Type` does not say.
562///
563/// A node with `/Kids` is a branch and one without is a page, whatever its
564/// `/Type` claims — files write the wrong type often enough that the
565/// structure is the more reliable witness.
566fn node_kind(node: &Dict) -> NodeKind {
567    match node.name(names::TYPE) {
568        Some(t) if t == names::PAGES => NodeKind::Branch,
569        Some(t) if t == names::PAGE => NodeKind::Leaf,
570        _ => {
571            if node.contains_key(names::KIDS) {
572                NodeKind::Branch
573            } else {
574                NodeKind::Leaf
575            }
576        }
577    }
578}
579
580impl Document {
581    /// How many pages the document has.
582    ///
583    /// A `u32`, deliberately, and not a
584    /// [`PageIndex`](pdfrum_common::PageIndex): a count answers "how many"
585    /// and an index answers "which one", and the last valid index of a
586    /// three-page document is 2, not 3. Giving them one type would let each be
587    /// passed where the other is meant, which is what the newtype exists to
588    /// stop.
589    #[must_use]
590    pub fn page_count(&self) -> u32 {
591        self.page_count
592    }
593
594    /// The page at `index`, counting from zero.
595    ///
596    /// The tree is walked in order and the pages found along the way are
597    /// remembered, so reading a document front to back costs one traversal.
598    /// A kid that will not load as a dictionary still **consumes its slot**:
599    /// a missing page leaves a hole rather than shifting every page after it.
600    ///
601    /// Takes `impl Into<PageIndex>`, so `doc.page(0)` reads as it always has.
602    ///
603    /// # Errors
604    ///
605    /// [`Error::NoPage`] for an index past the count, or one the walk could
606    /// not reach.
607    pub fn page(&self, index: impl Into<PageIndex>) -> Result<PageDict, Error> {
608        let index = index.into();
609        let slot = usize::try_from(index.get()).unwrap_or(usize::MAX);
610        if index.get() >= self.page_count {
611            return Err(Error::NoPage(index));
612        }
613        let Ok(mut pages) = self.pages.lock() else {
614            return Err(Error::NoPage(index));
615        };
616        if let Some(Some(found)) = pages.slots.get(slot) {
617            return Ok(found.clone());
618        }
619        if pages.poisoned {
620            return Err(Error::NoPage(index));
621        }
622
623        self.walk_pages(&mut pages);
624        pages
625            .slots
626            .get(slot)
627            .and_then(Clone::clone)
628            .ok_or(Error::NoPage(index))
629    }
630
631    /// Walk the whole tree once, filling every slot it can reach.
632    fn walk_pages(&self, pages: &mut PageCache) {
633        let Some(root) = self.trailer.dict.reference(names::ROOT) else {
634            return;
635        };
636        let Ok(catalog) = self.store.get(root.num) else {
637            return;
638        };
639        let Some(node) = catalog
640            .as_dict()
641            .and_then(|d| d.dict(names::PAGES, &*self.store))
642        else {
643            return;
644        };
645
646        let mut next: usize = 0;
647        let mut ancestors = Vec::new();
648        let objref = catalog.as_dict().and_then(|d| d.reference(names::PAGES));
649        self.visit(&node, objref, pages, &mut next, 0, &mut ancestors);
650    }
651
652    /// Depth-first, in order, filling slots as leaves are reached.
653    ///
654    /// `ancestors` is the cycle guard: the nodes on the path from the root to
655    /// here. A node that reappears as a *sibling* is a second page, not a
656    /// loop, so only an ancestor stops the descent.
657    fn visit(
658        &self,
659        node: &Dict,
660        reference: Option<ObjRef>,
661        pages: &mut PageCache,
662        next: &mut usize,
663        depth: u32,
664        ancestors: &mut Vec<Dict>,
665    ) {
666        if *next >= pages.slots.len() {
667            return;
668        }
669        // A node without `/Kids` is where the walk stops, whatever it claims
670        // to be — but a node that claims `/Type /Pages` and has no children
671        // is describing a subtree that is not there, so no page comes of it.
672        // (`page_count` still counts such a root as one page; the lookup is
673        // what fails.)
674        if node.raw(names::KIDS).is_none() {
675            if matches!(node_kind(node), NodeKind::Branch) {
676                return;
677            }
678            if let Some(slot) = pages.slots.get_mut(*next) {
679                *slot = Some(PageDict {
680                    dict: node.clone(),
681                    reference,
682                });
683            }
684            *next += 1;
685            return;
686        }
687
688        // Only a node with children can be too deep, so the cap is checked
689        // after the leaf case rather than on the way in. Exceeding it stops
690        // every later lookup too, not just this one.
691        if depth >= self.store.limits().max_page_tree_depth {
692            // The oracle records the same fact in the same shape:
693            // `reached_max_page_level_ = true` at `cpdf_document.cpp:281-283`,
694            // after which its own later lookups fail too. `poisoned` is that
695            // flag; the diagnostic is how a caller finds out *why* the pages
696            // stopped resolving.
697            self.store
698                .note(Severity::Suspicious, DiagKind::PageTreeDepthExceeded, None);
699            pages.poisoned = true;
700            return;
701        }
702
703        let Some(kids) = node.array(names::KIDS, &*self.store) else {
704            return;
705        };
706        ancestors.push(node.clone());
707        for kid in kids.iter() {
708            let kid_ref = kid.as_ref_id();
709            let loaded = kid
710                .resolve(&*self.store)
711                .ok()
712                .and_then(|k| k.as_dict().cloned());
713            let Some(loaded) = loaded else {
714                // A kid that will not load still costs a slot, so a missing
715                // page leaves a hole rather than shifting every page after it.
716                *next += 1;
717                continue;
718            };
719            // Only a kid that is already an ancestor would loop; the same
720            // node appearing twice as a sibling is two pages. PDFium skips a
721            // kid it has already visited for the same reason
722            // (`cpdf_document.cpp:87-88`), as part of the same pass that
723            // rewrites a wrong `/Count` (`:111`) and guesses a missing `/Type`
724            // (`:60`) — all of it "fix the in-memory representation for page
725            // tree nodes that violate the spec".
726            if ancestors.contains(&loaded) {
727                self.store
728                    .note(Severity::Recovered, DiagKind::PageTreeRepaired, None);
729                continue;
730            }
731            self.visit(&loaded, kid_ref, pages, next, depth + 1, ancestors);
732            if *next >= pages.slots.len() {
733                break;
734            }
735        }
736        ancestors.pop();
737    }
738
739    /// The trailer dictionary, merged across every section.
740    #[must_use]
741    pub fn trailer(&self) -> &Dict {
742        &self.trailer.dict
743    }
744
745    /// The object number the trailer came from; zero for a bare `trailer`
746    /// dictionary.
747    #[must_use]
748    pub fn trailer_object_number(&self) -> u32 {
749        self.trailer.object_number
750    }
751
752    /// The document catalog.
753    ///
754    /// # Errors
755    ///
756    /// [`Error::NoCatalog`] when the trailer names none.
757    pub fn catalog(&self) -> Result<Dict, Error> {
758        let root = self
759            .trailer
760            .dict
761            .reference(names::ROOT)
762            .ok_or(Error::NoCatalog)?;
763        self.store
764            .get(root.num)
765            .ok()
766            .and_then(|c| c.as_dict().cloned())
767            .ok_or(Error::NoCatalog)
768    }
769
770    /// The version the header declared: [`PdfVersion::PDF_1_7`] for
771    /// `%PDF-1.7`.
772    ///
773    /// Never validated — a header claiming 9.9 opens like any other and
774    /// reports 9.9. `None` means the header carried no readable digits at
775    /// all, which a file with no `%PDF` line and one with `%PDF-x.y` both
776    /// produce; the writer's fallback for that case is 1.7.
777    #[must_use]
778    pub fn version(&self) -> Option<PdfVersion> {
779        self.version
780    }
781
782    /// Where the `%PDF` header sat in the original file. Non-zero means
783    /// everything before it was ignored.
784    #[must_use]
785    pub fn header_offset(&self) -> u64 {
786        self.header_offset
787    }
788
789    /// What the object store has repaired since the file opened, as a running
790    /// total — read it *after* the work, not at load.
791    ///
792    /// [`Document::diags`] is the load-time snapshot and never changes. This
793    /// one grows, because the store is lazy: a wrong `/Length` or a bad table
794    /// offset is only discovered when a caller first reaches that object. It
795    /// clones rather than draining, so asking twice between two fetches gives
796    /// the same answer twice.
797    ///
798    /// ```
799    /// # use std::sync::Arc;
800    /// # use pdfrum_parser::{LoadOptions, load};
801    /// let bytes: Arc<[u8]> = Arc::from(&include_bytes!("../tests/files/minimal.pdf")[..]);
802    /// let doc = load(bytes, &LoadOptions::default())?;
803    /// // A clean file repairs nothing, before or after its pages are read.
804    /// let _ = doc.page(0);
805    /// assert!(doc.lazy_diagnostics().is_empty());
806    /// # Ok::<(), pdfrum_parser::LoadError>(())
807    /// ```
808    #[must_use]
809    pub fn lazy_diagnostics(&self) -> Diagnostics {
810        self.store.peek_diags()
811    }
812
813    /// Whether the cross-reference table came from the recovery scan rather
814    /// than the file's own sections. An incremental save is unsafe when it
815    /// did.
816    #[must_use]
817    pub fn xref_was_rebuilt(&self) -> bool {
818        self.xref_shape.rebuilt
819    }
820
821    /// Byte offset of the newest cross-reference section the load chained
822    /// from, or **0** when the table was rebuilt by scanning.
823    ///
824    /// This is what an incremental update writes as its `/Prev`, so the zero
825    /// carries meaning rather than being an absence: a rebuilt document has
826    /// no previous section worth naming, and the writer answers by emitting a
827    /// full table after the original bytes instead of a delta.
828    ///
829    /// ```
830    /// # use std::sync::Arc;
831    /// # use pdfrum_parser::{LoadOptions, load};
832    /// let bytes: Arc<[u8]> = Arc::from(&include_bytes!("../tests/files/minimal.pdf")[..]);
833    /// let doc = load(bytes, &LoadOptions::default())?;
834    /// assert!(doc.last_xref_offset() > 0);
835    /// assert!(!doc.xref_was_rebuilt());
836    /// # Ok::<(), pdfrum_parser::LoadError>(())
837    /// ```
838    #[must_use]
839    pub fn last_xref_offset(&self) -> u64 {
840        self.xref_shape.last_offset
841    }
842
843    /// Whether the document's **main** cross-reference — the newest section,
844    /// the one `startxref` names — was a stream rather than a classic table.
845    ///
846    /// Not "the chain contained a stream somewhere": a hybrid file whose
847    /// newest section is a classic table answers `false`. The writer reads it
848    /// to decide whether an incremental update appends a classic delta table
849    /// or folds the cross-reference into a stream object, and matching the
850    /// original keeps a reader that only understands one of the two working.
851    #[must_use]
852    pub fn main_xref_is_stream(&self) -> bool {
853        self.xref_shape.main_is_stream
854    }
855
856    /// The `/Encrypt` dictionary as the file wrote it, and whether the
857    /// trailer held it **directly** rather than by reference.
858    ///
859    /// Returned raw and undecrypted, because the encryption dictionary is the
860    /// one object in a document that is always plaintext. `Some` here does
861    /// not imply the document opened encrypted — a file can declare a handler
862    /// this reader answered with [`SecurityHandler::Identity`].
863    #[must_use]
864    pub fn encrypt_dict(&self) -> Option<(&Dict, bool)> {
865        self.encrypt.as_ref().map(|(d, inline)| (d, *inline))
866    }
867
868    /// What the document permits, for the password that opened it.
869    ///
870    /// The owner's own unrestricted view is
871    /// [`Document::owner_permissions`]. An unencrypted document permits
872    /// everything.
873    #[must_use]
874    pub fn permissions(&self) -> Permissions {
875        self.store.security().permissions()
876    }
877
878    /// What the document permits under the owner's view.
879    ///
880    /// Every permission, for a document the owner password opened; otherwise
881    /// the same answer as [`Document::permissions`].
882    #[must_use]
883    pub fn owner_permissions(&self) -> Permissions {
884        self.store.security().owner_permissions()
885    }
886
887    /// Whether the document is encrypted.
888    #[must_use]
889    pub fn is_encrypted(&self) -> bool {
890        !matches!(self.store.security(), SecurityHandler::Identity)
891    }
892
893    /// The security handler the password opened this document with.
894    ///
895    /// [`SecurityHandler::Identity`] for an unencrypted document, and for one
896    /// whose crypt filter is `/Identity`.
897    ///
898    /// The writer needs this to save an encrypted document *as encrypted*: it
899    /// re-enciphers every string and stream under the same handler, so the
900    /// result opens with the same password. It carries the file key, so it
901    /// is deliberately not `Clone`-friendly to hold onto — borrow it for the
902    /// length of a save and let it go.
903    #[must_use]
904    pub fn security_handler(&self) -> &SecurityHandler {
905        self.store.security()
906    }
907
908    /// The file, from its header onwards.
909    #[must_use]
910    pub fn bytes(&self) -> &[u8] {
911        &self.bytes
912    }
913
914    /// The object store, for fetching references.
915    #[must_use]
916    pub fn store(&self) -> &Arc<ObjectStore> {
917        &self.store
918    }
919
920    /// Where every object lives.
921    #[must_use]
922    pub fn xref(&self) -> &Xref {
923        self.store.xref()
924    }
925}
926
927impl Resolve for Document {
928    fn fetch(&self, r: ObjRef) -> Result<Arc<Object>, pdfrum_object::Error> {
929        self.store.fetch(r)
930    }
931}
932
933/// The content of `dict`'s page, decoded and joined, with the end offset of
934/// each `/Contents` element — read through `r`.
935pub fn content_segments(
936    dict: &PageDict,
937    r: &impl Resolve,
938    limits: &Limits,
939    diags: &mut Diagnostics,
940) -> (Vec<u8>, Vec<usize>) {
941    let Some(contents) = dict.dict.get(&pdfrum_object::Name::from("Contents"), r) else {
942        return (Vec::new(), Vec::new());
943    };
944    let mut out = Vec::new();
945    let mut ends = Vec::new();
946    let mut push = |object: &pdfrum_object::Object, out: &mut Vec<u8>, ends: &mut Vec<usize>| {
947        if let Some(stream) = object.as_stream() {
948            let decoded = pdfrum_filters::decode_chain(stream, 0, r, limits, diags);
949            out.extend_from_slice(&decoded.data);
950            // The separating space belongs to the element before it: it is
951            // what terminates a stream ending mid-token.
952            out.push(b' ');
953        }
954        ends.push(out.len());
955    };
956    let Some(direct) = contents.as_direct() else {
957        return (out, ends);
958    };
959    match direct {
960        pdfrum_object::Object::Stream(_) => push(direct, &mut out, &mut ends),
961        pdfrum_object::Object::Array(array) => {
962            for element in array.iter() {
963                if let Ok(resolved) = element.resolve(r) {
964                    push(resolved.get(), &mut out, &mut ends);
965                } else {
966                    // A dangling element still occupies an index, so the
967                    // ones after it keep their numbers.
968                    ends.push(out.len());
969                }
970            }
971        }
972        _ => {}
973    }
974    (out, ends)
975}
976
977/// The byte after the `%%EOF` that closes the revision whose `startxref`
978/// names `offset`.
979pub fn revision_end(bytes: &[u8], offset: u64) -> Option<usize> {
980    let needle = b"startxref";
981    let mut at = 0;
982    while let Some(found) = find(bytes.get(at..)?, needle) {
983        let start = at + found + needle.len();
984        let rest = bytes.get(start..)?;
985        let digits: String = rest
986            .iter()
987            .skip_while(|b| b.is_ascii_whitespace())
988            .take_while(|b| b.is_ascii_digit())
989            .map(|&b| char::from(b))
990            .collect();
991        if digits.parse::<u64>().ok() == Some(offset) {
992            let eof = find(rest, b"%%EOF")?;
993            let mut end = start + eof + b"%%EOF".len();
994            while bytes.get(end).is_some_and(|b| *b == b'\r' || *b == b'\n') {
995                end += 1;
996            }
997            return Some(end);
998        }
999        at = start;
1000    }
1001    None
1002}
1003
1004fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
1005    haystack.windows(needle.len()).position(|w| w == needle)
1006}
1007
1008#[cfg(test)]
1009mod tests {
1010    use super::{LoadError, LoadOptions, find_header, load, read_version};
1011    use pdfrum_common::{DiagKind, Limits, PdfVersion};
1012    use pdfrum_crypt::Permissions;
1013    use pdfrum_object::{Name, names};
1014    use std::sync::Arc;
1015
1016    fn open(bytes: &[u8]) -> Result<super::Document, LoadError> {
1017        load(Arc::from(bytes), &LoadOptions::default())
1018    }
1019
1020    /// A document with `count` pages under one `/Pages` node.
1021    fn build(count: usize) -> Vec<u8> {
1022        let mut out = Vec::new();
1023        out.extend_from_slice(b"%PDF-1.7\n");
1024        let mut offsets = vec![0usize];
1025
1026        offsets.push(out.len());
1027        out.extend_from_slice(b"1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n");
1028
1029        offsets.push(out.len());
1030        let kids: Vec<String> = (0..count).map(|i| format!("{} 0 R", i + 3)).collect();
1031        out.extend_from_slice(
1032            format!(
1033                "2 0 obj\n<< /Type /Pages /Count {count} /Kids [{}] /MediaBox [0 0 612 792] >>\nendobj\n",
1034                kids.join(" ")
1035            )
1036            .as_bytes(),
1037        );
1038
1039        for i in 0..count {
1040            offsets.push(out.len());
1041            out.extend_from_slice(
1042                format!(
1043                    "{} 0 obj\n<< /Type /Page /Parent 2 0 R /PageNumber {i} >>\nendobj\n",
1044                    i + 3
1045                )
1046                .as_bytes(),
1047            );
1048        }
1049
1050        let xref_at = out.len();
1051        out.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
1052        out.extend_from_slice(b"0000000000 65535 f \n");
1053        for offset in offsets.iter().skip(1) {
1054            out.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
1055        }
1056        out.extend_from_slice(
1057            format!(
1058                "trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{xref_at}\n%%EOF\n",
1059                offsets.len()
1060            )
1061            .as_bytes(),
1062        );
1063        out
1064    }
1065
1066    #[test]
1067    fn finds_a_header_at_the_start_or_after_junk() {
1068        assert_eq!(find_header(b"%PDF-1.7\n", &Limits::default()), Some(0));
1069        assert_eq!(find_header(b"junk%PDF-1.7\n", &Limits::default()), Some(4));
1070        assert_eq!(find_header(b"no header", &Limits::default()), None);
1071    }
1072
1073    #[test]
1074    fn version_digits_are_read_not_validated() {
1075        assert_eq!(read_version(b"%PDF-1.7\n"), Some(PdfVersion::PDF_1_7));
1076        assert_eq!(read_version(b"%PDF-2.0\n"), Some(PdfVersion::PDF_2_0));
1077        assert_eq!(read_version(b"%PDF-x.y\n"), None);
1078        // The private packing round-trips every digit pair the header can
1079        // spell, and only `0.0` — which is unreachable, since a `0` packed
1080        // value is reported as "no version" — is not a `Some`.
1081        for major in 0..=9u8 {
1082            for minor in 0..=9u8 {
1083                let header = format!("%PDF-{major}.{minor}\n");
1084                let expected = (major, minor) != (0, 0);
1085                assert_eq!(
1086                    read_version(header.as_bytes()),
1087                    expected.then(|| PdfVersion::new(major, minor)),
1088                    "header {header:?}"
1089                );
1090            }
1091        }
1092    }
1093
1094    #[test]
1095    fn a_file_without_a_header_is_not_a_pdf() {
1096        assert_eq!(open(b"just some bytes").err(), Some(LoadError::NotPdf));
1097        // A header at the very end with no room for a document.
1098        assert_eq!(open(b"%PDF").err(), Some(LoadError::NotPdf));
1099    }
1100
1101    #[test]
1102    fn opens_a_document_and_counts_its_pages() {
1103        let doc = open(&build(3)).expect("document");
1104        assert_eq!(doc.page_count(), 3);
1105        assert_eq!(doc.version(), Some(PdfVersion::PDF_1_7));
1106        assert!(!doc.xref_was_rebuilt());
1107        assert!(!doc.is_encrypted());
1108        assert_eq!(doc.permissions(), Permissions::ALL);
1109    }
1110
1111    #[test]
1112    fn reads_pages_in_order() {
1113        let doc = open(&build(5)).expect("document");
1114        let number = Name::from("PageNumber");
1115        for i in 0..5u32 {
1116            let page = doc.page(i).expect("page");
1117            assert_eq!(page.dict.direct_int(&number), Some(i64::from(i)));
1118        }
1119        assert!(doc.page(5).is_err());
1120    }
1121
1122    #[test]
1123    fn reads_pages_in_reverse_and_out_of_order() {
1124        let doc = open(&build(5)).expect("document");
1125        let number = Name::from("PageNumber");
1126        for i in (0..5u32).rev() {
1127            assert_eq!(
1128                doc.page(i).expect("page").dict.direct_int(&number),
1129                Some(i64::from(i))
1130            );
1131        }
1132        // An out-of-range lookup must not poison the ones after it.
1133        assert!(doc.page(99).is_err());
1134        assert_eq!(doc.page(3).expect("page").dict.direct_int(&number), Some(3));
1135    }
1136
1137    #[test]
1138    fn a_count_larger_than_the_tree_reports_the_lie() {
1139        let text = String::from_utf8_lossy(&build(3)).replace("/Count 3", "/Count 9");
1140        let doc = open(text.as_bytes()).expect("document");
1141        // The claimed count is what the document reports...
1142        assert_eq!(doc.page_count(), 9);
1143        // ...and the pages that exist still resolve.
1144        assert!(doc.page(0).is_ok());
1145        assert!(doc.page(2).is_ok());
1146        // The ones past the real tree do not.
1147        assert!(doc.page(3).is_err());
1148        assert!(doc.page(8).is_err());
1149        // And the real ones still work afterwards.
1150        assert!(doc.page(2).is_ok());
1151    }
1152
1153    #[test]
1154    fn a_kids_less_pages_node_counts_as_a_page_it_cannot_produce() {
1155        // Counting and looking up disagree here, and both are right. The
1156        // count treats a `/Pages` node with no `/Kids` as the document's one
1157        // page, but the walk refuses to hand back a node that calls itself a
1158        // branch — so the document reports one page and has none.
1159        let file = b"%PDF-1.7\n\
1160                     1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n\
1161                     2 0 obj\n<< /Type /Pages /Count 3 >>\nendobj\n\
1162                     trailer\n<< /Root 1 0 R >>\nstartxref\n0\n%%EOF\n";
1163        let doc = open(file).expect("document");
1164        assert_eq!(doc.page_count(), 1);
1165        assert!(doc.page(0).is_err());
1166        assert!(doc.page(1).is_err());
1167    }
1168
1169    #[test]
1170    fn a_kids_less_node_that_does_not_claim_to_be_a_branch_is_a_page() {
1171        // The same shape without the `/Type /Pages` claim: the node has no
1172        // children, so it is the page itself and the lookup succeeds.
1173        let file = b"%PDF-1.7\n\
1174                     1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n\
1175                     2 0 obj\n<< /MediaBox [0 0 10 10] >>\nendobj\n\
1176                     trailer\n<< /Root 1 0 R >>\nstartxref\n0\n%%EOF\n";
1177        let doc = open(file).expect("document");
1178        assert_eq!(doc.page_count(), 1);
1179        assert!(doc.page(0).is_ok());
1180    }
1181
1182    #[test]
1183    fn a_catalog_without_pages_will_not_open() {
1184        let file = b"%PDF-1.7\n\
1185                     1 0 obj\n<< /Type /Catalog >>\nendobj\n\
1186                     trailer\n<< /Root 1 0 R >>\nstartxref\n0\n%%EOF\n";
1187        assert!(matches!(open(file), Err(LoadError::Broken(_))));
1188    }
1189
1190    #[test]
1191    fn a_root_written_inline_does_not_name_a_catalog() {
1192        let file = b"%PDF-1.7\n\
1193                     1 0 obj\n<< /Type /Page >>\nendobj\n\
1194                     trailer\n<< /Root << /Type /Catalog /Pages 2 0 R >> >>\n\
1195                     startxref\n0\n%%EOF\n";
1196        assert!(matches!(open(file), Err(LoadError::Broken(_))));
1197    }
1198
1199    #[test]
1200    fn a_broken_start_xref_still_opens_the_document() {
1201        let text = String::from_utf8_lossy(&build(2)).into_owned();
1202        let broken = text
1203            .replace("%%EOF", "")
1204            .replace("startxref\n", "startxref\n1\n");
1205        let doc = open(broken.as_bytes()).expect("document");
1206        assert!(doc.xref_was_rebuilt());
1207        assert_eq!(doc.page_count(), 2);
1208        assert!(doc.diags.contains(&DiagKind::XrefRebuilt));
1209    }
1210
1211    #[test]
1212    fn a_header_after_junk_shifts_every_offset() {
1213        let mut file = vec![b'x'; 100];
1214        file.extend_from_slice(&build(2));
1215        let doc = open(&file).expect("document");
1216        assert_eq!(doc.header_offset(), 100);
1217        assert_eq!(doc.page_count(), 2);
1218        assert!(doc.diags.contains(&DiagKind::HeaderOffset));
1219    }
1220
1221    #[test]
1222    fn inheritable_attributes_come_from_the_parent() {
1223        let doc = open(&build(2)).expect("document");
1224        let page = doc.page(0).expect("page");
1225        // The page states no /MediaBox; its /Pages parent does.
1226        let inherited = page
1227            .inherited(names::MEDIA_BOX, &doc)
1228            .expect("inherited media box");
1229        let array = inherited.as_array().expect("array");
1230        assert_eq!(array.number_at(2), Some(612.0));
1231        assert_eq!(array.number_at(3), Some(792.0));
1232        // A key nobody states is absent.
1233        assert!(page.inherited(names::ROTATE, &doc).is_none());
1234    }
1235
1236    #[test]
1237    fn a_pages_reference_is_reported() {
1238        let doc = open(&build(1)).expect("document");
1239        assert_eq!(doc.page(0).expect("page").reference.map(|r| r.num), Some(3));
1240    }
1241
1242    #[test]
1243    fn documents_are_send_and_sync() {
1244        fn assert_both<T: Send + Sync>() {}
1245        assert_both::<super::Document>();
1246    }
1247
1248    #[test]
1249    fn never_panics_on_arbitrary_bytes() {
1250        let seeds: &[&[u8]] = &[
1251            b"",
1252            b"%PDF",
1253            b"%PDF-1.7",
1254            b"%PDF-1.7\nstartxref\n0\n%%EOF",
1255            b"%PDF-1.7\ntrailer<</Root 1 0 R>>",
1256            b"%PDF-1.7\n1 0 obj<</Length 1 0 R>>stream\n",
1257            b"%PDF-1.7\n\x00\xff\x80\x0b",
1258        ];
1259        for seed in seeds {
1260            let _ = open(seed);
1261        }
1262    }
1263}