Skip to main content

microsandbox_image/erofs/
reader.rs

1//! Minimal EROFS reader for extracting file contents from our own images.
2//!
3//! Only supports the subset of EROFS that our writer produces:
4//! - Extended inodes (64 bytes)
5//! - Uncompressed data (FLAT_PLAIN or FLAT_INLINE)
6//! - Sorted directory entries (binary search)
7//! - No shared xattrs, no compression, no chunks
8
9use std::collections::HashSet;
10use std::io::Read;
11use std::path::Path;
12use std::{fs::File, io, path::PathBuf};
13
14use super::format::{
15    EROFS_BLKSIZ, EROFS_DIRENT_SIZE, EROFS_INODE_EXTENDED_SIZE, EROFS_INODE_FLAT_INLINE,
16    EROFS_INODE_FLAT_PLAIN, EROFS_NULL_ADDR, EROFS_SUPER_OFFSET, EROFS_XATTR_IBODY_HEADER_SIZE,
17    EROFS_XATTR_INDEX_SECURITY, EROFS_XATTR_INDEX_TRUSTED, EROFS_XATTR_INDEX_USER, S_IFBLK,
18    S_IFCHR, S_IFDIR, S_IFIFO, S_IFLNK, S_IFMT, S_IFREG, S_IFSOCK, erofs_xattr_align,
19};
20use crate::path_bytes::os_string_from_vec;
21use crate::tree::{InodeMetadata, Xattr};
22
23//--------------------------------------------------------------------------------------------------
24// Types
25//--------------------------------------------------------------------------------------------------
26
27/// A handle to an open EROFS image for reading.
28pub struct ErofsReader {
29    file: File,
30    meta_blkaddr: u32,
31    root_nid: u32,
32}
33
34#[derive(Debug, Clone, Copy, PartialEq, Eq)]
35pub enum ErofsEntryKind {
36    RegularFile,
37    Directory,
38    Symlink,
39    CharDevice,
40    BlockDevice,
41    Fifo,
42    Socket,
43}
44
45#[derive(Debug, Clone, PartialEq, Eq)]
46pub struct ErofsEntryInfo {
47    pub kind: ErofsEntryKind,
48    pub opaque: bool,
49    pub whiteout: bool,
50}
51
52/// A filesystem entry discovered while walking an EROFS image.
53#[derive(Clone)]
54pub struct ErofsTreeEntry {
55    /// Path relative to the image root.
56    pub path: PathBuf,
57    /// Stable EROFS inode identifier.
58    pub nid: u32,
59    /// Entry kind.
60    pub kind: ErofsEntryKind,
61    /// POSIX inode metadata.
62    pub metadata: InodeMetadata,
63    /// Inline xattrs stored on the inode.
64    pub xattrs: Vec<Xattr>,
65    /// File or symlink data size.
66    pub size: u64,
67    /// Device major/minor for device nodes.
68    pub rdev: Option<(u32, u32)>,
69}
70
71/// Streaming reader for a regular file stored inside an EROFS image.
72pub struct ErofsFileDataReader {
73    file: File,
74    segments: Vec<(u64, u64)>,
75    segment_index: usize,
76    segment_offset: u64,
77}
78
79#[cfg(test)]
80#[derive(Debug, Clone, Copy, PartialEq, Eq)]
81pub(crate) struct ErofsInodeDebugInfo {
82    pub nid: u32,
83    pub nlink: u32,
84    pub size: u64,
85    pub data_layout: u8,
86}
87
88//--------------------------------------------------------------------------------------------------
89// Methods
90//--------------------------------------------------------------------------------------------------
91
92impl ErofsReader {
93    /// Open an EROFS image by parsing the superblock.
94    pub fn new(file: File) -> io::Result<Self> {
95        let mut sb = [0u8; 128];
96        read_exact_at(&file, EROFS_SUPER_OFFSET, &mut sb)?;
97
98        let magic = u32::from_le_bytes([sb[0], sb[1], sb[2], sb[3]]);
99        if magic != 0xE0F5_E1E2 {
100            return Err(io::Error::new(
101                io::ErrorKind::InvalidData,
102                format!("bad EROFS magic: {magic:#x}"),
103            ));
104        }
105
106        let root_nid = u16::from_le_bytes([sb[0x0E], sb[0x0F]]) as u32;
107        let meta_blkaddr = u32::from_le_bytes([sb[0x28], sb[0x29], sb[0x2A], sb[0x2B]]);
108
109        Ok(Self {
110            file,
111            meta_blkaddr,
112            root_nid,
113        })
114    }
115
116    /// Read a file by path from the EROFS image. Returns the file data.
117    pub fn read_file(&mut self, path: &str) -> io::Result<Vec<u8>> {
118        let target_inode = self.lookup_path(path)?;
119        if (target_inode.mode & S_IFMT) != S_IFREG {
120            return Err(io::Error::new(
121                io::ErrorKind::InvalidInput,
122                "target is not a regular file",
123            ));
124        }
125        self.read_inode_data(&target_inode)
126    }
127
128    /// Read a symlink target by path from the EROFS image.
129    pub fn read_link(&mut self, path: &str) -> io::Result<Vec<u8>> {
130        let target_inode = self.lookup_path(path)?;
131        if (target_inode.mode & S_IFMT) != S_IFLNK {
132            return Err(io::Error::new(
133                io::ErrorKind::InvalidInput,
134                "target is not a symlink",
135            ));
136        }
137        self.read_inode_data(&target_inode)
138    }
139
140    pub fn entry_info(&mut self, path: &str) -> io::Result<ErofsEntryInfo> {
141        let inode = self.lookup_path(path)?;
142        let kind = inode_kind(&inode)?;
143        let opaque = if kind == ErofsEntryKind::Directory {
144            self.inode_is_opaque(&inode)?
145        } else {
146            false
147        };
148        let whiteout = kind == ErofsEntryKind::CharDevice && inode.rdev == 0;
149
150        Ok(ErofsEntryInfo {
151            kind,
152            opaque,
153            whiteout,
154        })
155    }
156
157    /// Walk all entries in the image in stable path order.
158    pub fn walk(&mut self) -> io::Result<Vec<ErofsTreeEntry>> {
159        let root = self.read_inode(self.root_nid)?;
160        let mut entries = Vec::new();
161        let mut visited = HashSet::new();
162        self.walk_dir(&root, Vec::new(), &mut entries, &mut visited)?;
163        Ok(entries)
164    }
165
166    /// Walk all entries in stable path order, invoking a callback for each entry.
167    pub fn walk_entries<E, F>(&mut self, mut visit: F) -> Result<(), E>
168    where
169        E: From<io::Error>,
170        F: FnMut(&mut Self, ErofsTreeEntry) -> Result<(), E>,
171    {
172        self.walk_entries_with_path_bytes(|reader, _path, entry| visit(reader, entry))
173    }
174
175    /// Walk all entries while retaining canonical guest path bytes for internal image pipelines.
176    pub(crate) fn walk_entries_with_path_bytes<E, F>(&mut self, mut visit: F) -> Result<(), E>
177    where
178        E: From<io::Error>,
179        F: FnMut(&mut Self, &[u8], ErofsTreeEntry) -> Result<(), E>,
180    {
181        let root = self.read_inode(self.root_nid)?;
182        let mut visited = HashSet::new();
183        self.walk_dir_entries(&root, Vec::new(), &mut visited, &mut visit)
184    }
185
186    /// Create a streaming reader for a regular file inode by NID.
187    pub fn file_data_reader(&mut self, nid: u32) -> io::Result<ErofsFileDataReader> {
188        let inode = self.read_inode(nid)?;
189        if (inode.mode & S_IFMT) != S_IFREG {
190            return Err(io::Error::new(
191                io::ErrorKind::InvalidInput,
192                "target is not a regular file",
193            ));
194        }
195
196        Ok(ErofsFileDataReader {
197            file: self.file.try_clone()?,
198            segments: self.inode_data_segments(&inode)?,
199            segment_index: 0,
200            segment_offset: 0,
201        })
202    }
203
204    /// Return the block mapping recorded for a regular file in an image produced by our writer.
205    ///
206    /// Regular files are deliberately emitted as `FLAT_PLAIN`, which lets fsmeta be rebuilt from
207    /// a cached EROFS layer without retaining or re-downloading its source tarball.
208    pub(crate) fn file_block_mapping(&mut self, nid: u32) -> io::Result<(u32, u64)> {
209        let inode = self.read_inode(nid)?;
210        if (inode.mode & S_IFMT) != S_IFREG {
211            return Err(io::Error::new(
212                io::ErrorKind::InvalidInput,
213                "block mapping is available only for regular files",
214            ));
215        }
216        if inode.size == 0 {
217            return Ok((EROFS_NULL_ADDR, 0));
218        }
219        if inode.data_layout != EROFS_INODE_FLAT_PLAIN || inode.startblk_lo == EROFS_NULL_ADDR {
220            return Err(io::Error::new(
221                io::ErrorKind::InvalidData,
222                "regular file does not use the expected flat-plain EROFS layout",
223            ));
224        }
225        Ok((inode.startblk_lo, inode.size))
226    }
227
228    /// Return metadata and xattrs for the filesystem root directory.
229    pub(crate) fn root_directory_metadata(&mut self) -> io::Result<(InodeMetadata, Vec<Xattr>)> {
230        let inode = self.read_inode(self.root_nid)?;
231        if (inode.mode & S_IFMT) != S_IFDIR {
232            return Err(io::Error::new(
233                io::ErrorKind::InvalidData,
234                "EROFS root inode is not a directory",
235            ));
236        }
237        let xattrs = self
238            .read_inode_xattrs(&inode)?
239            .into_iter()
240            .map(|(name, value)| Xattr { name, value })
241            .collect();
242        Ok((inode.metadata(), xattrs))
243    }
244
245    /// Return the inode metadata and xattrs for a path. The mode includes the
246    /// file type bits.
247    pub fn entry_metadata(&mut self, path: &str) -> io::Result<(InodeMetadata, Vec<Xattr>)> {
248        let inode = self.lookup_path(path)?;
249        let xattrs = self
250            .read_inode_xattrs(&inode)?
251            .into_iter()
252            .map(|(name, value)| Xattr { name, value })
253            .collect();
254        Ok((inode.metadata(), xattrs))
255    }
256
257    /// Read a symlink target by NID.
258    pub fn read_link_by_nid(&mut self, nid: u32) -> io::Result<Vec<u8>> {
259        let inode = self.read_inode(nid)?;
260        if (inode.mode & S_IFMT) != S_IFLNK {
261            return Err(io::Error::new(
262                io::ErrorKind::InvalidInput,
263                "target is not a symlink",
264            ));
265        }
266        self.read_inode_data(&inode)
267    }
268
269    #[cfg(test)]
270    pub(crate) fn inode_debug_info(&mut self, path: &str) -> io::Result<ErofsInodeDebugInfo> {
271        let inode = self.lookup_path(path)?;
272        Ok(ErofsInodeDebugInfo {
273            nid: inode.nid,
274            nlink: inode.nlink,
275            size: inode.size,
276            data_layout: inode.data_layout,
277        })
278    }
279
280    fn inode_offset(&self, nid: u32) -> u64 {
281        (self.meta_blkaddr as u64) * (EROFS_BLKSIZ as u64) + (nid as u64) * 32
282    }
283
284    fn read_inode(&mut self, nid: u32) -> io::Result<InodeInfo> {
285        let offset = self.inode_offset(nid);
286
287        let mut buf = [0u8; EROFS_INODE_EXTENDED_SIZE as usize];
288        read_exact_at(&self.file, offset, &mut buf)?;
289
290        let i_format = u16::from_le_bytes([buf[0], buf[1]]);
291        let i_xattr_icount = u16::from_le_bytes([buf[2], buf[3]]);
292        let mode = u16::from_le_bytes([buf[4], buf[5]]);
293        let size = u64::from_le_bytes([
294            buf[8], buf[9], buf[10], buf[11], buf[12], buf[13], buf[14], buf[15],
295        ]);
296        let i_u = u32::from_le_bytes([buf[16], buf[17], buf[18], buf[19]]);
297        let nlink = u32::from_le_bytes([buf[44], buf[45], buf[46], buf[47]]);
298        let uid = u32::from_le_bytes([buf[24], buf[25], buf[26], buf[27]]);
299        let gid = u32::from_le_bytes([buf[28], buf[29], buf[30], buf[31]]);
300        let mtime = u64::from_le_bytes([
301            buf[32], buf[33], buf[34], buf[35], buf[36], buf[37], buf[38], buf[39],
302        ]);
303        let mtime_nsec = u32::from_le_bytes([buf[40], buf[41], buf[42], buf[43]]);
304
305        let data_layout = ((i_format >> 1) & 0x07) as u8;
306
307        // Compute xattr ibody size to know where inline data starts.
308        // Formula from EROFS spec: ibody = 12-byte header + (i_xattr_icount - 1) * 4 bytes.
309        // The "- 1" accounts for the header occupying the first count unit.
310        let xattr_ibody_size = if i_xattr_icount == 0 {
311            0u32
312        } else {
313            12 + ((i_xattr_icount as u32) - 1) * 4
314        };
315
316        Ok(InodeInfo {
317            nid,
318            mode,
319            size,
320            nlink,
321            uid,
322            gid,
323            mtime,
324            mtime_nsec,
325            data_layout,
326            startblk_lo: i_u,
327            rdev: i_u,
328            xattr_ibody_size,
329        })
330    }
331
332    fn lookup_path(&mut self, path: &str) -> io::Result<InodeInfo> {
333        let components: Vec<&str> = path
334            .trim_start_matches('/')
335            .split('/')
336            .filter(|c| !c.is_empty())
337            .collect();
338
339        if components.is_empty() {
340            if path == "/" {
341                return self.read_inode(self.root_nid);
342            }
343            return Err(io::Error::new(io::ErrorKind::InvalidInput, "empty path"));
344        }
345
346        let mut current_nid = self.root_nid;
347        for (i, component) in components.iter().enumerate() {
348            let inode = self.read_inode(current_nid)?;
349            let mode_type = inode.mode & S_IFMT;
350
351            if mode_type != S_IFDIR {
352                return Err(io::Error::new(
353                    io::ErrorKind::NotFound,
354                    format!("not a directory at component '{component}'"),
355                ));
356            }
357
358            let target_nid = self.lookup_in_dir(&inode, component)?;
359            if i + 1 == components.len() {
360                return self.read_inode(target_nid);
361            }
362
363            current_nid = target_nid;
364        }
365
366        Err(io::Error::new(io::ErrorKind::NotFound, "path not found"))
367    }
368
369    /// Look up a named entry in a directory inode's data.
370    ///
371    /// EROFS directory data is organized as self-contained blocks. Each block
372    /// starts with a packed array of 12-byte dirent headers, followed by the
373    /// concatenated name strings. The first dirent's `nameoff` field divided
374    /// by 12 gives the number of dirents in that block (the kernel uses this
375    /// same trick). Name lengths are derived from consecutive `nameoff`
376    /// values; the last entry's name extends to the end of valid data.
377    fn lookup_in_dir(&mut self, dir_inode: &InodeInfo, name: &str) -> io::Result<u32> {
378        let blksiz = EROFS_BLKSIZ as usize;
379        let target = name.as_bytes();
380        let block_count = self.checked_inode_data_len(dir_inode)?.div_ceil(blksiz);
381        let mut left = 0usize;
382        let mut right = block_count;
383
384        while left < right {
385            let mid = (left + right) / 2;
386            let block = self.read_inode_data_block(dir_inode, mid)?;
387            let dirent_count = dir_block_dirent_count(&block)?;
388            let first_name = dirent_name(&block, 0, dirent_count)?;
389            let last_name = dirent_name(&block, dirent_count - 1, dirent_count)?;
390
391            if target < first_name {
392                right = mid;
393                continue;
394            }
395
396            if target > last_name {
397                left = mid + 1;
398                continue;
399            }
400
401            return lookup_in_dir_block(&block, dirent_count, target)?.ok_or_else(|| {
402                io::Error::new(
403                    io::ErrorKind::NotFound,
404                    format!("entry '{name}' not found in directory"),
405                )
406            });
407        }
408
409        Err(io::Error::new(
410            io::ErrorKind::NotFound,
411            format!("entry '{name}' not found in directory"),
412        ))
413    }
414
415    fn walk_dir(
416        &mut self,
417        dir_inode: &InodeInfo,
418        dir_path: Vec<u8>,
419        entries: &mut Vec<ErofsTreeEntry>,
420        visited: &mut HashSet<u32>,
421    ) -> io::Result<()> {
422        if !visited.insert(dir_inode.nid) {
423            return Err(io::Error::new(
424                io::ErrorKind::InvalidData,
425                "cycle detected while walking EROFS directory tree",
426            ));
427        }
428
429        self.visit_dir_entries::<io::Error, _>(dir_inode, &mut |reader, name, nid| {
430            if name == b"." || name == b".." {
431                return Ok(());
432            }
433
434            let path = join_image_path(&dir_path, name)?;
435            let inode = reader.read_inode(nid)?;
436            let entry = reader.tree_entry(path.clone(), &inode)?;
437            let is_dir = entry.kind == ErofsEntryKind::Directory;
438            entries.push(entry);
439
440            if is_dir {
441                reader.walk_dir(&inode, path, entries, visited)?;
442            }
443            Ok(())
444        })?;
445
446        Ok(())
447    }
448
449    fn walk_dir_entries<E, F>(
450        &mut self,
451        dir_inode: &InodeInfo,
452        dir_path: Vec<u8>,
453        visited: &mut HashSet<u32>,
454        visit: &mut F,
455    ) -> Result<(), E>
456    where
457        E: From<io::Error>,
458        F: FnMut(&mut Self, &[u8], ErofsTreeEntry) -> Result<(), E>,
459    {
460        if !visited.insert(dir_inode.nid) {
461            return Err(io::Error::new(
462                io::ErrorKind::InvalidData,
463                "cycle detected while walking EROFS directory tree",
464            )
465            .into());
466        }
467
468        self.visit_dir_entries::<E, _>(dir_inode, &mut |reader, name, nid| {
469            if name == b"." || name == b".." {
470                return Ok(());
471            }
472
473            let path = join_image_path(&dir_path, name)?;
474            let inode = reader.read_inode(nid)?;
475            let entry = reader.tree_entry(path.clone(), &inode)?;
476            let is_dir = entry.kind == ErofsEntryKind::Directory;
477            visit(reader, &path, entry)?;
478
479            if is_dir {
480                reader.walk_dir_entries(&inode, path, visited, visit)?;
481            }
482            Ok(())
483        })?;
484
485        Ok(())
486    }
487
488    fn visit_dir_entries<E, F>(&mut self, dir_inode: &InodeInfo, visit: &mut F) -> Result<(), E>
489    where
490        E: From<io::Error>,
491        F: FnMut(&mut Self, &[u8], u32) -> Result<(), E>,
492    {
493        if (dir_inode.mode & S_IFMT) != S_IFDIR {
494            return Err(
495                io::Error::new(io::ErrorKind::InvalidInput, "target is not a directory").into(),
496            );
497        }
498
499        let blksiz = EROFS_BLKSIZ as usize;
500        let block_count = self.checked_inode_data_len(dir_inode)?.div_ceil(blksiz);
501
502        for block_index in 0..block_count {
503            let block = self.read_inode_data_block(dir_inode, block_index)?;
504            if block.is_empty() {
505                continue;
506            }
507            let dirent_count = dir_block_dirent_count(&block)?;
508            for idx in 0..dirent_count {
509                let name = dirent_name(&block, idx, dirent_count)?;
510                if name.is_empty() {
511                    continue;
512                }
513                visit(self, name, dirent_nid(&block, idx)?)?;
514            }
515        }
516
517        Ok(())
518    }
519
520    fn tree_entry(&mut self, path: Vec<u8>, inode: &InodeInfo) -> io::Result<ErofsTreeEntry> {
521        let kind = inode_kind(inode)?;
522        let rdev = if matches!(
523            kind,
524            ErofsEntryKind::CharDevice | ErofsEntryKind::BlockDevice
525        ) {
526            Some(decode_dev(inode.rdev))
527        } else {
528            None
529        };
530
531        Ok(ErofsTreeEntry {
532            path: PathBuf::from(os_string_from_vec(path)?),
533            nid: inode.nid,
534            kind,
535            metadata: inode.metadata(),
536            xattrs: self
537                .read_inode_xattrs(inode)?
538                .into_iter()
539                .map(|(name, value)| Xattr { name, value })
540                .collect(),
541            size: inode.size,
542            rdev,
543        })
544    }
545
546    fn read_inode_data(&mut self, inode: &InodeInfo) -> io::Result<Vec<u8>> {
547        let size = self.checked_inode_data_len(inode)?;
548        if size == 0 {
549            return Ok(Vec::new());
550        }
551
552        let blksiz = EROFS_BLKSIZ as usize;
553
554        match inode.data_layout {
555            EROFS_INODE_FLAT_PLAIN => {
556                if inode.startblk_lo == EROFS_NULL_ADDR {
557                    return Ok(Vec::new());
558                }
559                let data_offset = (inode.startblk_lo as u64) * (EROFS_BLKSIZ as u64);
560                let mut data = vec![0u8; size];
561                read_exact_at(&self.file, data_offset, &mut data)?;
562                Ok(data)
563            }
564            EROFS_INODE_FLAT_INLINE => {
565                let full_blocks = size / blksiz;
566                let tail_size = size % blksiz;
567                let mut data = Vec::with_capacity(size);
568
569                // Read full blocks from data area.
570                if full_blocks > 0 && inode.startblk_lo != EROFS_NULL_ADDR {
571                    let data_offset = (inode.startblk_lo as u64) * (EROFS_BLKSIZ as u64);
572                    let mut block_data = vec![0u8; full_blocks * blksiz];
573                    read_exact_at(&self.file, data_offset, &mut block_data)?;
574                    data.extend_from_slice(&block_data);
575                }
576
577                // Read inline tail from after inode metadata.
578                if tail_size > 0 {
579                    let inline_offset = self.inode_offset(inode.nid)
580                        + EROFS_INODE_EXTENDED_SIZE as u64
581                        + inode.xattr_ibody_size as u64;
582                    let mut tail = vec![0u8; tail_size];
583                    read_exact_at(&self.file, inline_offset, &mut tail)?;
584                    data.extend_from_slice(&tail);
585                }
586
587                Ok(data)
588            }
589            _ => Err(io::Error::new(
590                io::ErrorKind::Unsupported,
591                format!("unsupported data layout: {}", inode.data_layout),
592            )),
593        }
594    }
595
596    fn read_inode_data_block(&self, inode: &InodeInfo, block_index: usize) -> io::Result<Vec<u8>> {
597        let blksiz = EROFS_BLKSIZ as usize;
598        let size = self.checked_inode_data_len(inode)?;
599        let start = block_index.checked_mul(blksiz).ok_or_else(|| {
600            io::Error::new(io::ErrorKind::InvalidData, "directory block overflow")
601        })?;
602        if start >= size {
603            return Ok(Vec::new());
604        }
605
606        let remaining = size - start;
607        let len = remaining.min(blksiz);
608        self.read_inode_data_range(inode, start as u64, len)
609    }
610
611    fn read_inode_data_range(
612        &self,
613        inode: &InodeInfo,
614        start: u64,
615        len: usize,
616    ) -> io::Result<Vec<u8>> {
617        let size = self.checked_inode_data_len(inode)? as u64;
618        let end = start
619            .checked_add(len as u64)
620            .ok_or_else(|| io::Error::new(io::ErrorKind::InvalidData, "inode range overflow"))?;
621        if end > size {
622            return Err(io::Error::new(
623                io::ErrorKind::InvalidData,
624                "inode data range exceeds inode size",
625            ));
626        }
627
628        let segments = self.inode_data_segments(inode)?;
629        let mut data = vec![0u8; len];
630        let mut copied = 0usize;
631        let mut logical_start = 0u64;
632
633        for (file_offset, segment_len) in segments {
634            let logical_end = logical_start.checked_add(segment_len).ok_or_else(|| {
635                io::Error::new(io::ErrorKind::InvalidData, "inode segment range overflow")
636            })?;
637            let overlap_start = start.max(logical_start);
638            let overlap_end = end.min(logical_end);
639
640            if overlap_start < overlap_end {
641                let dst_start = (overlap_start - start) as usize;
642                let read_len = (overlap_end - overlap_start) as usize;
643                let source_offset = file_offset
644                    .checked_add(overlap_start - logical_start)
645                    .ok_or_else(|| {
646                        io::Error::new(io::ErrorKind::InvalidData, "inode file offset overflow")
647                    })?;
648                read_exact_at(
649                    &self.file,
650                    source_offset,
651                    &mut data[dst_start..dst_start + read_len],
652                )?;
653                copied += read_len;
654            }
655
656            logical_start = logical_end;
657            if logical_start >= end {
658                break;
659            }
660        }
661
662        if copied != len {
663            return Err(io::Error::new(
664                io::ErrorKind::UnexpectedEof,
665                "inode data range is not fully backed",
666            ));
667        }
668
669        Ok(data)
670    }
671
672    fn checked_inode_data_len(&self, inode: &InodeInfo) -> io::Result<usize> {
673        let file_len = self.file.metadata()?.len();
674        if inode.size > file_len {
675            return Err(io::Error::new(
676                io::ErrorKind::InvalidData,
677                "inode data size exceeds EROFS image size",
678            ));
679        }
680
681        usize::try_from(inode.size).map_err(|_| {
682            io::Error::new(
683                io::ErrorKind::InvalidData,
684                "inode data size does not fit in memory",
685            )
686        })
687    }
688
689    fn inode_data_segments(&self, inode: &InodeInfo) -> io::Result<Vec<(u64, u64)>> {
690        let size = inode.size;
691        if size == 0 {
692            return Ok(Vec::new());
693        }
694
695        let blksiz = EROFS_BLKSIZ as u64;
696        match inode.data_layout {
697            EROFS_INODE_FLAT_PLAIN => {
698                if inode.startblk_lo == EROFS_NULL_ADDR {
699                    Ok(Vec::new())
700                } else {
701                    Ok(vec![((inode.startblk_lo as u64) * blksiz, size)])
702                }
703            }
704            EROFS_INODE_FLAT_INLINE => {
705                let full_blocks = size / blksiz;
706                let tail_size = size % blksiz;
707                let mut segments = Vec::new();
708                if full_blocks > 0 && inode.startblk_lo != EROFS_NULL_ADDR {
709                    segments.push(((inode.startblk_lo as u64) * blksiz, full_blocks * blksiz));
710                }
711                if tail_size > 0 {
712                    segments.push((
713                        self.inode_offset(inode.nid)
714                            + EROFS_INODE_EXTENDED_SIZE as u64
715                            + inode.xattr_ibody_size as u64,
716                        tail_size,
717                    ));
718                }
719                Ok(segments)
720            }
721            _ => Err(io::Error::new(
722                io::ErrorKind::Unsupported,
723                format!("unsupported data layout: {}", inode.data_layout),
724            )),
725        }
726    }
727
728    fn inode_is_opaque(&mut self, inode: &InodeInfo) -> io::Result<bool> {
729        for (name, value) in self.read_inode_xattrs(inode)? {
730            if name == b"trusted.overlay.opaque" && value == b"y" {
731                return Ok(true);
732            }
733        }
734
735        Ok(false)
736    }
737
738    fn read_inode_xattrs(&mut self, inode: &InodeInfo) -> io::Result<Vec<(Vec<u8>, Vec<u8>)>> {
739        if inode.xattr_ibody_size == 0 {
740            return Ok(Vec::new());
741        }
742
743        let total = inode.xattr_ibody_size as usize;
744        if total < EROFS_XATTR_IBODY_HEADER_SIZE as usize {
745            return Err(io::Error::new(
746                io::ErrorKind::InvalidData,
747                "xattr ibody smaller than header",
748            ));
749        }
750
751        let mut offset = self.inode_offset(inode.nid)
752            + EROFS_INODE_EXTENDED_SIZE as u64
753            + EROFS_XATTR_IBODY_HEADER_SIZE as u64;
754        let mut remaining = total - EROFS_XATTR_IBODY_HEADER_SIZE as usize;
755        let mut xattrs = Vec::new();
756
757        while remaining > 0 {
758            if remaining < 4 {
759                return Err(io::Error::new(
760                    io::ErrorKind::InvalidData,
761                    "truncated xattr entry header",
762                ));
763            }
764
765            let mut entry = [0u8; 4];
766            read_exact_at(&self.file, offset, &mut entry)?;
767
768            let name_len = entry[0] as usize;
769            let name_index = entry[1];
770            let value_len = u16::from_le_bytes([entry[2], entry[3]]) as usize;
771            let entry_size = 4 + name_len + value_len;
772            let aligned_size = erofs_xattr_align(entry_size);
773
774            if aligned_size > remaining {
775                return Err(io::Error::new(
776                    io::ErrorKind::InvalidData,
777                    "xattr entry exceeds ibody size",
778                ));
779            }
780
781            let mut suffix = vec![0u8; name_len];
782            read_exact_at(&self.file, offset + 4, &mut suffix)?;
783            let mut value = vec![0u8; value_len];
784            read_exact_at(&self.file, offset + 4 + name_len as u64, &mut value)?;
785
786            let name = match name_index {
787                EROFS_XATTR_INDEX_USER => [b"user.".as_slice(), suffix.as_slice()].concat(),
788                EROFS_XATTR_INDEX_TRUSTED => [b"trusted.".as_slice(), suffix.as_slice()].concat(),
789                EROFS_XATTR_INDEX_SECURITY => [b"security.".as_slice(), suffix.as_slice()].concat(),
790                other => {
791                    return Err(io::Error::new(
792                        io::ErrorKind::InvalidData,
793                        format!("unsupported xattr name index: {other}"),
794                    ));
795                }
796            };
797
798            xattrs.push((name, value));
799            offset += aligned_size as u64;
800            remaining -= aligned_size;
801        }
802
803        Ok(xattrs)
804    }
805}
806
807//--------------------------------------------------------------------------------------------------
808// Types: Internal
809//--------------------------------------------------------------------------------------------------
810
811struct InodeInfo {
812    nid: u32,
813    mode: u16,
814    size: u64,
815    #[allow(dead_code)]
816    nlink: u32,
817    uid: u32,
818    gid: u32,
819    mtime: u64,
820    mtime_nsec: u32,
821    data_layout: u8,
822    startblk_lo: u32,
823    rdev: u32,
824    xattr_ibody_size: u32,
825}
826
827impl InodeInfo {
828    fn metadata(&self) -> InodeMetadata {
829        InodeMetadata {
830            uid: self.uid,
831            gid: self.gid,
832            mode: self.mode,
833            mtime: self.mtime,
834            mtime_nsec: self.mtime_nsec,
835        }
836    }
837}
838
839impl ErofsTreeEntry {
840    /// Return true if this directory carries the overlay opaque marker.
841    pub fn is_opaque(&self) -> bool {
842        self.xattrs
843            .iter()
844            .any(|x| x.name == b"trusted.overlay.opaque" && x.value == b"y")
845    }
846}
847
848impl Read for ErofsFileDataReader {
849    fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
850        if buf.is_empty() {
851            return Ok(0);
852        }
853
854        while self.segment_index < self.segments.len() {
855            let (offset, len) = self.segments[self.segment_index];
856            if self.segment_offset >= len {
857                self.segment_index += 1;
858                self.segment_offset = 0;
859                continue;
860            }
861
862            let remaining = (len - self.segment_offset) as usize;
863            let to_read = remaining.min(buf.len());
864            let read = read_at_file(
865                &self.file,
866                &mut buf[..to_read],
867                offset + self.segment_offset,
868            )?;
869            self.segment_offset += read as u64;
870            return Ok(read);
871        }
872
873        Ok(0)
874    }
875}
876
877//--------------------------------------------------------------------------------------------------
878// Functions
879//--------------------------------------------------------------------------------------------------
880
881/// Append one EROFS directory entry using the image's canonical separator.
882///
883/// Image paths belong to the Linux guest namespace, so host-native path joining
884/// must not turn `/` into `\` when materialization runs on Windows.
885fn join_image_path(parent: &[u8], name: &[u8]) -> io::Result<Vec<u8>> {
886    if name.is_empty() || name.contains(&b'/') || name.contains(&0) {
887        return Err(io::Error::new(
888            io::ErrorKind::InvalidData,
889            "EROFS directory entry contains an invalid name",
890        ));
891    }
892
893    let mut path = Vec::with_capacity(parent.len() + usize::from(!parent.is_empty()) + name.len());
894    path.extend_from_slice(parent);
895    if !parent.is_empty() {
896        path.push(b'/');
897    }
898    path.extend_from_slice(name);
899    Ok(path)
900}
901
902fn read_exact_at(file: &File, offset: u64, mut buf: &mut [u8]) -> io::Result<()> {
903    let mut current_offset = offset;
904    while !buf.is_empty() {
905        let read = read_at_file(file, buf, current_offset)?;
906        if read == 0 {
907            return Err(io::Error::new(
908                io::ErrorKind::UnexpectedEof,
909                "unexpected EOF",
910            ));
911        }
912        current_offset += read as u64;
913        buf = &mut buf[read..];
914    }
915
916    Ok(())
917}
918
919#[cfg(unix)]
920fn read_at_file(file: &File, buf: &mut [u8], offset: u64) -> io::Result<usize> {
921    use std::os::unix::fs::FileExt;
922
923    file.read_at(buf, offset)
924}
925
926#[cfg(windows)]
927fn read_at_file(file: &File, buf: &mut [u8], offset: u64) -> io::Result<usize> {
928    use std::os::windows::fs::FileExt;
929
930    file.seek_read(buf, offset)
931}
932
933fn dir_block_dirent_count(block: &[u8]) -> io::Result<usize> {
934    if block.len() < EROFS_DIRENT_SIZE as usize {
935        return Err(io::Error::new(
936            io::ErrorKind::InvalidData,
937            "directory block smaller than one dirent",
938        ));
939    }
940
941    let first_nameoff = u16::from_le_bytes([block[8], block[9]]) as usize;
942    let dirent_size = EROFS_DIRENT_SIZE as usize;
943    if first_nameoff < dirent_size
944        || !first_nameoff.is_multiple_of(dirent_size)
945        || first_nameoff > block.len()
946    {
947        return Err(io::Error::new(
948            io::ErrorKind::InvalidData,
949            "invalid first dirent name offset",
950        ));
951    }
952
953    Ok(first_nameoff / dirent_size)
954}
955
956fn dirent_name(block: &[u8], idx: usize, dirent_count: usize) -> io::Result<&[u8]> {
957    let dirent_size = EROFS_DIRENT_SIZE as usize;
958    let dirent_off = idx
959        .checked_mul(dirent_size)
960        .ok_or_else(|| io::Error::new(io::ErrorKind::InvalidData, "dirent offset overflow"))?;
961
962    if idx >= dirent_count || dirent_off + dirent_size > block.len() {
963        return Err(io::Error::new(
964            io::ErrorKind::InvalidData,
965            "dirent index out of bounds",
966        ));
967    }
968
969    let nameoff = u16::from_le_bytes([block[dirent_off + 8], block[dirent_off + 9]]) as usize;
970    let mut name_end = if idx + 1 < dirent_count {
971        let next_off = dirent_off + dirent_size;
972        u16::from_le_bytes([block[next_off + 8], block[next_off + 9]]) as usize
973    } else {
974        block.len()
975    };
976
977    if nameoff > name_end || name_end > block.len() {
978        return Err(io::Error::new(
979            io::ErrorKind::InvalidData,
980            "dirent name range out of bounds",
981        ));
982    }
983
984    while name_end > nameoff && block[name_end - 1] == 0 {
985        name_end -= 1;
986    }
987
988    Ok(&block[nameoff..name_end])
989}
990
991fn dirent_nid(block: &[u8], idx: usize) -> io::Result<u32> {
992    let dirent_size = EROFS_DIRENT_SIZE as usize;
993    let dirent_off = idx
994        .checked_mul(dirent_size)
995        .ok_or_else(|| io::Error::new(io::ErrorKind::InvalidData, "dirent offset overflow"))?;
996    if dirent_off + dirent_size > block.len() {
997        return Err(io::Error::new(
998            io::ErrorKind::InvalidData,
999            "dirent NID out of bounds",
1000        ));
1001    }
1002
1003    let nid = u64::from_le_bytes([
1004        block[dirent_off],
1005        block[dirent_off + 1],
1006        block[dirent_off + 2],
1007        block[dirent_off + 3],
1008        block[dirent_off + 4],
1009        block[dirent_off + 5],
1010        block[dirent_off + 6],
1011        block[dirent_off + 7],
1012    ]);
1013    u32::try_from(nid)
1014        .map_err(|_| io::Error::new(io::ErrorKind::InvalidData, "dirent NID overflow"))
1015}
1016
1017fn lookup_in_dir_block(
1018    block: &[u8],
1019    dirent_count: usize,
1020    target: &[u8],
1021) -> io::Result<Option<u32>> {
1022    let mut left = 0usize;
1023    let mut right = dirent_count;
1024
1025    while left < right {
1026        let mid = (left + right) / 2;
1027        match target.cmp(dirent_name(block, mid, dirent_count)?) {
1028            std::cmp::Ordering::Less => right = mid,
1029            std::cmp::Ordering::Greater => left = mid + 1,
1030            std::cmp::Ordering::Equal => return dirent_nid(block, mid).map(Some),
1031        }
1032    }
1033
1034    Ok(None)
1035}
1036
1037fn inode_kind(inode: &InodeInfo) -> io::Result<ErofsEntryKind> {
1038    match inode.mode & S_IFMT {
1039        S_IFREG => Ok(ErofsEntryKind::RegularFile),
1040        S_IFDIR => Ok(ErofsEntryKind::Directory),
1041        S_IFLNK => Ok(ErofsEntryKind::Symlink),
1042        S_IFCHR => Ok(ErofsEntryKind::CharDevice),
1043        S_IFBLK => Ok(ErofsEntryKind::BlockDevice),
1044        S_IFIFO => Ok(ErofsEntryKind::Fifo),
1045        S_IFSOCK => Ok(ErofsEntryKind::Socket),
1046        other => Err(io::Error::new(
1047            io::ErrorKind::InvalidData,
1048            format!("unsupported inode mode type: {other:#o}"),
1049        )),
1050    }
1051}
1052
1053fn decode_dev(encoded: u32) -> (u32, u32) {
1054    let major = (encoded >> 8) & 0x0000_0fff;
1055    let minor = (encoded & 0x0000_00ff) | ((encoded >> 12) & 0xffff_ff00);
1056    (major, minor)
1057}
1058
1059/// Read a file from an EROFS image file on disk.
1060pub fn read_file_from_erofs(image_path: &Path, file_path: &str) -> io::Result<Vec<u8>> {
1061    let file = std::fs::File::open(image_path)?;
1062    let mut reader = ErofsReader::new(file)?;
1063    reader.read_file(file_path)
1064}
1065
1066pub fn entry_info_from_erofs(image_path: &Path, file_path: &str) -> io::Result<ErofsEntryInfo> {
1067    let file = std::fs::File::open(image_path)?;
1068    let mut reader = ErofsReader::new(file)?;
1069    reader.entry_info(file_path)
1070}
1071
1072//--------------------------------------------------------------------------------------------------
1073// Tests
1074//--------------------------------------------------------------------------------------------------
1075
1076#[cfg(test)]
1077mod tests {
1078    use std::{fs::File, io, path::PathBuf};
1079
1080    use tempfile::tempdir;
1081
1082    use super::ErofsReader;
1083    use crate::{
1084        erofs::write_erofs,
1085        path_bytes::path_bytes,
1086        tree::{FileData, FileTree, InodeMetadata, RegularFileId, RegularFileNode, TreeNode},
1087    };
1088
1089    fn make_regular_file(data: &[u8]) -> TreeNode {
1090        make_regular_file_with_id(data, RegularFileId::new())
1091    }
1092
1093    fn make_regular_file_with_id(data: &[u8], id: RegularFileId) -> TreeNode {
1094        TreeNode::RegularFile(RegularFileNode {
1095            id,
1096            metadata: InodeMetadata::default(),
1097            xattrs: Vec::new(),
1098            data: FileData::Memory(data.to_vec()),
1099            nlink: 1,
1100        })
1101    }
1102
1103    #[test]
1104    fn lookup_path_resolves_large_multi_block_directory() {
1105        let mut tree = FileTree::new();
1106        for i in 0..5000 {
1107            let path = format!("dir/file-{i:04}.txt");
1108            tree.insert(path.as_bytes(), make_regular_file(b"x"))
1109                .expect("insert file");
1110        }
1111
1112        let output_dir = tempdir().expect("tempdir");
1113        let output = output_dir.path().join("large-dir.erofs");
1114        write_erofs(&tree, &output).expect("write erofs");
1115
1116        let file = File::open(&output).expect("open erofs");
1117        let mut reader = ErofsReader::new(file).expect("reader");
1118
1119        assert_eq!(reader.read_file("/dir/file-0000.txt").expect("first"), b"x");
1120        assert_eq!(
1121            reader.read_file("/dir/file-2500.txt").expect("middle"),
1122            b"x"
1123        );
1124        assert_eq!(reader.read_file("/dir/file-4999.txt").expect("last"), b"x");
1125
1126        let err = reader
1127            .entry_info("/dir/file-9999.txt")
1128            .expect_err("missing entry should fail");
1129        assert_eq!(err.kind(), io::ErrorKind::NotFound);
1130    }
1131
1132    #[test]
1133    fn walk_uses_guest_separators_on_every_host() {
1134        let mut tree = FileTree::new();
1135        tree.insert(b"etc/passwd", make_regular_file(b"root:x:0:0"))
1136            .expect("insert nested file");
1137
1138        let output_dir = tempdir().expect("tempdir");
1139        let output = output_dir.path().join("nested.erofs");
1140        write_erofs(&tree, &output).expect("write erofs");
1141
1142        let file = File::open(&output).expect("open erofs");
1143        let mut reader = ErofsReader::new(file).expect("reader");
1144        let paths = reader
1145            .walk()
1146            .expect("walk erofs")
1147            .into_iter()
1148            .map(|entry| path_bytes(&entry.path).to_vec())
1149            .collect::<Vec<_>>();
1150
1151        assert!(paths.iter().any(|path| path == b"etc/passwd"));
1152        assert!(!paths.iter().any(|path| path == b"etc\\passwd"));
1153
1154        let mut byte_paths = Vec::new();
1155        reader
1156            .walk_entries_with_path_bytes::<io::Error, _>(|_, path, _| {
1157                byte_paths.push(path.to_vec());
1158                Ok(())
1159            })
1160            .expect("walk erofs with canonical bytes");
1161        assert!(byte_paths.iter().any(|path| path == b"etc/passwd"));
1162        assert!(!byte_paths.iter().any(|path| path == b"etc\\passwd"));
1163    }
1164
1165    #[test]
1166    fn hardlinked_regular_files_share_inode_and_data_blocks() {
1167        let mut tree = FileTree::new();
1168        let file_id = RegularFileId::new();
1169
1170        tree.insert(b"alpha", make_regular_file_with_id(b"shared", file_id))
1171            .expect("insert alpha");
1172        tree.insert(b"beta", make_regular_file_with_id(b"shared", file_id))
1173            .expect("insert beta");
1174
1175        let output_dir = tempdir().expect("tempdir");
1176        let output = output_dir.path().join("hardlinks.erofs");
1177        let data_map = write_erofs(&tree, &output).expect("write erofs");
1178        let alpha_path = PathBuf::from("alpha");
1179        let beta_path = PathBuf::from("beta");
1180
1181        assert_eq!(
1182            data_map
1183                .file_blocks
1184                .get(&alpha_path)
1185                .copied()
1186                .expect("alpha data map"),
1187            data_map
1188                .file_blocks
1189                .get(&beta_path)
1190                .copied()
1191                .expect("beta data map")
1192        );
1193
1194        let file = File::open(&output).expect("open erofs");
1195        let mut reader = ErofsReader::new(file).expect("reader");
1196        let alpha = reader.inode_debug_info("/alpha").expect("alpha inode");
1197        let beta = reader.inode_debug_info("/beta").expect("beta inode");
1198
1199        assert_eq!(alpha.nid, beta.nid);
1200        assert_eq!(alpha.nlink, 2);
1201        assert_eq!(beta.nlink, 2);
1202        assert_eq!(alpha.size, b"shared".len() as u64);
1203        assert_eq!(reader.read_file("/alpha").expect("read alpha"), b"shared");
1204        assert_eq!(reader.read_file("/beta").expect("read beta"), b"shared");
1205    }
1206}