1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
use std::collections::HashMap;
// Darwin's `nlink_t` is 16 bits, so this marker cannot collide with an actual link count.
#[cfg(target_os = "macos")]
const UNRESOLVED_DIRECTORY_LINKS: u64 = u64::MAX;
/// Tracks hard-linked inodes and, on macOS, fully shared APFS data streams.
#[derive(Debug, Default, Clone)]
pub(crate) struct InodeFilter {
inner: HashMap<(u64, u64), u64>,
#[cfg(target_os = "macos")]
apfs: apfs::ApfsFilter,
}
impl InodeFilter {
#[cfg(unix)]
/// Register file metadata and return `true` if this link should be counted.
pub(crate) fn add(
&mut self,
entry: &crate::walk::Entry,
metadata: &crate::walk::Metadata,
) -> bool {
#[cfg(not(target_os = "macos"))]
use std::os::unix::fs::MetadataExt;
#[cfg(target_os = "macos")]
if entry.file_type.is_dir() {
return self.add_directory(entry, metadata);
}
#[cfg(not(target_os = "macos"))]
let _ = entry;
self.add_dev_inode((metadata.dev(), metadata.ino()), metadata.nlink())
}
#[cfg(windows)]
/// Register file metadata and return `true` if this link should be counted.
pub(crate) fn add(
&mut self,
_entry: &crate::walk::Entry,
metadata: &crate::walk::Metadata,
) -> bool {
metadata
.hard_link_id()
.is_none_or(|id| self.inner.insert(id, 0).is_none())
}
#[cfg(not(any(unix, windows)))]
/// Register file metadata and return `true` if this link should be counted.
pub(crate) fn add(
&mut self,
_entry: &crate::walk::Entry,
metadata: &std::fs::Metadata,
) -> bool {
true
}
/// Register a directory whose bulk link count may differ from `stat(2)` semantics.
///
/// `ATTR_DIR_LINKCOUNT` omits the synthetic `.` and `..` links, so a bulk-enumerated directory
/// can report one link where `stat(2)` reports more. The first ambiguous observation is counted
/// and marked unresolved. Only a repeated `(device, inode)` pays for `symlink_metadata`; that
/// result initializes the ordinary link-count cycle without counting the first observation
/// twice.
///
/// Returns `true` when the current directory observation should contribute to size/count.
#[cfg(target_os = "macos")]
fn add_directory(
&mut self,
entry: &crate::walk::Entry,
metadata: &crate::walk::Metadata,
) -> bool {
let dev_inode = (metadata.dev(), metadata.ino());
let link_count = metadata.nlink();
match self.inner.get(&dev_inode).copied() {
None if link_count <= 1 => {
// ATTR_DIR_LINKCOUNT does not include the synthetic links reported by stat.
// Resolve the actual count only if an overlapping root revisits this directory.
self.inner.insert(dev_inode, UNRESOLVED_DIRECTORY_LINKS);
true
}
None => self.add_dev_inode(dev_inode, link_count),
Some(UNRESOLVED_DIRECTORY_LINKS) => {
let nlinks = if link_count > 1 {
link_count
} else {
use std::os::unix::fs::MetadataExt;
std::fs::symlink_metadata(entry.path())
.ok()
.filter(|actual| (actual.dev(), actual.ino()) == dev_inode)
.map_or(link_count, |actual| actual.nlink())
};
self.inner.remove(&dev_inode);
if nlinks <= 1 {
return true;
}
// The first observation already contributed its size. Consume the current
// observation through the ordinary counter to retain its reset behavior.
self.inner.insert(dev_inode, nlinks - 1);
self.add_dev_inode(dev_inode, nlinks)
}
Some(remaining) => self.add_dev_inode(dev_inode, remaining + 1),
}
}
/// Register a `(device, inode)` with its hard-link count.
///
/// Returns `true` for the first observation that should contribute to size/count,
/// and `false` for subsequent links.
#[cfg(any(unix, test))]
pub(crate) fn add_dev_inode(&mut self, dev_inode: (u64, u64), nlinks: u64) -> bool {
if nlinks <= 1 {
return true;
}
match self.inner.get_mut(&dev_inode) {
Some(1) => {
self.inner.remove(&dev_inode);
false
}
Some(count) => {
*count -= 1;
false
}
None => {
self.inner.insert(dev_inode, nlinks - 1);
true
}
}
}
}
#[cfg(target_os = "macos")]
mod apfs {
use std::collections::HashSet;
impl super::InodeFilter {
/// Return allocated bytes while counting a cloned data stream only once.
pub(crate) fn allocated_size(&mut self, metadata: &crate::walk::Metadata) -> u64 {
if self.apfs.add_clone(metadata) {
metadata.allocated_size()
} else {
// Clone identifiers cover only the data fork; resource forks remain private.
metadata
.allocated_size()
.saturating_sub(metadata.data_allocated_size())
}
}
}
#[derive(Debug, Default, Clone)]
pub(super) struct ApfsFilter {
streams: HashSet<(u64, u64)>,
inodes: HashSet<(u64, u64)>,
}
impl ApfsFilter {
fn add_clone(&mut self, metadata: &crate::walk::Metadata) -> bool {
if metadata.allocated_size() == 0 {
return true;
}
let Some(clone_id) = metadata.clone_id() else {
return true;
};
let device = metadata.dev();
// Hard links and overlapping roots revisit one inode, not distinct cloned files.
if !self.inodes.insert((device, metadata.ino())) {
return true;
}
self.streams.insert((device, clone_id.get()))
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn it_filters_inodes() {
let mut inodes = InodeFilter::default();
assert!(inodes.add_dev_inode((1, 1), 2));
assert!(inodes.add_dev_inode((2, 1), 2));
assert!(!inodes.add_dev_inode((1, 1), 2));
assert!(!inodes.add_dev_inode((2, 1), 2));
assert!(inodes.add_dev_inode((1, 1), 3));
assert!(!inodes.add_dev_inode((1, 1), 3));
assert!(!inodes.add_dev_inode((1, 1), 3));
assert!(inodes.add_dev_inode((1, 1), 1));
assert!(inodes.add_dev_inode((1, 1), 1));
}
}