Skip to main content

fdu_core/
scan.rs

1//! The scan layer: walking a tree, producing observations, and applying reconciliation.
2//!
3//! Public scans emit upsert observations, and a revalidation sweep is the diff between
4//! what the index believes and what the filesystem says. Both speak the same
5//! [`Observation`] vocabulary as the watch layer. A detached one-shot index may consume
6//! equivalent parent-first directory groups privately because no observer can see its
7//! construction; every later mutation still crosses the shared observation boundary.
8//!
9//! # Status
10//!
11//! The portable `read_dir` plus non-following metadata reference is what every listing
12//! falls back to. Parallel scans use it on most platforms; on macOS they first try a
13//! measured `getattrlistbulk` backend that returns directory entries and stat-tier
14//! metadata together. Unsupported filesystems, malformed results, mount points, and
15//! firmlinks fail closed to the portable path for the complete containing directory.
16//! On Linux with glibc every route that lists a directory (the serial and concurrent
17//! walks, revalidation, reconciliation, opened discovery) first tries a reader that
18//! lists with raw `getdents64` and stats with `statx` against the listing's descriptor,
19//! passing `AT_NO_AUTOMOUNT`; an open or enumeration failure, a malformed record, or a
20//! kernel without `statx` falls back the same way, and the fallback's own stats then
21//! pass the flag too (`observe_dir_entry`), as does every stat of a path a route
22//! verifies by itself (`observe_path`). Only the walk root is resolved through a mount
23//! (`root_device`). So an autofs tree answers the same on every route and delivery.
24//! Every backend produces the same [`Observation`] contract.
25
26use std::collections::{BTreeMap, BTreeSet, VecDeque};
27use std::ffi::{OsStr, OsString};
28use std::fmt::Write as _;
29use std::fs;
30use std::io::Read as _;
31use std::path::{Component, Path, PathBuf};
32
33use crate::ApplyStats;
34use crate::engine_contract::{
35    Attrs, Commit, EntryKind, Error, Observation, ObservationOp, Op, PathExpectation, PathState,
36    Result, ScanScope,
37};
38use crate::execution::TreeRetention;
39use crate::index::{
40    DetachedIndexBuilder, Index, IndexHandle, ReconcileErrors, ReconcileFinish,
41    collect_child_expectations,
42};
43use crate::query::ScopeAxis;
44use crate::stored_state::{ControlTierIdentity, EntryScope, EntryTierIdentity, SnapshotIdentity};
45
46// Keep the FFI exception at the platform boundary. The rest of the engine, including
47// every consumer of these observations, remains under the workspace's unsafe-code
48// denial.
49#[cfg(target_os = "macos")]
50#[allow(unsafe_code)]
51mod macos_bulk;
52
53// glibc builds only: `libc` defines `struct statx` for glibc, not for default musl.
54#[cfg(all(target_os = "linux", target_env = "gnu"))]
55#[allow(unsafe_code)]
56mod linux_dents;
57
58#[cfg(windows)]
59#[allow(unsafe_code)]
60mod windows_metadata;
61
62/// How many ops accumulate before an observation is handed to the sink.
63///
64/// Batching matters for more than syscall economy: consumers coalesce per path within a
65/// batch and stat once per batch, and a live UI wants partial results while a large tree
66/// is still being walked rather than one delta at the end.
67const DEFAULT_BATCH_SIZE: usize = crate::platform_tuning::tuning().batch_size.get();
68
69/// Largest producer batch accepted before work must be published incrementally.
70pub const MAX_SCAN_BATCH_SIZE: usize = 64 * 1024;
71
72/// Most changed paths an exclusive parallel reconciliation may defer before applying.
73///
74/// Workers compare against one immutable index image, so mutations wait until the wave
75/// joins. Bounding that change set keeps a churned tree from turning the fast unchanged
76/// path into an unbounded allocation; overflow discards the wave and retries through
77/// the incremental serial reconciler.
78const MAX_DEFERRED_RECONCILE_OPS: usize = MAX_SCAN_BATCH_SIZE;
79
80/// Directories compared against one immutable index baseline before changes are applied.
81///
82/// The wave is large enough to amortize scoped worker creation and small enough that a
83/// changed tree publishes progress throughout a long reconciliation.
84const RECONCILE_WAVE_DIRECTORIES: usize =
85    crate::platform_tuning::tuning().reconcile_wave_directories.get();
86
87/// Identity of the fixed stat-tier reducer set.
88const REDUCERS_FINGERPRINT: u64 = 1;
89
90/// The order directories are visited in.
91///
92/// This changes *when* observations are produced, never *which* ones: both orders
93/// visit every entry exactly once and leave an identical index behind. It therefore
94/// stays out of [`ScanScope`] and cannot invalidate a cache, exactly like the worker
95/// count.
96///
97/// The choice only matters to a consumer that reads the index while the walk is still
98/// running, and there it matters a great deal.
99///
100/// # Strength of the guarantee
101///
102/// **These are scheduling preferences, not strict orders, whenever more than one worker
103/// is running** — which is the default.
104///
105/// The queue is ordered, but the *claims* are not. Workers take directories from the
106/// shared queue in the policy's order; a worker that finishes early can enqueue its
107/// children and another worker can claim them while a slower worker still holds
108/// unfinished work from a shallower level. Nothing releases a level barrier, because
109/// a barrier would idle every fast worker at each level boundary and give back most of
110/// the parallel producer's win.
111///
112/// So:
113///
114/// - With `threads: Some(1)`, [`ScanOrder::BreadthFirst`] is strict: no directory is
115///   read before one closer to the root.
116/// - With several workers it is *shallow-first*: shallow work is always preferred when
117///   a worker chooses, and deeper observations can still interleave.
118///
119/// That weaker property is what the browser use case actually needs — every top-level
120/// subtree starts filling early, so a mid-scan ranking is meaningful — and it is the
121/// property the tests pin. A caller that needs strict level order must ask for one
122/// worker and pay for it.
123#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
124pub enum ScanOrder {
125    /// Shallow directories before deep ones.
126    ///
127    /// The default, because it is the order whose partial results mean something.
128    /// Roll-ups are maintained per directory as the walk proceeds, so a consumer that
129    /// looks mid-scan sees top-level totals grow together — bars fill, rankings
130    /// converge — instead of one subtree finishing while its siblings read zero.
131    /// Interrupting early leaves a usefully complete picture of the top of the tree.
132    ///
133    /// Under several workers this is a preference rather than a guarantee; see the
134    /// type-level note above.
135    ///
136    /// Note that totals only grow *while an additive walk is running*. Monotonicity
137    /// comes from the producer being additive, not from the order — the order decides
138    /// which subtrees get to grow early.
139    #[default]
140    BreadthFirst,
141    /// One subtree toward completion before starting the next.
142    ///
143    /// Lower peak memory, since the frontier is bounded by depth rather than by the
144    /// width of a level, and better locality within a subtree. The cost is that
145    /// partial results are actively misleading: one child of the root approaches its
146    /// final total while its siblings read zero, so anything ranking by size mid-scan
147    /// ranks confidently and wrongly. Correct for a caller that only reads the
148    /// finished index and wants the smallest footprint.
149    ///
150    /// Under several workers this too is a preference: several subtrees will be in
151    /// flight at once, one per worker.
152    DepthFirst,
153}
154
155/// Knobs for a scan.
156#[derive(Clone, Debug)]
157// Four booleans, each an independent admission or observation switch with its own
158// semantic-scope consequence, not an enum in disguise: any combination is legal and
159// means what its fields say. The lint suspects flag-soup states; this is a config
160// surface whose fields are documented one by one.
161#[allow(clippy::struct_excessive_bools)]
162pub struct ScanConfig {
163    /// Maximum relative entry depth to retain. Zero keeps only the index root and `None`
164    /// means unlimited.
165    pub max_depth: Option<usize>,
166    /// Ops per emitted observation. Must be between one and [`MAX_SCAN_BATCH_SIZE`].
167    pub batch_size: usize,
168    /// Follow symlinks to directories. Off by default: following them turns a tree walk
169    /// into a graph walk with cycles, and every surveyed tool defaults to off.
170    pub follow_symlinks: bool,
171    /// Stay on the filesystem the root lives on.
172    pub one_filesystem: bool,
173    /// Hidden-component admission, or `None` to retain every component.
174    pub hidden: Option<std::sync::Arc<crate::admission::HiddenPolicy>>,
175    /// Exclude filesystem objects other than files, directories, and symlinks.
176    pub exclude_special: bool,
177    /// Directory-reading worker threads.
178    ///
179    /// A tree walk is a pile of independent, latency-bound directory reads, so it
180    /// scales with threads far better than most work does. One means the serial
181    /// walker, which stays the reference implementation and the thing every result is
182    /// checked against, with two exceptions that take the concurrent walker with one
183    /// worker instead: the detached index build, and a transient summary that reads
184    /// `.gitignore`, which must deliver each directory's control ahead of its entries
185    /// as that build consumes them. [`None`] asks for a bounded default derived from the
186    /// machine's available parallelism. The automatic pool starts conservatively and
187    /// unlocks more latency-hiding workers only when initial chunk timing identifies a
188    /// slow filesystem path.
189    ///
190    /// This is an operational knob, not a semantic one: it changes how fast the same
191    /// observations are produced, never which observations they are. That is why it
192    /// stays out of [`ScanScope`] and cannot invalidate a cache.
193    pub threads: Option<usize>,
194    /// The order directories are visited in. See [`ScanOrder`].
195    pub order: ScanOrder,
196    /// File-type rules to classify against, or `None` for the ones compiled into fdu.
197    ///
198    /// Unlike [`Self::threads`] this *is* semantic: a different taxonomy classifies the
199    /// same tree differently, which is why its fingerprint rides in [`ScanScope`] and a
200    /// change to it invalidates a snapshot. Shared rather than owned because a scan
201    /// clones its config per wave and a registry is read-only once built.
202    pub types: Option<std::sync::Arc<crate::classify::TypeRegistry>>,
203    /// Observe `.gitignore` control files and retain ignore classification.
204    ///
205    /// On by default on every surface: an [`Index`] from [`crate::open`] or a scan keeps
206    /// the exact control state it exposes and a watch maintains -- which entries are
207    /// ignored, and the ignored and unignored partitions of every roll-up -- and a one-shot
208    /// report from [`crate::prepare_report`] shows the ignored share of every row
209    /// (fdu-elnn). It costs a read of every `.gitignore` in the tree. A file past the
210    /// [`Self::control_limits`] is refused and named in [`Index::control_coverage`] rather
211    /// than ending the scan; a file that cannot be read is an error at its path, which
212    /// makes the result partial.
213    ///
214    /// Off, the scan performs no control-file I/O and retains no control table, and that is
215    /// stamped into [`ScanScope`], so an index-returning call never serves a snapshot taken
216    /// one way as the other. An [`Index`] built that way answers [`Index::is_ignored`],
217    /// [`Index::controls`], and the partition accessors with
218    /// [`crate::Error::ControlStateNotObserved`], never with "not ignored", and refuses
219    /// control input; a report's rows carry no ignored share, and a selection by ignored
220    /// state is refused. The command line spells it `--no-gitignore`.
221    ///
222    /// An opened root ([`crate::OpenedIndex`]) always observes control state, because its
223    /// ignored and unignored partitions are part of what it serves.
224    pub read_controls: bool,
225    /// Which ignored population shapes retained entries and content candidates.
226    pub population: crate::query::IgnoredEntries,
227    /// The budget and the line limit `.gitignore` files are applied under, each a size or
228    /// unbounded. See [`crate::control::ControlLimits`].
229    ///
230    /// A source that would take the table past the budget, or that has a line longer than
231    /// the line limit, is refused: its rules do not apply, the scan continues with every
232    /// size exact, and [`Index::control_coverage`] names it and the limit that fired. The
233    /// command line spells these `--gitignore-budget SIZE|all` and
234    /// `--gitignore-line-limit SIZE|all`; the Python API spells them `control_budget` and
235    /// `control_line_limit`.
236    ///
237    /// Semantic, like [`Self::read_controls`]: the limits decide which rules apply, so both
238    /// are part of [`ScanScope`] and a snapshot taken under other limits is not reused.
239    /// Ignored when control state is not observed.
240    pub control_limits: crate::control::ControlLimits,
241    /// Where to report how much of the walk has been done, or `None` to report nothing.
242    ///
243    /// An observer rather than a knob: it changes neither which observations a walk
244    /// produces nor how it produces them, so it is no part of [`ScanScope`] or of any
245    /// snapshot identity, and two configs that differ only here are the same scan.
246    /// Honoured by every walker in this module -- the cold scans, the summary fold,
247    /// [`revalidate`], and each `reconcile` entry point -- which enter
248    /// [`ProgressPhase::Scanning`](crate::ProgressPhase) or
249    /// [`ProgressPhase::Revalidating`](crate::ProgressPhase) and add their counts once
250    /// per chunk of directories, never per entry. See [`crate::Progress`] for what the
251    /// counts mean and what holds when a walk returns.
252    pub progress: Option<crate::Progress>,
253}
254
255impl Default for ScanConfig {
256    fn default() -> Self {
257        Self {
258            max_depth: None,
259            batch_size: DEFAULT_BATCH_SIZE,
260            follow_symlinks: false,
261            one_filesystem: false,
262            hidden: None,
263            exclude_special: false,
264            threads: None,
265            order: ScanOrder::default(),
266            types: None,
267            read_controls: crate::query::Request::DEFAULTS.read_controls,
268            population: crate::query::IgnoredEntries::Include,
269            control_limits: crate::query::Request::DEFAULTS.control_limits,
270            progress: None,
271        }
272    }
273}
274
275/// Why watching cannot narrow its structural scan boundary, said once for every surface.
276/// Ignored population is a supported retained-scope choice because control edits
277/// reconcile the governing directory and rebuild that population.
278///
279/// The CLI used to carry this guidance and the library carried "requires event-scope
280/// filtering", which names the implementation rather than the caller's next move -- so a
281/// library caller hitting the same wall got jargon and the CLI user got help. Two
282/// messages for one rule also drift, and the parity harness could not tell they were the
283/// same rule.
284///
285/// The knobs are named by the calling surface: `--scan-depth` on the command line,
286/// `max_depth` through the API. Everything else is identical, so the harness can verify
287/// mechanically that both surfaces state the same rule.
288pub const WATCH_SCOPE_GUIDANCE: &str = concat!(
289    "watching requires full scope and cannot be combined with max_depth or one_filesystem: ",
290    "a watcher cannot filter backend events against a narrowed boundary. Selection such as ",
291    "depth, include, and modified_since does work while watching, because it filters the ",
292    "retained index rather than narrowing the scan"
293);
294
295impl ScanConfig {
296    /// Classify this scan with `types` and include their derived identity in its scope.
297    #[must_use]
298    pub fn with_types(mut self, types: std::sync::Arc<crate::classify::TypeRegistry>) -> Self {
299        self.types = Some(types);
300        self
301    }
302
303    /// The file-type rules in effect: the supplied registry, or the compiled default.
304    pub fn types(&self) -> &crate::classify::TypeRegistry {
305        match &self.types {
306            Some(types) => types,
307            None => crate::classify::TypeRegistry::compiled(),
308        }
309    }
310
311    /// Share the file-type rules with an index that retains them.
312    pub(crate) fn types_shared(&self) -> std::sync::Arc<crate::classify::TypeRegistry> {
313        self.types
314            .as_ref()
315            .map_or_else(crate::classify::TypeRegistry::compiled_shared, std::sync::Arc::clone)
316    }
317
318    /// Hidden-component policy in effect.
319    pub fn hidden(&self) -> &crate::admission::HiddenPolicy {
320        self.hidden.as_deref().unwrap_or_else(|| crate::admission::HiddenPolicy::keep_all())
321    }
322
323    /// Semantic cache identity, excluding operational batching choices.
324    ///
325    /// Composed from [`Self::snapshot_identity`], so the scope an index records and the
326    /// tier identities a snapshot of it carries are one value in two shapes.
327    ///
328    /// No longer `const`: the type-rule fingerprint is now a property of the registry in
329    /// effect rather than a compiled-in constant, which is the whole point of letting a
330    /// caller supply one. A snapshot taken under different rules must not be reused.
331    pub fn scope(&self) -> ScanScope {
332        self.snapshot_identity().scan_scope()
333    }
334
335    /// Which entries this scan retains, the part of its scope no `.gitignore` setting
336    /// changes.
337    pub fn entry_scope(&self) -> EntryScope {
338        EntryScope {
339            max_depth: self.max_depth,
340            follow_symlinks: self.follow_symlinks,
341            one_filesystem: self.one_filesystem,
342            hidden_fingerprint: self.hidden().fingerprint(),
343            exclude_special: self.exclude_special,
344            population: self.population,
345            control_fingerprint: if self.population == crate::query::IgnoredEntries::Include {
346                0
347            } else {
348                self.control_identity().ignore_rules_fingerprint()
349            },
350        }
351    }
352
353    /// Whether this scan observes `.gitignore` control state, and under which limits.
354    ///
355    /// The limits are part of the identity only when control state is observed: a scan
356    /// that reads no control file applies none, whatever [`Self::control_limits`] says.
357    pub fn control_identity(&self) -> ControlTierIdentity {
358        if self.read_controls {
359            ControlTierIdentity::Observed { limits: self.control_limits }
360        } else {
361            ControlTierIdentity::NotObserved
362        }
363    }
364
365    /// The identity of every tier a snapshot of this scan holds.
366    pub fn snapshot_identity(&self) -> SnapshotIdentity {
367        SnapshotIdentity {
368            entries: EntryTierIdentity {
369                engine: crate::snapshot::engine_fingerprint(),
370                scope: self.entry_scope(),
371                type_rules_fingerprint: self.types().fingerprint(),
372                reducers_fingerprint: REDUCERS_FINGERPRINT,
373            },
374            controls: self.control_identity(),
375        }
376    }
377
378    /// Resolve [`Self::threads`] to the workers active when a scan begins.
379    #[cfg(any(target_os = "macos", test))]
380    fn worker_threads(&self) -> usize {
381        self.worker_pool().initial
382    }
383
384    /// Resolve the worker count for immutable-baseline reconciliation waves.
385    fn reconciliation_worker_threads(&self) -> usize {
386        match self.threads {
387            Some(threads) => threads.clamp(1, MAX_SCAN_THREADS),
388            None => std::thread::available_parallelism()
389                .map_or(1, std::num::NonZero::get)
390                .clamp(1, DEFAULT_RECONCILE_THREADS_CAP),
391        }
392    }
393
394    /// Resolve the initial and maximum worker counts for one scan.
395    #[cfg(any(target_os = "macos", test))]
396    fn worker_pool(&self) -> WorkerPool {
397        self.worker_pool_for(std::thread::available_parallelism().map_or(1, std::num::NonZero::get))
398    }
399
400    /// Resolve the worker pool from one captured operating-system parallelism value.
401    fn worker_pool_for(&self, available_parallelism: usize) -> WorkerPool {
402        match self.threads {
403            Some(threads) => WorkerPool::fixed(threads.clamp(1, MAX_SCAN_THREADS)),
404            None => automatic_worker_pool(available_parallelism),
405        }
406    }
407
408    /// The scope axis this build cannot honour, if any.
409    ///
410    /// The one statement of the capability rule, so it is asked rather than restated.
411    /// [`Request::validate`](crate::query::Request::validate) asks it before any stored
412    /// state is read, which is what makes a scope this build cannot honour refuse the same
413    /// way on every route, every cache policy, and both surfaces; [`Self::validate`] asks
414    /// it for the engine-internal callers -- a bound root, a raw scan, an observation --
415    /// that never carry a request.
416    pub(crate) const fn unsupported_axis(&self) -> Option<ScopeAxis> {
417        if self.follow_symlinks {
418            return Some(ScopeAxis::FollowSymlinks);
419        }
420        #[cfg(not(unix))]
421        if self.one_filesystem {
422            return Some(ScopeAxis::OneFilesystem);
423        }
424        None
425    }
426
427    pub(crate) fn validate(&self) -> Result<()> {
428        if !self.read_controls && self.population != crate::query::IgnoredEntries::Include {
429            return Err(Error::UnsupportedScanConfig(
430                "ignored population requires .gitignore observation",
431            ));
432        }
433        if self.batch_size == 0 || self.batch_size > MAX_SCAN_BATCH_SIZE {
434            return Err(Error::UnsupportedScanConfig(
435                "batch_size must be nonzero and no greater than MAX_SCAN_BATCH_SIZE",
436            ));
437        }
438        if let Some(axis) = self.unsupported_axis() {
439            return Err(Error::UnsupportedScanConfig(axis.reason()));
440        }
441        Ok(())
442    }
443
444    pub(crate) fn validate_for_scope(&self, indexed: ScanScope) -> Result<()> {
445        self.validate()?;
446        let requested = self.scope();
447        if indexed != requested {
448            return Err(Error::ScanScopeMismatch { indexed, requested });
449        }
450        Ok(())
451    }
452
453    /// Scope equality, plus the boundary a watcher cannot filter its backend's events
454    /// against.
455    ///
456    /// The rule belongs to the request model, which refuses a watch of a narrowed scope
457    /// before anything is opened ([`RequestError::WatchScope`](crate::query::RequestError));
458    /// this is the same rule where a watcher is bound without a request -- an opened root
459    /// that observes, and each batch the adapter applies -- and it renders the one
460    /// guidance string the model renders.
461    #[cfg(feature = "watch")]
462    pub(crate) fn validate_for_watch_scope(&self, indexed: ScanScope) -> Result<()> {
463        self.validate_for_scope(indexed)?;
464        if self.max_depth.is_some() || self.one_filesystem {
465            return Err(Error::UnsupportedScanConfig(WATCH_SCOPE_GUIDANCE));
466        }
467        Ok(())
468    }
469}
470
471impl Default for ScanScope {
472    fn default() -> Self {
473        ScanConfig::default().scope()
474    }
475}
476
477/// What a scan did, including the errors it walked past.
478///
479/// Unreadable directories are skipped rather than aborting the scan — a permission-denied
480/// subdirectory should not cost you the other 499,000 files — but they are reported
481/// rather than swallowed, so a caller can tell a complete answer from a partial one.
482#[derive(Debug, Default)]
483pub struct ScanReport {
484    /// Directories successfully listed.
485    pub dirs_read: u64,
486    /// Entries observed, directories included.
487    pub entries: u64,
488    /// Regular files whose metadata was observed.
489    pub files_walked: u64,
490    /// Apparent bytes represented by the regular files whose metadata was observed,
491    /// saturating at `u64::MAX`.
492    ///
493    /// A measure of the walk's work, not an answer: a tree whose total no `u64` can hold
494    /// is refused where it is counted ([`crate::Error::UnrepresentableTotal`]), and this
495    /// tally stops at the bound rather than wrapping or panicking before that refusal.
496    pub bytes_walked: u64,
497    /// Allocated bytes of those files: what the default size metric counts, and what a
498    /// sparse disk image or a clone makes far smaller than their apparent bytes. It
499    /// saturates as `bytes_walked` does.
500    pub allocated_walked: u64,
501    /// Paths that could not be read, with the reason.
502    pub errors: Vec<Error>,
503    /// Where the walk's time went, summed across workers.
504    pub attribution: WalkAttribution,
505}
506
507impl ScanReport {
508    /// True when every directory in scope was read successfully.
509    pub fn is_complete(&self) -> bool {
510        self.errors.is_empty()
511    }
512
513    /// Fold one worker's share of a parallel walk into the whole-walk report.
514    fn absorb(&mut self, other: Self) {
515        self.dirs_read += other.dirs_read;
516        self.entries += other.entries;
517        self.files_walked += other.files_walked;
518        self.bytes_walked = self.bytes_walked.saturating_add(other.bytes_walked);
519        self.allocated_walked = self.allocated_walked.saturating_add(other.allocated_walked);
520        self.errors.extend(other.errors);
521        self.attribution.absorb(other.attribution);
522    }
523
524    /// Record one successfully stated directory entry.
525    fn observe(&mut self, kind: EntryKind, attrs: Attrs) {
526        self.entries += 1;
527        if kind == EntryKind::File {
528            self.files_walked += 1;
529            self.bytes_walked = self.bytes_walked.saturating_add(attrs.size);
530            self.allocated_walked = self.allocated_walked.saturating_add(attrs.allocated);
531        }
532    }
533}
534
535/// One walker's running share of the progress counters.
536///
537/// Each walker keeps its own [`ScanReport`]; this remembers how much of that report it
538/// has already added to the shared [`crate::Progress`] cells, so each addition is the
539/// difference since the last. It lives on the worker's stack beside the report rather
540/// than inside it, so a report absorbed into another never carries a stale baseline.
541struct ProgressTally<'a> {
542    progress: Option<&'a crate::Progress>,
543    directories: u64,
544    files: u64,
545    bytes: u64,
546    allocated: u64,
547}
548
549impl<'a> ProgressTally<'a> {
550    const fn new(progress: Option<&'a crate::Progress>) -> Self {
551        Self { progress, directories: 0, files: 0, bytes: 0, allocated: 0 }
552    }
553
554    /// Add what `report` has counted since the last call.
555    ///
556    /// The one `Option` check is the whole cost when no handle is attached. Called once
557    /// per chunk of directories a walker hands over, never per entry.
558    fn flush(&mut self, report: &ScanReport) {
559        let Some(progress) = self.progress else { return };
560        let directories = report.dirs_read - self.directories;
561        let files = report.files_walked - self.files;
562        let bytes = report.bytes_walked - self.bytes;
563        let allocated = report.allocated_walked - self.allocated;
564        if directories != 0 || files != 0 || bytes != 0 || allocated != 0 {
565            progress.add_walked(directories, files, bytes, allocated);
566            self.directories = report.dirs_read;
567            self.files = report.files_walked;
568            self.bytes = report.bytes_walked;
569            self.allocated = report.allocated_walked;
570        }
571    }
572
573    /// Treat everything `report` holds as already added.
574    ///
575    /// For a walker that continues a report whose counts other workers added themselves.
576    fn skip_to(&mut self, report: &ScanReport) {
577        self.directories = report.dirs_read;
578        self.files = report.files_walked;
579        self.bytes = report.bytes_walked;
580        self.allocated = report.allocated_walked;
581    }
582}
583
584/// Normalize filesystem failures before one of the bounded status collectors retains them.
585///
586/// A walk may encounter the same inaccessible path from several worker paths. The report is
587/// already the full, transient set for this pass, so sorting and deduplicating it here avoids
588/// allocating or formatting a second unbounded set solely to decide which 64 details survive.
589/// I/O causes are keyed by their native root-relative path and the issue category, exactly the
590/// cause identity retained by an index. Other engine failures are left distinct: walker errors
591/// are I/O failures, and treating arbitrary engine errors as equivalent without constructing
592/// their bounded issue representation would lose information.
593pub(crate) fn normalize_walk_errors(root: &Path, errors: &mut Vec<Error>) {
594    errors.sort_by(|left, right| match (left, right) {
595        (
596            Error::Io { path: left_path, source: left_source },
597            Error::Io { path: right_path, source: right_source },
598        ) => left_path
599            .strip_prefix(root)
600            .unwrap_or(left_path)
601            .cmp(right_path.strip_prefix(root).unwrap_or(right_path))
602            .then_with(|| {
603                walk_issue_kind_rank(left_source).cmp(&walk_issue_kind_rank(right_source))
604            }),
605        (Error::Io { .. }, _) => std::cmp::Ordering::Less,
606        (_, Error::Io { .. }) => std::cmp::Ordering::Greater,
607        _ => std::cmp::Ordering::Equal,
608    });
609    errors.dedup_by(|right, left| match (left, right) {
610        (
611            Error::Io { path: left_path, source: left_source },
612            Error::Io { path: right_path, source: right_source },
613        ) => {
614            left_path.strip_prefix(root).unwrap_or(left_path)
615                == right_path.strip_prefix(root).unwrap_or(right_path)
616                && walk_issue_kind_rank(left_source) == walk_issue_kind_rank(right_source)
617        }
618        _ => false,
619    });
620}
621
622fn walk_issue_kind_rank(error: &std::io::Error) -> u8 {
623    match error.kind() {
624        std::io::ErrorKind::PermissionDenied => 0,
625        std::io::ErrorKind::NotFound => 1,
626        std::io::ErrorKind::InvalidData | std::io::ErrorKind::InvalidInput => 2,
627        _ => 5,
628    }
629}
630
631/// Schema carried by [`ScanDiagnostics`].
632///
633/// Diagnostics are an opt-in measurement contract rather than stable human output.
634/// Consumers must reject an unknown schema instead of guessing that fields retained
635/// their meaning.
636pub const SCAN_DIAGNOSTICS_SCHEMA: &str = "fdu-scan-diagnostics-v1";
637
638/// Maximum policy-window records retained by one diagnostic scan.
639///
640/// The bound is on controller evaluations, not filesystem entries. A controller that
641/// needs more history must mark the artifact truncated; claim-grade consumers reject
642/// that artifact rather than silently analyzing an incomplete policy history.
643const MAX_POLICY_TRACE_EVENTS: usize = 256;
644
645/// Opt-in, run-scoped evidence about a filesystem scan.
646///
647/// Obtain this through [`scan_with_diagnostics`] or
648/// [`scan_into_index_with_diagnostics`]. Keeping it out of [`ScanReport`] preserves the
649/// existing scan API and keeps ordinary callers off the measurement path entirely.
650#[derive(Clone, Debug, PartialEq, Eq)]
651pub struct ScanDiagnostics {
652    /// Version of this diagnostic contract.
653    pub schema: &'static str,
654    /// Automatic worker-controller history and queue state.
655    pub worker_policy: WorkerPolicyDiagnostics,
656    /// Directory-enumeration backends used by this run.
657    pub backend: ScanBackendDiagnostics,
658}
659
660/// Repository-only controller variants used by the performance evidence probe.
661///
662/// These variants are not selected by [`scan`] or [`scan_with_diagnostics`]; both keep
663/// the shipped one-shot policy. The explicit experimental APIs make candidate behavior
664/// measurable without hiding a production change behind an environment variable.
665#[doc(hidden)]
666#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
667pub enum WorkerPolicyExperiment {
668    /// The production controller: one prefix window and at most one expansion.
669    #[default]
670    ShippedOneShot,
671    /// Re-evaluate independent windows until a slow phase requests the full reserve.
672    RepeatedWindows,
673    /// Re-evaluate independent windows, gate on useful frontier/handoff backlog, and grow
674    /// the pool in stages.
675    StagedGatedWindows,
676}
677
678/// Final state of the automatic worker controller.
679#[derive(Clone, Copy, Debug, PartialEq, Eq)]
680pub enum WorkerPolicyOutcome {
681    /// The scan did no walking, for example because `max_depth` was zero.
682    NotRun,
683    /// A fixed pool had no adaptive decision to make.
684    Fixed,
685    /// The walk ended before an adaptive window became observable.
686    Undecided,
687    /// The controller measured a window and retained the initial pool.
688    Held,
689    /// The controller requested and activated the reserve workers.
690    ScaledUp,
691    /// Slow work was observed only after no useful queued or in-flight work remained.
692    HeldNoUsefulWork,
693}
694
695/// One controller evaluation over a half-open range of completed entry ordinals.
696#[derive(Clone, Debug, PartialEq, Eq)]
697pub struct WorkerPolicyWindow {
698    /// Monotonic record number within this scan.
699    pub sequence: u64,
700    /// First completed-entry ordinal represented by this window, inclusive.
701    pub start_entry_ordinal: u64,
702    /// Ordinal immediately after the last represented entry.
703    pub end_entry_ordinal: u64,
704    /// Entries contributing to the service-time signal.
705    pub observed_entries: u64,
706    /// Completed directory claims contributing to the service-time signal.
707    pub observed_chunks: u64,
708    /// Worker time contributing to the service-time signal.
709    pub observed_work_ns: u64,
710    /// Derived service time, or null when no entry made the signal observable.
711    pub work_ns_per_entry: Option<u64>,
712    /// Why `work_ns_per_entry` is null.
713    pub work_ns_per_entry_unavailable_reason: Option<&'static str>,
714    /// Directories ready to claim when the controller evaluated the window.
715    pub ready_directories: usize,
716    /// Claimed directories still being processed at that point.
717    pub in_flight_directories: usize,
718    /// Live worker threads at that point, including workers waiting for a claim.
719    pub active_workers: usize,
720    /// Observation batches sent but not yet received by the consumer.
721    pub handoff_backlog: usize,
722    /// Worker target requested by a scale decision.
723    pub requested_workers: Option<usize>,
724    /// What the controller concluded from this window.
725    pub decision: WorkerPolicyDecision,
726}
727
728/// Decision represented by a [`WorkerPolicyWindow`].
729#[derive(Clone, Copy, Debug, PartialEq, Eq)]
730pub enum WorkerPolicyDecision {
731    /// The walk ended before the window could support a decision.
732    Undecided,
733    /// The observed window retained the current pool.
734    Hold,
735    /// The observed window activated reserve workers.
736    ScaleUp,
737    /// The trigger fired after all useful work had drained.
738    HoldNoUsefulWork,
739    /// A complete window held because reserve workers had no useful frontier to claim.
740    HoldInsufficientFrontier,
741    /// A complete window held because the unbounded handoff backlog was already high.
742    HoldHandoffBacklog,
743    /// A post-decision observation window remained below the slow threshold.
744    ObserveFast,
745    /// A post-decision observation window met the slow threshold.
746    ObserveSlow,
747    /// A trailing partial window carried no new terminal decision.
748    Incomplete,
749    /// A trailing partial post-decision observation carried no policy decision.
750    ObserveIncomplete,
751}
752
753/// Worker-controller configuration, trace, and terminal queue state.
754#[derive(Clone, Debug, PartialEq, Eq)]
755pub struct WorkerPolicyDiagnostics {
756    /// Controller variant exercised by this scan.
757    pub controller: &'static str,
758    /// Parallelism reported by the operating system when the scan began.
759    pub available_parallelism: usize,
760    /// Workers in the pool before any adaptive decision.
761    pub initial_workers: usize,
762    /// Hard maximum workers this scan could activate.
763    pub maximum_workers: usize,
764    /// Entry target for an adaptive window, or null for a fixed pool.
765    pub calibration_window_entries: Option<u64>,
766    /// Slow-service trigger, or null for a fixed pool.
767    pub slow_threshold_ns_per_entry: Option<u64>,
768    /// Directory chunks folded into live controller windows.
769    pub calibration_chunks: u64,
770    /// Entries folded into live controller windows.
771    pub calibration_entries: u64,
772    /// Worker time folded into live controller windows.
773    pub calibration_work_ns: u64,
774    /// Expansion messages that caused the consumer to create more workers.
775    pub worker_expansions: u64,
776    /// Terminal policy outcome.
777    pub outcome: WorkerPolicyOutcome,
778    /// Explanation when no adaptive evaluation exists.
779    pub outcome_reason: Option<&'static str>,
780    /// Total worker threads created during the walk.
781    pub workers_spawned: usize,
782    /// Maximum simultaneously live worker threads, including workers waiting for work.
783    pub peak_active_workers: usize,
784    /// Ready directories at scan completion; a complete walk must leave zero.
785    pub ready_directories_at_finish: usize,
786    /// In-flight directories at scan completion; a complete walk must leave zero.
787    pub in_flight_directories_at_finish: usize,
788    /// Observation batches outstanding at scan completion.
789    pub handoff_backlog_at_finish: usize,
790    /// Maximum outstanding observation batches during the scan.
791    pub handoff_backlog_high_water: usize,
792    /// Bounded controller history.
793    pub windows: Vec<WorkerPolicyWindow>,
794    /// True when controller history exceeded the 256-event diagnostic bound.
795    pub events_truncated: bool,
796}
797
798/// Directory enumeration backends used by one scan.
799///
800/// Each native reader's listings are counted in its own fields, and a directory it
801/// declines is counted as its fallback and then as a portable attempt. On Linux, for the
802/// native reader (`getdents64` and `statx`, glibc builds):
803///
804/// - `linux_dents_attempts` = `linux_dents_successes` + `linux_dents_fallbacks`;
805/// - the report's `dirs_read` = `linux_dents_successes` + `portable_directory_reads`.
806///
807/// `unavailable_reason` describes only the macOS fields.
808///
809/// Non-exhaustive, as [`crate::counters::Counts`] is: a diagnostics record grows with the
810/// backends it describes, so a later field is an additive change. Code outside the engine
811/// reads its fields; only the engine builds one.
812#[derive(Clone, Debug, PartialEq, Eq)]
813#[non_exhaustive]
814pub struct ScanBackendDiagnostics {
815    /// Portable `read_dir` calls attempted.
816    pub portable_attempts: u64,
817    /// Portable directory listings completed successfully.
818    pub portable_directory_reads: u64,
819    /// macOS bulk enumeration attempts, or null off macOS.
820    pub macos_bulk_attempts: Option<u64>,
821    /// Successful macOS bulk listings, or null off macOS.
822    pub macos_bulk_successes: Option<u64>,
823    /// Bulk attempts that fell back to portable enumeration, or null off macOS.
824    pub macos_bulk_fallbacks: Option<u64>,
825    /// Why macOS fields are null.
826    pub unavailable_reason: Option<&'static str>,
827    /// Linux native `getdents64` listing attempts, or null off Linux (and on a Linux
828    /// build without glibc, where the native reader is not compiled).
829    pub linux_dents_attempts: Option<u64>,
830    /// Successful Linux native listings, or null off Linux.
831    pub linux_dents_successes: Option<u64>,
832    /// Linux native attempts that fell back to portable enumeration, or null off Linux.
833    pub linux_dents_fallbacks: Option<u64>,
834}
835
836impl ScanDiagnostics {
837    /// Serialize this versioned diagnostic contract as compact JSON.
838    ///
839    /// This deliberately lives beside the contract instead of in a benchmark binary:
840    /// claim-grade installed-command measurements and the repository probe must emit
841    /// byte-for-byte equivalent evidence without adding a serialization dependency to
842    /// the core crate.
843    pub fn to_json(&self) -> String {
844        let policy = &self.worker_policy;
845        let backend = &self.backend;
846        let mut windows = String::from("[");
847        for (index, window) in policy.windows.iter().enumerate() {
848            if index > 0 {
849                windows.push(',');
850            }
851            let _ = write!(
852                windows,
853                concat!(
854                    "{{\"active_workers\":{},\"decision\":\"{}\",",
855                    "\"end_entry_ordinal\":{},\"handoff_backlog\":{},",
856                    "\"in_flight_directories\":{},\"observed_chunks\":{},",
857                    "\"observed_entries\":{},",
858                    "\"observed_work_ns\":{},\"ready_directories\":{},",
859                    "\"requested_workers\":{},\"sequence\":{},\"start_entry_ordinal\":{},",
860                    "\"work_ns_per_entry\":{},",
861                    "\"work_ns_per_entry_unavailable_reason\":{}}}"
862                ),
863                window.active_workers,
864                worker_policy_decision_name(window.decision),
865                window.end_entry_ordinal,
866                window.handoff_backlog,
867                window.in_flight_directories,
868                window.observed_chunks,
869                window.observed_entries,
870                window.observed_work_ns,
871                window.ready_directories,
872                json_optional_usize(window.requested_workers),
873                window.sequence,
874                window.start_entry_ordinal,
875                json_optional_u64(window.work_ns_per_entry),
876                json_optional_string(window.work_ns_per_entry_unavailable_reason),
877            );
878        }
879        windows.push(']');
880        format!(
881            concat!(
882                "{{\"backend\":{{\"linux_dents_attempts\":{},",
883                "\"linux_dents_fallbacks\":{},\"linux_dents_successes\":{},",
884                "\"macos_bulk_attempts\":{},",
885                "\"macos_bulk_fallbacks\":{},\"macos_bulk_successes\":{},",
886                "\"portable_attempts\":{},\"portable_directory_reads\":{},",
887                "\"unavailable_reason\":{}}},",
888                "\"schema\":\"{}\",\"worker_policy\":{{",
889                "\"available_parallelism\":{},\"calibration_chunks\":{},",
890                "\"calibration_entries\":{},\"calibration_window_entries\":{},",
891                "\"calibration_work_ns\":{},",
892                "\"controller\":\"{}\",",
893                "\"events_truncated\":{},\"handoff_backlog_at_finish\":{},",
894                "\"handoff_backlog_high_water\":{},\"in_flight_directories_at_finish\":{},",
895                "\"initial_workers\":{},\"maximum_workers\":{},\"outcome\":\"{}\",",
896                "\"outcome_reason\":{},\"peak_active_workers\":{},",
897                "\"ready_directories_at_finish\":{},\"slow_threshold_ns_per_entry\":{},",
898                "\"windows\":{},\"worker_expansions\":{},\"workers_spawned\":{}}}}}"
899            ),
900            json_optional_u64(backend.linux_dents_attempts),
901            json_optional_u64(backend.linux_dents_fallbacks),
902            json_optional_u64(backend.linux_dents_successes),
903            json_optional_u64(backend.macos_bulk_attempts),
904            json_optional_u64(backend.macos_bulk_fallbacks),
905            json_optional_u64(backend.macos_bulk_successes),
906            backend.portable_attempts,
907            backend.portable_directory_reads,
908            json_optional_string(backend.unavailable_reason),
909            self.schema,
910            policy.available_parallelism,
911            policy.calibration_chunks,
912            policy.calibration_entries,
913            json_optional_u64(policy.calibration_window_entries),
914            policy.calibration_work_ns,
915            policy.controller,
916            policy.events_truncated,
917            policy.handoff_backlog_at_finish,
918            policy.handoff_backlog_high_water,
919            policy.in_flight_directories_at_finish,
920            policy.initial_workers,
921            policy.maximum_workers,
922            worker_policy_outcome_name(policy.outcome),
923            json_optional_string(policy.outcome_reason),
924            policy.peak_active_workers,
925            policy.ready_directories_at_finish,
926            json_optional_u64(policy.slow_threshold_ns_per_entry),
927            windows,
928            policy.worker_expansions,
929            policy.workers_spawned,
930        )
931    }
932}
933
934const fn worker_policy_outcome_name(value: WorkerPolicyOutcome) -> &'static str {
935    match value {
936        WorkerPolicyOutcome::NotRun => "not_run",
937        WorkerPolicyOutcome::Fixed => "fixed",
938        WorkerPolicyOutcome::Undecided => "undecided",
939        WorkerPolicyOutcome::Held => "held",
940        WorkerPolicyOutcome::ScaledUp => "scaled_up",
941        WorkerPolicyOutcome::HeldNoUsefulWork => "held_no_useful_work",
942    }
943}
944
945const fn worker_policy_decision_name(value: WorkerPolicyDecision) -> &'static str {
946    match value {
947        WorkerPolicyDecision::Undecided => "undecided",
948        WorkerPolicyDecision::Hold => "hold",
949        WorkerPolicyDecision::ScaleUp => "scale_up",
950        WorkerPolicyDecision::HoldNoUsefulWork => "hold_no_useful_work",
951        WorkerPolicyDecision::HoldInsufficientFrontier => "hold_insufficient_frontier",
952        WorkerPolicyDecision::HoldHandoffBacklog => "hold_handoff_backlog",
953        WorkerPolicyDecision::ObserveFast => "observe_fast",
954        WorkerPolicyDecision::ObserveSlow => "observe_slow",
955        WorkerPolicyDecision::Incomplete => "incomplete",
956        WorkerPolicyDecision::ObserveIncomplete => "observe_incomplete",
957    }
958}
959
960fn json_optional_string(value: Option<&str>) -> String {
961    value.map_or_else(|| "null".into(), |value| format!("\"{}\"", json_escape(value)))
962}
963
964fn json_optional_u64(value: Option<u64>) -> String {
965    value.map_or_else(|| "null".into(), |value| value.to_string())
966}
967
968fn json_optional_usize(value: Option<usize>) -> String {
969    value.map_or_else(|| "null".into(), |value| value.to_string())
970}
971
972fn json_escape(value: &str) -> String {
973    let mut escaped = String::new();
974    for character in value.chars() {
975        match character {
976            '"' => escaped.push_str("\\\""),
977            '\\' => escaped.push_str("\\\\"),
978            '\u{08}' => escaped.push_str("\\b"),
979            '\u{0c}' => escaped.push_str("\\f"),
980            '\n' => escaped.push_str("\\n"),
981            '\r' => escaped.push_str("\\r"),
982            '\t' => escaped.push_str("\\t"),
983            character if character <= '\u{1f}' => {
984                let _ = write!(escaped, "\\u{:04x}", u32::from(character));
985            }
986            character => escaped.push(character),
987        }
988    }
989    escaped
990}
991
992/// Where a walk's time went, so "blocked" is never one undifferentiated number.
993///
994/// The performance loop's standing question is whether a walk is bound by disk I/O,
995/// by CPU, or by coordination, and process-level counters cannot answer it: user and
996/// system time say how much CPU was burned, but a fused "blocked" number cannot say
997/// whether workers were waiting on the filesystem, on the queue lock, or on nothing
998/// at all because the queue was empty. These counters split that out at the source.
999///
1000/// Everything is measured in *chunks*, never per file: one timing pair per claimed
1001/// run of directories, per contended lock, per batch handoff. On the 60k-entry
1002/// reference tree that is a few thousand `Instant` reads against hundreds of
1003/// milliseconds of walking — the instrumentation follows the same amortization rule
1004/// it exists to verify.
1005///
1006/// In a parallel walk the fields sum over workers, so `wall_ns` is worker-seconds
1007/// (it can exceed the scan's wall clock) and every other duration is a disjoint
1008/// slice of it: `work_ns + starved_ns + lock_wait_ns + send_ns <= wall_ns`, with the
1009/// remainder being uninstrumented odds and ends (uncontended lock ops, loop
1010/// bookkeeping). A serial walk fills only `wall_ns`, `work_ns`, and `send_ns` —
1011/// there is no coordination to attribute.
1012#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
1013pub struct WalkAttribution {
1014    /// Total time workers spent in the walk loop, summed across workers.
1015    pub wall_ns: u64,
1016    /// Reading directories and stating entries — the real work, syscalls plus the
1017    /// compute between them. Separating disk from CPU *within* this span needs the
1018    /// process-level user/system counters alongside; per-syscall timing would break
1019    /// the chunk-amortization rule.
1020    pub work_ns: u64,
1021    /// Waiting on the queue's condvar because no work was available. Starvation:
1022    /// either the frontier is momentarily narrower than the worker pool, or the walk
1023    /// is ending.
1024    pub starved_ns: u64,
1025    /// Waiting to acquire the queue lock when another worker held it. This is the
1026    /// contention the shared-queue design bets stays negligible; now it is measured
1027    /// instead of argued.
1028    pub lock_wait_ns: u64,
1029    /// Handing observation batches to the consumer: the channel send in a parallel
1030    /// walk, the inline sink call — which is the consumer actually running — in a
1031    /// serial one.
1032    pub send_ns: u64,
1033    /// Chunks of directories claimed from the queue.
1034    pub claims: u64,
1035    /// Queue lock acquisitions, contended or not.
1036    pub lock_ops: u64,
1037    /// Lock acquisitions that found the lock already held.
1038    pub lock_contended: u64,
1039}
1040
1041impl WalkAttribution {
1042    /// Fold one worker's counters into the whole-walk totals.
1043    fn absorb(&mut self, other: Self) {
1044        self.wall_ns += other.wall_ns;
1045        self.work_ns += other.work_ns;
1046        self.starved_ns += other.starved_ns;
1047        self.lock_wait_ns += other.lock_wait_ns;
1048        self.send_ns += other.send_ns;
1049        self.claims += other.claims;
1050        self.lock_ops += other.lock_ops;
1051        self.lock_contended += other.lock_contended;
1052    }
1053
1054    /// Time attributed to a named cause, as opposed to `wall_ns`'s total.
1055    pub fn accounted_ns(&self) -> u64 {
1056        self.work_ns + self.starved_ns + self.lock_wait_ns + self.send_ns
1057    }
1058}
1059
1060/// Filesystem and index effects from an applying reconciliation pass.
1061#[derive(Debug, Default)]
1062pub struct ReconcileReport {
1063    /// Filesystem walk effects and partial errors.
1064    ///
1065    /// [`ScanReport::attribution`] remains zero for reconciliation because neither the
1066    /// serial nor parallel path has complete, comparable instrumentation yet. Zero
1067    /// means "not measured" here, not "no work".
1068    pub scan: ScanReport,
1069    /// Index arbitration and mutation effects.
1070    pub apply: ApplyStats,
1071    /// Exact producer operations considered, including no-op controls that do not
1072    /// increment an effect counter or create a commit.
1073    pub(crate) observations: u64,
1074    /// Directories this pass listed in full, with no error inside them, that the index did
1075    /// not yet hold as complete.
1076    ///
1077    /// The closing commit records each one's child set as authoritative, as discovery's
1078    /// own listing commit does, whether or not the rest of the pass completed: one transient
1079    /// child error
1080    /// elsewhere used to keep every directory the pass listed incomplete, and a directory
1081    /// first listed by such a pass stayed `Unknown { Building }` under a complete root.
1082    pub(crate) listed_incomplete: Vec<PathBuf>,
1083    /// Ownership epoch for conditional reconciliation batches.
1084    reconcile_epoch: Option<u64>,
1085    /// Retry after bounded verification evidence was superseded.
1086    retry_required: bool,
1087}
1088
1089impl ReconcileReport {
1090    /// True when the filesystem walk was complete and no conditional observation lost
1091    /// a race with another producer.
1092    pub fn is_complete(&self) -> bool {
1093        self.scan.is_complete()
1094            && self.apply.stale == 0
1095            && self.apply.resource_refused == 0
1096            && !self.retry_required
1097    }
1098
1099    /// True when a newer verification retired this pass's bounded evidence before it
1100    /// closed, so its scope is published partial and must be walked again.
1101    pub(crate) const fn retry_required(&self) -> bool {
1102        self.retry_required
1103    }
1104
1105    /// The directories whose listings this pass can vouch for, taken out of the report.
1106    ///
1107    /// None when a conditional commit lost a race or was refused: a child of any listed
1108    /// directory may then be missing from the index until the retry that race earns, and
1109    /// the retry records completeness for what it lists.
1110    pub(crate) fn take_recordable_completeness(&mut self) -> Vec<PathBuf> {
1111        let listed = std::mem::take(&mut self.listed_incomplete);
1112        if self.apply.stale > 0 || self.apply.resource_refused > 0 { Vec::new() } else { listed }
1113    }
1114}
1115
1116enum ReconcileTarget<'a> {
1117    Direct(&'a mut Index),
1118    Shared(&'a IndexHandle),
1119    Controlled { handle: &'a IndexHandle, control: &'a dyn ReconcileControl },
1120}
1121
1122/// Lifecycle checkpoints used by an owned long-running reconciliation.
1123///
1124/// The ordinary one-shot APIs use no controller. An [`crate::OpenedIndex`] supplies one
1125/// so close can stop a refresh before another write, and deterministic tests can pause
1126/// after filesystem verification but before conditional arbitration.
1127pub(crate) trait ReconcileControl {
1128    /// Fail when the owning operation may no longer publish state.
1129    fn check_active(&self) -> Result<()>;
1130
1131    /// Boundary after filesystem verification and before a conditional fact commit.
1132    fn before_conditional_commit(&self) -> Result<()>;
1133
1134    /// Atomic file-retention limit shared with every producer for this opened root.
1135    fn max_files(&self) -> Option<u64>;
1136}
1137
1138impl ReconcileTarget<'_> {
1139    fn scope(&self) -> Result<ScanScope> {
1140        match self {
1141            Self::Direct(index) => Ok(index.scope()),
1142            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.scope(),
1143        }
1144    }
1145
1146    fn root_path(&self) -> Result<PathBuf> {
1147        match self {
1148            Self::Direct(index) => Ok(index.root_path().to_path_buf()),
1149            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.root_path(),
1150        }
1151    }
1152
1153    fn expectation(&self, path: &Path) -> Result<PathExpectation> {
1154        match self {
1155            Self::Direct(index) => Ok(index.expectation(path)),
1156            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.expectation(path),
1157        }
1158    }
1159
1160    fn child_states(&self, path: &Path) -> Result<BTreeMap<OsString, PathExpectation>> {
1161        match self {
1162            Self::Direct(index) => Ok(collect_child_expectations(index, path)),
1163            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.child_states(path),
1164        }
1165    }
1166
1167    /// Child baselines for one directory listing, and whether a complete listing of it
1168    /// would be news to the index's directory completeness.
1169    ///
1170    /// A directory whose upsert has not been flushed yet is not held at all and counts as
1171    /// incomplete.
1172    fn listing_baseline(&self, path: &Path) -> Result<(BTreeMap<OsString, PathExpectation>, bool)> {
1173        match self {
1174            Self::Direct(index) => Ok((
1175                collect_child_expectations(index, path),
1176                index.directory_complete(path) != Some(true),
1177            )),
1178            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.listing_baseline(path),
1179        }
1180    }
1181
1182    fn has_control(&self, path: &Path) -> Result<bool> {
1183        match self {
1184            Self::Direct(index) => Ok(index.control_table().contains(path)),
1185            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.has_control(path),
1186        }
1187    }
1188
1189    fn control_table(&self) -> Result<crate::control::ControlTable> {
1190        match self {
1191            Self::Direct(index) => Ok(index.control_table().clone()),
1192            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1193                handle.read_with(|index| index.control_table().clone())
1194            }
1195        }
1196    }
1197
1198    fn control_classification_known(&self, path: &Path) -> Result<bool> {
1199        match self {
1200            Self::Direct(index) => Ok(index.control_classification_known(path)),
1201            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1202                handle.read_with(|index| index.control_classification_known(path))
1203            }
1204        }
1205    }
1206
1207    fn apply(&mut self, started_at: u64, observation: &Observation) -> Result<crate::ApplyOutcome> {
1208        match self {
1209            Self::Direct(index) => index.apply(observation),
1210            Self::Shared(handle) => handle.apply_reconcile(started_at, observation),
1211            Self::Controlled { handle, control } => {
1212                control.before_conditional_commit()?;
1213                handle.apply_opened_reconcile(started_at, observation, control.max_files())
1214            }
1215        }
1216    }
1217
1218    fn direct_upsert_is_unchanged(
1219        &self,
1220        baseline: PathExpectation,
1221        kind: EntryKind,
1222        attrs: Attrs,
1223    ) -> bool {
1224        matches!(self, Self::Direct(_)) && baseline.state == (PathState::Present { kind, attrs })
1225    }
1226
1227    fn take_pending_invalidations(&mut self) -> Result<Vec<(PathBuf, crate::InvalidateReason)>> {
1228        match self {
1229            Self::Direct(index) => Ok(index.take_pending_invalidations()),
1230            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1231                handle.take_pending_invalidations()
1232            }
1233        }
1234    }
1235
1236    fn restore_pending_invalidations(
1237        &mut self,
1238        invalidations: Vec<(PathBuf, crate::InvalidateReason)>,
1239    ) -> Result<()> {
1240        match self {
1241            Self::Direct(index) => index.restore_pending_invalidations(invalidations),
1242            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1243                handle.restore_pending_invalidations(invalidations)?;
1244            }
1245        }
1246        Ok(())
1247    }
1248
1249    /// Whether an invalidation whose reconciliation came back incomplete is queued again.
1250    ///
1251    /// A caller of the one-shot API owns its index exclusively and drains the queue when it
1252    /// chooses, so an unreadable subtree stays queued for it to retry. The shared API and an
1253    /// opened root are drained after every observed event -- by `Watcher::apply_next` and by
1254    /// the opened root's observer -- where that retry is a full walk of the same unreadable
1255    /// subtree per unrelated event, for the life of the session. There only a lost race is
1256    /// worth retrying: a stale conditional commit, or one the budget refused. A scan error
1257    /// is a settled boundary: the subtree stays partial, as it does at the observation
1258    /// handoff, and the report names the error once.
1259    fn retries_incomplete(&self, report: &ReconcileReport) -> bool {
1260        match self {
1261            Self::Direct(_) => !report.is_complete(),
1262            Self::Shared(_) | Self::Controlled { .. } => {
1263                report.apply.stale > 0 || report.apply.resource_refused > 0 || report.retry_required
1264            }
1265        }
1266    }
1267
1268    fn begin_reconcile(&mut self, path: &Path) -> Result<(u64, Option<Commit>)> {
1269        match self {
1270            Self::Direct(index) => index.begin_reconcile(path),
1271            Self::Shared(handle) => handle.begin_reconcile(path),
1272            Self::Controlled { handle, control } => {
1273                control.check_active()?;
1274                handle.begin_reconcile(path)
1275            }
1276        }
1277    }
1278
1279    fn finish_reconcile(
1280        &mut self,
1281        path: &Path,
1282        started_at: u64,
1283        complete: bool,
1284        listed_incomplete: &[PathBuf],
1285        failed_paths: &[PathBuf],
1286        errors: ReconcileErrors<'_>,
1287    ) -> Result<ReconcileFinish> {
1288        match self {
1289            Self::Direct(index) => index.finish_reconcile(
1290                path,
1291                started_at,
1292                complete,
1293                listed_incomplete,
1294                failed_paths,
1295                errors,
1296            ),
1297            Self::Shared(handle) => handle.finish_reconcile(
1298                path,
1299                started_at,
1300                complete,
1301                listed_incomplete,
1302                failed_paths,
1303                errors,
1304            ),
1305            Self::Controlled { handle, control } => {
1306                control.check_active()?;
1307                handle.finish_reconcile(
1308                    path,
1309                    started_at,
1310                    complete,
1311                    listed_incomplete,
1312                    failed_paths,
1313                    errors,
1314                )
1315            }
1316        }
1317    }
1318}
1319
1320#[cfg(unix)]
1321pub(crate) fn metadata_for_fingerprint(entry: &fs::DirEntry) -> std::io::Result<fs::Metadata> {
1322    crate::counters::bump(|c| c.stats += 1);
1323    entry.metadata()
1324}
1325
1326#[cfg(any(all(windows, test), not(any(unix, windows))))]
1327pub(crate) fn metadata_for_fingerprint(entry: &fs::DirEntry) -> std::io::Result<fs::Metadata> {
1328    crate::counters::bump(|c| c.stats += 1);
1329    // Windows serves DirEntry metadata from directory-enumeration data, which the
1330    // platform permits to be stale. Fingerprints need a fresh non-following query.
1331    fs::symlink_metadata(entry.path())
1332}
1333
1334#[cfg(test)]
1335type WalkHook = std::sync::Arc<dyn Fn(WalkHookPoint<'_>) -> Option<std::io::Error> + Send + Sync>;
1336
1337/// Where a test hook runs in a listing walk.
1338#[cfg(test)]
1339#[derive(Clone, Copy, Debug)]
1340pub(crate) enum WalkHookPoint<'a> {
1341    /// Before the metadata lookup of the listed child at this absolute path; an error
1342    /// stands in for the lookup's.
1343    ChildMetadata(&'a Path),
1344    /// After a reconciliation's listing of a directory returns its last entry; an error is
1345    /// read as one more listing item, which leaves the listing incomplete.
1346    ListingEnd,
1347    /// After a lookup of a directory's canonical control path, at this absolute path, has
1348    /// returned and before its answer is used; an error stands in for that answer.
1349    ControlLookup(&'a Path),
1350}
1351
1352/// Hooks run at each [`WalkHookPoint`], each for the paths under its root.
1353///
1354/// Process-wide, because a parallel walk looks children up on its worker threads; keyed
1355/// by root, because tests run in parallel and each walks its own temporary directory.
1356#[cfg(test)]
1357static WALK_HOOKS: std::sync::RwLock<Vec<(Vec<PathBuf>, WalkHook)>> =
1358    std::sync::RwLock::new(Vec::new());
1359
1360/// Removes its hook from [`WALK_HOOKS`] when dropped.
1361#[cfg(test)]
1362#[must_use = "the hook is removed as soon as the guard is dropped"]
1363pub(crate) struct WalkHookGuard(WalkHook);
1364
1365#[cfg(test)]
1366impl Drop for WalkHookGuard {
1367    fn drop(&mut self) {
1368        WALK_HOOKS
1369            .write()
1370            .unwrap_or_else(std::sync::PoisonError::into_inner)
1371            .retain(|(_, hook)| !std::sync::Arc::ptr_eq(hook, &self.0));
1372    }
1373}
1374
1375/// Run `hook` at every [`WalkHookPoint`] under `root`, on any thread, until the guard drops.
1376///
1377/// The hook may also change the tree before it returns. `root` matches as given and
1378/// canonical, since an opened root and a detached scan walk the canonical path.
1379#[cfg(test)]
1380pub(crate) fn install_walk_hook(
1381    root: &Path,
1382    hook: impl Fn(WalkHookPoint<'_>) -> Option<std::io::Error> + Send + Sync + 'static,
1383) -> WalkHookGuard {
1384    let mut roots = vec![root.to_path_buf()];
1385    if let Ok(canonical) = root.canonicalize() {
1386        roots.push(canonical);
1387    }
1388    let hook: WalkHook = std::sync::Arc::new(hook);
1389    WALK_HOOKS
1390        .write()
1391        .unwrap_or_else(std::sync::PoisonError::into_inner)
1392        .push((roots, std::sync::Arc::clone(&hook)));
1393    WalkHookGuard(hook)
1394}
1395
1396/// Run `hook` before every listed child's metadata lookup under `root`, with the child's
1397/// absolute path, until the guard drops.
1398#[cfg(test)]
1399pub(crate) fn install_child_metadata_hook(
1400    root: &Path,
1401    hook: impl Fn(&Path) -> Option<std::io::Error> + Send + Sync + 'static,
1402) -> WalkHookGuard {
1403    install_walk_hook(root, move |point| match point {
1404        WalkHookPoint::ChildMetadata(path) => hook(path),
1405        WalkHookPoint::ListingEnd | WalkHookPoint::ControlLookup(_) => None,
1406    })
1407}
1408
1409/// The hook installed for a root containing `path`, if any.
1410#[cfg(test)]
1411fn walk_hook(path: &Path) -> Option<WalkHook> {
1412    WALK_HOOKS
1413        .read()
1414        .unwrap_or_else(std::sync::PoisonError::into_inner)
1415        .iter()
1416        .find(|(roots, _)| roots.iter().any(|root| path.starts_with(root)))
1417        .map(|(_, hook)| std::sync::Arc::clone(hook))
1418}
1419
1420/// A reconciliation's listing of `dir`, followed by any error a test hook injects.
1421///
1422/// Callers bind the result to `listing` and iterate it as `for … in listing`, because the
1423/// admission audit (`scripts/check-admission-sites.mjs`) counts routed listing loops by that
1424/// shape. Keep the binding when editing a call site; dropping it silently removes the loop
1425/// from the audit, whose expected count would then look too high rather than wrong.
1426#[cfg(test)]
1427fn reconcile_listing<'r>(listing: Listing<'r>, dir: &Path) -> impl Iterator<Item = Listed<'r>> {
1428    let injected = walk_hook(dir).and_then(|hook| hook(WalkHookPoint::ListingEnd));
1429    listing.chain(injected.map(Listed::Failed))
1430}
1431
1432/// A reconciliation's listing of `dir`.
1433#[cfg(not(test))]
1434fn reconcile_listing<'r>(listing: Listing<'r>, _dir: &Path) -> Listing<'r> {
1435    listing
1436}
1437
1438/// The error a test hook injects for the metadata lookup of the listed child at `path`.
1439#[cfg(all(test, not(windows)))]
1440fn child_metadata_hook(path: &Path) -> Option<std::io::Error> {
1441    walk_hook(path).and_then(|hook| hook(WalkHookPoint::ChildMetadata(path)))
1442}
1443
1444/// Whether std's stat of a listed child, `DirEntry::file_type` of a `DT_UNKNOWN` entry
1445/// included, is one the kernel treats as `AT_NO_AUTOMOUNT`.
1446///
1447/// It is `fstatat` everywhere but glibc, where it is `statx` without the flag wherever
1448/// `statx` is served, and `fstatat` only once the reader has found it unavailable
1449/// (`linux_dents`).
1450#[cfg(all(target_os = "linux", target_env = "gnu"))]
1451fn std_child_stat_never_automounts() -> bool {
1452    linux_dents::statx_unavailable()
1453}
1454
1455#[cfg(not(any(windows, all(target_os = "linux", target_env = "gnu"))))]
1456const fn std_child_stat_never_automounts() -> bool {
1457    true
1458}
1459
1460/// Kind and attributes for one listed child.
1461///
1462/// The transient summary fold counts directories and ignores symlink attributes, so a
1463/// listing `file_type` (`d_type` on Linux) is enough for those kinds when the walk is
1464/// not bound to one filesystem. Files and specials still need a metadata lookup for
1465/// size, allocated bytes, and mtime. `one_filesystem` still stats directories because
1466/// descent compares `attrs.dev` to the root device, and `dev == 0` would otherwise
1467/// cross a mount. The listing's kind is taken only once the listing has proved its
1468/// directory searchable ([`Searchability`]); until then every child is stated, and the
1469/// skip applies to what the stat found.
1470///
1471/// Where `d_type` is `DT_UNKNOWN` (XFS without `ftype`, some FUSE/NFS mounts, older
1472/// ext3), std's `file_type` performs the non-following stat itself, and the skip then
1473/// applies to the kind it found. On glibc that stat would trigger an automount, so
1474/// while `statx` is served the listing's `file_type` is not consulted at all: every
1475/// child is stated by [`observe_dir_entry`], which never automounts, and the skip
1476/// applies to what that stat found, as it does to a `DT_UNKNOWN` entry. That route is
1477/// the fallback for a directory the native reader declined; the reader itself keeps
1478/// the `d_type` skip.
1479///
1480/// Windows never takes the skip: its observation contract reads every listed entry
1481/// through a fresh non-following handle ([`observe_dir_entry`]), so the transient fold
1482/// there performs exactly the observations the retained walk performs.
1483fn listed_child_kind_and_attrs(
1484    entry: &fs::DirEntry,
1485    policy: ListingPolicy,
1486    searchability: &mut Searchability,
1487) -> std::io::Result<Option<(EntryKind, Attrs)>> {
1488    #[cfg(not(windows))]
1489    {
1490        let listing_kind_suffices = policy.skip_dir_symlink_stat
1491            && *searchability == Searchability::Proven
1492            && std_child_stat_never_automounts();
1493        if listing_kind_suffices {
1494            if let Ok(file_type) = entry.file_type() {
1495                if file_type.is_dir() && !policy.one_filesystem {
1496                    return Ok(Some((EntryKind::Dir, Attrs::default())));
1497                }
1498                if file_type.is_symlink() {
1499                    return Ok(Some((EntryKind::Symlink, Attrs::default())));
1500                }
1501            }
1502        }
1503        let observed = observe_dir_entry(entry)?;
1504        if observed.is_some() {
1505            *searchability = Searchability::Proven;
1506        }
1507        if policy.skip_dir_symlink_stat && !listing_kind_suffices {
1508            return Ok(observed.map(|(kind, attrs)| match kind {
1509                EntryKind::Dir if !policy.one_filesystem => (kind, Attrs::default()),
1510                EntryKind::Symlink => (kind, Attrs::default()),
1511                EntryKind::Dir | EntryKind::File | EntryKind::Other => (kind, attrs),
1512            }));
1513        }
1514        Ok(observed)
1515    }
1516    #[cfg(windows)]
1517    {
1518        // Windows observes every listed entry through a fresh handle on both routes, so
1519        // the skip does not apply there; a successful observation proves the directory
1520        // searchable all the same, so the listing's state means the same on every host.
1521        let _ = (policy.skip_dir_symlink_stat, policy.one_filesystem);
1522        let observed = observe_dir_entry(entry)?;
1523        if observed.is_some() {
1524            *searchability = Searchability::Proven;
1525        }
1526        Ok(observed)
1527    }
1528}
1529
1530/// How a listing observes its children.
1531#[derive(Clone, Copy, Debug)]
1532pub(crate) struct ListingPolicy {
1533    /// H72: directory and symlink kinds come from the listing without a stat, once the
1534    /// listing has proved its directory searchable ([`Searchability`]).
1535    pub(crate) skip_dir_symlink_stat: bool,
1536    /// Descent compares `attrs.dev`, so directories are stated even under the skip.
1537    pub(crate) one_filesystem: bool,
1538}
1539
1540/// Whether one listing has proved its directory searchable, which the skip of
1541/// [`ListingPolicy::skip_dir_symlink_stat`] requires before it takes a child's kind from
1542/// the listing alone. Per listing: it starts unproven with each directory.
1543///
1544/// A stat is also an observation of failure. A directory that is readable but not
1545/// searchable (mode `0400`) opens and lists, and then every child's stat fails with
1546/// `EACCES`; a walk that stats every child reports each child as an error and holds no
1547/// entry for it. A `DT_DIR` child admitted on its `d_type` alone would be a directory
1548/// entry the full index does not have, queued and then reported again when it fails to
1549/// open, and a `DT_LNK` child a symlink entry with no error at all, so the folded tree
1550/// and the summary would count one directory more and report one error fewer than the
1551/// full index (R163-1). Under the skip, therefore, every child is stated until one stat
1552/// in the listing succeeds, which proves the directory searchable, as the skip assumes;
1553/// only then are directory and symlink kinds taken from the listing. The cost is one
1554/// stat per listing whose first children are directories or symlinks. A child that
1555/// vanished before its stat proves nothing here, which costs a stat and never an entry.
1556#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
1557pub(crate) enum Searchability {
1558    /// No child's stat has succeeded yet, so every child is stated.
1559    #[default]
1560    Unproven,
1561    /// A child's stat succeeded, so the listing's kind stands for a directory or symlink.
1562    Proven,
1563}
1564
1565impl ListingPolicy {
1566    /// Every child stated: what the retained index, a reconciliation, and discovery need.
1567    pub(crate) const fn every_child_stated(one_filesystem: bool) -> Self {
1568        Self { skip_dir_symlink_stat: false, one_filesystem }
1569    }
1570}
1571
1572/// The platform readers a listing loop keeps between the directories it lists.
1573///
1574/// One per walker or reconciliation worker: the Linux reader holds a 64 KiB record
1575/// buffer that each of its listings borrows for the directory's duration. Elsewhere
1576/// there is nothing to keep, and every listing is portable.
1577pub(crate) struct Readers {
1578    #[cfg(all(target_os = "linux", target_env = "gnu"))]
1579    dents: linux_dents::Reader,
1580}
1581
1582impl Readers {
1583    pub(crate) fn new() -> Self {
1584        Self {
1585            #[cfg(all(target_os = "linux", target_env = "gnu"))]
1586            dents: linux_dents::Reader::new(),
1587        }
1588    }
1589}
1590
1591/// One item of a directory listing, whichever backend served it.
1592pub(crate) enum Listed<'a> {
1593    /// The listing failed to yield an item, so the directory is listed incompletely.
1594    Failed(std::io::Error),
1595    /// A child by name, with its kind and attributes, or `Ok(None)` when it vanished
1596    /// between the listing and its stat, or the error that stat failed with.
1597    ///
1598    /// `NotFound` for a name the listing just returned means the entry was deleted in
1599    /// between, and every walk records it as it records a name the listing never
1600    /// returned: a cold walk has nothing to record, and a reconciliation removes what
1601    /// its baseline held. Reported as an error, it would make a walk over a tree being
1602    /// cleaned partial, and in a reconciliation it would settle as a phantom entry with
1603    /// permanent partial freshness. A native listing does not yield such a child at all.
1604    /// Any other error means the entry is present but unreadable.
1605    Child {
1606        name: std::borrow::Cow<'a, OsStr>,
1607        observed: std::io::Result<Option<(EntryKind, Attrs)>>,
1608    },
1609}
1610
1611/// One directory's children, natively where a reader served it and portably otherwise.
1612pub(crate) enum Listing<'r> {
1613    #[cfg(all(target_os = "linux", target_env = "gnu"))]
1614    Native(linux_dents::Listing<'r>),
1615    Portable {
1616        entries: fs::ReadDir,
1617        policy: ListingPolicy,
1618        /// What this listing has proved of its directory so far.
1619        searchability: Searchability,
1620        readers: std::marker::PhantomData<&'r mut Readers>,
1621    },
1622}
1623
1624impl<'r> Iterator for Listing<'r> {
1625    type Item = Listed<'r>;
1626
1627    fn next(&mut self) -> Option<Listed<'r>> {
1628        match self {
1629            #[cfg(all(target_os = "linux", target_env = "gnu"))]
1630            Self::Native(listing) => {
1631                let entry = listing.next()?;
1632                let observed = match entry.outcome {
1633                    linux_dents::Outcome::Observed { kind, attrs } => Ok(Some((kind, attrs))),
1634                    linux_dents::Outcome::Failed(error) => Err(error),
1635                };
1636                Some(Listed::Child { name: std::borrow::Cow::Borrowed(entry.name), observed })
1637            }
1638            Self::Portable { entries, policy, searchability, .. } => {
1639                let item = match entries.next()? {
1640                    Ok(item) => item,
1641                    Err(error) => return Some(Listed::Failed(error)),
1642                };
1643                crate::counters::bump(|c| c.dir_entries += 1);
1644                let observed = listed_child_kind_and_attrs(&item, *policy, searchability);
1645                Some(Listed::Child { name: std::borrow::Cow::Owned(item.file_name()), observed })
1646            }
1647        }
1648    }
1649}
1650
1651/// List `abs_dir` for a walk, a reconciliation, or discovery.
1652///
1653/// On Linux with glibc the native reader serves the directory unless it declines or a
1654/// test hook covers the directory; the portable `read_dir` answers otherwise, and an
1655/// error opening it is the caller's to report. Each backend produces the same items:
1656/// a native listing counts itself, and the portable one is counted here, as a
1657/// `dir_opens` and a portable attempt in `diagnostics`.
1658pub(crate) fn list_directory<'r>(
1659    readers: &'r mut Readers,
1660    abs_dir: &Path,
1661    policy: ListingPolicy,
1662    diagnostics: Option<&ScanDiagnosticsRecorder>,
1663) -> std::io::Result<Listing<'r>> {
1664    #[cfg(all(target_os = "linux", target_env = "gnu"))]
1665    {
1666        // Plain `if`, not `bool::then(|| …)`: the listing borrows the reader.
1667        if !walk_hook_covers(abs_dir) {
1668            if let Some(diagnostics) = diagnostics {
1669                diagnostics.linux_dents_attempted();
1670            }
1671            let native = linux_dents::StatPolicy {
1672                skip_dir_symlink_stat: policy.skip_dir_symlink_stat,
1673                one_filesystem: policy.one_filesystem,
1674            };
1675            if let Some(listing) = readers.dents.read(abs_dir, native) {
1676                if let Some(diagnostics) = diagnostics {
1677                    diagnostics.linux_dents_succeeded();
1678                }
1679                return Ok(Listing::Native(listing));
1680            }
1681            // A declined directory is counted as a fallback here and as the portable
1682            // attempt below, as the walker counts it.
1683            if let Some(diagnostics) = diagnostics {
1684                diagnostics.linux_dents_fell_back();
1685            }
1686        }
1687    }
1688    #[cfg(not(all(target_os = "linux", target_env = "gnu")))]
1689    let _ = &readers;
1690    crate::counters::bump(|c| c.dir_opens += 1);
1691    if let Some(diagnostics) = diagnostics {
1692        diagnostics.portable_attempted();
1693    }
1694    let entries = fs::read_dir(abs_dir)?;
1695    if let Some(diagnostics) = diagnostics {
1696        diagnostics.portable_succeeded();
1697    }
1698    Ok(Listing::Portable {
1699        entries,
1700        policy,
1701        searchability: Searchability::Unproven,
1702        readers: std::marker::PhantomData,
1703    })
1704}
1705
1706fn missing_as_none<T>(lookup: std::io::Result<T>) -> std::io::Result<Option<T>> {
1707    match lookup {
1708        Ok(metadata) => Ok(Some(metadata)),
1709        Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(None),
1710        Err(error) => Err(error),
1711    }
1712}
1713
1714/// Whether a test hook observes lookups or listings under `path`, which a native read
1715/// would not make.
1716#[cfg(all(test, any(target_os = "macos", all(target_os = "linux", target_env = "gnu"))))]
1717fn walk_hook_covers(path: &Path) -> bool {
1718    walk_hook(path).is_some()
1719}
1720
1721#[cfg(all(not(test), any(target_os = "macos", all(target_os = "linux", target_env = "gnu"))))]
1722const fn walk_hook_covers(_path: &Path) -> bool {
1723    false
1724}
1725
1726/// Owned output from the filesystem walker before it crosses a public mutation boundary.
1727///
1728/// Only the scan and opened-discovery producers construct this type. Their admission,
1729/// depth, filesystem, and symlink checks have already selected every operation, and the
1730/// index consumes the owned paths while proving their parent identities under its write
1731/// boundary. Public scan callers receive an [`Observation`] instead and therefore keep
1732/// the full public normalization and atomic-validation contract.
1733#[derive(Debug)]
1734pub(crate) struct ScannerBatch {
1735    ops: Vec<ObservationOp>,
1736    /// When set, the consumer must return `ops` through this sender instead of dropping
1737    /// them. Workers allocate the `PathBuf`s; returning the drained vec lets glibc free
1738    /// those arenas on the producing thread. The public [`scan`] path leaves this unset.
1739    recycle: Option<std::sync::mpsc::Sender<Vec<ObservationOp>>>,
1740}
1741
1742impl ScannerBatch {
1743    pub(crate) const fn new(ops: Vec<ObservationOp>) -> Self {
1744        Self { ops, recycle: None }
1745    }
1746
1747    fn with_recycle(self, recycle: std::sync::mpsc::Sender<Vec<ObservationOp>>) -> Self {
1748        Self { recycle: Some(recycle), ..self }
1749    }
1750
1751    #[cfg(test)]
1752    pub(crate) fn from_ops(ops: Vec<Op>) -> Self {
1753        Self { ops: ops.into_iter().map(ObservationOp::unconditional).collect(), recycle: None }
1754    }
1755
1756    pub(crate) fn len(&self) -> usize {
1757        self.ops.len()
1758    }
1759
1760    pub(crate) fn ops(&self) -> &[ObservationOp] {
1761        &self.ops
1762    }
1763
1764    pub(crate) fn into_ops(self) -> Vec<ObservationOp> {
1765        self.ops
1766    }
1767
1768    fn into_observation(self) -> Observation {
1769        Observation::from_ops(self.ops)
1770    }
1771
1772    fn recycle(self) {
1773        if let Some(recycle) = self.recycle {
1774            let _ = recycle.send(self.ops);
1775        }
1776    }
1777}
1778
1779/// One direct child retained by the private detached cold-bootstrap builder.
1780///
1781/// The worker owns the component once. Unlike [`ScannerBatch`], this record does not
1782/// manufacture a full relative path or a public observation for every entry.
1783#[derive(Debug)]
1784pub(crate) struct DetachedChild {
1785    pub(crate) name: OsString,
1786    pub(crate) kind: EntryKind,
1787    pub(crate) attrs: Attrs,
1788    /// Enumeration order within the listing. An enumerator can repeat a name while its
1789    /// directory is modified, and the builder keeps the later observation, as a
1790    /// streaming re-upsert does.
1791    pub(crate) position: u32,
1792}
1793
1794/// One directory listing retained by a worker for detached bootstrap consolidation.
1795///
1796/// `path` is paid once per directory. Its children remain grouped exactly as the
1797/// filesystem enumerator produced them, so consolidation resolves the parent once and
1798/// never reconstructs a child path for nondirectories. A fixed control is retained
1799/// separately so the consumer can install the directory's complete control state
1800/// before it classifies any sibling or makes descendants visible.
1801#[derive(Debug)]
1802pub(crate) struct DetachedDirectory {
1803    pub(crate) path: PathBuf,
1804    pub(crate) children: Vec<DetachedChild>,
1805    pub(crate) control: Option<Op>,
1806}
1807
1808/// Walk `root` and emit observations describing everything found.
1809pub fn scan(
1810    root: &Path,
1811    config: &ScanConfig,
1812    sink: &mut dyn FnMut(Observation),
1813) -> Result<ScanReport> {
1814    let mut public_sink = |batch: ScannerBatch| sink(batch.into_observation());
1815    let (mut report, _diagnostics) = scan_internal(
1816        root,
1817        config,
1818        &mut public_sink,
1819        false,
1820        WorkerPolicyExperiment::ShippedOneShot,
1821        SinkMode::Retained,
1822    )?;
1823    normalize_walk_errors(root, &mut report.errors);
1824    Ok(report)
1825}
1826
1827/// Walk `root` for the transient summary tier, folding each op without retaining it.
1828///
1829/// The public [`scan`] path hands each batch to the caller as an [`Observation`], so
1830/// worker-allocated `PathBuf`s are freed on the consumer thread. This path returns
1831/// drained batches to the producing worker so each arena is allocated and freed on one
1832/// thread. Tallies must match [`scan`], and so must the normalized error set: the
1833/// summary report's status is built from these errors exactly as a retained walk's is.
1834/// A scan that reads `.gitignore` delivers each directory's control ahead of every entry
1835/// it governs, so the fold can classify each entry as the index would
1836/// ([`SinkMode::groups_directories`]).
1837pub(crate) fn scan_summary_fold(
1838    root: &Path,
1839    config: &ScanConfig,
1840    fold: &mut dyn FnMut(&ObservationOp),
1841) -> Result<ScanReport> {
1842    let mut sink = |batch: ScannerBatch| {
1843        for op in batch.ops() {
1844            fold(op);
1845        }
1846        batch.recycle();
1847    };
1848    let (mut report, _diagnostics) = scan_internal(
1849        root,
1850        config,
1851        &mut sink,
1852        false,
1853        WorkerPolicyExperiment::ShippedOneShot,
1854        SinkMode::TransientFold,
1855    )?;
1856    normalize_walk_errors(root, &mut report.errors);
1857    Ok(report)
1858}
1859
1860/// [`scan_summary_fold`] plus the diagnostic trace [`scan_with_diagnostics`] collects.
1861pub(crate) fn scan_summary_fold_with_diagnostics(
1862    root: &Path,
1863    config: &ScanConfig,
1864    fold: &mut dyn FnMut(&ObservationOp),
1865) -> Result<(ScanReport, ScanDiagnostics)> {
1866    let mut sink = |batch: ScannerBatch| {
1867        for op in batch.ops() {
1868            fold(op);
1869        }
1870        batch.recycle();
1871    };
1872    let (mut report, diagnostics) = scan_internal(
1873        root,
1874        config,
1875        &mut sink,
1876        true,
1877        WorkerPolicyExperiment::ShippedOneShot,
1878        SinkMode::TransientFold,
1879    )?;
1880    normalize_walk_errors(root, &mut report.errors);
1881    Ok((report, diagnostics.expect("diagnostic scan creates a recorder")))
1882}
1883
1884/// Walk `root`, emitting observations and a bounded run-scoped diagnostic trace.
1885///
1886/// This is the measurement counterpart to [`scan`]. It produces the same observation
1887/// stream and report while recording controller and backend evidence that ordinary
1888/// scans intentionally do not collect.
1889pub fn scan_with_diagnostics(
1890    root: &Path,
1891    config: &ScanConfig,
1892    sink: &mut dyn FnMut(Observation),
1893) -> Result<(ScanReport, ScanDiagnostics)> {
1894    scan_with_policy_diagnostics(root, config, sink, WorkerPolicyExperiment::ShippedOneShot)
1895}
1896
1897/// Exercise a repository-only worker-controller candidate and retain its trace.
1898#[doc(hidden)]
1899pub fn scan_with_policy_diagnostics(
1900    root: &Path,
1901    config: &ScanConfig,
1902    sink: &mut dyn FnMut(Observation),
1903    policy: WorkerPolicyExperiment,
1904) -> Result<(ScanReport, ScanDiagnostics)> {
1905    let mut public_sink = |batch: ScannerBatch| sink(batch.into_observation());
1906    let (mut report, diagnostics) =
1907        scan_internal(root, config, &mut public_sink, true, policy, SinkMode::Retained)?;
1908    normalize_walk_errors(root, &mut report.errors);
1909    Ok((report, diagnostics.expect("diagnostic scan creates a recorder")))
1910}
1911
1912/// What the caller does with each batch of observations.
1913///
1914/// Two measured keeps hang off this one concept, and both were measured on the
1915/// transient fold alone: returning drained batches to the producing worker (H147,
1916/// exp-151) and taking directory and symlink kind from the listing without a stat
1917/// (H72, exp-153). They are named here as properties of the mode rather than passed as
1918/// one flag under one of their names, so a measurement on another platform can move
1919/// one without silently moving the other.
1920///
1921/// A third property is semantic rather than measured: a transient fold that observes
1922/// `.gitignore` classifies on its consumer, and that needs each directory's control
1923/// before its entries ([`Self::groups_directories`]).
1924#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1925enum SinkMode {
1926    /// The consumer keeps the observations: the public [`scan`] and the index.
1927    Retained,
1928    /// The consumer folds each batch and drops it: the transient summary tier.
1929    TransientFold,
1930}
1931
1932impl SinkMode {
1933    /// Drained batches go back to the worker that allocated them (H147).
1934    fn recycles_batches(self) -> bool {
1935        self == Self::TransientFold
1936    }
1937
1938    /// Directory and symlink kind come from the listing without a stat (H72).
1939    fn skips_dir_symlink_stat(self) -> bool {
1940        self == Self::TransientFold
1941    }
1942
1943    /// Each directory's control reaches the consumer ahead of every entry it governs
1944    /// (fdu-1ovb).
1945    ///
1946    /// The transient summary classifies every entry against `.gitignore` on its
1947    /// consumer, as the detached index builder does, and the builder applies a
1948    /// directory's control before it classifies any child because it receives each
1949    /// listing whole. A streaming batch keeps listing order, where `.gitignore` can come
1950    /// last, so a worker holds a listing's observations and moves the control ahead of
1951    /// them when the listing ends, or, if the batch fills first, reads the directory's
1952    /// control directly and lets that read stand for the listing
1953    /// (`StreamingEmission::send_if_full`). The retained stream keeps listing order: the
1954    /// index reclassifies the subtree a control governs when that control arrives.
1955    fn groups_directories(self, config: &ScanConfig) -> bool {
1956        self == Self::TransientFold && config.read_controls
1957    }
1958}
1959
1960fn scan_internal(
1961    root: &Path,
1962    config: &ScanConfig,
1963    sink: &mut dyn FnMut(ScannerBatch),
1964    collect_diagnostics: bool,
1965    policy: WorkerPolicyExperiment,
1966    sink_mode: SinkMode,
1967) -> Result<(ScanReport, Option<ScanDiagnostics>)> {
1968    config.validate()?;
1969    if let Some(progress) = &config.progress {
1970        progress.enter(crate::ProgressPhase::Scanning);
1971    }
1972    let root_meta = {
1973        crate::counters::bump(|c| c.stats += 1);
1974        fs::symlink_metadata(root)
1975    }
1976    .map_err(|e| Error::io(root, e))?;
1977    if !root_meta.is_dir() {
1978        return Err(Error::io(
1979            root,
1980            std::io::Error::new(std::io::ErrorKind::NotADirectory, "scan root is not a directory"),
1981        ));
1982    }
1983    let root_dev = root_device(root, &root_meta).map_err(|error| Error::io(root, error))?;
1984    let available_parallelism =
1985        std::thread::available_parallelism().map_or(1, std::num::NonZero::get);
1986    // Narrow populations need control admission before child enumeration. Keep one ordered
1987    // producer and control table so a bounded budget makes the same decisions as the
1988    // index that consumes the observations.
1989    let pool = if config.population == crate::query::IgnoredEntries::Include {
1990        config.worker_pool_for(available_parallelism)
1991    } else {
1992        WorkerPool::fixed(1)
1993    };
1994    let diagnostics = collect_diagnostics
1995        .then(|| ScanDiagnosticsRecorder::new(pool, available_parallelism, policy));
1996
1997    // The serial walk below emits in listing order and never groups a directory, so a
1998    // transient fold that classifies takes the concurrent walk even with one worker. That
1999    // is the walk the detached index takes at every worker count, and one worker visits
2000    // directories in the order that builder consumes them, so a control budget admits
2001    // the same files on both routes. A narrowed population keeps the serial walk, which
2002    // reads each directory's control before listing it.
2003    let groups = sink_mode.groups_directories(config)
2004        && config.population == crate::query::IgnoredEntries::Include;
2005    if config.max_depth != Some(0) && (pool.initial > 1 || groups) {
2006        let report = scan_concurrent(
2007            root,
2008            config,
2009            root_dev,
2010            sink,
2011            pool,
2012            diagnostics.as_ref(),
2013            policy,
2014            sink_mode,
2015        );
2016        return Ok((report, diagnostics.as_ref().map(|value| value.finish())));
2017    }
2018
2019    let mut report = ScanReport::default();
2020    if config.max_depth == Some(0) {
2021        if let Some(diagnostics) = &diagnostics {
2022            diagnostics.mark_not_run();
2023            diagnostics.record_queue_finish(0, 0);
2024        }
2025        return Ok((report, diagnostics.as_ref().map(|value| value.finish())));
2026    }
2027    let worker_guard = diagnostics.as_ref().map(ScanDiagnosticsRecorder::worker_guard);
2028    let walk_started = std::time::Instant::now();
2029    let mut batch: Vec<ObservationOp> = Vec::with_capacity(config.batch_size);
2030    let mut queue: VecDeque<(PathBuf, usize)> = VecDeque::from(vec![(PathBuf::new(), 0)]);
2031    let mut controls = (config.population != crate::query::IgnoredEntries::Include)
2032        .then(|| crate::control::ControlTable::with_limits(config.control_limits));
2033    let mut unreadable_controls = std::collections::BTreeSet::new();
2034    let mut tally = ProgressTally::new(config.progress.as_ref());
2035    let mut readers = Readers::new();
2036    let policy = ListingPolicy {
2037        skip_dir_symlink_stat: sink_mode.skips_dir_symlink_stat(),
2038        one_filesystem: config.one_filesystem,
2039    };
2040    // Every batch leaves through here, so the batch is where the serial walk reports
2041    // its progress: the handoff the consumer already pays for, never the entry.
2042    let mut emit = |ops: Vec<ObservationOp>, report: &mut ScanReport| {
2043        let send_started = std::time::Instant::now();
2044        sink(ScannerBatch::new(ops));
2045        report.attribution.send_ns += elapsed_ns(send_started);
2046        tally.flush(report);
2047    };
2048
2049    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
2050        let abs_dir = root.join(&rel_dir);
2051        if let Some(controls) = controls.as_mut() {
2052            let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
2053            match read_directory_control(config, root, &control_path) {
2054                Ok(Some(op)) => {
2055                    apply_discovery_control(controls, &op)?;
2056                    batch.push(ObservationOp::unconditional(op));
2057                }
2058                Ok(None) => {}
2059                Err(error) => {
2060                    unreadable_controls.insert(rel_dir.clone());
2061                    report.errors.push(error);
2062                }
2063            }
2064        }
2065        let listing = match list_directory(&mut readers, &abs_dir, policy, diagnostics.as_deref()) {
2066            Ok(listing) => listing,
2067            Err(e) => {
2068                report.errors.push(Error::io(abs_dir, e));
2069                continue;
2070            }
2071        };
2072        report.dirs_read += 1;
2073
2074        for item in listing {
2075            let (name, observed) = match item {
2076                Listed::Child { name, observed } => (name, observed),
2077                Listed::Failed(e) => {
2078                    report.errors.push(Error::io(&abs_dir, e));
2079                    continue;
2080                }
2081            };
2082            let rel_path = rel_dir.join(&name);
2083            let (kind, attrs) = match observed {
2084                Ok(Some(observed)) => observed,
2085                Ok(None) => continue,
2086                Err(error) => {
2087                    report.errors.push(Error::io(abs_dir.join(&name), error));
2088                    continue;
2089                }
2090            };
2091            let disposition =
2092                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
2093            if disposition == crate::admission::Disposition::Reject {
2094                continue;
2095            }
2096            if population_prunes(
2097                config.population,
2098                &rel_path,
2099                kind,
2100                disposition,
2101                controls.as_ref(),
2102                &unreadable_controls,
2103            ) {
2104                continue;
2105            }
2106            // A narrowed walk looked the directory's control up before listing it, and that
2107            // lookup stands for every spelling the listing shows.
2108            let control =
2109                match if controls.is_some() && crate::control::control_spelling(&name).is_some() {
2110                    Ok(None)
2111                } else {
2112                    read_control_op(config, root, &rel_path, kind)
2113                } {
2114                    Ok(control) => control,
2115                    Err(error) => {
2116                        report.errors.push(error);
2117                        None
2118                    }
2119                };
2120            if disposition == crate::admission::Disposition::ControlOnly {
2121                if let Some(control) = control {
2122                    batch.push(ObservationOp::unconditional(control));
2123                    if batch.len() >= config.batch_size {
2124                        emit(std::mem::take(&mut batch), &mut report);
2125                        batch.reserve(config.batch_size);
2126                    }
2127                }
2128                continue;
2129            }
2130            report.observe(kind, attrs);
2131            batch.push(ObservationOp::unconditional(Op::Upsert {
2132                path: rel_path.clone(),
2133                kind,
2134                attrs,
2135            }));
2136            if batch.len() >= config.batch_size {
2137                emit(std::mem::take(&mut batch), &mut report);
2138                batch.reserve(config.batch_size);
2139            }
2140            if let Some(control) = control {
2141                batch.push(ObservationOp::unconditional(control));
2142                if batch.len() >= config.batch_size {
2143                    emit(std::mem::take(&mut batch), &mut report);
2144                    batch.reserve(config.batch_size);
2145                }
2146            }
2147
2148            if should_descend(kind, attrs, depth, root_dev, config) {
2149                queue.push_back((rel_path, depth + 1));
2150            }
2151        }
2152    }
2153
2154    if !batch.is_empty() {
2155        emit(batch, &mut report);
2156    }
2157    // A walk whose last directories filled no batch has counted them and sent nothing.
2158    tally.flush(&report);
2159    // A serial walk has no coordination to attribute: wall is the loop, "send" is the
2160    // inline sink — which is the consumer actually running — and work is the rest.
2161    report.attribution.wall_ns = elapsed_ns(walk_started);
2162    report.attribution.work_ns =
2163        report.attribution.wall_ns.saturating_sub(report.attribution.send_ns);
2164    drop(worker_guard);
2165    if let Some(diagnostics) = &diagnostics {
2166        diagnostics.record_queue_finish(0, 0);
2167    }
2168    Ok((report, diagnostics.as_ref().map(|value| value.finish())))
2169}
2170
2171/// Take the next directory in the configured order.
2172///
2173/// Both orders push to the back; only the end they are taken from differs, which is
2174/// what keeps this a one-line policy rather than two walkers.
2175fn take_next(queue: &mut VecDeque<(PathBuf, usize)>, order: ScanOrder) -> Option<(PathBuf, usize)> {
2176    match order {
2177        ScanOrder::BreadthFirst => queue.pop_front(),
2178        ScanOrder::DepthFirst => queue.pop_back(),
2179    }
2180}
2181
2182/// Largest worker pool a caller may ask for explicitly.
2183///
2184/// Well past anything measured to help. It exists so a caller that computes a thread
2185/// count from something silly cannot spawn thousands of threads.
2186const MAX_SCAN_THREADS: usize = 32;
2187
2188/// Ceiling on the workers active at the start of an automatic scan.
2189///
2190/// Measured, not guessed. On a 10-core machine walking a 60k-entry `node_modules`
2191/// tree, wall time fell 37% at two workers and 50% at four, then stopped improving:
2192/// six matched four within noise and eight was 4% worse than four. The walk becomes
2193/// bound by the single index consumer, so past this point extra workers buy queue
2194/// contention and efficiency-core scheduling rather than throughput. See
2195/// `docs/project/reports/report-2026-08-10-fdu-performance-experiments.md`.
2196///
2197/// That measurement was taken on macOS, and every constant in this group now reads its
2198/// value from [`crate::platform_tuning`], which records per platform whether the number
2199/// was measured there or inherited. On Linux these are inherited.
2200const DEFAULT_SCAN_THREADS_CAP: usize = crate::platform_tuning::tuning().scan_threads_cap.get();
2201
2202/// Ceiling on automatic workers for an immutable-baseline reconciliation wave.
2203///
2204/// Reconciliation reads both filesystem and index state. Its measured knee arrives
2205/// before the cold producer's because additional metadata calls amplify kernel work
2206/// after the index comparisons already saturate the performance cores.
2207const DEFAULT_RECONCILE_THREADS_CAP: usize =
2208    crate::platform_tuning::tuning().reconcile_threads_cap.get();
2209
2210/// Ceiling an automatic scan may unlock after it establishes that the tree is large.
2211///
2212/// Sixteen was the knee on the 720k-entry cache-pressure corpus in exp-015. Thirty-two
2213/// did not improve on it and spent substantially more worker time waiting at the end.
2214const ADAPTIVE_SCAN_THREADS_CAP: usize =
2215    crate::platform_tuning::tuning().adaptive_scan_threads_cap.get();
2216
2217/// Maximum reserve depth relative to the host's reported parallelism.
2218const ADAPTIVE_SCAN_PARALLELISM_MULTIPLIER: usize =
2219    crate::platform_tuning::tuning().adaptive_scan_parallelism_multiplier.get();
2220
2221/// Entries used to calibrate the initial workers' filesystem service time.
2222const ADAPTIVE_SCAN_CALIBRATION_ENTRIES: u64 =
2223    crate::platform_tuning::tuning().adaptive_scan_calibration_entries.get();
2224
2225/// Average worker time per observed entry that identifies a latency-bound scan.
2226///
2227/// Whole-run attribution separated the measured regimes: roughly 18 microseconds on
2228/// the 60k tree, 22 on the 120k boundary, and 42 or more on the 720k cache-pressure
2229/// tree. Thirty leaves margin between them. The calibration uses the same chunk timing
2230/// already collected for attribution, so it adds no per-entry clock reads.
2231///
2232/// Those are APFS regimes. The Linux warm floor is about 1.5 µs per entry, twenty times
2233/// below this threshold, so the trigger may never fire there — which is exactly the kind
2234/// of inherited constant [`crate::platform_tuning`] exists to make visible (H84).
2235const ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY: u64 =
2236    crate::platform_tuning::tuning().adaptive_scan_slow_work_ns_per_entry.get();
2237
2238#[derive(Clone, Copy, Debug, PartialEq, Eq)]
2239struct WorkerPool {
2240    initial: usize,
2241    maximum: usize,
2242    calibration: Option<WorkerCalibration>,
2243}
2244
2245#[derive(Clone, Copy, Debug, PartialEq, Eq)]
2246struct WorkerCalibration {
2247    minimum_entries: u64,
2248    slow_work_ns_per_entry: u64,
2249    entries: u64,
2250    work_ns: u64,
2251    chunks: u64,
2252}
2253
2254#[derive(Clone, Copy, Debug)]
2255struct PolicyWindowSnapshot {
2256    sequence: u64,
2257    start_entry_ordinal: u64,
2258    end_entry_ordinal: u64,
2259    observed_entries: u64,
2260    observed_chunks: u64,
2261    observed_work_ns: u64,
2262    ready_directories: usize,
2263    in_flight_directories: usize,
2264    active_workers: usize,
2265    handoff_backlog: usize,
2266    requested_workers: Option<usize>,
2267    decision: WorkerPolicyDecision,
2268}
2269
2270struct PolicyTraceState {
2271    outcome: WorkerPolicyOutcome,
2272    outcome_reason: Option<&'static str>,
2273    outcome_sequence: Option<u64>,
2274    windows: Vec<WorkerPolicyWindow>,
2275    events_truncated: bool,
2276    ready_directories_at_finish: usize,
2277    in_flight_directories_at_finish: usize,
2278}
2279
2280/// Shared state used only by the opt-in diagnostic scan APIs.
2281///
2282/// Normal scans pass no recorder and therefore never touch these atomics or locks. The
2283/// trace mutex is deliberately separate from the directory queue: recording a policy
2284/// window may add diagnostic cost, but it cannot alter the queue's synchronization or
2285/// the controller's decision.
2286pub(crate) struct ScanDiagnosticsRecorder {
2287    available_parallelism: usize,
2288    pool: WorkerPool,
2289    policy: WorkerPolicyExperiment,
2290    trace: std::sync::Mutex<PolicyTraceState>,
2291    workers_spawned: std::sync::atomic::AtomicUsize,
2292    active_workers: std::sync::atomic::AtomicUsize,
2293    peak_active_workers: std::sync::atomic::AtomicUsize,
2294    handoff_backlog: std::sync::atomic::AtomicUsize,
2295    handoff_backlog_high_water: std::sync::atomic::AtomicUsize,
2296    calibration_chunks: std::sync::atomic::AtomicU64,
2297    calibration_entries: std::sync::atomic::AtomicU64,
2298    calibration_work_ns: std::sync::atomic::AtomicU64,
2299    worker_expansions: std::sync::atomic::AtomicU64,
2300    portable_attempts: std::sync::atomic::AtomicU64,
2301    portable_successes: std::sync::atomic::AtomicU64,
2302    #[cfg(target_os = "macos")]
2303    macos_bulk_attempts: std::sync::atomic::AtomicU64,
2304    #[cfg(target_os = "macos")]
2305    macos_bulk_successes: std::sync::atomic::AtomicU64,
2306    #[cfg(target_os = "macos")]
2307    macos_bulk_fallbacks: std::sync::atomic::AtomicU64,
2308    #[cfg(all(target_os = "linux", target_env = "gnu"))]
2309    linux_dents_attempts: std::sync::atomic::AtomicU64,
2310    #[cfg(all(target_os = "linux", target_env = "gnu"))]
2311    linux_dents_successes: std::sync::atomic::AtomicU64,
2312    #[cfg(all(target_os = "linux", target_env = "gnu"))]
2313    linux_dents_fallbacks: std::sync::atomic::AtomicU64,
2314}
2315
2316impl ScanDiagnosticsRecorder {
2317    fn new(
2318        pool: WorkerPool,
2319        available_parallelism: usize,
2320        policy: WorkerPolicyExperiment,
2321    ) -> std::sync::Arc<Self> {
2322        let (outcome, outcome_reason) = if pool.calibration.is_some() {
2323            (
2324                WorkerPolicyOutcome::Undecided,
2325                Some("the adaptive calibration window has not completed"),
2326            )
2327        } else {
2328            (WorkerPolicyOutcome::Fixed, Some("this worker pool has no adaptive reserve"))
2329        };
2330        std::sync::Arc::new(Self {
2331            available_parallelism,
2332            pool,
2333            policy,
2334            trace: std::sync::Mutex::new(PolicyTraceState {
2335                outcome,
2336                outcome_reason,
2337                outcome_sequence: None,
2338                windows: Vec::new(),
2339                events_truncated: false,
2340                ready_directories_at_finish: 0,
2341                in_flight_directories_at_finish: 0,
2342            }),
2343            workers_spawned: std::sync::atomic::AtomicUsize::new(0),
2344            active_workers: std::sync::atomic::AtomicUsize::new(0),
2345            peak_active_workers: std::sync::atomic::AtomicUsize::new(0),
2346            handoff_backlog: std::sync::atomic::AtomicUsize::new(0),
2347            handoff_backlog_high_water: std::sync::atomic::AtomicUsize::new(0),
2348            calibration_chunks: std::sync::atomic::AtomicU64::new(0),
2349            calibration_entries: std::sync::atomic::AtomicU64::new(0),
2350            calibration_work_ns: std::sync::atomic::AtomicU64::new(0),
2351            worker_expansions: std::sync::atomic::AtomicU64::new(0),
2352            portable_attempts: std::sync::atomic::AtomicU64::new(0),
2353            portable_successes: std::sync::atomic::AtomicU64::new(0),
2354            #[cfg(target_os = "macos")]
2355            macos_bulk_attempts: std::sync::atomic::AtomicU64::new(0),
2356            #[cfg(target_os = "macos")]
2357            macos_bulk_successes: std::sync::atomic::AtomicU64::new(0),
2358            #[cfg(target_os = "macos")]
2359            macos_bulk_fallbacks: std::sync::atomic::AtomicU64::new(0),
2360            #[cfg(all(target_os = "linux", target_env = "gnu"))]
2361            linux_dents_attempts: std::sync::atomic::AtomicU64::new(0),
2362            #[cfg(all(target_os = "linux", target_env = "gnu"))]
2363            linux_dents_successes: std::sync::atomic::AtomicU64::new(0),
2364            #[cfg(all(target_os = "linux", target_env = "gnu"))]
2365            linux_dents_fallbacks: std::sync::atomic::AtomicU64::new(0),
2366        })
2367    }
2368
2369    fn worker_guard(self: &std::sync::Arc<Self>) -> ScanWorkerGuard {
2370        let active = self
2371            .active_workers
2372            .fetch_add(1, std::sync::atomic::Ordering::Relaxed)
2373            .saturating_add(1);
2374        self.workers_spawned.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2375        atomic_update_max(&self.peak_active_workers, active);
2376        ScanWorkerGuard { recorder: self.clone() }
2377    }
2378
2379    fn record_policy_window(&self, snapshot: PolicyWindowSnapshot) {
2380        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2381        let sequence = snapshot.sequence;
2382        let supersedes = trace.outcome_sequence.is_none_or(|current| sequence >= current);
2383        match snapshot.decision {
2384            WorkerPolicyDecision::Undecided if supersedes => {
2385                trace.outcome = WorkerPolicyOutcome::Undecided;
2386                trace.outcome_reason =
2387                    Some("the walk ended before the adaptive calibration window completed");
2388                trace.outcome_sequence = Some(sequence);
2389            }
2390            WorkerPolicyDecision::Hold
2391                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2392            {
2393                trace.outcome = WorkerPolicyOutcome::Held;
2394                trace.outcome_reason = None;
2395                trace.outcome_sequence = Some(sequence);
2396            }
2397            WorkerPolicyDecision::ScaleUp => {
2398                trace.outcome = WorkerPolicyOutcome::ScaledUp;
2399                trace.outcome_reason = None;
2400                trace.outcome_sequence = Some(sequence);
2401            }
2402            WorkerPolicyDecision::HoldNoUsefulWork
2403                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2404            {
2405                trace.outcome = WorkerPolicyOutcome::HeldNoUsefulWork;
2406                trace.outcome_reason =
2407                    Some("the slow trigger fired only after ready and in-flight work had drained");
2408                trace.outcome_sequence = Some(sequence);
2409            }
2410            WorkerPolicyDecision::HoldInsufficientFrontier
2411                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2412            {
2413                trace.outcome = WorkerPolicyOutcome::Held;
2414                trace.outcome_reason =
2415                    Some("the observed frontier could not use additional workers");
2416                trace.outcome_sequence = Some(sequence);
2417            }
2418            WorkerPolicyDecision::HoldHandoffBacklog
2419                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2420            {
2421                trace.outcome = WorkerPolicyOutcome::Held;
2422                trace.outcome_reason =
2423                    Some("the observation handoff backlog was already at the controller limit");
2424                trace.outcome_sequence = Some(sequence);
2425            }
2426            WorkerPolicyDecision::Incomplete
2427                if supersedes && trace.outcome == WorkerPolicyOutcome::Undecided =>
2428            {
2429                trace.outcome_reason =
2430                    Some("the walk ended before any adaptive calibration window completed");
2431                trace.outcome_sequence = Some(sequence);
2432            }
2433            _ => {}
2434        }
2435        if sequence >= MAX_POLICY_TRACE_EVENTS as u64 {
2436            trace.events_truncated = true;
2437            return;
2438        }
2439        let work_ns_per_entry = (snapshot.observed_entries > 0)
2440            .then(|| snapshot.observed_work_ns / snapshot.observed_entries);
2441        trace.windows.push(WorkerPolicyWindow {
2442            sequence,
2443            start_entry_ordinal: snapshot.start_entry_ordinal,
2444            end_entry_ordinal: snapshot.end_entry_ordinal,
2445            observed_entries: snapshot.observed_entries,
2446            observed_chunks: snapshot.observed_chunks,
2447            observed_work_ns: snapshot.observed_work_ns,
2448            work_ns_per_entry,
2449            work_ns_per_entry_unavailable_reason: work_ns_per_entry
2450                .is_none()
2451                .then_some("the window observed no entries"),
2452            ready_directories: snapshot.ready_directories,
2453            in_flight_directories: snapshot.in_flight_directories,
2454            active_workers: snapshot.active_workers,
2455            handoff_backlog: snapshot.handoff_backlog,
2456            requested_workers: snapshot.requested_workers,
2457            decision: snapshot.decision,
2458        });
2459    }
2460
2461    fn mark_not_run(&self) {
2462        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2463        trace.outcome = WorkerPolicyOutcome::NotRun;
2464        trace.outcome_reason = Some("max_depth zero requested no directory walk");
2465    }
2466
2467    fn record_queue_finish(&self, ready_directories: usize, in_flight_directories: usize) {
2468        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2469        trace.ready_directories_at_finish = ready_directories;
2470        trace.in_flight_directories_at_finish = in_flight_directories;
2471    }
2472
2473    fn handoff_sent(&self) {
2474        let backlog = self
2475            .handoff_backlog
2476            .fetch_add(1, std::sync::atomic::Ordering::Relaxed)
2477            .saturating_add(1);
2478        atomic_update_max(&self.handoff_backlog_high_water, backlog);
2479    }
2480
2481    fn handoff_received(&self) {
2482        let previous = self.handoff_backlog.fetch_sub(1, std::sync::atomic::Ordering::Relaxed);
2483        debug_assert!(previous > 0, "received handoff must have been sent");
2484    }
2485
2486    fn calibration_chunk(&self, entries: u64, work_ns: u64) {
2487        self.calibration_chunks.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2488        self.calibration_entries.fetch_add(entries, std::sync::atomic::Ordering::Relaxed);
2489        self.calibration_work_ns.fetch_add(work_ns, std::sync::atomic::Ordering::Relaxed);
2490    }
2491
2492    fn worker_expanded(&self) {
2493        self.worker_expansions.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2494    }
2495
2496    fn portable_attempted(&self) {
2497        self.portable_attempts.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2498    }
2499
2500    fn portable_succeeded(&self) {
2501        self.portable_successes.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2502    }
2503
2504    #[cfg(target_os = "macos")]
2505    fn macos_bulk_attempted(&self) {
2506        self.macos_bulk_attempts.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2507    }
2508
2509    #[cfg(target_os = "macos")]
2510    fn macos_bulk_succeeded(&self) {
2511        self.macos_bulk_successes.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2512    }
2513
2514    #[cfg(target_os = "macos")]
2515    fn macos_bulk_fell_back(&self) {
2516        self.macos_bulk_fallbacks.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2517    }
2518
2519    #[cfg(all(target_os = "linux", target_env = "gnu"))]
2520    fn linux_dents_attempted(&self) {
2521        self.linux_dents_attempts.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2522    }
2523
2524    #[cfg(all(target_os = "linux", target_env = "gnu"))]
2525    fn linux_dents_succeeded(&self) {
2526        self.linux_dents_successes.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2527    }
2528
2529    #[cfg(all(target_os = "linux", target_env = "gnu"))]
2530    fn linux_dents_fell_back(&self) {
2531        self.linux_dents_fallbacks.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2532    }
2533
2534    fn finish(&self) -> ScanDiagnostics {
2535        let trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2536        let calibration = self.pool.calibration;
2537        let backend = ScanBackendDiagnostics {
2538            portable_attempts: self.portable_attempts.load(std::sync::atomic::Ordering::Relaxed),
2539            portable_directory_reads: self
2540                .portable_successes
2541                .load(std::sync::atomic::Ordering::Relaxed),
2542            #[cfg(target_os = "macos")]
2543            macos_bulk_attempts: Some(
2544                self.macos_bulk_attempts.load(std::sync::atomic::Ordering::Relaxed),
2545            ),
2546            #[cfg(not(target_os = "macos"))]
2547            macos_bulk_attempts: None,
2548            #[cfg(target_os = "macos")]
2549            macos_bulk_successes: Some(
2550                self.macos_bulk_successes.load(std::sync::atomic::Ordering::Relaxed),
2551            ),
2552            #[cfg(not(target_os = "macos"))]
2553            macos_bulk_successes: None,
2554            #[cfg(target_os = "macos")]
2555            macos_bulk_fallbacks: Some(
2556                self.macos_bulk_fallbacks.load(std::sync::atomic::Ordering::Relaxed),
2557            ),
2558            #[cfg(not(target_os = "macos"))]
2559            macos_bulk_fallbacks: None,
2560            #[cfg(target_os = "macos")]
2561            unavailable_reason: None,
2562            #[cfg(not(target_os = "macos"))]
2563            unavailable_reason: Some(
2564                "macOS bulk directory enumeration is unavailable on this platform",
2565            ),
2566            #[cfg(all(target_os = "linux", target_env = "gnu"))]
2567            linux_dents_attempts: Some(
2568                self.linux_dents_attempts.load(std::sync::atomic::Ordering::Relaxed),
2569            ),
2570            #[cfg(not(all(target_os = "linux", target_env = "gnu")))]
2571            linux_dents_attempts: None,
2572            #[cfg(all(target_os = "linux", target_env = "gnu"))]
2573            linux_dents_successes: Some(
2574                self.linux_dents_successes.load(std::sync::atomic::Ordering::Relaxed),
2575            ),
2576            #[cfg(not(all(target_os = "linux", target_env = "gnu")))]
2577            linux_dents_successes: None,
2578            #[cfg(all(target_os = "linux", target_env = "gnu"))]
2579            linux_dents_fallbacks: Some(
2580                self.linux_dents_fallbacks.load(std::sync::atomic::Ordering::Relaxed),
2581            ),
2582            #[cfg(not(all(target_os = "linux", target_env = "gnu")))]
2583            linux_dents_fallbacks: None,
2584        };
2585        ScanDiagnostics {
2586            schema: SCAN_DIAGNOSTICS_SCHEMA,
2587            worker_policy: WorkerPolicyDiagnostics {
2588                controller: worker_policy_experiment_name(self.policy),
2589                available_parallelism: self.available_parallelism,
2590                initial_workers: self.pool.initial,
2591                maximum_workers: self.pool.maximum,
2592                calibration_window_entries: calibration.map(|value| value.minimum_entries),
2593                slow_threshold_ns_per_entry: calibration.map(|value| value.slow_work_ns_per_entry),
2594                calibration_chunks: self
2595                    .calibration_chunks
2596                    .load(std::sync::atomic::Ordering::Relaxed),
2597                calibration_entries: self
2598                    .calibration_entries
2599                    .load(std::sync::atomic::Ordering::Relaxed),
2600                calibration_work_ns: self
2601                    .calibration_work_ns
2602                    .load(std::sync::atomic::Ordering::Relaxed),
2603                worker_expansions: self
2604                    .worker_expansions
2605                    .load(std::sync::atomic::Ordering::Relaxed),
2606                outcome: trace.outcome,
2607                outcome_reason: trace.outcome_reason,
2608                workers_spawned: self.workers_spawned.load(std::sync::atomic::Ordering::Relaxed),
2609                peak_active_workers: self
2610                    .peak_active_workers
2611                    .load(std::sync::atomic::Ordering::Relaxed),
2612                ready_directories_at_finish: trace.ready_directories_at_finish,
2613                in_flight_directories_at_finish: trace.in_flight_directories_at_finish,
2614                handoff_backlog_at_finish: self
2615                    .handoff_backlog
2616                    .load(std::sync::atomic::Ordering::Relaxed),
2617                handoff_backlog_high_water: self
2618                    .handoff_backlog_high_water
2619                    .load(std::sync::atomic::Ordering::Relaxed),
2620                windows: {
2621                    let mut windows = trace.windows.clone();
2622                    windows.sort_by_key(|window| window.sequence);
2623                    windows
2624                },
2625                events_truncated: trace.events_truncated,
2626            },
2627            backend,
2628        }
2629    }
2630}
2631
2632const fn worker_policy_experiment_name(value: WorkerPolicyExperiment) -> &'static str {
2633    match value {
2634        WorkerPolicyExperiment::ShippedOneShot => "shipped_one_shot",
2635        WorkerPolicyExperiment::RepeatedWindows => "repeated_windows",
2636        WorkerPolicyExperiment::StagedGatedWindows => "staged_gated_windows",
2637    }
2638}
2639
2640struct ScanWorkerGuard {
2641    recorder: std::sync::Arc<ScanDiagnosticsRecorder>,
2642}
2643
2644impl Drop for ScanWorkerGuard {
2645    fn drop(&mut self) {
2646        let previous =
2647            self.recorder.active_workers.fetch_sub(1, std::sync::atomic::Ordering::Relaxed);
2648        debug_assert!(previous > 0, "worker guard must balance worker start");
2649    }
2650}
2651
2652fn atomic_update_max(target: &std::sync::atomic::AtomicUsize, value: usize) {
2653    let mut observed = target.load(std::sync::atomic::Ordering::Relaxed);
2654    while value > observed {
2655        match target.compare_exchange_weak(
2656            observed,
2657            value,
2658            std::sync::atomic::Ordering::Relaxed,
2659            std::sync::atomic::Ordering::Relaxed,
2660        ) {
2661            Ok(_) => break,
2662            Err(actual) => observed = actual,
2663        }
2664    }
2665}
2666
2667enum WalkMessage {
2668    Batch(ScannerBatch),
2669    DetachedDirectories {
2670        directories: Vec<DetachedDirectory>,
2671        /// The worker that allocated `directories`. The consumer drains each listing
2672        /// into the index and sends the emptied listings back here, so their path and
2673        /// child buffers are freed or reused on that worker's thread (H159).
2674        recycle: std::sync::mpsc::Sender<Vec<DetachedDirectory>>,
2675    },
2676    ScaleUp {
2677        sender: std::sync::mpsc::Sender<Self>,
2678        target_workers: usize,
2679    },
2680}
2681
2682impl WorkerPool {
2683    const fn fixed(workers: usize) -> Self {
2684        Self { initial: workers, maximum: workers, calibration: None }
2685    }
2686}
2687
2688impl WorkerCalibration {
2689    const fn new(minimum_entries: u64, slow_work_ns_per_entry: u64) -> Self {
2690        Self { minimum_entries, slow_work_ns_per_entry, entries: 0, work_ns: 0, chunks: 0 }
2691    }
2692
2693    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<bool> {
2694        self.chunks = self.chunks.saturating_add(1);
2695        self.entries = self.entries.saturating_add(entries);
2696        self.work_ns = self.work_ns.saturating_add(work_ns);
2697        (self.entries >= self.minimum_entries)
2698            .then(|| self.work_ns / self.entries >= self.slow_work_ns_per_entry)
2699    }
2700}
2701
2702#[derive(Clone, Copy, Debug)]
2703struct CalibrationWindow {
2704    start_entry_ordinal: u64,
2705    end_entry_ordinal: u64,
2706    entries: u64,
2707    chunks: u64,
2708    work_ns: u64,
2709    slow: bool,
2710}
2711
2712#[derive(Debug)]
2713struct RepeatedCalibration {
2714    minimum_entries: u64,
2715    slow_work_ns_per_entry: u64,
2716    window_start: u64,
2717    entries: u64,
2718    chunks: u64,
2719    work_ns: u64,
2720    completed_windows: u64,
2721}
2722
2723impl RepeatedCalibration {
2724    const fn new(calibration: WorkerCalibration) -> Self {
2725        Self {
2726            minimum_entries: calibration.minimum_entries,
2727            slow_work_ns_per_entry: calibration.slow_work_ns_per_entry,
2728            window_start: 0,
2729            entries: 0,
2730            chunks: 0,
2731            work_ns: 0,
2732            completed_windows: 0,
2733        }
2734    }
2735
2736    const fn starting_at(calibration: WorkerCalibration, window_start: u64) -> Self {
2737        let mut repeated = Self::new(calibration);
2738        repeated.window_start = window_start;
2739        repeated
2740    }
2741
2742    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<CalibrationWindow> {
2743        self.chunks = self.chunks.saturating_add(1);
2744        self.entries = self.entries.saturating_add(entries);
2745        self.work_ns = self.work_ns.saturating_add(work_ns);
2746        if self.entries < self.minimum_entries {
2747            return None;
2748        }
2749        let end_entry_ordinal = self.window_start.saturating_add(self.entries);
2750        let window = CalibrationWindow {
2751            start_entry_ordinal: self.window_start,
2752            end_entry_ordinal,
2753            entries: self.entries,
2754            chunks: self.chunks,
2755            work_ns: self.work_ns,
2756            slow: self.work_ns / self.entries >= self.slow_work_ns_per_entry,
2757        };
2758        self.window_start = end_entry_ordinal;
2759        self.entries = 0;
2760        self.chunks = 0;
2761        self.work_ns = 0;
2762        self.completed_windows = self.completed_windows.saturating_add(1);
2763        Some(window)
2764    }
2765}
2766
2767#[derive(Debug)]
2768enum WorkerController {
2769    OneShot(WorkerCalibration),
2770    Repeated { calibration: RepeatedCalibration, staged_gated: bool },
2771}
2772
2773impl WorkerController {
2774    fn new(calibration: WorkerCalibration, policy: WorkerPolicyExperiment) -> Self {
2775        match policy {
2776            WorkerPolicyExperiment::ShippedOneShot => Self::OneShot(calibration),
2777            WorkerPolicyExperiment::RepeatedWindows => Self::Repeated {
2778                calibration: RepeatedCalibration::new(calibration),
2779                staged_gated: false,
2780            },
2781            WorkerPolicyExperiment::StagedGatedWindows => Self::Repeated {
2782                calibration: RepeatedCalibration::new(calibration),
2783                staged_gated: true,
2784            },
2785        }
2786    }
2787
2788    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<CalibrationWindow> {
2789        match self {
2790            Self::OneShot(calibration) => {
2791                let slow = calibration.observe(entries, work_ns)?;
2792                Some(CalibrationWindow {
2793                    start_entry_ordinal: 0,
2794                    end_entry_ordinal: calibration.entries,
2795                    entries: calibration.entries,
2796                    chunks: calibration.chunks,
2797                    work_ns: calibration.work_ns,
2798                    slow,
2799                })
2800            }
2801            Self::Repeated { calibration, .. } => calibration.observe(entries, work_ns),
2802        }
2803    }
2804
2805    fn partial_window(&self) -> (CalibrationWindow, WorkerPolicyDecision) {
2806        match self {
2807            Self::OneShot(calibration) => (
2808                CalibrationWindow {
2809                    start_entry_ordinal: 0,
2810                    end_entry_ordinal: calibration.entries,
2811                    entries: calibration.entries,
2812                    chunks: calibration.chunks,
2813                    work_ns: calibration.work_ns,
2814                    slow: false,
2815                },
2816                WorkerPolicyDecision::Undecided,
2817            ),
2818            Self::Repeated { calibration, .. } => (
2819                CalibrationWindow {
2820                    start_entry_ordinal: calibration.window_start,
2821                    end_entry_ordinal: calibration.window_start.saturating_add(calibration.entries),
2822                    entries: calibration.entries,
2823                    chunks: calibration.chunks,
2824                    work_ns: calibration.work_ns,
2825                    slow: false,
2826                },
2827                if calibration.completed_windows == 0 {
2828                    WorkerPolicyDecision::Undecided
2829                } else {
2830                    WorkerPolicyDecision::Incomplete
2831                },
2832            ),
2833        }
2834    }
2835
2836    const fn is_staged_gated(&self) -> bool {
2837        matches!(self, Self::Repeated { staged_gated: true, .. })
2838    }
2839
2840    const fn is_one_shot(&self) -> bool {
2841        matches!(self, Self::OneShot(_))
2842    }
2843
2844    const fn calibration_spec(&self) -> WorkerCalibration {
2845        match self {
2846            Self::OneShot(calibration) => WorkerCalibration::new(
2847                calibration.minimum_entries,
2848                calibration.slow_work_ns_per_entry,
2849            ),
2850            Self::Repeated { calibration, .. } => WorkerCalibration::new(
2851                calibration.minimum_entries,
2852                calibration.slow_work_ns_per_entry,
2853            ),
2854        }
2855    }
2856}
2857
2858fn automatic_worker_pool(available: usize) -> WorkerPool {
2859    let initial = available.clamp(1, DEFAULT_SCAN_THREADS_CAP);
2860    // Preserve the serial fallback when the platform cannot report more than one
2861    // available processor. There is no measured basis for inventing parallelism there.
2862    if initial == 1 {
2863        return WorkerPool::fixed(1);
2864    }
2865    let maximum = available
2866        .saturating_mul(ADAPTIVE_SCAN_PARALLELISM_MULTIPLIER)
2867        .clamp(initial, ADAPTIVE_SCAN_THREADS_CAP);
2868    WorkerPool {
2869        initial,
2870        maximum,
2871        calibration: (maximum > initial).then_some(WorkerCalibration::new(
2872            ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
2873            ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
2874        )),
2875    }
2876}
2877
2878/// Directories handed to a worker in one go.
2879///
2880/// Popping one directory at a time makes the queue lock the bottleneck on a wide,
2881/// shallow tree; taking a small run amortizes the lock without letting one worker
2882/// starve the others by hoarding the queue.
2883const DIR_CLAIM: usize = 4;
2884
2885/// A parallel directory walk that produces exactly the observations the serial walk does.
2886///
2887/// The shape is deliberate. Workers read directories and *produce* observations; they
2888/// never touch an index. A single consumer — the caller's sink, on this thread —
2889/// applies them. That keeps the crate's one mutation contract intact: parallelism is a
2890/// property of the producer, and the index still sees one ordered stream of observations.
2891///
2892/// Ordering across independent subtrees is not fixed, but a directory observation is
2893/// published before that directory becomes claimable. The index therefore sees a
2894/// parent-first causal stream without imposing a global level barrier or serializing
2895/// filesystem work. The resulting index is byte-identical to the serial walker's,
2896/// which the benchmark harness re-proves on every trial by comparing engine digests
2897/// against an independent oracle.
2898#[allow(clippy::too_many_arguments)]
2899fn scan_concurrent(
2900    root: &Path,
2901    config: &ScanConfig,
2902    root_dev: u64,
2903    sink: &mut dyn FnMut(ScannerBatch),
2904    pool: WorkerPool,
2905    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2906    policy: WorkerPolicyExperiment,
2907    sink_mode: SinkMode,
2908) -> ScanReport {
2909    let mut consume = |message| match message {
2910        WalkMessage::Batch(batch) => {
2911            if let Some(diagnostics) = diagnostics {
2912                diagnostics.handoff_received();
2913            }
2914            sink(batch);
2915        }
2916        WalkMessage::DetachedDirectories { .. } => {
2917            unreachable!("the streaming walker never publishes detached directories")
2918        }
2919        WalkMessage::ScaleUp { .. } => {
2920            unreachable!("the shared runner consumes scale-up messages")
2921        }
2922    };
2923    run_concurrent_walk(
2924        root,
2925        config,
2926        root_dev,
2927        pool,
2928        diagnostics,
2929        policy,
2930        match sink_mode {
2931            SinkMode::Retained => walk_worker,
2932            SinkMode::TransientFold => walk_worker_transient_fold,
2933        },
2934        &mut consume,
2935    )
2936}
2937
2938/// Parallel cold walk for a detached index that has no streaming consumer.
2939///
2940/// Workers publish directory-shaped facts before making their children claimable. The
2941/// caller consumes those groups into a private builder while filesystem work continues,
2942/// preserving parent-first causality and pipeline overlap without sending one full path
2943/// or public observation per entry.
2944fn scan_concurrent_detached(
2945    root: &Path,
2946    config: &ScanConfig,
2947    root_dev: u64,
2948    pool: WorkerPool,
2949    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2950    policy: WorkerPolicyExperiment,
2951    retention: Option<TreeRetention>,
2952) -> Result<(ScanReport, DetachedIndexBuilder)> {
2953    let mut builder = detached_builder(root, config, retention);
2954    let mut build_error = None;
2955    let output = {
2956        let mut consume = |message| match message {
2957            WalkMessage::Batch(_) => {
2958                unreachable!("the detached walker never publishes scanner batches")
2959            }
2960            WalkMessage::DetachedDirectories { mut directories, recycle } => {
2961                if let Some(diagnostics) = diagnostics {
2962                    diagnostics.handoff_received();
2963                }
2964                if build_error.is_none() {
2965                    for directory in &mut directories {
2966                        if let Err(error) = builder.push_directory(directory) {
2967                            build_error = Some(error);
2968                            break;
2969                        }
2970                    }
2971                }
2972                // A worker that has already left has dropped its receiver, and then the
2973                // listings are freed here, as every listing was before H159.
2974                let _ = recycle.send(directories);
2975            }
2976            WalkMessage::ScaleUp { .. } => {
2977                unreachable!("the shared runner consumes scale-up messages")
2978            }
2979        };
2980        // A folded index takes H72's listing policy (H185); the full index stats every
2981        // entry, since its directory attributes are the cache's freshness fingerprint.
2982        let worker: WalkWorker =
2983            if retention.is_some() { walk_detached_folding_worker } else { walk_detached_worker };
2984        run_concurrent_walk(root, config, root_dev, pool, diagnostics, policy, worker, &mut consume)
2985    };
2986    if let Some(error) = build_error {
2987        return Err(error);
2988    }
2989    Ok((output, builder))
2990}
2991
2992type WalkWorker = fn(
2993    &Path,
2994    &ScanConfig,
2995    u64,
2996    &DirectoryQueue,
2997    &std::sync::mpsc::Sender<WalkMessage>,
2998    Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2999) -> ScanReport;
3000
3001/// Run the shared pool, scaling controller, diagnostics, and report reduction.
3002///
3003/// Streaming and detached scans differ only in their worker emission and main-thread
3004/// consumer. Keeping orchestration here prevents fixes to termination, diagnostics, or
3005/// panic handling from diverging between the two cold paths.
3006#[allow(clippy::too_many_arguments)]
3007fn run_concurrent_walk<C>(
3008    root: &Path,
3009    config: &ScanConfig,
3010    root_dev: u64,
3011    pool: WorkerPool,
3012    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3013    policy: WorkerPolicyExperiment,
3014    worker: WalkWorker,
3015    consume: &mut C,
3016) -> ScanReport
3017where
3018    C: FnMut(WalkMessage),
3019{
3020    let diagnostics = diagnostics.cloned();
3021    let queue = DirectoryQueue::new_with_policy(
3022        (PathBuf::new(), 0),
3023        config.order,
3024        pool.calibration,
3025        diagnostics.clone(),
3026        pool.initial,
3027        pool.maximum,
3028        policy,
3029    );
3030    let (sender, receiver) = std::sync::mpsc::channel::<WalkMessage>();
3031
3032    let mut report = std::thread::scope(|scope| {
3033        let mut handles: Vec<_> = (0..pool.initial)
3034            .map(|_| {
3035                let sender = sender.clone();
3036                let queue = &queue;
3037                let diagnostics = diagnostics.clone();
3038                scope.spawn(move || {
3039                    worker(root, config, root_dev, queue, &sender, diagnostics.as_ref())
3040                })
3041            })
3042            .collect();
3043        // The loop below ends when every sender is gone, so this one must go first.
3044        drop(sender);
3045
3046        let mut spawned_workers = pool.initial;
3047        for message in receiver {
3048            match message {
3049                WalkMessage::ScaleUp { sender, target_workers }
3050                    if target_workers > spawned_workers =>
3051                {
3052                    let target_workers = target_workers.min(pool.maximum);
3053                    record_adaptive_worker_expansion(diagnostics.as_ref());
3054                    for _ in spawned_workers..target_workers {
3055                        let sender = sender.clone();
3056                        let queue = &queue;
3057                        let diagnostics = diagnostics.clone();
3058                        handles.push(scope.spawn(move || {
3059                            worker(root, config, root_dev, queue, &sender, diagnostics.as_ref())
3060                        }));
3061                    }
3062                    spawned_workers = target_workers;
3063                }
3064                WalkMessage::ScaleUp { .. } => {}
3065                output => consume(output),
3066            }
3067        }
3068
3069        // A walk that ends before its calibration window fills never observed enough to
3070        // decide anything. That is an *unobservable* policy, not a decision to hold the
3071        // initial pool, and an artifact that conflated the two would report a held pool
3072        // as if the walk had measured one and chosen it.
3073        let queue_finish = {
3074            let mut state = queue.lock();
3075            let mut trailing_window = None;
3076            if let Some(controller) = &state.controller {
3077                let (window, decision) = controller.partial_window();
3078                if decision == WorkerPolicyDecision::Undecided {
3079                    crate::counters::bump(|counts| {
3080                        counts.adaptive_policy_undecided =
3081                            counts.adaptive_policy_undecided.saturating_add(1);
3082                    });
3083                }
3084                if diagnostics.is_some() {
3085                    let sequence = state.allocate_policy_sequence();
3086                    trailing_window = Some(PolicyWindowSnapshot {
3087                        sequence,
3088                        start_entry_ordinal: window.start_entry_ordinal,
3089                        end_entry_ordinal: window.end_entry_ordinal,
3090                        observed_entries: window.entries,
3091                        observed_chunks: window.chunks,
3092                        observed_work_ns: window.work_ns,
3093                        ready_directories: state.ready_directories,
3094                        in_flight_directories: state.in_flight_directories,
3095                        active_workers: diagnostics.as_ref().map_or(0, |diagnostics| {
3096                            diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
3097                        }),
3098                        handoff_backlog: diagnostics.as_ref().map_or(0, |diagnostics| {
3099                            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
3100                        }),
3101                        requested_workers: None,
3102                        decision,
3103                    });
3104                }
3105            } else if diagnostics.is_some() {
3106                let shadow = state.shadow_calibration.as_ref().and_then(|shadow| {
3107                    (shadow.entries > 0).then_some((
3108                        shadow.window_start,
3109                        shadow.entries,
3110                        shadow.chunks,
3111                        shadow.work_ns,
3112                    ))
3113                });
3114                if let Some((window_start, entries, chunks, work_ns)) = shadow {
3115                    let sequence = state.allocate_policy_sequence();
3116                    trailing_window = Some(PolicyWindowSnapshot {
3117                        sequence,
3118                        start_entry_ordinal: window_start,
3119                        end_entry_ordinal: window_start.saturating_add(entries),
3120                        observed_entries: entries,
3121                        observed_chunks: chunks,
3122                        observed_work_ns: work_ns,
3123                        ready_directories: state.ready_directories,
3124                        in_flight_directories: state.in_flight_directories,
3125                        active_workers: diagnostics.as_ref().map_or(0, |diagnostics| {
3126                            diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
3127                        }),
3128                        handoff_backlog: diagnostics.as_ref().map_or(0, |diagnostics| {
3129                            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
3130                        }),
3131                        requested_workers: None,
3132                        decision: WorkerPolicyDecision::ObserveIncomplete,
3133                    });
3134                }
3135            }
3136            (state.ready_directories, state.in_flight_directories, trailing_window)
3137        };
3138        if let Some(diagnostics) = &diagnostics {
3139            diagnostics.record_queue_finish(queue_finish.0, queue_finish.1);
3140            if let Some(window) = queue_finish.2 {
3141                diagnostics.record_policy_window(window);
3142            }
3143        }
3144
3145        let mut report = ScanReport::default();
3146        for handle in handles {
3147            match handle.join() {
3148                Ok(worker) => report.absorb(worker),
3149                Err(_) => {
3150                    // A worker panic leaves directories unaccounted for. Preserve that
3151                    // as a partial scan instead of reporting a short tree as complete.
3152                    report.errors.push(Error::io(
3153                        root,
3154                        std::io::Error::other("a scan worker thread panicked"),
3155                    ));
3156                }
3157            }
3158        }
3159        report
3160    });
3161
3162    // Workers finish in filesystem order, so normalize errors before they escape.
3163    report.errors.sort_by_cached_key(ToString::to_string);
3164    report
3165}
3166
3167/// Compile-time adapter for the one directory walker.
3168///
3169/// The filesystem, queue, admission, and diagnostics logic stays singular. Generic
3170/// emission keeps the public streaming path and private detached path branch-free in
3171/// their per-entry loops after monomorphization.
3172trait WalkEmission {
3173    type Directory;
3174
3175    fn begin_directory(&mut self, path: &Path) -> Self::Directory;
3176
3177    #[allow(clippy::too_many_arguments)]
3178    fn record_entry(
3179        &mut self,
3180        root: &Path,
3181        rel_dir: &Path,
3182        depth: usize,
3183        region: RegionId,
3184        name: &OsStr,
3185        kind: EntryKind,
3186        attrs: Attrs,
3187        root_dev: u64,
3188        config: &ScanConfig,
3189        directory: &mut Self::Directory,
3190        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3191        report: &mut ScanReport,
3192        sender: &std::sync::mpsc::Sender<WalkMessage>,
3193        chunk_send_ns: &mut u64,
3194        diagnostics: Option<&ScanDiagnosticsRecorder>,
3195    ) -> bool;
3196
3197    fn finish_directory(&mut self, directory: Self::Directory);
3198
3199    fn publish_before_discovery(
3200        &mut self,
3201        has_discovered: bool,
3202        sender: &std::sync::mpsc::Sender<WalkMessage>,
3203        chunk_send_ns: &mut u64,
3204        diagnostics: Option<&ScanDiagnosticsRecorder>,
3205    ) -> bool;
3206
3207    fn finish(
3208        &mut self,
3209        sender: &std::sync::mpsc::Sender<WalkMessage>,
3210        report: &mut ScanReport,
3211        diagnostics: Option<&ScanDiagnosticsRecorder>,
3212    );
3213
3214    /// The transient summary and the folded index take directory and symlink kind from
3215    /// the listing, with default attributes (H72, H185); every other route stats them.
3216    fn skip_dir_symlink_stat(&self) -> bool {
3217        false
3218    }
3219}
3220
3221struct StreamingEmission {
3222    batch: Vec<ObservationOp>,
3223    batch_size: usize,
3224    recycle_tx: Option<std::sync::mpsc::Sender<Vec<ObservationOp>>>,
3225    recycle_rx: Option<std::sync::mpsc::Receiver<Vec<ObservationOp>>>,
3226    skip_dir_symlink_stat: bool,
3227    /// Whether each listing's control goes ahead of its entries
3228    /// ([`SinkMode::groups_directories`]).
3229    group_directories: bool,
3230    /// Where the listing being recorded begins in `batch`.
3231    directory_start: usize,
3232    /// Whether that listing's control already leads its observations, so the batch may be
3233    /// sent before the listing ends.
3234    listing_settled: bool,
3235    /// Where that listing stands on its directory's control.
3236    listing_control: ListingControl,
3237}
3238
3239/// Where the listing a grouping worker is recording stands on its directory's control.
3240#[derive(Clone, Copy, PartialEq, Eq, Debug)]
3241enum ListingControl {
3242    /// No listed spelling of `.gitignore` has read the directory's control yet.
3243    Unread,
3244    /// A listed spelling read the directory's control, or failed to; its observation, if
3245    /// any, is among the held ones.
3246    Listed,
3247    /// The held observations filled the batch before any listed spelling read the
3248    /// control, so the directory's control was looked up directly (`read_directory_control`)
3249    /// and stands for the whole listing, as in the narrowed-population walk.
3250    ///
3251    /// A spelling listed afterwards is recorded as a row and not read again, so the
3252    /// directory's control file is read once. On a tree nothing modifies during the walk
3253    /// the probed file is the listed one; if the file changes between the two, the listing
3254    /// keeps what the probe read, and an error the second read would have met is never
3255    /// met.
3256    Probed,
3257}
3258
3259impl StreamingEmission {
3260    /// An emission with the properties [`SinkMode`] names for `mode` under `config`.
3261    ///
3262    /// The recycle channel returns drained `PathBuf` arenas to this worker so glibc
3263    /// frees them on the thread that allocated them.
3264    fn for_sink(config: &ScanConfig, mode: SinkMode) -> Self {
3265        let batch_size = config.batch_size;
3266        let (recycle_tx, recycle_rx) = if mode.recycles_batches() {
3267            let (tx, rx) = std::sync::mpsc::channel();
3268            (Some(tx), Some(rx))
3269        } else {
3270            (None, None)
3271        };
3272        Self {
3273            batch: Vec::with_capacity(batch_size),
3274            batch_size,
3275            recycle_tx,
3276            recycle_rx,
3277            skip_dir_symlink_stat: mode.skips_dir_symlink_stat(),
3278            group_directories: mode.groups_directories(config),
3279            directory_start: 0,
3280            listing_settled: false,
3281            listing_control: ListingControl::Unread,
3282        }
3283    }
3284
3285    /// Send the batch once it is full, settling a grouped listing's control first.
3286    ///
3287    /// A grouped listing's observations are held until its control leads them: at the
3288    /// listing's end ([`WalkEmission::finish_directory`]), or here, when the batch fills
3289    /// first. So a batch never holds more than `batch_size` observations and one control,
3290    /// however long the listing, and a fill that finds no spelling of `.gitignore` listed
3291    /// yet probes for it once per listing: one metadata lookup, and a read on a hit
3292    /// (`read_directory_control`). Returns whether the consumer is still there.
3293    #[allow(clippy::too_many_arguments)]
3294    fn send_if_full(
3295        &mut self,
3296        root: &Path,
3297        rel_dir: &Path,
3298        config: &ScanConfig,
3299        report: &mut ScanReport,
3300        sender: &std::sync::mpsc::Sender<WalkMessage>,
3301        chunk_send_ns: &mut u64,
3302        diagnostics: Option<&ScanDiagnosticsRecorder>,
3303    ) -> bool {
3304        if self.batch.len() < self.batch_size {
3305            return true;
3306        }
3307        if self.group_directories && !self.listing_settled {
3308            self.settle_listing(root, rel_dir, config, report);
3309        }
3310        let send_started = std::time::Instant::now();
3311        let sent = self.send_full(sender, diagnostics);
3312        *chunk_send_ns += elapsed_ns(send_started);
3313        sent
3314    }
3315
3316    /// Put the current listing's control ahead of its held observations.
3317    fn settle_listing(
3318        &mut self,
3319        root: &Path,
3320        rel_dir: &Path,
3321        config: &ScanConfig,
3322        report: &mut ScanReport,
3323    ) {
3324        self.listing_settled = true;
3325        if self.listing_control != ListingControl::Unread {
3326            controls_first(&mut self.batch[self.directory_start..]);
3327            return;
3328        }
3329        self.listing_control = ListingControl::Probed;
3330        let control = rel_dir.join(crate::control::CONTROL_FILE_NAME);
3331        // The probe is the lookup the rule names, so a hit stands for the directory
3332        // whichever spelling it resolved to, exactly as a listed spelling's read would.
3333        match read_directory_control(config, root, &control) {
3334            Ok(Some(op)) => {
3335                self.batch.insert(self.directory_start, ObservationOp::unconditional(op));
3336            }
3337            Ok(None) => {}
3338            Err(error) => report.errors.push(error),
3339        }
3340    }
3341
3342    /// Whether the entries this listing still lists must not read their control: its
3343    /// directory's control was already probed, and one read stands for the listing, as in
3344    /// the narrowed-population walk (`read_listed_control_op`).
3345    fn control_probed(&self) -> bool {
3346        self.group_directories && self.listing_control == ListingControl::Probed
3347    }
3348
3349    /// Note that a listed entry read this listing's control, or failed to.
3350    fn note_listed_control(&mut self, control: Option<&Op>, control_error: Option<&Error>) {
3351        if control.is_some() || control_error.is_some() {
3352            self.listing_control = ListingControl::Listed;
3353        }
3354    }
3355
3356    fn wrap(&self, ops: Vec<ObservationOp>) -> ScannerBatch {
3357        match &self.recycle_tx {
3358            Some(recycle) => ScannerBatch::new(ops).with_recycle(recycle.clone()),
3359            None => ScannerBatch::new(ops),
3360        }
3361    }
3362
3363    fn next_vec(&self) -> Vec<ObservationOp> {
3364        // The retained path keeps its pre-H147 shape: an empty vec that grows by
3365        // doubling. Pre-sizing every batch there was never measured, and the public
3366        // `scan` is what a library caller pays for.
3367        let Some(recycle_rx) = &self.recycle_rx else {
3368            return Vec::new();
3369        };
3370        let mut kept = None;
3371        while let Ok(mut recycled) = recycle_rx.try_recv() {
3372            recycled.clear();
3373            kept = Some(recycled);
3374        }
3375        kept.unwrap_or_else(|| Vec::with_capacity(self.batch_size))
3376    }
3377
3378    fn send_full(
3379        &mut self,
3380        sender: &std::sync::mpsc::Sender<WalkMessage>,
3381        diagnostics: Option<&ScanDiagnosticsRecorder>,
3382    ) -> bool {
3383        let ops = std::mem::take(&mut self.batch);
3384        let sent = send_scanner_batch(sender, self.wrap(ops), diagnostics);
3385        self.batch = self.next_vec();
3386        // A listing sent partway continues at the start of the new batch.
3387        self.directory_start = 0;
3388        sent
3389    }
3390}
3391
3392impl WalkEmission for StreamingEmission {
3393    type Directory = ();
3394
3395    fn begin_directory(&mut self, _path: &Path) {
3396        self.directory_start = self.batch.len();
3397        self.listing_settled = false;
3398        self.listing_control = ListingControl::Unread;
3399    }
3400
3401    #[allow(clippy::too_many_arguments)]
3402    fn record_entry(
3403        &mut self,
3404        root: &Path,
3405        rel_dir: &Path,
3406        depth: usize,
3407        region: RegionId,
3408        name: &OsStr,
3409        kind: EntryKind,
3410        attrs: Attrs,
3411        root_dev: u64,
3412        config: &ScanConfig,
3413        _directory: &mut Self::Directory,
3414        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3415        report: &mut ScanReport,
3416        sender: &std::sync::mpsc::Sender<WalkMessage>,
3417        chunk_send_ns: &mut u64,
3418        diagnostics: Option<&ScanDiagnosticsRecorder>,
3419    ) -> bool {
3420        record_walk_entry(
3421            root,
3422            rel_dir,
3423            depth,
3424            region,
3425            name,
3426            kind,
3427            attrs,
3428            root_dev,
3429            config,
3430            self,
3431            discovered,
3432            report,
3433            sender,
3434            chunk_send_ns,
3435            diagnostics,
3436        )
3437    }
3438
3439    fn finish_directory(&mut self, _directory: Self::Directory) {
3440        if self.group_directories && !self.listing_settled {
3441            controls_first(&mut self.batch[self.directory_start..]);
3442        }
3443    }
3444
3445    fn publish_before_discovery(
3446        &mut self,
3447        has_discovered: bool,
3448        sender: &std::sync::mpsc::Sender<WalkMessage>,
3449        chunk_send_ns: &mut u64,
3450        diagnostics: Option<&ScanDiagnosticsRecorder>,
3451    ) -> bool {
3452        if self.batch.is_empty() || !has_discovered {
3453            return true;
3454        }
3455        let send_started = std::time::Instant::now();
3456        let sent = self.send_full(sender, diagnostics);
3457        *chunk_send_ns += elapsed_ns(send_started);
3458        sent
3459    }
3460
3461    fn finish(
3462        &mut self,
3463        sender: &std::sync::mpsc::Sender<WalkMessage>,
3464        report: &mut ScanReport,
3465        diagnostics: Option<&ScanDiagnosticsRecorder>,
3466    ) {
3467        if self.batch.is_empty() {
3468            return;
3469        }
3470        let send_started = std::time::Instant::now();
3471        // The walk is over: do not ask `send_full` for a replacement vec that no
3472        // later `record_entry` would use.
3473        let ops = std::mem::take(&mut self.batch);
3474        let _ = send_scanner_batch(sender, self.wrap(ops), diagnostics);
3475        self.batch = Vec::new();
3476        report.attribution.send_ns += elapsed_ns(send_started);
3477    }
3478
3479    fn skip_dir_symlink_stat(&self) -> bool {
3480        self.skip_dir_symlink_stat
3481    }
3482}
3483
3484/// Emptied listings one detached worker keeps for reuse: one chunk's worth.
3485///
3486/// A bound on retention, not a tuned speed value. A chunk claims at most [`DIR_CLAIM`]
3487/// directories, and the worker takes returned listings back once per chunk, so this
3488/// covers the next chunk; a listing returned past it is freed at once, still on its own
3489/// thread, which is the property H159 needs. Retained listings are memory the consumer
3490/// can no longer reuse for the index: four chunks' worth of listings up to 256 children
3491/// each cost 1.0 MiB (+1.4%) of peak RSS on a 158,705-entry macOS subject and 2.3 MiB
3492/// (+5.0%) on a 77,159-entry one.
3493const DETACHED_SPARE_LISTINGS: usize = DIR_CLAIM;
3494
3495/// The largest child buffer, in children, a spare listing may keep.
3496///
3497/// Also a bound on retention rather than a tuned value: most directories are small, and
3498/// a listing whose buffer grew past this, at about 80 bytes per child, is freed when it
3499/// comes back, on its own thread, instead of being pinned for the rest of the walk.
3500const DETACHED_SPARE_CHILD_CAPACITY: usize = 64;
3501
3502/// One worker's detached emission: listings built here and published to the consumer.
3503///
3504/// Every path and child buffer in a listing is allocated on this worker's thread. The
3505/// consumer drains each listing into the index and sends the emptied listings back
3506/// (H159), and the worker reuses them or frees them itself. Under glibc a chunk freed on
3507/// another thread goes back to the arena that allocated it, under that arena's lock,
3508/// while this worker is allocating from it; the 2026-09-27 Linux comparison's allocator
3509/// screen and context-switch profile point at that contention for the index tier's gap
3510/// to its peers. A reused listing carries exactly the facts a fresh one would: the same
3511/// path bytes, and children and control only from this directory's listing.
3512struct DetachedEmission {
3513    /// H72's listing policy for the folded index (H185): a directory or symlink takes its
3514    /// kind from `d_type` and default attributes, as the transient summary does. A
3515    /// one-shot tree report reads no directory's or symlink's own attributes: its rows
3516    /// carry roll-ups, `newest_mtime_ns` is the files', symlinks and other kinds
3517    /// contribute nothing, and `dev` is read only under `--one-filesystem`, where the
3518    /// policy keeps the stat. The full index keeps every stat: its directory attributes
3519    /// are the cache's freshness fingerprint.
3520    skip_dir_symlink_stat: bool,
3521    directories: Vec<DetachedDirectory>,
3522    /// Emptied listings ready to be reused by [`WalkEmission::begin_directory`].
3523    spare: Vec<DetachedDirectory>,
3524    /// An emptied list to publish the next chunk's listings in.
3525    spare_list: Vec<DetachedDirectory>,
3526    recycle_tx: std::sync::mpsc::Sender<Vec<DetachedDirectory>>,
3527    recycle_rx: std::sync::mpsc::Receiver<Vec<DetachedDirectory>>,
3528}
3529
3530impl DetachedEmission {
3531    fn new(skip_dir_symlink_stat: bool) -> Self {
3532        let (recycle_tx, recycle_rx) = std::sync::mpsc::channel();
3533        Self {
3534            skip_dir_symlink_stat,
3535            directories: Vec::new(),
3536            spare: Vec::new(),
3537            spare_list: Vec::new(),
3538            recycle_tx,
3539            recycle_rx,
3540        }
3541    }
3542
3543    /// Take back every list the consumer has returned since the last chunk.
3544    ///
3545    /// A listing the consumer skipped, after a build error or for a repeated directory,
3546    /// comes back with its children and control still in it; they are dropped here, on
3547    /// the thread that allocated them, before the listing can be reused.
3548    fn collect_returned(&mut self) {
3549        while let Ok(mut returned) = self.recycle_rx.try_recv() {
3550            for mut directory in returned.drain(..) {
3551                directory.children.clear();
3552                directory.control = None;
3553                if self.spare.len() < DETACHED_SPARE_LISTINGS
3554                    && directory.children.capacity() <= DETACHED_SPARE_CHILD_CAPACITY
3555                {
3556                    self.spare.push(directory);
3557                }
3558            }
3559            if self.spare_list.capacity() == 0 {
3560                self.spare_list = returned;
3561            }
3562        }
3563    }
3564}
3565
3566impl WalkEmission for DetachedEmission {
3567    type Directory = DetachedDirectory;
3568
3569    fn skip_dir_symlink_stat(&self) -> bool {
3570        self.skip_dir_symlink_stat
3571    }
3572
3573    fn begin_directory(&mut self, path: &Path) -> Self::Directory {
3574        let Some(mut directory) = self.spare.pop() else {
3575            return DetachedDirectory {
3576                path: path.to_path_buf(),
3577                children: Vec::new(),
3578                control: None,
3579            };
3580        };
3581        // The same bytes `to_path_buf` would copy, into a buffer this thread already owns.
3582        let buffer = directory.path.as_mut_os_string();
3583        buffer.clear();
3584        buffer.push(path);
3585        debug_assert!(directory.children.is_empty() && directory.control.is_none());
3586        directory
3587    }
3588
3589    #[allow(clippy::too_many_arguments)]
3590    fn record_entry(
3591        &mut self,
3592        root: &Path,
3593        rel_dir: &Path,
3594        depth: usize,
3595        region: RegionId,
3596        name: &OsStr,
3597        kind: EntryKind,
3598        attrs: Attrs,
3599        root_dev: u64,
3600        config: &ScanConfig,
3601        directory: &mut Self::Directory,
3602        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3603        report: &mut ScanReport,
3604        _sender: &std::sync::mpsc::Sender<WalkMessage>,
3605        _chunk_send_ns: &mut u64,
3606        _diagnostics: Option<&ScanDiagnosticsRecorder>,
3607    ) -> bool {
3608        record_detached_entry(
3609            root,
3610            rel_dir,
3611            depth,
3612            region,
3613            name,
3614            kind,
3615            attrs,
3616            root_dev,
3617            config,
3618            &mut directory.children,
3619            &mut directory.control,
3620            discovered,
3621            report,
3622        );
3623        true
3624    }
3625
3626    fn finish_directory(&mut self, directory: Self::Directory) {
3627        self.directories.push(directory);
3628    }
3629
3630    fn publish_before_discovery(
3631        &mut self,
3632        _has_discovered: bool,
3633        sender: &std::sync::mpsc::Sender<WalkMessage>,
3634        chunk_send_ns: &mut u64,
3635        diagnostics: Option<&ScanDiagnosticsRecorder>,
3636    ) -> bool {
3637        if self.directories.is_empty() {
3638            return true;
3639        }
3640        // Taking back returned listings is handoff work, timed with the send so the
3641        // chunk's work time, which calibrates the worker pool, stays the walk's own.
3642        let send_started = std::time::Instant::now();
3643        self.collect_returned();
3644        let next = std::mem::take(&mut self.spare_list);
3645        let directories = std::mem::replace(&mut self.directories, next);
3646        let sent =
3647            send_detached_directories(sender, directories, self.recycle_tx.clone(), diagnostics);
3648        *chunk_send_ns += elapsed_ns(send_started);
3649        sent
3650    }
3651
3652    fn finish(
3653        &mut self,
3654        _sender: &std::sync::mpsc::Sender<WalkMessage>,
3655        _report: &mut ScanReport,
3656        _diagnostics: Option<&ScanDiagnosticsRecorder>,
3657    ) {
3658    }
3659}
3660
3661fn walk_detached_worker(
3662    root: &Path,
3663    config: &ScanConfig,
3664    root_dev: u64,
3665    queue: &DirectoryQueue,
3666    sender: &std::sync::mpsc::Sender<WalkMessage>,
3667    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3668) -> ScanReport {
3669    walk_detached_with(root, config, root_dev, queue, sender, diagnostics, false)
3670}
3671
3672/// [`walk_detached_worker`] for a folded index, which describes each directory once, by
3673/// its own listing (H185): see [`DetachedEmission::skip_dir_symlink_stat`].
3674fn walk_detached_folding_worker(
3675    root: &Path,
3676    config: &ScanConfig,
3677    root_dev: u64,
3678    queue: &DirectoryQueue,
3679    sender: &std::sync::mpsc::Sender<WalkMessage>,
3680    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3681) -> ScanReport {
3682    walk_detached_with(root, config, root_dev, queue, sender, diagnostics, true)
3683}
3684
3685fn walk_detached_with(
3686    root: &Path,
3687    config: &ScanConfig,
3688    root_dev: u64,
3689    queue: &DirectoryQueue,
3690    sender: &std::sync::mpsc::Sender<WalkMessage>,
3691    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3692    skip_dir_symlink_stat: bool,
3693) -> ScanReport {
3694    let report = walk_worker_with(
3695        root,
3696        config,
3697        root_dev,
3698        queue,
3699        sender,
3700        diagnostics,
3701        DetachedEmission::new(skip_dir_symlink_stat),
3702    );
3703    // A walker leaves only when the queue is empty with nothing in flight, or when its
3704    // consumer is gone, so the walk is over. The index may still be assembling the
3705    // listings already sent; the counters have stopped, and the phase says why.
3706    if let Some(progress) = &config.progress {
3707        progress.enter(crate::ProgressPhase::Indexing);
3708    }
3709    report
3710}
3711
3712/// One worker's share of the public observation walk.
3713fn walk_worker(
3714    root: &Path,
3715    config: &ScanConfig,
3716    root_dev: u64,
3717    queue: &DirectoryQueue,
3718    sender: &std::sync::mpsc::Sender<WalkMessage>,
3719    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3720) -> ScanReport {
3721    walk_worker_with(
3722        root,
3723        config,
3724        root_dev,
3725        queue,
3726        sender,
3727        diagnostics,
3728        StreamingEmission::for_sink(config, SinkMode::Retained),
3729    )
3730}
3731
3732/// One worker's share of the transient summary walk.
3733fn walk_worker_transient_fold(
3734    root: &Path,
3735    config: &ScanConfig,
3736    root_dev: u64,
3737    queue: &DirectoryQueue,
3738    sender: &std::sync::mpsc::Sender<WalkMessage>,
3739    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3740) -> ScanReport {
3741    walk_worker_with(
3742        root,
3743        config,
3744        root_dev,
3745        queue,
3746        sender,
3747        diagnostics,
3748        StreamingEmission::for_sink(config, SinkMode::TransientFold),
3749    )
3750}
3751
3752fn walk_worker_with<E: WalkEmission>(
3753    root: &Path,
3754    config: &ScanConfig,
3755    root_dev: u64,
3756    queue: &DirectoryQueue,
3757    sender: &std::sync::mpsc::Sender<WalkMessage>,
3758    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3759    mut emission: E,
3760) -> ScanReport {
3761    let _counter_guard = crate::counters::thread_flush_guard();
3762    let _worker_guard = diagnostics.map(ScanDiagnosticsRecorder::worker_guard);
3763    let worker_started = std::time::Instant::now();
3764    let mut report = ScanReport::default();
3765    let mut tally = ProgressTally::new(config.progress.as_ref());
3766    let mut claimed: Vec<(PathBuf, usize, RegionId)> = Vec::with_capacity(DIR_CLAIM);
3767    let mut discovered: Vec<(PathBuf, usize, RegionId)> = Vec::new();
3768    let mut consumer_gone = false;
3769    #[cfg(target_os = "macos")]
3770    let mut bulk_reader = macos_bulk::Reader::new();
3771    #[cfg(all(target_os = "linux", target_env = "gnu"))]
3772    let mut dents_reader = linux_dents::Reader::new();
3773
3774    'walk: while let Some(claim) = queue.claim(&mut claimed, &mut report.attribution) {
3775        // One timing pair per claimed chunk, never per entry: the chunk is the unit
3776        // the amortization argument is made in, so it is the unit the evidence is
3777        // collected in.
3778        let chunk_started = std::time::Instant::now();
3779        let mut chunk_send_ns: u64 = 0;
3780        let entries_before = report.entries;
3781        for (rel_dir, depth, region) in claimed.drain(..) {
3782            let abs_dir = root.join(&rel_dir);
3783            let mut directory = emission.begin_directory(&rel_dir);
3784            #[cfg(target_os = "macos")]
3785            {
3786                if let Some(diagnostics) = diagnostics {
3787                    diagnostics.macos_bulk_attempted();
3788                }
3789                if let Some(entries) =
3790                    (!walk_hook_covers(&abs_dir)).then(|| bulk_reader.read(&abs_dir)).flatten()
3791                {
3792                    if let Some(diagnostics) = diagnostics {
3793                        diagnostics.macos_bulk_succeeded();
3794                    }
3795                    report.dirs_read += 1;
3796                    for entry in entries {
3797                        if !emission.record_entry(
3798                            root,
3799                            &rel_dir,
3800                            depth,
3801                            region,
3802                            &entry.name,
3803                            entry.kind,
3804                            entry.attrs,
3805                            root_dev,
3806                            config,
3807                            &mut directory,
3808                            &mut discovered,
3809                            &mut report,
3810                            sender,
3811                            &mut chunk_send_ns,
3812                            diagnostics.map(AsRef::as_ref),
3813                        ) {
3814                            consumer_gone = true;
3815                            break 'walk;
3816                        }
3817                    }
3818                    emission.finish_directory(directory);
3819                    continue;
3820                }
3821                if let Some(diagnostics) = diagnostics {
3822                    diagnostics.macos_bulk_fell_back();
3823                }
3824            }
3825            // Counted in the Linux backend fields: an attempt, then a success or a
3826            // fallback, and a fallback goes on to count as a portable attempt below, so
3827            // `dirs_read` is the native successes plus the portable reads.
3828            #[cfg(all(target_os = "linux", target_env = "gnu"))]
3829            {
3830                if let Some(diagnostics) = diagnostics {
3831                    diagnostics.linux_dents_attempted();
3832                }
3833                let policy = linux_dents::StatPolicy {
3834                    skip_dir_symlink_stat: emission.skip_dir_symlink_stat(),
3835                    one_filesystem: config.one_filesystem,
3836                };
3837                // Plain `if`, not `bool::then(|| …)`: the listing borrows the reader for the
3838                // loop.
3839                let listing = if walk_hook_covers(&abs_dir) {
3840                    None
3841                } else {
3842                    dents_reader.read(&abs_dir, policy)
3843                };
3844                if let Some(listing) = listing {
3845                    if let Some(diagnostics) = diagnostics {
3846                        diagnostics.linux_dents_succeeded();
3847                    }
3848                    report.dirs_read += 1;
3849                    for entry in listing {
3850                        let (kind, attrs) = match entry.outcome {
3851                            linux_dents::Outcome::Observed { kind, attrs } => (kind, attrs),
3852                            linux_dents::Outcome::Failed(error) => {
3853                                // The same path std's `DirEntry::path` builds: the listing
3854                                // path joined with the name.
3855                                report.errors.push(Error::io(abs_dir.join(entry.name), error));
3856                                continue;
3857                            }
3858                        };
3859                        if !emission.record_entry(
3860                            root,
3861                            &rel_dir,
3862                            depth,
3863                            region,
3864                            entry.name,
3865                            kind,
3866                            attrs,
3867                            root_dev,
3868                            config,
3869                            &mut directory,
3870                            &mut discovered,
3871                            &mut report,
3872                            sender,
3873                            &mut chunk_send_ns,
3874                            diagnostics.map(AsRef::as_ref),
3875                        ) {
3876                            consumer_gone = true;
3877                            break 'walk;
3878                        }
3879                    }
3880                    emission.finish_directory(directory);
3881                    continue;
3882                }
3883                if let Some(diagnostics) = diagnostics {
3884                    diagnostics.linux_dents_fell_back();
3885                }
3886            }
3887
3888            crate::counters::bump(|c| c.dir_opens += 1);
3889            if let Some(diagnostics) = diagnostics {
3890                diagnostics.portable_attempted();
3891            }
3892
3893            let listing = match fs::read_dir(&abs_dir) {
3894                Ok(listing) => {
3895                    if let Some(diagnostics) = diagnostics {
3896                        diagnostics.portable_succeeded();
3897                    }
3898                    listing
3899                }
3900                Err(e) => {
3901                    report.errors.push(Error::io(abs_dir, e));
3902                    continue;
3903                }
3904            };
3905            report.dirs_read += 1;
3906
3907            let policy = ListingPolicy {
3908                skip_dir_symlink_stat: emission.skip_dir_symlink_stat(),
3909                one_filesystem: config.one_filesystem,
3910            };
3911            let mut searchability = Searchability::Unproven;
3912            for item in listing {
3913                let item = match item {
3914                    Ok(item) => item,
3915                    Err(e) => {
3916                        report.errors.push(Error::io(&abs_dir, e));
3917                        continue;
3918                    }
3919                };
3920                crate::counters::bump(|c| c.dir_entries += 1);
3921                let name = item.file_name();
3922                let (kind, attrs) =
3923                    match listed_child_kind_and_attrs(&item, policy, &mut searchability) {
3924                        Ok(Some(observed)) => observed,
3925                        Ok(None) => continue,
3926                        Err(error) => {
3927                            report.errors.push(Error::io(item.path(), error));
3928                            continue;
3929                        }
3930                    };
3931                if !emission.record_entry(
3932                    root,
3933                    &rel_dir,
3934                    depth,
3935                    region,
3936                    &name,
3937                    kind,
3938                    attrs,
3939                    root_dev,
3940                    config,
3941                    &mut directory,
3942                    &mut discovered,
3943                    &mut report,
3944                    sender,
3945                    &mut chunk_send_ns,
3946                    diagnostics.map(AsRef::as_ref),
3947                ) {
3948                    // The consumer is gone; nothing further will be read.
3949                    consumer_gone = true;
3950                    break 'walk;
3951                }
3952            }
3953            emission.finish_directory(directory);
3954        }
3955        // Publish facts that authorize newly discovered directories before making
3956        // those directories claimable. Both emission modes preserve this boundary.
3957        if !emission.publish_before_discovery(
3958            !discovered.is_empty(),
3959            sender,
3960            &mut chunk_send_ns,
3961            diagnostics.map(AsRef::as_ref),
3962        ) {
3963            report.attribution.send_ns += chunk_send_ns;
3964            report.attribution.work_ns += elapsed_ns(chunk_started).saturating_sub(chunk_send_ns);
3965            consumer_gone = true;
3966            break 'walk;
3967        }
3968        report.attribution.send_ns += chunk_send_ns;
3969        let chunk_work_ns = elapsed_ns(chunk_started).saturating_sub(chunk_send_ns);
3970        report.attribution.work_ns += chunk_work_ns;
3971        // Progress is reported per chunk for the same reason timing is: the chunk is
3972        // the unit of handoff, so it is the unit the shared counters are touched in.
3973        tally.flush(&report);
3974
3975        // Publish new work before releasing the claim so a worker that finds nothing
3976        // new does not hold work that others could be doing.
3977        if !discovered.is_empty() {
3978            queue.extend(discovered.drain(..), &mut report.attribution);
3979        }
3980        if let Some(target_workers) = claim.release(
3981            report.entries.saturating_sub(entries_before),
3982            chunk_work_ns,
3983            &mut report.attribution,
3984        ) {
3985            // Carry a sender in-band so the consumer can create the reserve workers
3986            // without retaining a channel endpoint that would keep a small scan alive.
3987            // Only the release that completes a slow calibration returns true, so one
3988            // message expands the pool exactly once.
3989            let _ = sender.send(WalkMessage::ScaleUp { sender: sender.clone(), target_workers });
3990        }
3991    }
3992
3993    if !consumer_gone {
3994        emission.finish(sender, &mut report, diagnostics.map(AsRef::as_ref));
3995    }
3996    // A worker that left mid-chunk because its consumer was gone still read what it
3997    // read, and the report it returns says so.
3998    tally.flush(&report);
3999    report.attribution.wall_ns = elapsed_ns(worker_started);
4000    report
4001}
4002
4003#[allow(clippy::too_many_arguments)]
4004fn record_detached_entry(
4005    root: &Path,
4006    rel_dir: &Path,
4007    depth: usize,
4008    region: RegionId,
4009    name: &OsStr,
4010    kind: EntryKind,
4011    attrs: Attrs,
4012    root_dev: u64,
4013    config: &ScanConfig,
4014    children: &mut Vec<DetachedChild>,
4015    control: &mut Option<Op>,
4016    discovered: &mut Vec<(PathBuf, usize, RegionId)>,
4017    report: &mut ScanReport,
4018) {
4019    let disposition = crate::admission::decide(name, kind, config.hidden(), config.exclude_special);
4020    if disposition == crate::admission::Disposition::Reject {
4021        return;
4022    }
4023    // Construct a full path only for a spelling of the control name. The scanner's public
4024    // preparation builds one for every retained entry because that path escapes in an
4025    // observation; this private builder keeps ordinary children component-only.
4026    if config.read_controls && crate::control::control_spelling(name).is_some() {
4027        let path = rel_dir.join(name);
4028        match read_control_op(config, root, &path, kind) {
4029            // A listing can repeat the control name while the directory changes, and a
4030            // case-sensitive one can list `.gitignore` beside `.GITIGNORE`, whose lookup
4031            // reads the same file. The later read wins, as the builder keeps the later
4032            // observation of the entry.
4033            Ok(observed) => *control = observed,
4034            Err(error) => report.errors.push(error),
4035        }
4036    }
4037    if disposition != crate::admission::Disposition::Retain {
4038        return;
4039    }
4040    report.observe(kind, attrs);
4041    // Positions only order repeated names, and no real listing reaches `u32::MAX` entries.
4042    let position = u32::try_from(children.len()).unwrap_or(u32::MAX);
4043    children.push(DetachedChild { name: name.to_os_string(), kind, attrs, position });
4044    if should_descend(kind, attrs, depth, root_dev, config) {
4045        let child_region = if depth == 0 { RegionId::UNASSIGNED } else { region };
4046        discovered.push((rel_dir.join(name), depth + 1, child_region));
4047    }
4048}
4049
4050/// One filesystem entry after the scan's shared admission, control, and descent rules.
4051pub(crate) struct PreparedWalkEntry {
4052    pub(crate) path: PathBuf,
4053    pub(crate) kind: EntryKind,
4054    pub(crate) attrs: Attrs,
4055    pub(crate) retained: bool,
4056    pub(crate) control: Option<Op>,
4057    pub(crate) descend: bool,
4058    pub(crate) control_error: Option<Error>,
4059}
4060
4061/// Apply the producer-independent part of a directory walk to one verified entry.
4062///
4063/// Both blocking and opened-root scans call this after obtaining non-following metadata,
4064/// which keeps admission, fixed controls, and traversal boundaries from drifting.
4065#[allow(clippy::too_many_arguments)]
4066pub(crate) fn prepare_walk_entry(
4067    root: &Path,
4068    rel_dir: &Path,
4069    depth: usize,
4070    name: &OsStr,
4071    kind: EntryKind,
4072    attrs: Attrs,
4073    root_dev: u64,
4074    config: &ScanConfig,
4075) -> Option<PreparedWalkEntry> {
4076    prepare_walk_entry_reading(root, rel_dir, depth, name, kind, attrs, root_dev, config, true)
4077}
4078
4079/// [`prepare_walk_entry`], reading the entry's control only when `read_control` allows.
4080///
4081/// A grouping emission whose listing already probed its directory's control passes
4082/// `false`, so a `.gitignore` listed afterwards is not read a second time
4083/// (`StreamingEmission::control_probed`).
4084#[allow(clippy::too_many_arguments)]
4085fn prepare_walk_entry_reading(
4086    root: &Path,
4087    rel_dir: &Path,
4088    depth: usize,
4089    name: &OsStr,
4090    kind: EntryKind,
4091    attrs: Attrs,
4092    root_dev: u64,
4093    config: &ScanConfig,
4094    read_control: bool,
4095) -> Option<PreparedWalkEntry> {
4096    let disposition = crate::admission::decide(name, kind, config.hidden(), config.exclude_special);
4097    if disposition == crate::admission::Disposition::Reject {
4098        return None;
4099    }
4100    let path = join_listed_name(rel_dir, name);
4101    let (control, control_error) = if read_control {
4102        match read_named_control_op(config, root, &path, name, kind) {
4103            Ok(control) => (control, None),
4104            Err(error) => (None, Some(error)),
4105        }
4106    } else {
4107        (None, None)
4108    };
4109    Some(PreparedWalkEntry {
4110        path,
4111        kind,
4112        attrs,
4113        retained: disposition == crate::admission::Disposition::Retain,
4114        control,
4115        descend: should_descend(kind, attrs, depth, root_dev, config),
4116        control_error,
4117    })
4118}
4119
4120/// `rel_dir.join(name)`, allocated once at the joined length.
4121///
4122/// [`Path::join`] copies `rel_dir` at its exact length and pushes onto the copy, so the
4123/// separator grows it and, for a name longer than the directory, so does the name: a
4124/// `realloc` for nearly every entry a streaming walk prepares. The summary route's
4125/// walkers spent 760-820 instructions per entry joining, and 410-450 in `realloc`
4126/// against the detached route's 85-150 (H180). The same push onto a copy that already
4127/// fits both makes the same path, byte for byte, on every platform.
4128fn join_listed_name(rel_dir: &Path, name: &OsStr) -> PathBuf {
4129    let mut path = PathBuf::with_capacity(rel_dir.as_os_str().len() + 1 + name.len());
4130    path.as_mut_os_string().push(rel_dir);
4131    path.push(name);
4132    path
4133}
4134
4135#[allow(clippy::too_many_arguments)]
4136fn record_walk_entry(
4137    root: &Path,
4138    rel_dir: &Path,
4139    depth: usize,
4140    region: RegionId,
4141    name: &OsStr,
4142    kind: EntryKind,
4143    attrs: Attrs,
4144    root_dev: u64,
4145    config: &ScanConfig,
4146    emission: &mut StreamingEmission,
4147    discovered: &mut Vec<(PathBuf, usize, RegionId)>,
4148    report: &mut ScanReport,
4149    sender: &std::sync::mpsc::Sender<WalkMessage>,
4150    chunk_send_ns: &mut u64,
4151    diagnostics: Option<&ScanDiagnosticsRecorder>,
4152) -> bool {
4153    let read_control = !emission.control_probed();
4154    let Some(prepared) = prepare_walk_entry_reading(
4155        root,
4156        rel_dir,
4157        depth,
4158        name,
4159        kind,
4160        attrs,
4161        root_dev,
4162        config,
4163        read_control,
4164    ) else {
4165        return true;
4166    };
4167    emission.note_listed_control(prepared.control.as_ref(), prepared.control_error.as_ref());
4168    let (mut control, control_error) = (prepared.control, prepared.control_error);
4169    if let Some(error) = control_error {
4170        report.errors.push(error);
4171    }
4172    // A grouping emission holds an entry's control ahead of the entry itself, so no send
4173    // can take the entry, `.gitignore` included, before the control that governs it.
4174    if !prepared.retained || emission.group_directories {
4175        if let Some(control) = control.take() {
4176            emission.batch.push(ObservationOp::unconditional(control));
4177            if !emission.send_if_full(
4178                root,
4179                rel_dir,
4180                config,
4181                report,
4182                sender,
4183                chunk_send_ns,
4184                diagnostics,
4185            ) {
4186                return false;
4187            }
4188        }
4189        if !prepared.retained {
4190            return true;
4191        }
4192    }
4193    report.observe(kind, attrs);
4194    // Only a directory the walk descends into needs its path twice, in its observation
4195    // and in the queue. Every other entry's path moves into its observation, so it is
4196    // allocated once and freed with its batch rather than copied and freed at once
4197    // (H180).
4198    let (path, descend_path) = if prepared.descend {
4199        (prepared.path.clone(), Some(prepared.path))
4200    } else {
4201        (prepared.path, None)
4202    };
4203    emission.batch.push(ObservationOp::unconditional(Op::Upsert { path, kind, attrs }));
4204    if !emission.send_if_full(root, rel_dir, config, report, sender, chunk_send_ns, diagnostics) {
4205        return false;
4206    }
4207    if let Some(control) = control {
4208        emission.batch.push(ObservationOp::unconditional(control));
4209        if !emission.send_if_full(root, rel_dir, config, report, sender, chunk_send_ns, diagnostics)
4210        {
4211            return false;
4212        }
4213    }
4214    if let Some(path) = descend_path {
4215        // A child of the root seeds a new region; everything deeper inherits its
4216        // parent's. Region membership therefore costs one integer copy and never
4217        // inspects a path.
4218        let child_region = if depth == 0 { RegionId::UNASSIGNED } else { region };
4219        discovered.push((path, depth + 1, child_region));
4220    }
4221    true
4222}
4223
4224/// Move one listing's control observations ahead of its entries, each kind in its order.
4225///
4226/// A listing that repeats `.gitignore` while the directory changes keeps its reads in
4227/// order, so the later one still wins, as it does in the detached builder. Only the rare
4228/// listing that holds a control moves; every other listing costs one pass over the tags
4229/// it has just written.
4230fn controls_first(listing: &mut [ObservationOp]) {
4231    let mut placed = 0;
4232    for position in 0..listing.len() {
4233        if matches!(listing[position].op, Op::ControlUpsert { .. } | Op::ControlRemove { .. }) {
4234            listing[placed..=position].rotate_right(1);
4235            placed += 1;
4236        }
4237    }
4238}
4239
4240/// Observe the control an entry at `path` stands for, if the scan's policy asks for
4241/// control state at all.
4242///
4243/// Every control observation goes through here -- each walk and reconcile site, and the
4244/// watch layer's verification -- or through [`read_named_control_op`], which shares its
4245/// gate, so the policy cannot be forgotten at one of them. A watch must honor it like a
4246/// scan does: its scope has to equal the index's, the scope carries this bit, and a
4247/// verifier that read control files regardless would grow a partial rule set, from
4248/// whichever sources events touched, under a scope that says there is none.
4249///
4250/// It is also where a listed name becomes a control read, by the rule the module
4251/// documentation of [`crate::control`] states: the directory's control is what a lookup
4252/// of `<dir>/.gitignore` resolves to. An entry named exactly `.gitignore` is that
4253/// lookup's target on every filesystem, so it is read through its own path with the kind
4254/// its listing observed, and costs nothing more than it did. A case variant such as
4255/// `.GITIGNORE` may or may not be the target, so the canonical path is looked up
4256/// ([`read_directory_control`]): the entry decides only whether to look. The observation
4257/// names the canonical path either way, and its bytes are the ones git reads. A name
4258/// that spells no control is answered without a system call.
4259pub(crate) fn read_control_op(
4260    config: &ScanConfig,
4261    root: &Path,
4262    path: &Path,
4263    kind: EntryKind,
4264) -> Result<Option<Op>> {
4265    read_spelled_control_op(config, root, path, kind, crate::control::path_control_spelling)
4266}
4267
4268/// [`read_control_op`] for the entry `name` a listing just produced, at `path`, the
4269/// listed directory joined with `name`.
4270///
4271/// The walker already holds the name, so its spelling is tested on those bytes: a length
4272/// comparison for nearly every entry, as in the detached builder. Parsing the last
4273/// component back out of the joined path cost the transient summary about 270
4274/// instructions for every entry, on a tree with no `.gitignore` as much as on one with
4275/// many (H180). A listed name is one normal component, so it is the path's last one and
4276/// the decision is the same.
4277fn read_named_control_op(
4278    config: &ScanConfig,
4279    root: &Path,
4280    path: &Path,
4281    name: &OsStr,
4282    kind: EntryKind,
4283) -> Result<Option<Op>> {
4284    debug_assert_eq!(path.file_name(), Some(name), "a listed name ends its path");
4285    read_spelled_control_op(config, root, path, kind, |_| crate::control::control_spelling(name))
4286}
4287
4288/// [`read_control_op`], with the spelling of `path`'s last component found by `spelling`
4289/// once the policy allows a read at all.
4290fn read_spelled_control_op(
4291    config: &ScanConfig,
4292    root: &Path,
4293    path: &Path,
4294    kind: EntryKind,
4295    spelling: impl FnOnce(&Path) -> Option<crate::control::ControlSpelling>,
4296) -> Result<Option<Op>> {
4297    if !config.read_controls {
4298        return Ok(None);
4299    }
4300    match spelling(path) {
4301        Some(crate::control::ControlSpelling::Exact) => {
4302            read_control_op_unconditional(root, path, kind, config.control_limits.budget)
4303        }
4304        Some(crate::control::ControlSpelling::Variant) => {
4305            read_directory_control(config, root, &crate::control::sibling_control_path(path))
4306        }
4307        None => Ok(None),
4308    }
4309}
4310
4311/// Read a directory's control by looking up its canonical path, `<dir>/.gitignore`.
4312///
4313/// This is the rule itself: on a case-insensitive directory the lookup resolves to
4314/// whichever spelling the directory stores, as git's open does, and on a case-sensitive
4315/// one only to the exact name. `Ok(None)` means nothing resolves. It costs one metadata
4316/// lookup, and a read on a hit, so it runs only where a listing cannot stand in for it: a
4317/// narrowed population reads each directory's control before listing it, a classifying
4318/// transient fold whose batch fills first probes for it (`StreamingEmission::send_if_full`),
4319/// and a listed case variant resolves through it ([`read_control_op`]).
4320fn read_directory_control(
4321    config: &ScanConfig,
4322    root: &Path,
4323    control_path: &Path,
4324) -> Result<Option<Op>> {
4325    if !config.read_controls {
4326        return Ok(None);
4327    }
4328    let absolute = control_lookup_path(root, control_path);
4329    let found = look_up_control(&absolute, control_path, config.control_limits.budget);
4330    #[cfg(test)]
4331    {
4332        if let Some(error) =
4333            walk_hook(&absolute).and_then(|hook| hook(WalkHookPoint::ControlLookup(&absolute)))
4334        {
4335            return Err(Error::io(&absolute, error));
4336        }
4337    }
4338    found
4339}
4340
4341/// The lookup [`read_directory_control`] makes, at `absolute`, of the control it names
4342/// `control_path`.
4343fn look_up_control(
4344    absolute: &Path,
4345    control_path: &Path,
4346    budget: Option<usize>,
4347) -> Result<Option<Op>> {
4348    let kind = match observe_path(absolute) {
4349        Ok((EntryKind::File, _)) => EntryKind::File,
4350        Ok(_) => EntryKind::Other,
4351        Err(error) if error.kind() == std::io::ErrorKind::NotFound => return Ok(None),
4352        Err(error) => return Err(Error::io(absolute, error)),
4353    };
4354    read_control_source(absolute, control_path, kind, budget)
4355}
4356
4357/// [`read_directory_control`], answering a miss with the removal of the directory's rules.
4358///
4359/// For a producer that saw one spelling of the control name vanish or fail: whether the
4360/// directory still has a control is the lookup's question, and a removal of rules the
4361/// table does not hold is inert.
4362pub(crate) fn read_directory_control_or_removal(
4363    config: &ScanConfig,
4364    root: &Path,
4365    control_path: &Path,
4366) -> Result<Option<Op>> {
4367    if !config.read_controls {
4368        return Ok(None);
4369    }
4370    Ok(Some(
4371        read_directory_control(config, root, control_path)?
4372            .unwrap_or_else(|| Op::ControlRemove { path: control_path.to_path_buf() }),
4373    ))
4374}
4375
4376/// Where a lookup of the control path `control_path` under `root` goes.
4377#[cfg(not(test))]
4378fn control_lookup_path(root: &Path, control_path: &Path) -> PathBuf {
4379    root.join(control_path)
4380}
4381
4382/// Where a lookup of the control path `control_path` under `root` goes, which a test may
4383/// resolve as a case-insensitive directory would ([`install_case_folding_control_lookup`]).
4384#[cfg(test)]
4385fn control_lookup_path(root: &Path, control_path: &Path) -> PathBuf {
4386    let absolute = root.join(control_path);
4387    let folds = CASE_FOLDING_LOOKUPS
4388        .read()
4389        .unwrap_or_else(std::sync::PoisonError::into_inner)
4390        .iter()
4391        .any(|folded| absolute.starts_with(folded));
4392    let Some(directory) = absolute.parent().filter(|_| folds) else {
4393        return absolute;
4394    };
4395    // A case-insensitive directory holds at most one spelling, and the lookup returns it.
4396    // A test tree may hold two only where the host is case-sensitive, and then the exact
4397    // name is what the host itself resolves.
4398    if fs::symlink_metadata(&absolute).is_ok() {
4399        return absolute;
4400    }
4401    let Ok(mut names) = fs::read_dir(directory) else {
4402        return absolute;
4403    };
4404    names
4405        .find_map(|item| {
4406            let name = item.ok()?.file_name();
4407            crate::control::control_spelling(&name).map(|_| directory.join(name))
4408        })
4409        .unwrap_or(absolute)
4410}
4411
4412/// Roots whose control lookups [`control_lookup_path`] resolves case-insensitively.
4413#[cfg(test)]
4414static CASE_FOLDING_LOOKUPS: std::sync::RwLock<Vec<PathBuf>> = std::sync::RwLock::new(Vec::new());
4415
4416/// Removes its root from [`CASE_FOLDING_LOOKUPS`] when dropped.
4417#[cfg(test)]
4418#[must_use = "the lookups fold only until the guard is dropped"]
4419pub(crate) struct CaseFoldingGuard(Vec<PathBuf>);
4420
4421#[cfg(test)]
4422impl Drop for CaseFoldingGuard {
4423    fn drop(&mut self) {
4424        CASE_FOLDING_LOOKUPS
4425            .write()
4426            .unwrap_or_else(std::sync::PoisonError::into_inner)
4427            .retain(|root| !self.0.contains(root));
4428    }
4429}
4430
4431/// Resolve every control lookup under `root` as a case-insensitive directory would, until
4432/// the guard drops: `<dir>/.gitignore` opens whichever spelling `<dir>` stores.
4433///
4434/// A case-sensitive test host cannot make a case-insensitive directory (an ext4 casefold
4435/// directory needs a kernel built with Unicode support and an empty directory to flag), so
4436/// this is how the rule's case-insensitive branch runs on every CI runner. It changes only
4437/// the lookup the rule depends on: listings still show stored names, and every other path
4438/// resolves as the host resolves it. A real case-insensitive volume is tested where the
4439/// temporary directory is one. `root` matches as given and canonical.
4440#[cfg(test)]
4441pub(crate) fn install_case_folding_control_lookup(root: &Path) -> CaseFoldingGuard {
4442    let mut roots = vec![root.to_path_buf()];
4443    if let Ok(canonical) = root.canonicalize() {
4444        if canonical != root {
4445            roots.push(canonical);
4446        }
4447    }
4448    CASE_FOLDING_LOOKUPS
4449        .write()
4450        .unwrap_or_else(std::sync::PoisonError::into_inner)
4451        .extend(roots.iter().cloned());
4452    CaseFoldingGuard(roots)
4453}
4454
4455fn read_listed_control_op(
4456    config: &ScanConfig,
4457    root: &Path,
4458    path: &Path,
4459    kind: EntryKind,
4460) -> Result<Option<Op>> {
4461    if config.population != crate::query::IgnoredEntries::Include
4462        && crate::control::path_control_spelling(path).is_some()
4463    {
4464        // The exclusion walk already looked the directory's control up before any
4465        // sibling, whichever spelling holds it.
4466        Ok(None)
4467    } else {
4468        read_control_op(config, root, path, kind)
4469    }
4470}
4471
4472fn population_prunes(
4473    population: crate::query::IgnoredEntries,
4474    path: &Path,
4475    kind: EntryKind,
4476    disposition: crate::admission::Disposition,
4477    controls: Option<&crate::control::ControlTable>,
4478    unreadable_controls: &std::collections::BTreeSet<PathBuf>,
4479) -> bool {
4480    // The fixed control source stays in the entry tier even for `only`: removing its
4481    // entry would also remove the retained rule source during a watch reconciliation.
4482    // Every spelling is kept, because the listing cannot tell which one a case-insensitive
4483    // directory resolves `.gitignore` to without the lookup the walk made before it.
4484    if crate::control::path_control_spelling(path).is_some() {
4485        return false;
4486    }
4487    let Some(table) = controls else { return false };
4488    if !table.classification_known(path)
4489        || path.ancestors().skip(1).any(|ancestor| unreadable_controls.contains(ancestor))
4490    {
4491        return false;
4492    }
4493    let ignored = table.is_ignored(path, kind.is_dir());
4494    match population {
4495        crate::query::IgnoredEntries::Include => false,
4496        crate::query::IgnoredEntries::Exclude => ignored,
4497        crate::query::IgnoredEntries::Only => {
4498            !ignored && !kind.is_dir() && disposition != crate::admission::Disposition::ControlOnly
4499        }
4500    }
4501}
4502
4503fn apply_discovery_control(table: &mut crate::control::ControlTable, op: &Op) -> Result<()> {
4504    match op {
4505        Op::ControlUpsert { path, source } => {
4506            table.upsert(path, source.clone())?;
4507        }
4508        Op::ControlRemove { path } => {
4509            table.remove(path)?;
4510        }
4511        _ => unreachable!("directory control probe emits only control operations"),
4512    }
4513    Ok(())
4514}
4515
4516/// Read one fixed control source without allowing a raced or hostile file to allocate
4517/// beyond the index-wide control budget.
4518///
4519/// A file longer than the budget is read only to one byte past it. No table under that
4520/// budget can admit a source that long, since every retained byte is charged at least
4521/// once, so the truncated source it sends is refused for the budget rather than parsed.
4522///
4523/// Private to this module, so no caller elsewhere can step around the policy gate in
4524/// `read_control_op`.
4525fn read_control_op_unconditional(
4526    root: &Path,
4527    path: &Path,
4528    kind: EntryKind,
4529    budget: Option<usize>,
4530) -> Result<Option<Op>> {
4531    if !crate::control::is_control_file(path) {
4532        return Ok(None);
4533    }
4534    read_control_source(&root.join(path), path, kind, budget)
4535}
4536
4537/// The most a control file's read buffer is sized from its metadata length (H189): every
4538/// `.gitignore` in the nominated subjects is under 16 KiB, and a larger one grows its
4539/// buffer by reading, as any file did before.
4540const CONTROL_READ_RESERVE_BYTES: usize = 64 * 1024;
4541
4542/// Read the control at `absolute`, whose kind is `kind`, as an observation of `path`.
4543///
4544/// `absolute` is where the lookup went and `path` the canonical control path the
4545/// observation names; on a real filesystem the first is `root.join(path)`.
4546fn read_control_source(
4547    absolute: &Path,
4548    path: &Path,
4549    kind: EntryKind,
4550    budget: Option<usize>,
4551) -> Result<Option<Op>> {
4552    if kind != EntryKind::File {
4553        return Ok(Some(Op::ControlRemove { path: path.to_path_buf() }));
4554    }
4555    let file = open_control_file(absolute).map_err(|error| Error::io(absolute, error))?;
4556    let metadata = file.metadata().map_err(|error| Error::io(absolute, error))?;
4557    if !metadata.file_type().is_file() {
4558        return Ok(Some(Op::ControlRemove { path: path.to_path_buf() }));
4559    }
4560    let read_limit = budget
4561        .map_or(u64::MAX, |budget| u64::try_from(budget).unwrap_or(u64::MAX).saturating_add(1));
4562    // Room for the length the metadata already carries, within the read limit and a
4563    // fixed bound, plus the byte `read_to_end` reads to see the end (H189). `take` hides
4564    // the length from `read_to_end`, which otherwise grows from 32 bytes by doubling: eight
4565    // `read` calls for the 2 KiB root `.gitignore` of a kernel tree, two now. A file that
4566    // grew since its stat reads on as before; a longer one is bounded by the limit as
4567    // before, and its buffer is grown by `read_to_end` past the reservation, not sized
4568    // from its length.
4569    let reserve = usize::try_from(metadata.len().min(read_limit))
4570        .unwrap_or(usize::MAX)
4571        .min(CONTROL_READ_RESERVE_BYTES)
4572        .saturating_add(1);
4573    let mut source = Vec::with_capacity(reserve);
4574    file.take(read_limit).read_to_end(&mut source).map_err(|error| Error::io(absolute, error))?;
4575    crate::counters::bump(|counts| counts.control_reads = counts.control_reads.saturating_add(1));
4576    Ok(Some(Op::ControlUpsert { path: path.to_path_buf(), source }))
4577}
4578
4579#[cfg(unix)]
4580fn open_control_file(path: &Path) -> std::io::Result<fs::File> {
4581    use std::os::unix::fs::OpenOptionsExt as _;
4582
4583    fs::OpenOptions::new().read(true).custom_flags(libc::O_NONBLOCK | libc::O_NOFOLLOW).open(path)
4584}
4585
4586#[cfg(not(unix))]
4587fn open_control_file(path: &Path) -> std::io::Result<fs::File> {
4588    fs::File::open(path)
4589}
4590
4591fn send_scanner_batch(
4592    sender: &std::sync::mpsc::Sender<WalkMessage>,
4593    batch: ScannerBatch,
4594    diagnostics: Option<&ScanDiagnosticsRecorder>,
4595) -> bool {
4596    if let Some(diagnostics) = diagnostics {
4597        diagnostics.handoff_sent();
4598    }
4599    let sent = sender.send(WalkMessage::Batch(batch)).is_ok();
4600    if !sent {
4601        // Balance the reservation when the receiver disappeared before accepting it.
4602        if let Some(diagnostics) = diagnostics {
4603            diagnostics.handoff_received();
4604        }
4605    }
4606    sent
4607}
4608
4609fn send_detached_directories(
4610    sender: &std::sync::mpsc::Sender<WalkMessage>,
4611    directories: Vec<DetachedDirectory>,
4612    recycle: std::sync::mpsc::Sender<Vec<DetachedDirectory>>,
4613    diagnostics: Option<&ScanDiagnosticsRecorder>,
4614) -> bool {
4615    if let Some(diagnostics) = diagnostics {
4616        diagnostics.handoff_sent();
4617    }
4618    let sent = sender.send(WalkMessage::DetachedDirectories { directories, recycle }).is_ok();
4619    if !sent {
4620        if let Some(diagnostics) = diagnostics {
4621            diagnostics.handoff_received();
4622        }
4623    }
4624    sent
4625}
4626
4627/// Nanoseconds since `started`, saturating rather than panicking on the absurd.
4628fn elapsed_ns(started: std::time::Instant) -> u64 {
4629    u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)
4630}
4631
4632/// A top-level subtree, used to spread workers across the breadth of the tree.
4633///
4634/// Every directory below the root belongs to the region seeded by its depth-1
4635/// ancestor, inherited from its parent rather than recomputed from its path. The root
4636/// itself is [`RegionId::ROOT`], which exists only to bootstrap.
4637#[derive(Clone, Copy, PartialEq, Eq, Debug)]
4638struct RegionId(usize);
4639
4640impl RegionId {
4641    /// The root's own region, which exists only to bootstrap the walk.
4642    const ROOT: Self = Self(0);
4643    /// "Allocate a fresh region for this directory." Resolved by
4644    /// [`DirectoryQueueState::push`], which is the only place holding the lock that
4645    /// owns the region table.
4646    const UNASSIGNED: Self = Self(usize::MAX);
4647}
4648
4649/// Directories still to read, plus enough state to know when the walk is finished.
4650///
4651/// The termination condition is the only subtle part: the queue being empty does not
4652/// mean the walk is done, because a worker that is mid-directory may be about to push
4653/// its children. So a worker holds a claim from the moment it takes work until the
4654/// moment it has published everything that work produced, and the walk ends only when
4655/// the queue is empty *and* no claim is outstanding.
4656///
4657/// # Why breadth-first is region-scheduled rather than a global FIFO
4658///
4659/// A global FIFO orders the *queue*, but claims are unordered: workers take whatever
4660/// is at the front, which on a real tree means several workers grinding through the
4661/// same top-level subtree while others sit untouched. Measured on the branching
4662/// fixture, that left a global-FIFO walk starting the same 7–8 of 12 subtrees at the
4663/// halfway mark as depth-first did — the ordering bought nothing a consumer could see.
4664/// It also made the pending set hold a whole level of the tree, which is where the
4665/// +1.5–3.7% peak RSS in exp-012 came from.
4666///
4667/// So breadth-first keeps work in per-region buckets and hands each free worker a
4668/// *different* region, round-robin. Within a region the bucket is LIFO, which restores
4669/// depth-first's locality and spine-bounded memory. Nothing waits on a level boundary:
4670/// if only one region has work, every worker takes it. The result is a scheduler whose
4671/// shallow preference is expressed in *which subtree a worker picks up*, not in the
4672/// order a single queue drains — which is the property progressive consumers actually
4673/// need.
4674struct DirectoryQueue {
4675    state: std::sync::Mutex<DirectoryQueueState>,
4676    ready: std::sync::Condvar,
4677    order: ScanOrder,
4678    diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
4679}
4680
4681/// One outstanding claim, held for exactly as long as the worker owes the queue the
4682/// work it took.
4683///
4684/// Giving the claim back is the queue's liveness condition, not a courtesy: [`claim`]
4685/// parks every other worker on the condvar while `outstanding` is nonzero, so a single
4686/// claim that is never returned stops the whole walk and the scoped join that waits on
4687/// it. That makes `Drop` the only safe place to put the release, because the paths that
4688/// skip a hand-written call are exactly the ones that matter — the `break` taken when
4689/// the consumer disconnects, and an unwinding panic inside a directory read.
4690///
4691/// [`claim`]: DirectoryQueue::claim
4692struct DirectoryClaim<'a> {
4693    queue: &'a DirectoryQueue,
4694    /// Directories represented by the claim, for exact in-flight accounting.
4695    directories: usize,
4696    /// Whether the worker already returned this claim through [`Self::release`].
4697    released: bool,
4698}
4699
4700impl DirectoryClaim<'_> {
4701    /// Return the claim at the end of a completed chunk, feeding the chunk's own
4702    /// measurements to the shared calibration.
4703    ///
4704    /// Returns the queue's scale-up decision, which is why the normal path cannot be
4705    /// `Drop`: a destructor has neither the chunk's timing nor anywhere to put an
4706    /// answer.
4707    fn release(
4708        mut self,
4709        entries: u64,
4710        work_ns: u64,
4711        timing: &mut WalkAttribution,
4712    ) -> Option<usize> {
4713        self.released = true;
4714        self.queue.release(self.directories, entries, work_ns, timing)
4715    }
4716}
4717
4718impl Drop for DirectoryClaim<'_> {
4719    fn drop(&mut self) {
4720        if self.released {
4721            return;
4722        }
4723        // An abandoned chunk: the consumer went away mid-directory, or a read panicked.
4724        // Either way the partial timing describes an aborted chunk rather than the cost
4725        // of reading directories, so it must not reach the calibration that sizes the
4726        // worker pool. Returning the claim is the whole job.
4727        self.queue.abandon(self.directories);
4728    }
4729}
4730
4731struct DirectoryQueueState {
4732    /// Depth-first's single stack. Unused under breadth-first.
4733    pending: VecDeque<(PathBuf, usize, RegionId)>,
4734    /// Breadth-first's per-region work, indexed by [`RegionId`]. Each is a LIFO stack.
4735    regions: Vec<Vec<(PathBuf, usize, RegionId)>>,
4736    /// Regions with work, in round-robin order. A region appears at most once; the
4737    /// flag array is what keeps that true without scanning the ring.
4738    ready_ring: VecDeque<RegionId>,
4739    /// Whether each region is currently in `ready_ring`.
4740    enqueued: Vec<bool>,
4741    /// Directories currently available for a future claim.
4742    ready_directories: usize,
4743    /// Directories held by outstanding claims.
4744    in_flight_directories: usize,
4745    outstanding: usize,
4746    finished: bool,
4747    controller: Option<WorkerController>,
4748    /// Observation-only windows retained after the shipped one-shot decision.
4749    shadow_calibration: Option<RepeatedCalibration>,
4750    /// Completion-order sequence assigned under the queue lock.
4751    next_policy_sequence: u64,
4752    worker_target: usize,
4753    maximum_workers: usize,
4754}
4755
4756impl DirectoryQueueState {
4757    fn seeded(
4758        root: (PathBuf, usize),
4759        order: ScanOrder,
4760        calibration: Option<WorkerCalibration>,
4761        initial_workers: usize,
4762        maximum_workers: usize,
4763        policy: WorkerPolicyExperiment,
4764    ) -> Self {
4765        let mut state = Self {
4766            pending: VecDeque::new(),
4767            regions: Vec::new(),
4768            ready_ring: VecDeque::new(),
4769            enqueued: Vec::new(),
4770            ready_directories: 0,
4771            in_flight_directories: 0,
4772            outstanding: 0,
4773            finished: false,
4774            controller: calibration.map(|value| WorkerController::new(value, policy)),
4775            shadow_calibration: None,
4776            next_policy_sequence: 0,
4777            worker_target: initial_workers,
4778            maximum_workers,
4779        };
4780        state.push((root.0, root.1, RegionId::ROOT), order);
4781        state
4782    }
4783
4784    /// Push one directory into the structure the order uses.
4785    fn push(&mut self, item: (PathBuf, usize, RegionId), order: ScanOrder) {
4786        self.ready_directories = self.ready_directories.saturating_add(1);
4787        match order {
4788            ScanOrder::DepthFirst => self.pending.push_back(item),
4789            ScanOrder::BreadthFirst => {
4790                let region = if item.2 == RegionId::UNASSIGNED {
4791                    // One region per top-level subtree, numbered as they are found.
4792                    self.regions.len().max(1)
4793                } else {
4794                    item.2.0
4795                };
4796                if region >= self.regions.len() {
4797                    self.regions.resize_with(region + 1, Vec::new);
4798                    self.enqueued.resize(region + 1, false);
4799                }
4800                // Resolve the id *into* the item, so every directory discovered beneath
4801                // this one inherits a concrete region instead of the sentinel. Without
4802                // this the sentinel propagates and each directory allocates a region of
4803                // its own, degenerating the scheduler into round-robin over the whole
4804                // frontier.
4805                let mut item = item;
4806                item.2 = RegionId(region);
4807                self.regions[region].push(item);
4808                if !self.enqueued[region] {
4809                    self.enqueued[region] = true;
4810                    self.ready_ring.push_back(RegionId(region));
4811                }
4812            }
4813        }
4814    }
4815
4816    /// Whether any work is available.
4817    fn is_empty(&self, order: ScanOrder) -> bool {
4818        match order {
4819            ScanOrder::DepthFirst => self.pending.is_empty(),
4820            ScanOrder::BreadthFirst => self.ready_ring.is_empty(),
4821        }
4822    }
4823
4824    /// Take up to `limit` directories from the next region in the round-robin ring.
4825    ///
4826    /// Every region holding work is in the ring exactly once, so popping it always
4827    /// finds work and always moves to a *different* subtree than the previous claim.
4828    /// An earlier version preferred the caller's previous region for locality, which
4829    /// pinned each worker to one subtree: with twelve deep chains and six workers only
4830    /// six subtrees ever advanced, and depth-first — whose four-directory claims
4831    /// happen to fan across the root's children — spread wider than breadth-first did.
4832    /// Locality still comes from the claim being a run of directories out of one
4833    /// region; it must not come from a worker refusing to leave.
4834    fn take(
4835        &mut self,
4836        limit: usize,
4837        order: ScanOrder,
4838        into: &mut Vec<(PathBuf, usize, RegionId)>,
4839    ) -> usize {
4840        let before = into.len();
4841        match order {
4842            ScanOrder::DepthFirst => {
4843                let take = self.pending.len().min(limit);
4844                let start = self.pending.len() - take;
4845                into.extend(self.pending.drain(start..));
4846            }
4847            ScanOrder::BreadthFirst => {
4848                let Some(region) = self.ready_ring.pop_front() else { return 0 };
4849                self.enqueued[region.0] = false;
4850                let bucket = &mut self.regions[region.0];
4851                let take = bucket.len().min(limit);
4852                let start = bucket.len() - take;
4853                into.extend(bucket.drain(start..));
4854                // Re-arm the region only if work remains and it is not already queued,
4855                // so a busy region cannot appear twice and starve the others.
4856                if !bucket.is_empty() && !self.enqueued[region.0] {
4857                    self.enqueued[region.0] = true;
4858                    self.ready_ring.push_back(region);
4859                }
4860            }
4861        }
4862        into.len().saturating_sub(before)
4863    }
4864
4865    fn allocate_policy_sequence(&mut self) -> u64 {
4866        let sequence = self.next_policy_sequence;
4867        self.next_policy_sequence = self.next_policy_sequence.saturating_add(1);
4868        sequence
4869    }
4870}
4871
4872impl DirectoryQueue {
4873    /// Seed the queue with the root, which is region zero until its children fan out.
4874    #[cfg(test)]
4875    fn new(
4876        root: (PathBuf, usize),
4877        order: ScanOrder,
4878        calibration: Option<WorkerCalibration>,
4879        diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
4880    ) -> Self {
4881        Self::new_with_policy(
4882            root,
4883            order,
4884            calibration,
4885            diagnostics,
4886            1,
4887            2,
4888            WorkerPolicyExperiment::ShippedOneShot,
4889        )
4890    }
4891
4892    #[allow(clippy::too_many_arguments)]
4893    fn new_with_policy(
4894        root: (PathBuf, usize),
4895        order: ScanOrder,
4896        calibration: Option<WorkerCalibration>,
4897        diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
4898        initial_workers: usize,
4899        maximum_workers: usize,
4900        policy: WorkerPolicyExperiment,
4901    ) -> Self {
4902        let state = DirectoryQueueState::seeded(
4903            root,
4904            order,
4905            calibration,
4906            initial_workers,
4907            maximum_workers,
4908            policy,
4909        );
4910        Self {
4911            state: std::sync::Mutex::new(state),
4912            ready: std::sync::Condvar::new(),
4913            order,
4914            diagnostics,
4915        }
4916    }
4917
4918    /// Take up to [`DIR_CLAIM`] directories, blocking until there is work or the walk
4919    /// is over. Returns `None` once no more work will ever arrive.
4920    ///
4921    /// Time spent waiting is charged to `timing`: lock acquisition to `lock_wait_ns`
4922    /// when contended, condvar waits to `starved_ns`. The condvar span includes the
4923    /// lock re-acquisition on wake, which slightly overstates starvation rather than
4924    /// understating contention — the fail-honest direction for the number that is
4925    /// supposed to stay near zero.
4926    fn claim<'a>(
4927        &'a self,
4928        into: &mut Vec<(PathBuf, usize, RegionId)>,
4929        timing: &mut WalkAttribution,
4930    ) -> Option<DirectoryClaim<'a>> {
4931        let mut state = self.lock_timed(timing);
4932        loop {
4933            if !state.is_empty(self.order) {
4934                let directories = state.take(DIR_CLAIM, self.order, into);
4935                state.ready_directories = state.ready_directories.saturating_sub(directories);
4936                state.in_flight_directories =
4937                    state.in_flight_directories.saturating_add(directories);
4938                state.outstanding += 1;
4939                timing.claims += 1;
4940                return Some(DirectoryClaim { queue: self, directories, released: false });
4941            }
4942            if state.finished {
4943                return None;
4944            }
4945            if state.outstanding == 0 {
4946                state.finished = true;
4947                self.ready.notify_all();
4948                return None;
4949            }
4950            let started = std::time::Instant::now();
4951            state = self.ready.wait(state).unwrap_or_else(std::sync::PoisonError::into_inner);
4952            timing.starved_ns += elapsed_ns(started);
4953        }
4954    }
4955
4956    fn extend(
4957        &self,
4958        directories: impl Iterator<Item = (PathBuf, usize, RegionId)>,
4959        timing: &mut WalkAttribution,
4960    ) {
4961        let mut state = self.lock_timed(timing);
4962        for item in directories {
4963            state.push(item, self.order);
4964        }
4965        drop(state);
4966        self.ready.notify_all();
4967    }
4968
4969    /// Give up a claim whose chunk never finished. Wakes everyone if it was the last.
4970    ///
4971    /// Reached only from [`DirectoryClaim::drop`], where there is no `WalkAttribution`
4972    /// to charge and nothing worth charging: an abandoned chunk read some unknown
4973    /// fraction of its directories, so its lock wait says nothing about contention
4974    /// during the walk.
4975    fn abandon(&self, directories: usize) {
4976        let mut state = self.lock();
4977        state.outstanding -= 1;
4978        state.in_flight_directories = state.in_flight_directories.saturating_sub(directories);
4979        if state.outstanding == 0 && state.is_empty(self.order) {
4980            state.finished = true;
4981            drop(state);
4982            self.ready.notify_all();
4983        }
4984    }
4985
4986    /// Give up a claim taken by [`claim`]. Wakes everyone if this was the last one.
4987    ///
4988    /// The chunk's own entry count and work time feed the shared service-time
4989    /// calibration, so the decision uses the timing the walk already collects for
4990    /// attribution rather than a second clock. Returns the new worker target when a
4991    /// controller requests expansion. A release that ends the walk returns no target,
4992    /// because there is no longer useful work for a reserve worker to take.
4993    fn release(
4994        &self,
4995        directories: usize,
4996        observed_entries: u64,
4997        observed_work_ns: u64,
4998        timing: &mut WalkAttribution,
4999    ) -> Option<usize> {
5000        let mut state = self.lock_timed(timing);
5001        // Whether this chunk reached the calibration at all. Only chunks released while
5002        // it is still live are policy history; later ones are ordinary walk work.
5003        let calibrating = state.controller.is_some();
5004        let one_shot = state.controller.as_ref().is_some_and(WorkerController::is_one_shot);
5005        let controller_spec = state.controller.as_ref().map(WorkerController::calibration_spec);
5006        let staged_gated = state.controller.as_ref().is_some_and(WorkerController::is_staged_gated);
5007        let completed_window = state
5008            .controller
5009            .as_mut()
5010            .and_then(|value| value.observe(observed_entries, observed_work_ns));
5011        let observed_window = state
5012            .shadow_calibration
5013            .as_mut()
5014            .and_then(|value| value.observe(observed_entries, observed_work_ns));
5015        let window_completed = completed_window.is_some();
5016        state.outstanding -= 1;
5017        state.in_flight_directories = state.in_flight_directories.saturating_sub(directories);
5018        let finished = state.outstanding == 0 && state.is_empty(self.order);
5019        if finished {
5020            state.finished = true;
5021        }
5022        let handoff_backlog = self.diagnostics.as_ref().map_or(0, |diagnostics| {
5023            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
5024        });
5025        let mut requested_workers = None;
5026        let policy_window = completed_window.map(|window| {
5027            let useful_frontier =
5028                state.ready_directories.saturating_add(state.in_flight_directories);
5029            let decision = if !window.slow {
5030                WorkerPolicyDecision::Hold
5031            } else if finished {
5032                WorkerPolicyDecision::HoldNoUsefulWork
5033            } else if staged_gated && useful_frontier <= state.worker_target {
5034                WorkerPolicyDecision::HoldInsufficientFrontier
5035            } else if staged_gated && handoff_backlog >= state.worker_target {
5036                WorkerPolicyDecision::HoldHandoffBacklog
5037            } else {
5038                let target = if staged_gated {
5039                    state.worker_target.saturating_mul(2).min(state.maximum_workers)
5040                } else {
5041                    state.maximum_workers
5042                };
5043                if target > state.worker_target {
5044                    state.worker_target = target;
5045                    requested_workers = Some(target);
5046                    WorkerPolicyDecision::ScaleUp
5047                } else {
5048                    WorkerPolicyDecision::HoldInsufficientFrontier
5049                }
5050            };
5051            let sequence = state.allocate_policy_sequence();
5052            PolicyWindowSnapshot {
5053                sequence,
5054                start_entry_ordinal: window.start_entry_ordinal,
5055                end_entry_ordinal: window.end_entry_ordinal,
5056                observed_entries: window.entries,
5057                observed_chunks: window.chunks,
5058                observed_work_ns: window.work_ns,
5059                ready_directories: state.ready_directories,
5060                in_flight_directories: state.in_flight_directories,
5061                active_workers: self.diagnostics.as_ref().map_or(0, |diagnostics| {
5062                    diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
5063                }),
5064                handoff_backlog,
5065                requested_workers,
5066                decision,
5067            }
5068        });
5069        let shadow_window = observed_window.map(|window| {
5070            let sequence = state.allocate_policy_sequence();
5071            PolicyWindowSnapshot {
5072                sequence,
5073                start_entry_ordinal: window.start_entry_ordinal,
5074                end_entry_ordinal: window.end_entry_ordinal,
5075                observed_entries: window.entries,
5076                observed_chunks: window.chunks,
5077                observed_work_ns: window.work_ns,
5078                ready_directories: state.ready_directories,
5079                in_flight_directories: state.in_flight_directories,
5080                active_workers: self.diagnostics.as_ref().map_or(0, |diagnostics| {
5081                    diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
5082                }),
5083                handoff_backlog,
5084                requested_workers: None,
5085                decision: if window.slow {
5086                    WorkerPolicyDecision::ObserveSlow
5087                } else {
5088                    WorkerPolicyDecision::ObserveFast
5089                },
5090            }
5091        });
5092        let controller_terminates = (one_shot || finished) && window_completed
5093            || requested_workers.is_some_and(|target| target == state.maximum_workers);
5094        if controller_terminates {
5095            // Continue observing after every terminal decision, including a candidate
5096            // that reached the maximum pool. Without this shadow history a slow-prefix
5097            // expansion makes a later fast phase unobservable, precisely the
5098            // irreversible over-expansion case the evidence matrix must detect.
5099            if let (Some(spec), Some(window)) = (controller_spec, completed_window) {
5100                if self.diagnostics.is_some() && !finished {
5101                    state.shadow_calibration =
5102                        Some(RepeatedCalibration::starting_at(spec, window.end_entry_ordinal));
5103                }
5104            }
5105            state.controller = None;
5106        }
5107        drop(state);
5108
5109        // The sequence was assigned while the queue was locked, but the trace lock and
5110        // bounded-vector update stay outside that critical section. Recorder arrival
5111        // may differ from completion order; `finish` sorts the retained prefix.
5112        if let (Some(diagnostics), Some(window)) = (&self.diagnostics, policy_window) {
5113            diagnostics.record_policy_window(window);
5114        }
5115        if let (Some(diagnostics), Some(window)) = (&self.diagnostics, shadow_window) {
5116            diagnostics.record_policy_window(window);
5117        }
5118
5119        // Recorded outside the lock: the sampling is off by default, and a disabled
5120        // counter must not lengthen the critical section it observes.
5121        if calibrating {
5122            record_adaptive_calibration_chunk(
5123                self.diagnostics.as_ref(),
5124                observed_entries,
5125                observed_work_ns,
5126            );
5127        }
5128
5129        if finished {
5130            self.ready.notify_all();
5131        }
5132        requested_workers
5133    }
5134
5135    /// Acquire the state lock, charging any contention to `timing`.
5136    ///
5137    /// The fast path is a `try_lock` that succeeds and costs one counter increment;
5138    /// only the contended path pays for reading the clock. Poisoning is tolerated for
5139    /// the same reason as [`Self::lock`].
5140    fn lock_timed(
5141        &self,
5142        timing: &mut WalkAttribution,
5143    ) -> std::sync::MutexGuard<'_, DirectoryQueueState> {
5144        timing.lock_ops += 1;
5145        match self.state.try_lock() {
5146            Ok(guard) => guard,
5147            Err(std::sync::TryLockError::Poisoned(poisoned)) => poisoned.into_inner(),
5148            Err(std::sync::TryLockError::WouldBlock) => {
5149                timing.lock_contended += 1;
5150                let started = std::time::Instant::now();
5151                let guard = self.lock();
5152                timing.lock_wait_ns += elapsed_ns(started);
5153                guard
5154            }
5155        }
5156    }
5157
5158    /// A poisoned queue means a worker panicked mid-walk. The data behind the lock is
5159    /// a plain work list with no invariant that a panic could have broken, and the
5160    /// caller already reports the panic as a scan error, so recovering the list is
5161    /// strictly better than propagating a second panic into every other worker.
5162    fn lock(&self) -> std::sync::MutexGuard<'_, DirectoryQueueState> {
5163        self.state.lock().unwrap_or_else(std::sync::PoisonError::into_inner)
5164    }
5165}
5166
5167fn record_adaptive_calibration_chunk(
5168    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
5169    entries: u64,
5170    work_ns: u64,
5171) {
5172    if let Some(diagnostics) = diagnostics {
5173        diagnostics.calibration_chunk(entries, work_ns);
5174    }
5175    crate::counters::bump(|counts| {
5176        counts.adaptive_calibration_chunks = counts.adaptive_calibration_chunks.saturating_add(1);
5177        counts.adaptive_calibration_entries =
5178            counts.adaptive_calibration_entries.saturating_add(entries);
5179        counts.adaptive_calibration_work_us =
5180            counts.adaptive_calibration_work_us.saturating_add(work_ns / 1_000);
5181    });
5182}
5183
5184fn record_adaptive_worker_expansion(diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>) {
5185    if let Some(diagnostics) = diagnostics {
5186        diagnostics.worker_expanded();
5187    }
5188    crate::counters::bump(|counts| {
5189        counts.adaptive_scale_ups = counts.adaptive_scale_ups.saturating_add(1);
5190    });
5191}
5192
5193/// The private builder a detached walk consumes its listings into: a full index, or with
5194/// `retention` a folded one.
5195fn detached_builder(
5196    root: &Path,
5197    config: &ScanConfig,
5198    retention: Option<TreeRetention>,
5199) -> DetachedIndexBuilder {
5200    let builder = DetachedIndexBuilder::new(root, config.scope(), config.types_shared())
5201        .with_control_limits(config.control_limits);
5202    match retention {
5203        Some(retention) => builder.folding(retention),
5204        None => builder,
5205    }
5206}
5207
5208fn scan_detached_directories(
5209    root: &Path,
5210    config: &ScanConfig,
5211    collect_diagnostics: bool,
5212    policy: WorkerPolicyExperiment,
5213    retention: Option<TreeRetention>,
5214) -> Result<(ScanReport, DetachedIndexBuilder, Option<ScanDiagnostics>)> {
5215    if let Some(progress) = &config.progress {
5216        progress.enter(crate::ProgressPhase::Scanning);
5217    }
5218    let root_metadata = {
5219        crate::counters::bump(|counts| counts.stats += 1);
5220        fs::symlink_metadata(root)
5221    }
5222    .map_err(|error| Error::io(root, error))?;
5223    if !root_metadata.is_dir() {
5224        return Err(Error::io(
5225            root,
5226            std::io::Error::new(std::io::ErrorKind::NotADirectory, "scan root is not a directory"),
5227        ));
5228    }
5229    let root_dev = root_device(root, &root_metadata).map_err(|error| Error::io(root, error))?;
5230    let available_parallelism =
5231        std::thread::available_parallelism().map_or(1, std::num::NonZero::get);
5232    let pool = config.worker_pool_for(available_parallelism);
5233    let diagnostics = collect_diagnostics
5234        .then(|| ScanDiagnosticsRecorder::new(pool, available_parallelism, policy));
5235
5236    if config.max_depth == Some(0) {
5237        if let Some(diagnostics) = &diagnostics {
5238            diagnostics.mark_not_run();
5239            diagnostics.record_queue_finish(0, 0);
5240        }
5241        return Ok((
5242            ScanReport::default(),
5243            detached_builder(root, config, retention),
5244            diagnostics.as_ref().map(|value| value.finish()),
5245        ));
5246    }
5247
5248    let walk_started = crate::counters::enabled().then(std::time::Instant::now);
5249    let (output, builder) = scan_concurrent_detached(
5250        root,
5251        config,
5252        root_dev,
5253        pool,
5254        diagnostics.as_ref(),
5255        policy,
5256        retention,
5257    )?;
5258    // Also reached by a walk no worker left, such as a single-threaded one, so every
5259    // cold index ends its walk in the same phase.
5260    if let Some(progress) = &config.progress {
5261        progress.enter(crate::ProgressPhase::Indexing);
5262    }
5263    if let Some(started) = walk_started {
5264        let elapsed = elapsed_ns(started) / 1_000;
5265        crate::counters::bump(|counts| {
5266            counts.detached_walk_us = counts.detached_walk_us.saturating_add(elapsed);
5267        });
5268    }
5269    Ok((output, builder, diagnostics.as_ref().map(|value| value.finish())))
5270}
5271
5272fn consolidate_detached_index(
5273    mut output: ScanReport,
5274    builder: DetachedIndexBuilder,
5275) -> (Index, ScanReport) {
5276    let entries = output.entries;
5277    let consolidate_started = crate::counters::enabled().then(std::time::Instant::now);
5278    let mut index = builder.finish();
5279    if let Some(started) = consolidate_started {
5280        let elapsed = elapsed_ns(started) / 1_000;
5281        crate::counters::bump(|counts| {
5282            counts.detached_builds = counts.detached_builds.saturating_add(1);
5283            counts.detached_entries = counts.detached_entries.saturating_add(entries);
5284            counts.detached_finish_us = counts.detached_finish_us.saturating_add(elapsed);
5285        });
5286    }
5287    index.record_walk_errors(&mut output.errors);
5288    index.set_initial_detached_scan_freshness(&output.errors);
5289    (index, output)
5290}
5291
5292/// Walk `root` and return a fully populated index.
5293pub fn scan_into_index(root: &Path, config: &ScanConfig) -> Result<(Index, ScanReport)> {
5294    config.validate()?;
5295    let root = root.canonicalize().map_err(|error| Error::io(root, error))?;
5296    if config.population != crate::query::IgnoredEntries::Include {
5297        let (index, report, _) = scan_into_index_with_scanner(
5298            &root,
5299            config,
5300            false,
5301            WorkerPolicyExperiment::ShippedOneShot,
5302        )?;
5303        return Ok((index, report));
5304    }
5305    let (output, builder, _diagnostics) = scan_detached_directories(
5306        &root,
5307        config,
5308        false,
5309        WorkerPolicyExperiment::ShippedOneShot,
5310        None,
5311    )?;
5312    Ok(consolidate_detached_index(output, builder))
5313}
5314
5315/// Walk the canonical `root` into a folded index that keeps what `retention` names
5316/// ([`crate::execution::RetainedState::Tree`]), with the diagnostic trace when asked.
5317///
5318/// The same detached walk and builder as [`scan_into_index`], which a folded index
5319/// differs from only in the files it keeps. Crate-private because a folded index answers
5320/// one tree report and nothing else; its one caller reports from it and frees it.
5321pub(crate) fn scan_into_folded_index(
5322    root: &Path,
5323    config: &ScanConfig,
5324    retention: TreeRetention,
5325    collect_diagnostics: bool,
5326) -> Result<(Index, ScanReport, Option<ScanDiagnostics>)> {
5327    config.validate()?;
5328    debug_assert_eq!(
5329        config.population,
5330        crate::query::IgnoredEntries::Include,
5331        "a folded index keeps the whole population; a narrowed one takes the scanner"
5332    );
5333    let (output, builder, diagnostics) = scan_detached_directories(
5334        root,
5335        config,
5336        collect_diagnostics,
5337        WorkerPolicyExperiment::ShippedOneShot,
5338        Some(retention),
5339    )?;
5340    let (index, report) = consolidate_detached_index(output, builder);
5341    Ok((index, report, diagnostics))
5342}
5343
5344fn scan_into_index_with_scanner(
5345    root: &Path,
5346    config: &ScanConfig,
5347    collect_diagnostics: bool,
5348    policy: WorkerPolicyExperiment,
5349) -> Result<(Index, ScanReport, Option<ScanDiagnostics>)> {
5350    let mut index = Index::new_with_scope_and_types(root, config.scope(), config.types_shared());
5351    index.set_control_limits(config.control_limits);
5352    let mut apply_error: Option<Error> = None;
5353    let (mut report, diagnostics) = scan_internal(
5354        root,
5355        config,
5356        &mut |batch| {
5357            if apply_error.is_none() {
5358                if let Err(error) = index.apply_scanner_baseline(batch) {
5359                    apply_error = Some(error);
5360                }
5361            }
5362        },
5363        collect_diagnostics,
5364        policy,
5365        SinkMode::Retained,
5366    )?;
5367    if let Some(error) = apply_error {
5368        return Err(error);
5369    }
5370    index.record_walk_errors(&mut report.errors);
5371    index.set_initial_scan_freshness(&report.errors);
5372    Ok((index, report, diagnostics))
5373}
5374
5375#[cfg(test)]
5376fn scan_into_index_via_scanner(root: &Path, config: &ScanConfig) -> Result<(Index, ScanReport)> {
5377    let (index, report, _) =
5378        scan_into_index_with_scanner(root, config, false, WorkerPolicyExperiment::ShippedOneShot)?;
5379    Ok((index, report))
5380}
5381
5382/// Walk `root` into an index and retain the opt-in diagnostic trace for that run.
5383///
5384/// The index and [`ScanReport`] have the same semantics as [`scan_into_index`]. The
5385/// additional trace is versioned independently so evidence tooling can fail closed on
5386/// changes without coupling cache or engine behavior to measurement details.
5387pub fn scan_into_index_with_diagnostics(
5388    root: &Path,
5389    config: &ScanConfig,
5390) -> Result<(Index, ScanReport, ScanDiagnostics)> {
5391    scan_into_index_with_policy_diagnostics(root, config, WorkerPolicyExperiment::ShippedOneShot)
5392}
5393
5394/// Exercise a repository-only worker-controller candidate while building an index.
5395#[doc(hidden)]
5396pub fn scan_into_index_with_policy_diagnostics(
5397    root: &Path,
5398    config: &ScanConfig,
5399    policy: WorkerPolicyExperiment,
5400) -> Result<(Index, ScanReport, ScanDiagnostics)> {
5401    config.validate()?;
5402    let root = root.canonicalize().map_err(|error| Error::io(root, error))?;
5403    if config.population != crate::query::IgnoredEntries::Include {
5404        let (index, report, diagnostics) =
5405            scan_into_index_with_scanner(&root, config, true, policy)?;
5406        return Ok((index, report, diagnostics.expect("diagnostic scanner creates a recorder")));
5407    }
5408    let (output, builder, diagnostics) =
5409        scan_detached_directories(&root, config, true, policy, None)?;
5410    let (index, report) = consolidate_detached_index(output, builder);
5411    Ok((index, report, diagnostics.expect("diagnostic detached scan creates a recorder")))
5412}
5413
5414/// Diff the filesystem against an existing index and emit conditional observations.
5415///
5416/// This is cache tier 2: after a snapshot is loaded, a sweep like this is what makes the
5417/// answer trustworthy rather than merely fast. Unchanged entries produce upserts whose
5418/// fingerprints already match, which the index discards as no-ops, so the caller can
5419/// apply the whole stream without filtering it first.
5420///
5421/// Entries the index holds but the filesystem no longer has become [`Op::Remove`],
5422/// detected per directory rather than by accumulating every visited path in memory.
5423///
5424/// This observation-only reference API assumes its emitted stream is applied to the same
5425/// unchanged baseline after the borrow ends. Use [`reconcile`] or [`reconcile_handle`]
5426/// when other producers can write concurrently; those paths capture the stronger
5427/// generation/revision/absence expectations returned by [`Index::expectation`].
5428pub fn revalidate(
5429    index: &Index,
5430    config: &ScanConfig,
5431    sink: &mut dyn FnMut(Observation),
5432) -> Result<ScanReport> {
5433    config.validate_for_scope(index.scope())?;
5434    let root = index.root_path().to_path_buf();
5435    let root_meta = {
5436        crate::counters::bump(|c| c.stats += 1);
5437        fs::symlink_metadata(&root)
5438    }
5439    .map_err(|error| Error::io(&root, error))?;
5440    if !root_meta.is_dir() {
5441        return Err(Error::io(
5442            &root,
5443            std::io::Error::new(
5444                std::io::ErrorKind::NotADirectory,
5445                "revalidation root is not a directory",
5446            ),
5447        ));
5448    }
5449    let root_dev = root_device(&root, &root_meta).map_err(|error| Error::io(&root, error))?;
5450    if let Some(progress) = &config.progress {
5451        progress.enter(crate::ProgressPhase::Revalidating);
5452    }
5453    let mut report = ScanReport::default();
5454    let mut tally = ProgressTally::new(config.progress.as_ref());
5455    let batch_limit = config.batch_size.max(1);
5456    let mut batch: Vec<ObservationOp> = Vec::with_capacity(batch_limit);
5457    if config.max_depth == Some(0) {
5458        if let Some(children) = index.children(Path::new("")) {
5459            for (name, _) in children {
5460                let path = PathBuf::from(name);
5461                batch.push(ObservationOp::if_state(
5462                    Op::Remove { path: path.clone() },
5463                    index.relaxed_expectation(&path),
5464                ));
5465                if batch.len() >= batch_limit {
5466                    sink(Observation::from_ops(std::mem::take(&mut batch)));
5467                    batch.reserve(batch_limit);
5468                }
5469            }
5470        }
5471        if !batch.is_empty() {
5472            sink(Observation::from_ops(batch));
5473        }
5474        return Ok(report);
5475    }
5476    let mut queue: VecDeque<(PathBuf, usize)> = VecDeque::from(vec![(PathBuf::new(), 0)]);
5477    let mut controls = (config.population != crate::query::IgnoredEntries::Include)
5478        .then(|| index.control_table().clone());
5479    let mut unreadable_controls = std::collections::BTreeSet::new();
5480    let mut readers = Readers::new();
5481    let policy = ListingPolicy::every_child_stated(config.one_filesystem);
5482
5483    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
5484        let abs_dir = root.join(&rel_dir);
5485        let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
5486        let mut had_control = index.control_table().contains(&control_path);
5487        let mut listed_control = ListedControl::default();
5488        if let Some(table) = controls.as_mut() {
5489            let baseline = index.relaxed_expectation(&control_path);
5490            let lookup = read_directory_control(config, &root, &control_path);
5491            listed_control.lookup(&lookup);
5492            match lookup {
5493                Ok(Some(op)) => {
5494                    apply_discovery_control(table, &op)?;
5495                    batch.push(ObservationOp::if_state(op, baseline));
5496                }
5497                Ok(None) => {
5498                    table.remove(&control_path)?;
5499                    if had_control {
5500                        batch.push(ObservationOp::if_state(
5501                            Op::ControlRemove { path: control_path.clone() },
5502                            baseline,
5503                        ));
5504                        had_control = false;
5505                    }
5506                }
5507                Err(error) => {
5508                    unreadable_controls.insert(rel_dir.clone());
5509                    report.errors.push(error);
5510                }
5511            }
5512        }
5513        let listing = match list_directory(&mut readers, &abs_dir, policy, None) {
5514            Ok(listing) => listing,
5515            Err(e) => {
5516                report.errors.push(Error::io(abs_dir, e));
5517                continue;
5518            }
5519        };
5520        report.dirs_read += 1;
5521
5522        let mut seen: BTreeSet<OsString> = BTreeSet::new();
5523        let mut listing_complete = true;
5524        let listing = reconcile_listing(listing, &abs_dir);
5525        for item in listing {
5526            let (name, observed) = match item {
5527                Listed::Child { name, observed } => (name, observed),
5528                Listed::Failed(e) => {
5529                    listing_complete = false;
5530                    report.errors.push(Error::io(&abs_dir, e));
5531                    continue;
5532                }
5533            };
5534            // Seeing the name proves it is not absent even when a following metadata
5535            // lookup fails. Record it before any fallible per-entry work so an
5536            // operational error cannot become a false removal in the missing sweep.
5537            seen.insert(name.to_os_string());
5538            let rel_path = rel_dir.join(&name);
5539            let baseline = index.relaxed_expectation(&rel_path);
5540            let (kind, attrs) = match observed {
5541                Ok(Some(observed)) => observed,
5542                Ok(None) => {
5543                    let entry_held = baseline.state != PathState::Absent;
5544                    for removal in
5545                        vanished_child_removals(&rel_dir, &name, entry_held, &mut had_control)
5546                            .into_iter()
5547                            .flatten()
5548                    {
5549                        batch.push(ObservationOp::if_state(removal, baseline));
5550                    }
5551                    if batch.len() >= batch_limit {
5552                        sink(Observation::from_ops(std::mem::take(&mut batch)));
5553                        batch.reserve(batch_limit);
5554                    }
5555                    continue;
5556                }
5557                Err(e) => {
5558                    listed_control.listed(&name);
5559                    report.errors.push(Error::io(abs_dir.join(&name), e));
5560                    continue;
5561                }
5562            };
5563            listed_control.listed(&name);
5564            let disposition =
5565                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
5566            if population_prunes(
5567                config.population,
5568                &rel_path,
5569                kind,
5570                disposition,
5571                controls.as_ref(),
5572                &unreadable_controls,
5573            ) {
5574                if baseline.state != PathState::Absent {
5575                    batch.push(ObservationOp::if_state(Op::Remove { path: rel_path }, baseline));
5576                }
5577                continue;
5578            }
5579            let read = read_listed_control_op(config, &root, &rel_path, kind);
5580            listed_control.read(&name, &read);
5581            let control = match read {
5582                Ok(control) => control,
5583                Err(error) => {
5584                    report.errors.push(error);
5585                    None
5586                }
5587            };
5588            if disposition != crate::admission::Disposition::Retain {
5589                if baseline.state != PathState::Absent {
5590                    batch.push(ObservationOp::if_state(
5591                        Op::Remove { path: rel_path.clone() },
5592                        baseline,
5593                    ));
5594                }
5595                if disposition == crate::admission::Disposition::ControlOnly {
5596                    if let Some(control) = control {
5597                        let guard = control_guard(&rel_path, baseline, |path| {
5598                            index.relaxed_expectation(path)
5599                        });
5600                        batch.push(ObservationOp::if_state(control, guard));
5601                    }
5602                }
5603                if batch.len() >= batch_limit {
5604                    sink(Observation::from_ops(std::mem::take(&mut batch)));
5605                    batch.reserve(batch_limit);
5606                }
5607                continue;
5608            }
5609            report.observe(kind, attrs);
5610            batch.push(ObservationOp::if_state(
5611                Op::Upsert { path: rel_path.clone(), kind, attrs },
5612                baseline,
5613            ));
5614            if batch.len() >= batch_limit {
5615                sink(Observation::from_ops(std::mem::take(&mut batch)));
5616                batch.reserve(batch_limit);
5617            }
5618            if let Some(control) = control {
5619                let guard =
5620                    control_guard(&rel_path, baseline, |path| index.relaxed_expectation(path));
5621                batch.push(ObservationOp::if_state(control, guard));
5622                if batch.len() >= batch_limit {
5623                    sink(Observation::from_ops(std::mem::take(&mut batch)));
5624                    batch.reserve(batch_limit);
5625                }
5626            }
5627
5628            if should_descend(kind, attrs, depth, root_dev, config) {
5629                queue.push_back((rel_path, depth + 1));
5630            } else if kind.is_dir() {
5631                if let Some(children) = index.children(&rel_path) {
5632                    for (child_name, _) in children {
5633                        let child_path = rel_path.join(child_name);
5634                        batch.push(ObservationOp::if_state(
5635                            Op::Remove { path: child_path.clone() },
5636                            index.relaxed_expectation(&child_path),
5637                        ));
5638                        if batch.len() >= batch_limit {
5639                            sink(Observation::from_ops(std::mem::take(&mut batch)));
5640                            batch.reserve(batch_limit);
5641                        }
5642                    }
5643                }
5644            }
5645        }
5646
5647        // Anything the index still lists here but the filesystem did not return is gone.
5648        if listing_complete {
5649            if let Some(known) = index.children(&rel_dir) {
5650                for (name, _) in known {
5651                    if !seen.contains(name) {
5652                        let path = rel_dir.join(name);
5653                        batch.push(ObservationOp::if_state(
5654                            Op::Remove { path: path.clone() },
5655                            index.relaxed_expectation(&path),
5656                        ));
5657                        // In the same batch, so the rules the removal drops are back
5658                        // before any commit shows the directory without them.
5659                        if let Some(control) = listed_control.restatement_after_removing(name) {
5660                            batch.push(ObservationOp::if_state(
5661                                control,
5662                                index.relaxed_expectation(&control_path),
5663                            ));
5664                        }
5665                    }
5666                }
5667            }
5668            if had_control && !listed_control.seen {
5669                batch.push(ObservationOp::if_state(
5670                    Op::ControlRemove { path: control_path.clone() },
5671                    index.relaxed_expectation(&control_path),
5672                ));
5673            }
5674        }
5675        // Per directory: an unchanged tree fills no batch, so the batch cannot be the
5676        // unit here without the counters standing still for the whole walk.
5677        tally.flush(&report);
5678    }
5679
5680    if !batch.is_empty() {
5681        sink(Observation::from_ops(batch));
5682    }
5683    tally.flush(&report);
5684    Ok(report)
5685}
5686
5687/// Reconcile the full index and publish each exact commit as it lands.
5688pub fn reconcile(
5689    index: &mut Index,
5690    config: &ScanConfig,
5691    sink: &mut dyn FnMut(&Commit),
5692) -> Result<ReconcileReport> {
5693    reconcile_subtree(index, Path::new(""), config, sink)
5694}
5695
5696/// Reconcile one relative subtree, applying effective changes during the walk.
5697///
5698/// If an ancestor vanished or became a non-directory, reconciliation widens to that
5699/// ancestor so a child invalidation can converge instead of retrying `ENOTDIR` forever.
5700pub fn reconcile_subtree(
5701    index: &mut Index,
5702    subtree: &Path,
5703    config: &ScanConfig,
5704    sink: &mut dyn FnMut(&Commit),
5705) -> Result<ReconcileReport> {
5706    reconcile_target(&mut ReconcileTarget::Direct(index), subtree, config, sink)
5707}
5708
5709/// Reconcile a shared index while allowing readers between applied batches.
5710pub fn reconcile_handle(
5711    handle: &IndexHandle,
5712    config: &ScanConfig,
5713    sink: &mut dyn FnMut(&Commit),
5714) -> Result<ReconcileReport> {
5715    reconcile_subtree_handle(handle, Path::new(""), config, sink)
5716}
5717
5718/// Reconcile one subtree of a shared index, widening to a missing/non-directory ancestor
5719/// when necessary.
5720pub fn reconcile_subtree_handle(
5721    handle: &IndexHandle,
5722    subtree: &Path,
5723    config: &ScanConfig,
5724    sink: &mut dyn FnMut(&Commit),
5725) -> Result<ReconcileReport> {
5726    reconcile_target(&mut ReconcileTarget::Shared(handle), subtree, config, sink)
5727}
5728
5729/// Internal effects of one opened-root multi-path reconciliation.
5730#[derive(Debug, Default)]
5731pub(crate) struct ReconcilePathsReport {
5732    pub(crate) reconciliation: ReconcileReport,
5733    pub(crate) accepted: Vec<PathBuf>,
5734    pub(crate) rejected: Vec<crate::RejectedRefreshPath>,
5735}
5736
5737/// Reconcile one bounded path set under an opened-root lifecycle controller.
5738///
5739/// Classification precedes I/O, overlapping descendants fold into one walk, and all
5740/// surviving scopes enter `Reconciling` before the first is read. `forbid_expansion`
5741/// is the conservative resource-stop rule: removals and same-file verification remain
5742/// legal, while work that could retain another file or discover children is refused.
5743pub(crate) fn reconcile_paths_handle_controlled(
5744    handle: &IndexHandle,
5745    paths: &[PathBuf],
5746    config: &ScanConfig,
5747    forbid_expansion: bool,
5748    control: &dyn ReconcileControl,
5749    sink: &mut dyn FnMut(&Commit),
5750) -> Result<ReconcilePathsReport> {
5751    let mut target = ReconcileTarget::Controlled { handle, control };
5752    reconcile_paths_target(&mut target, paths, config, forbid_expansion, sink)
5753}
5754
5755fn reconcile_paths_target(
5756    target: &mut ReconcileTarget<'_>,
5757    paths: &[PathBuf],
5758    config: &ScanConfig,
5759    forbid_expansion: bool,
5760    sink: &mut dyn FnMut(&Commit),
5761) -> Result<ReconcilePathsReport> {
5762    config.validate_for_scope(target.scope()?)?;
5763    let mut report = ReconcilePathsReport::default();
5764    let mut accepted = BTreeSet::new();
5765
5766    for requested in paths {
5767        let reject = |reason| crate::RejectedRefreshPath { path: requested.clone(), reason };
5768        let Ok(path) = normalize_subtree(requested) else {
5769            report.rejected.push(reject(crate::RefreshRejection::OutsideRoot));
5770            continue;
5771        };
5772        if config.max_depth.is_some_and(|maximum| path.components().count() > maximum) {
5773            report.rejected.push(reject(crate::RefreshRejection::BeyondDepth));
5774            continue;
5775        }
5776        // This is lexical admission before the final kind is observed. Treating the
5777        // boundary as a file preserves the fixed hidden `.gitignore` control exception;
5778        // the verified walk still applies the real kind and special-object policy.
5779        if crate::admission::decide_path(&path, EntryKind::File, config.hidden(), false)
5780            == crate::admission::Disposition::Reject
5781        {
5782            report.rejected.push(reject(crate::RefreshRejection::NotAdmitted));
5783            continue;
5784        }
5785        if forbid_expansion && refresh_may_expand(target, &path, &mut report.reconciliation.scan)? {
5786            report.rejected.push(reject(crate::RefreshRejection::ResourceBudget));
5787            continue;
5788        }
5789        accepted.insert(path);
5790    }
5791
5792    report.accepted = accepted.into_iter().collect();
5793    let mut resolved = Vec::new();
5794    let mut unsafe_roots = Vec::new();
5795    for requested_root in covering_roots(report.accepted.clone()) {
5796        match resolve_subtree_root(target, &requested_root, config) {
5797            Ok(root) => resolved.push(root),
5798            Err(Error::SubtreeOutsideScanScope { .. }) => unsafe_roots.push(requested_root),
5799            Err(error) => return Err(error),
5800        }
5801    }
5802    if !unsafe_roots.is_empty() {
5803        let mut retained = Vec::with_capacity(report.accepted.len());
5804        for path in std::mem::take(&mut report.accepted) {
5805            if unsafe_roots.iter().any(|root| path.starts_with(root)) {
5806                report.rejected.push(crate::RejectedRefreshPath {
5807                    path,
5808                    reason: crate::RefreshRejection::UnsafeAncestry,
5809                });
5810            } else {
5811                retained.push(path);
5812            }
5813        }
5814        report.accepted = retained;
5815    }
5816    let walked = covering_roots(resolved);
5817    if walked.is_empty() {
5818        return Ok(report);
5819    }
5820
5821    let mut opened = Vec::with_capacity(walked.len());
5822    for subtree in walked {
5823        let (started_at, commit) = target.begin_reconcile(&subtree)?;
5824        if let Some(commit) = commit.as_ref() {
5825            sink(commit);
5826        }
5827        opened.push((subtree, started_at));
5828    }
5829
5830    // Each subtree closes on its own walk's outcome, so a subtree that could not be read
5831    // neither marks a verified sibling partial nor withholds the completeness its listing
5832    // earned.
5833    let mut failure = None;
5834    let mut outcomes = Vec::with_capacity(opened.len());
5835    for (subtree, started_at) in &opened {
5836        if failure.is_some() {
5837            outcomes.push((false, false));
5838            continue;
5839        }
5840        match reconcile_target_inner(
5841            target,
5842            subtree,
5843            *started_at,
5844            config,
5845            MAX_DEFERRED_RECONCILE_OPS,
5846            sink,
5847        ) {
5848            Ok(mut reconciliation) => {
5849                outcomes.push((
5850                    reconciliation.is_complete(),
5851                    reconciliation.apply.stale == 0 && reconciliation.apply.resource_refused == 0,
5852                ));
5853                reconciliation.listed_incomplete = reconciliation.take_recordable_completeness();
5854                merge_reconcile_report(&mut report.reconciliation, reconciliation);
5855            }
5856            Err(error) => {
5857                outcomes.push((false, false));
5858                failure = Some(error);
5859            }
5860        }
5861    }
5862
5863    let listed_incomplete = std::mem::take(&mut report.reconciliation.listed_incomplete);
5864    let root = target.root_path()?;
5865    normalize_walk_errors(&root, &mut report.reconciliation.scan.errors);
5866    let failed_paths = failure_paths(target, &report.reconciliation.scan.errors)?;
5867    for ((subtree, started_at), (complete, disproves_old)) in opened.into_iter().zip(outcomes) {
5868        let commit = target.finish_reconcile(
5869            &subtree,
5870            started_at,
5871            complete,
5872            &listed_incomplete,
5873            &failed_paths,
5874            ReconcileErrors {
5875                errors: &report.reconciliation.scan.errors,
5876                terminal: failure.as_ref(),
5877                disproves_old,
5878            },
5879        )?;
5880        if let Some(commit) = commit.commit.as_ref() {
5881            sink(commit);
5882        }
5883        report.reconciliation.retry_required |= commit.retry;
5884    }
5885
5886    match failure {
5887        Some(error) => Err(error),
5888        None => Ok(report),
5889    }
5890}
5891
5892/// Root-relative paths whose filesystem facts a failed reconciliation could not verify.
5893///
5894/// An unscoped error returns an empty set, which makes the closer conservatively mark the
5895/// whole requested subtree partial. Precise I/O paths let verified siblings remain fresh.
5896fn failure_paths(target: &ReconcileTarget<'_>, errors: &[Error]) -> Result<Vec<PathBuf>> {
5897    if errors.is_empty() {
5898        return Ok(Vec::new());
5899    }
5900    let root = target.root_path()?;
5901    let mut paths = Vec::with_capacity(errors.len());
5902    for error in errors {
5903        let Some(path) = crate::Issue::from_error_under(&root, error).path else {
5904            return Ok(Vec::new());
5905        };
5906        if path.is_absolute() {
5907            return Ok(Vec::new());
5908        }
5909        paths.push(path);
5910    }
5911    paths.sort();
5912    paths.dedup();
5913    Ok(paths)
5914}
5915
5916/// Whether verification could increase the retained-file set.
5917///
5918/// This deliberately recognizes only cases that prove non-expansion. At a resource
5919/// boundary, uncertainty is a refusal rather than permission to exceed the bound.
5920fn refresh_may_expand(
5921    target: &ReconcileTarget<'_>,
5922    path: &Path,
5923    work: &mut ScanReport,
5924) -> Result<bool> {
5925    let current = target.expectation(path)?.state;
5926    let absolute = target.root_path()?.join(path);
5927    let observed = match observe_path(&absolute) {
5928        Ok((kind, attrs)) => {
5929            work.observe(kind, attrs);
5930            Some(kind)
5931        }
5932        Err(error)
5933            if matches!(
5934                error.kind(),
5935                std::io::ErrorKind::NotFound | std::io::ErrorKind::NotADirectory
5936            ) =>
5937        {
5938            None
5939        }
5940        Err(_) => return Ok(true),
5941    };
5942    Ok(!matches!(
5943        (current, observed),
5944        (PathState::Present { kind: EntryKind::File, .. }, Some(EntryKind::File)) | (_, None)
5945    ))
5946}
5947
5948/// Drop every path covered by a shallower member of the same sorted set.
5949fn covering_roots(mut paths: Vec<PathBuf>) -> Vec<PathBuf> {
5950    paths.sort();
5951    paths.dedup();
5952    if paths.first().is_some_and(|first| first.as_os_str().is_empty()) {
5953        return vec![PathBuf::new()];
5954    }
5955    let mut roots: Vec<PathBuf> = Vec::with_capacity(paths.len());
5956    for path in paths {
5957        if roots.last().is_some_and(|kept| path.starts_with(kept)) {
5958            continue;
5959        }
5960        roots.push(path);
5961    }
5962    roots
5963}
5964
5965fn reconcile_target(
5966    target: &mut ReconcileTarget<'_>,
5967    subtree: &Path,
5968    config: &ScanConfig,
5969    sink: &mut dyn FnMut(&Commit),
5970) -> Result<ReconcileReport> {
5971    config.validate_for_scope(target.scope()?)?;
5972    if let Some(progress) = &config.progress {
5973        progress.enter(crate::ProgressPhase::Revalidating);
5974    }
5975    let subtree = normalize_subtree(subtree)?;
5976    if config.max_depth.is_some_and(|maximum| subtree.components().count() > maximum) {
5977        return Err(Error::SubtreeOutsideScanScope { path: subtree, scope: config.scope() });
5978    }
5979    let subtree = if config.population != crate::query::IgnoredEntries::Include
5980        && !subtree.as_os_str().is_empty()
5981    {
5982        // A narrowed tier may have pruned the requested entry or an ancestor. Start
5983        // from its nearest retained parent so the governing control is read before the
5984        // directory listing decides whether the boundary itself belongs in the tier.
5985        // A control-file edit also needs this parent listing to discover siblings that
5986        // were absent under the previous rule.
5987        let mut parent = subtree.parent().map_or_else(PathBuf::new, Path::to_path_buf);
5988        while !parent.as_os_str().is_empty()
5989            && (target.expectation(&parent)?.state == PathState::Absent
5990                || !target.control_classification_known(&parent)?)
5991        {
5992            parent = parent.parent().map_or_else(PathBuf::new, Path::to_path_buf);
5993        }
5994        parent
5995    } else {
5996        subtree
5997    };
5998    let subtree = resolve_subtree_root(target, &subtree, config)?;
5999    let (started_at, started) = target.begin_reconcile(&subtree)?;
6000    if let Some(commit) = started.as_ref() {
6001        sink(commit);
6002    }
6003    match reconcile_target_inner(
6004        target,
6005        &subtree,
6006        started_at,
6007        config,
6008        MAX_DEFERRED_RECONCILE_OPS,
6009        sink,
6010    ) {
6011        Ok(mut report) => {
6012            let root = target.root_path()?;
6013            normalize_walk_errors(&root, &mut report.scan.errors);
6014            let listed_incomplete = report.take_recordable_completeness();
6015            let failed_paths = failure_paths(target, &report.scan.errors)?;
6016            let finished = target.finish_reconcile(
6017                &subtree,
6018                started_at,
6019                report.is_complete(),
6020                &listed_incomplete,
6021                &failed_paths,
6022                ReconcileErrors {
6023                    errors: &report.scan.errors,
6024                    terminal: None,
6025                    disproves_old: report.apply.stale == 0 && report.apply.resource_refused == 0,
6026                },
6027            )?;
6028            if let Some(commit) = finished.commit.as_ref() {
6029                sink(commit);
6030            }
6031            report.retry_required |= finished.retry;
6032            Ok(report)
6033        }
6034        Err(error) => {
6035            let finished = target.finish_reconcile(
6036                &subtree,
6037                started_at,
6038                false,
6039                &[],
6040                &[],
6041                ReconcileErrors { errors: &[], terminal: Some(&error), disproves_old: false },
6042            )?;
6043            if let Some(commit) = finished.commit.as_ref() {
6044                sink(commit);
6045            }
6046            Err(error)
6047        }
6048    }
6049}
6050
6051fn reconcile_target_inner(
6052    target: &mut ReconcileTarget<'_>,
6053    subtree: &Path,
6054    started_at: u64,
6055    config: &ScanConfig,
6056    max_deferred_ops: usize,
6057    sink: &mut dyn FnMut(&Commit),
6058) -> Result<ReconcileReport> {
6059    let root = target.root_path()?;
6060    let root_meta = {
6061        crate::counters::bump(|c| c.stats += 1);
6062        fs::symlink_metadata(&root)
6063    }
6064    .map_err(|error| Error::io(&root, error))?;
6065    if !root_meta.is_dir() {
6066        return Err(Error::io(
6067            &root,
6068            std::io::Error::new(
6069                std::io::ErrorKind::NotADirectory,
6070                "reconciliation root is not a directory",
6071            ),
6072        ));
6073    }
6074    let root_dev = root_device(&root, &root_meta).map_err(|error| Error::io(&root, error))?;
6075    let start_depth = subtree.components().count();
6076    let mut report =
6077        ReconcileReport { reconcile_epoch: Some(started_at), ..ReconcileReport::default() };
6078    let mut tally = ProgressTally::new(config.progress.as_ref());
6079    let mut retry_frontier = None;
6080    let mut batch: Vec<ObservationOp> = Vec::with_capacity(config.batch_size.max(1));
6081
6082    if config.max_depth == Some(0) {
6083        remove_known_children(target, Path::new(""), config, &mut batch, sink, &mut report)?;
6084        return Ok(report);
6085    }
6086
6087    if !subtree.as_os_str().is_empty() {
6088        let baseline = target.expectation(subtree)?;
6089        let absolute = root.join(subtree);
6090        let (kind, attrs) = match observe_path(&absolute) {
6091            Ok(observed) => observed,
6092            Err(error) if error.kind() == std::io::ErrorKind::NotFound => {
6093                batch.push(ObservationOp::if_state(
6094                    Op::Remove { path: subtree.to_path_buf() },
6095                    baseline,
6096                ));
6097                push_lost_spelling_control(
6098                    target,
6099                    &root,
6100                    config,
6101                    subtree,
6102                    baseline,
6103                    &mut batch,
6104                    &mut report,
6105                )?;
6106                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6107                return Ok(report);
6108            }
6109            Err(error) => {
6110                report.scan.errors.push(Error::io(&absolute, error));
6111                if baseline.state != PathState::Absent {
6112                    batch.push(ObservationOp::if_state(
6113                        Op::Remove { path: subtree.to_path_buf() },
6114                        baseline,
6115                    ));
6116                }
6117                push_lost_spelling_control(
6118                    target,
6119                    &root,
6120                    config,
6121                    subtree,
6122                    baseline,
6123                    &mut batch,
6124                    &mut report,
6125                )?;
6126                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6127                return Ok(report);
6128            }
6129        };
6130        let disposition =
6131            crate::admission::decide_path(subtree, kind, config.hidden(), config.exclude_special);
6132        if disposition != crate::admission::Disposition::Retain {
6133            if baseline.state != PathState::Absent {
6134                batch.push(ObservationOp::if_state(
6135                    Op::Remove { path: subtree.to_path_buf() },
6136                    baseline,
6137                ));
6138            }
6139            if disposition == crate::admission::Disposition::ControlOnly {
6140                let read = read_control_op(config, &root, subtree, kind);
6141                push_read_control(target, subtree, baseline, read, &mut batch, &mut report)?;
6142            }
6143            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6144            return Ok(report);
6145        }
6146        report.scan.observe(kind, attrs);
6147        push_reconcile_upsert(target, subtree, kind, attrs, baseline, &mut batch, &mut report);
6148        // A retained control file at the root of the walk reads its rules here, as the
6149        // listing walk does for every retained entry it lists: a file does not descend,
6150        // so nothing below would read them, and the table kept the old source while the
6151        // pass reported complete and marked the path fresh. In the same batch as the
6152        // upsert, so both are arbitrated against one baseline.
6153        let read = read_control_op(config, &root, subtree, kind);
6154        push_read_control(target, subtree, baseline, read, &mut batch, &mut report)?;
6155        flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6156        if !should_descend(kind, attrs, start_depth.saturating_sub(1), root_dev, config) {
6157            if kind.is_dir() {
6158                remove_known_children(target, subtree, config, &mut batch, sink, &mut report)?;
6159            }
6160            tally.flush(&report.scan);
6161            return Ok(report);
6162        }
6163    }
6164
6165    if subtree.as_os_str().is_empty()
6166        && config.population == crate::query::IgnoredEntries::Include
6167        && config.reconciliation_worker_threads() > 1
6168    {
6169        if let ReconcileTarget::Direct(index) = target {
6170            match reconcile_direct_parallel(index, &root, root_dev, config, max_deferred_ops, sink)?
6171            {
6172                DirectParallelOutcome::Complete(parallel) => return Ok(parallel),
6173                DirectParallelOutcome::RetrySerial { prefix, remaining } => {
6174                    report = prefix;
6175                    report.reconcile_epoch = Some(started_at);
6176                    retry_frontier = Some(remaining);
6177                    // The wave workers reported the prefix themselves, and the wave
6178                    // that overflowed as well: progress counts that wave's reads twice,
6179                    // once there and once as the serial retry rereads it, while this
6180                    // report counts each directory once. Work done, not the answer.
6181                    tally.skip_to(&report.scan);
6182                }
6183            }
6184        }
6185    }
6186
6187    let mut queue: VecDeque<(PathBuf, usize)> = retry_frontier
6188        .unwrap_or_else(|| VecDeque::from(vec![(subtree.to_path_buf(), start_depth)]));
6189    let mut controls = if config.population == crate::query::IgnoredEntries::Include {
6190        None
6191    } else {
6192        Some(target.control_table()?)
6193    };
6194    let mut unreadable_controls = std::collections::BTreeSet::new();
6195    #[cfg(target_os = "macos")]
6196    let mut bulk_reader = (config.worker_threads() > 1).then(macos_bulk::Reader::new);
6197    let mut readers = Readers::new();
6198    let policy = ListingPolicy::every_child_stated(config.one_filesystem);
6199    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
6200        let errors_before = report.scan.errors.len();
6201        let abs_dir = root.join(&rel_dir);
6202        let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
6203        let mut had_control = target.has_control(&control_path)?;
6204        let mut listed_control = ListedControl::default();
6205        if let Some(table) = controls.as_mut() {
6206            let lookup = read_directory_control(config, &root, &control_path);
6207            listed_control.lookup(&lookup);
6208            match lookup {
6209                Ok(Some(op)) => {
6210                    let baseline = target.expectation(&control_path)?;
6211                    apply_discovery_control(table, &op)?;
6212                    batch.push(ObservationOp::if_state(op, baseline));
6213                }
6214                Ok(None) => {
6215                    table.remove(&control_path)?;
6216                    if had_control {
6217                        let baseline = target.expectation(&control_path)?;
6218                        batch.push(ObservationOp::if_state(
6219                            Op::ControlRemove { path: control_path.clone() },
6220                            baseline,
6221                        ));
6222                        had_control = false;
6223                    }
6224                }
6225                Err(error) => {
6226                    unreadable_controls.insert(rel_dir.clone());
6227                    report.scan.errors.push(error);
6228                }
6229            }
6230            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6231            if report.apply.stale > 0 || report.apply.resource_refused > 0 {
6232                // This directory's population decisions use the source just read.
6233                // If its conditional control commit lost ownership, walking siblings
6234                // against the local copy could prune under rules the index rejected.
6235                report.retry_required = true;
6236                return Ok(report);
6237            }
6238        }
6239        let (mut known, records_completeness) = target.listing_baseline(&rel_dir)?;
6240        let mut listing_complete = true;
6241        let process_entry = |name: OsString,
6242                             kind: EntryKind,
6243                             attrs: Attrs,
6244                             baseline: PathExpectation,
6245                             listed_control: &mut ListedControl,
6246                             target: &mut ReconcileTarget<'_>,
6247                             queue: &mut VecDeque<(PathBuf, usize)>,
6248                             batch: &mut Vec<ObservationOp>,
6249                             sink: &mut dyn FnMut(&Commit),
6250                             report: &mut ReconcileReport|
6251         -> Result<()> {
6252            let rel_path = rel_dir.join(&name);
6253            listed_control.listed(&name);
6254            let disposition =
6255                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
6256            if population_prunes(
6257                config.population,
6258                &rel_path,
6259                kind,
6260                disposition,
6261                controls.as_ref(),
6262                &unreadable_controls,
6263            ) {
6264                if baseline.state != PathState::Absent {
6265                    batch.push(ObservationOp::if_state(Op::Remove { path: rel_path }, baseline));
6266                }
6267                return Ok(());
6268            }
6269            if disposition != crate::admission::Disposition::Retain {
6270                if baseline.state != PathState::Absent {
6271                    batch.push(ObservationOp::if_state(
6272                        Op::Remove { path: rel_path.clone() },
6273                        baseline,
6274                    ));
6275                }
6276                if disposition == crate::admission::Disposition::ControlOnly {
6277                    let read = read_listed_control_op(config, &root, &rel_path, kind);
6278                    listed_control.read(&name, &read);
6279                    push_read_control(target, &rel_path, baseline, read, batch, report)?;
6280                }
6281                if batch.len() >= config.batch_size.max(1) {
6282                    flush_reconcile_batch(target, batch, sink, report)?;
6283                }
6284                return Ok(());
6285            }
6286            report.scan.observe(kind, attrs);
6287            push_reconcile_upsert(target, &rel_path, kind, attrs, baseline, batch, report);
6288            if batch.len() >= config.batch_size.max(1) {
6289                flush_reconcile_batch(target, batch, sink, report)?;
6290            }
6291            let read = read_listed_control_op(config, &root, &rel_path, kind);
6292            listed_control.read(&name, &read);
6293            push_read_control(target, &rel_path, baseline, read, batch, report)?;
6294            if batch.len() >= config.batch_size.max(1) {
6295                flush_reconcile_batch(target, batch, sink, report)?;
6296            }
6297
6298            if should_descend(kind, attrs, depth, root_dev, config) {
6299                queue.push_back((rel_path, depth + 1));
6300            } else if kind.is_dir() {
6301                remove_known_children(target, &rel_path, config, batch, sink, report)?;
6302            }
6303            Ok(())
6304        };
6305
6306        #[cfg(target_os = "macos")]
6307        let used_bulk = if walk_hook_covers(&abs_dir) {
6308            false
6309        } else if let Some(entries) = bulk_reader.as_mut().and_then(|reader| reader.read(&abs_dir))
6310        {
6311            report.scan.dirs_read += 1;
6312            for entry in entries {
6313                let baseline = match known.remove(&entry.name) {
6314                    Some(baseline) => baseline,
6315                    None => target.expectation(&rel_dir.join(&entry.name))?,
6316                };
6317                process_entry(
6318                    entry.name,
6319                    entry.kind,
6320                    entry.attrs,
6321                    baseline,
6322                    &mut listed_control,
6323                    target,
6324                    &mut queue,
6325                    &mut batch,
6326                    sink,
6327                    &mut report,
6328                )?;
6329            }
6330            true
6331        } else {
6332            false
6333        };
6334        #[cfg(not(target_os = "macos"))]
6335        let used_bulk = false;
6336
6337        if !used_bulk {
6338            let listing = match list_directory(&mut readers, &abs_dir, policy, None) {
6339                Ok(listing) => listing,
6340                Err(error) => {
6341                    report.scan.errors.push(Error::io(&abs_dir, error));
6342                    remove_known_children(target, &rel_dir, config, &mut batch, sink, &mut report)?;
6343                    if had_control {
6344                        let baseline = target.expectation(&control_path)?;
6345                        batch.push(ObservationOp::if_state(
6346                            Op::ControlRemove { path: control_path },
6347                            baseline,
6348                        ));
6349                        flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6350                    }
6351                    continue;
6352                }
6353            };
6354            report.scan.dirs_read += 1;
6355            let listing = reconcile_listing(listing, &abs_dir);
6356            for item in listing {
6357                let (name, observed) = match item {
6358                    Listed::Child { name, observed } => (name.into_owned(), observed),
6359                    Listed::Failed(error) => {
6360                        listing_complete = false;
6361                        report.scan.errors.push(Error::io(&abs_dir, error));
6362                        continue;
6363                    }
6364                };
6365                // Seeing the name proves it is not absent even if the following
6366                // metadata lookup fails. Remove it from the missing set before that
6367                // fallible lookup so an operational error cannot turn an existing
6368                // entry into a deletion.
6369                let baseline = match known.remove(&name) {
6370                    Some(baseline) => baseline,
6371                    None => target.expectation(&rel_dir.join(&name))?,
6372                };
6373                let (kind, attrs) = match observed {
6374                    Ok(Some(observed)) => observed,
6375                    Ok(None) => {
6376                        let entry_held = baseline.state != PathState::Absent;
6377                        for removal in
6378                            vanished_child_removals(&rel_dir, &name, entry_held, &mut had_control)
6379                                .into_iter()
6380                                .flatten()
6381                        {
6382                            batch.push(ObservationOp::if_state(removal, baseline));
6383                        }
6384                        if batch.len() >= config.batch_size.max(1) {
6385                            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6386                        }
6387                        continue;
6388                    }
6389                    Err(error) => {
6390                        listed_control.listed(&name);
6391                        report.scan.errors.push(Error::io(abs_dir.join(&name), error));
6392                        if baseline.state != PathState::Absent {
6393                            batch.push(ObservationOp::if_state(
6394                                Op::Remove { path: rel_dir.join(&name) },
6395                                baseline,
6396                            ));
6397                        }
6398                        if name == crate::control::CONTROL_FILE_NAME && had_control {
6399                            batch.push(ObservationOp::if_state(
6400                                Op::ControlRemove { path: rel_dir.join(&name) },
6401                                baseline,
6402                            ));
6403                            had_control = false;
6404                        }
6405                        if batch.len() >= config.batch_size.max(1) {
6406                            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6407                        }
6408                        continue;
6409                    }
6410                };
6411                process_entry(
6412                    name,
6413                    kind,
6414                    attrs,
6415                    baseline,
6416                    &mut listed_control,
6417                    target,
6418                    &mut queue,
6419                    &mut batch,
6420                    sink,
6421                    &mut report,
6422                )?;
6423            }
6424        }
6425
6426        for (name, baseline) in known {
6427            let restatement = listed_control.restatement_after_removing(&name);
6428            batch.push(ObservationOp::if_state(Op::Remove { path: rel_dir.join(name) }, baseline));
6429            // Before the batch can be flushed, so the rules the removal drops are back in the
6430            // same commit.
6431            if let Some(control) = restatement {
6432                let guard = target.expectation(&control_path)?;
6433                batch.push(ObservationOp::if_state(control, guard));
6434            }
6435            if batch.len() >= config.batch_size.max(1) {
6436                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6437            }
6438        }
6439        if had_control && !listed_control.seen {
6440            let baseline = target.expectation(&control_path)?;
6441            batch.push(ObservationOp::if_state(Op::ControlRemove { path: control_path }, baseline));
6442        }
6443        if listing_complete {
6444            // Only a directory with no error inside its own processing vouches for its
6445            // child set; an error under a sibling or a descendant is that directory's
6446            // to answer for, as discovery decides completeness per directory.
6447            if records_completeness && report.scan.errors.len() == errors_before {
6448                report.listed_incomplete.push(rel_dir);
6449            }
6450        }
6451        // Per directory, as in `revalidate`: an unchanged tree hands the sink nothing.
6452        tally.flush(&report.scan);
6453    }
6454
6455    flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
6456    report.scan.errors.sort_by_cached_key(ToString::to_string);
6457    tally.flush(&report.scan);
6458    Ok(report)
6459}
6460
6461#[derive(Debug, Default)]
6462struct DeferredReconcile {
6463    scan: ScanReport,
6464    unchanged: u64,
6465    operations: Vec<Op>,
6466    /// Each removal of a stale `.gitignore` entry whose rules a case variant now holds,
6467    /// followed by the restatement of those rules (`ListedControl`). Applied after the
6468    /// wave's sorted operations, in one batch, because the sort moves every removal after
6469    /// every other operation and a batch boundary could fall between the two.
6470    restated: Vec<Op>,
6471    discovered: Vec<(PathBuf, usize, RegionId)>,
6472    listed_incomplete: Vec<PathBuf>,
6473}
6474
6475enum DirectParallelOutcome {
6476    Complete(ReconcileReport),
6477    RetrySerial { prefix: ReconcileReport, remaining: VecDeque<(PathBuf, usize)> },
6478}
6479
6480/// Reconcile an exclusive full tree in bounded immutable-baseline waves.
6481///
6482/// No index write occurs while a wave's workers hold shared baseline references. That
6483/// lets each worker discard exact no-ops where they are observed instead of funnelling
6484/// every entry through one consumer. Effective changes still enter through ordinary
6485/// observations between waves, preserving both the index's sole mutation contract and
6486/// progressive delta delivery.
6487///
6488/// Unlike every other reconciliation path, the operations a wave defers are
6489/// unconditional: they carry no [`ObservationOp::if_state`] guard. Three properties
6490/// have to hold together for that to be safe, and a change to any one of them puts the
6491/// guards back. The target is an exclusive `&mut Index`, so no other producer can
6492/// commit between a worker's read and the wave's write. Nothing is applied while
6493/// workers run, so no baseline a worker compared against can go stale beneath it. And
6494/// a directory is only reconciled in a wave after the wave that discovered it has
6495/// committed, so a parent is never absent when its children arrive.
6496fn reconcile_direct_parallel(
6497    index: &mut Index,
6498    root: &Path,
6499    root_dev: u64,
6500    config: &ScanConfig,
6501    max_deferred_ops: usize,
6502    sink: &mut dyn FnMut(&Commit),
6503) -> Result<DirectParallelOutcome> {
6504    let mut frontier = DirectoryQueueState::seeded(
6505        (PathBuf::new(), 0),
6506        config.order,
6507        None,
6508        1,
6509        1,
6510        WorkerPolicyExperiment::ShippedOneShot,
6511    );
6512    let mut report = ReconcileReport::default();
6513    while !frontier.is_empty(config.order) {
6514        let mut wave = Vec::with_capacity(RECONCILE_WAVE_DIRECTORIES);
6515        while wave.len() < RECONCILE_WAVE_DIRECTORIES && !frontier.is_empty(config.order) {
6516            let remaining = RECONCILE_WAVE_DIRECTORIES - wave.len();
6517            frontier.take(remaining.min(DIR_CLAIM), config.order, &mut wave);
6518        }
6519
6520        let next = std::sync::atomic::AtomicUsize::new(0);
6521        let deferred_count = std::sync::atomic::AtomicUsize::new(0);
6522        let overflowed = std::sync::atomic::AtomicBool::new(false);
6523        let workers = config.reconciliation_worker_threads().min(wave.len());
6524        let baseline: &Index = index;
6525        let results: Vec<DeferredReconcile> = std::thread::scope(|scope| {
6526            let handles: Vec<_> = (0..workers)
6527                .map(|_| {
6528                    scope.spawn(|| {
6529                        reconcile_wave_worker(
6530                            baseline,
6531                            root,
6532                            root_dev,
6533                            config,
6534                            &wave,
6535                            &next,
6536                            &deferred_count,
6537                            &overflowed,
6538                            max_deferred_ops,
6539                        )
6540                    })
6541                })
6542                .collect();
6543
6544            handles
6545                .into_iter()
6546                .map(|handle| {
6547                    if let Ok(worker) = handle.join() {
6548                        return worker;
6549                    }
6550                    let mut worker = DeferredReconcile::default();
6551                    worker.scan.errors.push(Error::io(
6552                        root,
6553                        std::io::Error::other("a reconciliation worker thread panicked"),
6554                    ));
6555                    worker
6556                })
6557                .collect()
6558        });
6559
6560        // Nothing from an overflowing wave was applied, so the ordinary incremental
6561        // reconciler can resume at that wave. Completed waves and their statistics are
6562        // retained exactly once; restarting from the root would count their unchanged
6563        // entries again and misreport the logical reconciliation pass.
6564        if overflowed.load(std::sync::atomic::Ordering::Relaxed) {
6565            let mut remaining: VecDeque<_> =
6566                wave.into_iter().map(|(path, depth, _region)| (path, depth)).collect();
6567            let mut deferred = Vec::with_capacity(DIR_CLAIM);
6568            while !frontier.is_empty(config.order) {
6569                frontier.take(DIR_CLAIM, config.order, &mut deferred);
6570                remaining.extend(deferred.drain(..).map(|(path, depth, _region)| (path, depth)));
6571            }
6572            return Ok(DirectParallelOutcome::RetrySerial { prefix: report, remaining });
6573        }
6574
6575        let operation_count = deferred_count.load(std::sync::atomic::Ordering::Relaxed);
6576        let mut operations = Vec::with_capacity(operation_count);
6577        let mut restated = Vec::new();
6578        for worker in results {
6579            report.listed_incomplete.extend(worker.listed_incomplete);
6580            report.scan.absorb(worker.scan);
6581            report.apply.unchanged += worker.unchanged;
6582            report.observations = report.observations.saturating_add(worker.unchanged);
6583            operations.extend(worker.operations);
6584            restated.extend(worker.restated);
6585            for directory in worker.discovered {
6586                frontier.push(directory, config.order);
6587            }
6588        }
6589        apply_deferred_reconcile(
6590            index,
6591            &mut operations,
6592            restated,
6593            config,
6594            sink,
6595            &mut report.apply,
6596            &mut report.observations,
6597        )?;
6598    }
6599    report.scan.errors.sort_by_cached_key(ToString::to_string);
6600    Ok(DirectParallelOutcome::Complete(report))
6601}
6602
6603fn apply_deferred_reconcile(
6604    index: &mut Index,
6605    operations: &mut Vec<Op>,
6606    restated: Vec<Op>,
6607    config: &ScanConfig,
6608    sink: &mut dyn FnMut(&Commit),
6609    stats: &mut ApplyStats,
6610    observations: &mut u64,
6611) -> Result<()> {
6612    // Parent upserts establish real directory attributes before children arrive.
6613    // Removals run deepest first so a parent removal never precedes an independently
6614    // observed descendant operation. Deterministic causal order also makes emitted
6615    // commits stable for callers.
6616    operations.sort_by(|left, right| {
6617        let left_remove = matches!(left, Op::Remove { .. });
6618        let right_remove = matches!(right, Op::Remove { .. });
6619        left_remove.cmp(&right_remove).then_with(|| {
6620            let left_depth = left.path().components().count();
6621            let right_depth = right.path().components().count();
6622            if left_remove {
6623                right_depth.cmp(&left_depth).then_with(|| left.path().cmp(right.path()))
6624            } else {
6625                left_depth.cmp(&right_depth).then_with(|| left.path().cmp(right.path()))
6626            }
6627        })
6628    });
6629
6630    let batch_limit = config.batch_size.max(1);
6631    let mut batch = Vec::with_capacity(batch_limit.min(operations.len()));
6632    for operation in operations.drain(..) {
6633        batch.push(operation);
6634        if batch.len() >= batch_limit {
6635            flush_direct_reconcile_batch(index, &mut batch, sink, stats, observations)?;
6636        }
6637    }
6638    flush_direct_reconcile_batch(index, &mut batch, sink, stats, observations)?;
6639    // Whole, so no commit shows a directory without the rules a restatement puts back.
6640    let mut restated = restated;
6641    flush_direct_reconcile_batch(index, &mut restated, sink, stats, observations)
6642}
6643
6644#[allow(clippy::too_many_arguments)]
6645fn reconcile_wave_worker(
6646    index: &Index,
6647    root: &Path,
6648    root_dev: u64,
6649    config: &ScanConfig,
6650    wave: &[(PathBuf, usize, RegionId)],
6651    next: &std::sync::atomic::AtomicUsize,
6652    deferred_count: &std::sync::atomic::AtomicUsize,
6653    overflowed: &std::sync::atomic::AtomicBool,
6654    max_deferred_ops: usize,
6655) -> DeferredReconcile {
6656    let _counter_guard = crate::counters::thread_flush_guard();
6657    let mut result = DeferredReconcile::default();
6658    let mut tally = ProgressTally::new(config.progress.as_ref());
6659    #[cfg(target_os = "macos")]
6660    let mut bulk_reader = macos_bulk::Reader::new();
6661    let mut readers = Readers::new();
6662    let policy = ListingPolicy::every_child_stated(config.one_filesystem);
6663
6664    loop {
6665        let start = next.fetch_add(DIR_CLAIM, std::sync::atomic::Ordering::Relaxed);
6666        if start >= wave.len() {
6667            break;
6668        }
6669        let end = start.saturating_add(DIR_CLAIM).min(wave.len());
6670        for (rel_dir, depth, region) in &wave[start..end] {
6671            let errors_before = result.scan.errors.len();
6672            let mut known = collect_child_expectations(index, rel_dir);
6673            let abs_dir = root.join(rel_dir);
6674            let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
6675            let mut had_control = index.control_table().contains(&control_path);
6676            let mut listed_control = ListedControl::default();
6677            let mut control_errors = Vec::new();
6678            let mut vanished = Vec::new();
6679            let mut unverified = Vec::new();
6680            let mut control_read_failed = false;
6681            let mut listing_open_failed = false;
6682
6683            {
6684                let mut process_entry =
6685                    |name: OsString,
6686                     kind: EntryKind,
6687                     attrs: Attrs,
6688                     baseline: PathExpectation,
6689                     listed_control: &mut ListedControl| {
6690                        let rel_path = rel_dir.join(&name);
6691                        listed_control.listed(&name);
6692                        let disposition = crate::admission::decide(
6693                            &name,
6694                            kind,
6695                            config.hidden(),
6696                            config.exclude_special,
6697                        );
6698                        if disposition == crate::admission::Disposition::Reject {
6699                            if baseline.state != PathState::Absent {
6700                                defer_reconcile_op(
6701                                    Op::Remove { path: rel_path },
6702                                    &mut result.operations,
6703                                    deferred_count,
6704                                    overflowed,
6705                                    max_deferred_ops,
6706                                );
6707                            }
6708                            return;
6709                        }
6710                        if disposition == crate::admission::Disposition::ControlOnly {
6711                            let read = read_control_op(config, root, &rel_path, kind);
6712                            listed_control.read(&name, &read);
6713                            match read {
6714                                Ok(Some(Op::ControlUpsert { path, source })) => {
6715                                    if !index.control_table().source_is(&path, &source) {
6716                                        defer_reconcile_op(
6717                                            Op::ControlUpsert { path, source },
6718                                            &mut result.operations,
6719                                            deferred_count,
6720                                            overflowed,
6721                                            max_deferred_ops,
6722                                        );
6723                                    }
6724                                }
6725                                Ok(Some(Op::ControlRemove { path })) => {
6726                                    if index.control_table().contains(&path) {
6727                                        defer_reconcile_op(
6728                                            Op::ControlRemove { path },
6729                                            &mut result.operations,
6730                                            deferred_count,
6731                                            overflowed,
6732                                            max_deferred_ops,
6733                                        );
6734                                    }
6735                                }
6736                                Ok(Some(_) | None) => {}
6737                                Err(error) => {
6738                                    control_errors.push(error);
6739                                    control_read_failed = true;
6740                                }
6741                            }
6742                            return;
6743                        }
6744                        result.scan.entries += 1;
6745                        if kind == EntryKind::File {
6746                            result.scan.files_walked += 1;
6747                            result.scan.bytes_walked =
6748                                result.scan.bytes_walked.saturating_add(attrs.size);
6749                            result.scan.allocated_walked =
6750                                result.scan.allocated_walked.saturating_add(attrs.allocated);
6751                        }
6752                        if baseline.state == (PathState::Present { kind, attrs }) {
6753                            result.unchanged += 1;
6754                        } else {
6755                            defer_reconcile_op(
6756                                Op::Upsert { path: rel_path.clone(), kind, attrs },
6757                                &mut result.operations,
6758                                deferred_count,
6759                                overflowed,
6760                                max_deferred_ops,
6761                            );
6762                        }
6763                        let read = read_control_op(config, root, &rel_path, kind);
6764                        listed_control.read(&name, &read);
6765                        match read {
6766                            Ok(Some(Op::ControlUpsert { path, source })) => {
6767                                if !index.control_table().source_is(&path, &source) {
6768                                    defer_reconcile_op(
6769                                        Op::ControlUpsert { path, source },
6770                                        &mut result.operations,
6771                                        deferred_count,
6772                                        overflowed,
6773                                        max_deferred_ops,
6774                                    );
6775                                }
6776                            }
6777                            // The exact name's upsert as a non-file drops its rules by
6778                            // itself; a case variant's does not, so its lookup's removal
6779                            // is sent.
6780                            Ok(Some(Op::ControlRemove { path }))
6781                                if crate::control::control_spelling(&name)
6782                                    == Some(crate::control::ControlSpelling::Variant)
6783                                    && index.control_table().contains(&path) =>
6784                            {
6785                                defer_reconcile_op(
6786                                    Op::ControlRemove { path },
6787                                    &mut result.operations,
6788                                    deferred_count,
6789                                    overflowed,
6790                                    max_deferred_ops,
6791                                );
6792                            }
6793                            Ok(Some(_) | None) => {}
6794                            Err(error) => {
6795                                control_errors.push(error);
6796                                // The failed read was of the directory's control, through
6797                                // the entry for the exact name and a lookup for a variant.
6798                                control_read_failed = true;
6799                            }
6800                        }
6801
6802                        if should_descend(kind, attrs, *depth, root_dev, config) {
6803                            let child_region =
6804                                if *depth == 0 { RegionId::UNASSIGNED } else { *region };
6805                            result.discovered.push((rel_path, depth + 1, child_region));
6806                        } else if kind.is_dir() {
6807                            for name in collect_child_expectations(index, &rel_path).into_keys() {
6808                                defer_reconcile_op(
6809                                    Op::Remove { path: rel_path.join(name) },
6810                                    &mut result.operations,
6811                                    deferred_count,
6812                                    overflowed,
6813                                    max_deferred_ops,
6814                                );
6815                            }
6816                        }
6817                    };
6818
6819                #[cfg(target_os = "macos")]
6820                let used_bulk = if let Some(entries) =
6821                    (!walk_hook_covers(&abs_dir)).then(|| bulk_reader.read(&abs_dir)).flatten()
6822                {
6823                    result.scan.dirs_read += 1;
6824                    for entry in entries {
6825                        let baseline = known
6826                            .remove(&entry.name)
6827                            .unwrap_or_else(|| index.expectation(&rel_dir.join(&entry.name)));
6828                        process_entry(
6829                            entry.name,
6830                            entry.kind,
6831                            entry.attrs,
6832                            baseline,
6833                            &mut listed_control,
6834                        );
6835                    }
6836                    true
6837                } else {
6838                    false
6839                };
6840                #[cfg(not(target_os = "macos"))]
6841                let used_bulk = false;
6842
6843                if !used_bulk {
6844                    let listing = match list_directory(&mut readers, &abs_dir, policy, None) {
6845                        Ok(listing) => Some(listing),
6846                        Err(error) => {
6847                            result.scan.errors.push(Error::io(&abs_dir, error));
6848                            listing_open_failed = true;
6849                            None
6850                        }
6851                    };
6852                    if let Some(listing) = listing {
6853                        result.scan.dirs_read += 1;
6854                        let listing = reconcile_listing(listing, &abs_dir);
6855                        for item in listing {
6856                            let (name, observed) = match item {
6857                                Listed::Child { name, observed } => (name.into_owned(), observed),
6858                                Listed::Failed(error) => {
6859                                    result.scan.errors.push(Error::io(&abs_dir, error));
6860                                    continue;
6861                                }
6862                            };
6863                            // Match the serial path: an entry whose name was enumerated is
6864                            // not missing merely because its metadata could not be read.
6865                            let baseline = known
6866                                .remove(&name)
6867                                .unwrap_or_else(|| index.expectation(&rel_dir.join(&name)));
6868                            let (kind, attrs) = match observed {
6869                                Ok(Some(observed)) => observed,
6870                                Ok(None) => {
6871                                    // Removed once this directory's listing is done.
6872                                    vanished.push((name, baseline.state != PathState::Absent));
6873                                    continue;
6874                                }
6875                                Err(error) => {
6876                                    listed_control.listed(&name);
6877                                    result.scan.errors.push(Error::io(abs_dir.join(&name), error));
6878                                    unverified.push((name, baseline.state != PathState::Absent));
6879                                    continue;
6880                                }
6881                            };
6882                            process_entry(name, kind, attrs, baseline, &mut listed_control);
6883                        }
6884                    }
6885                }
6886            }
6887            control_read_failed |= listing_open_failed && had_control;
6888            result.scan.errors.append(&mut control_errors);
6889            if index.directory_complete(rel_dir) != Some(true)
6890                && result.scan.errors.len() == errors_before
6891            {
6892                result.listed_incomplete.push(rel_dir.clone());
6893            }
6894            for (name, entry_held) in vanished {
6895                for removal in vanished_child_removals(rel_dir, &name, entry_held, &mut had_control)
6896                    .into_iter()
6897                    .flatten()
6898                {
6899                    defer_reconcile_op(
6900                        removal,
6901                        &mut result.operations,
6902                        deferred_count,
6903                        overflowed,
6904                        max_deferred_ops,
6905                    );
6906                }
6907            }
6908            for (name, entry_held) in unverified {
6909                if entry_held {
6910                    defer_reconcile_op(
6911                        Op::Remove { path: rel_dir.join(&name) },
6912                        &mut result.operations,
6913                        deferred_count,
6914                        overflowed,
6915                        max_deferred_ops,
6916                    );
6917                }
6918                if name == crate::control::CONTROL_FILE_NAME {
6919                    control_read_failed = true;
6920                }
6921            }
6922            for (name, _) in known {
6923                let restatement = listed_control.restatement_after_removing(&name);
6924                let removal = Op::Remove { path: rel_dir.join(name) };
6925                match restatement {
6926                    Some(control) => {
6927                        for operation in [removal, control] {
6928                            defer_reconcile_op(
6929                                operation,
6930                                &mut result.restated,
6931                                deferred_count,
6932                                overflowed,
6933                                max_deferred_ops,
6934                            );
6935                        }
6936                    }
6937                    None => defer_reconcile_op(
6938                        removal,
6939                        &mut result.operations,
6940                        deferred_count,
6941                        overflowed,
6942                        max_deferred_ops,
6943                    ),
6944                }
6945            }
6946            if had_control && (!listed_control.seen || control_read_failed) {
6947                defer_reconcile_op(
6948                    Op::ControlRemove { path: control_path },
6949                    &mut result.operations,
6950                    deferred_count,
6951                    overflowed,
6952                    max_deferred_ops,
6953                );
6954            }
6955        }
6956        // Once per claimed chunk, as the cold walker reports, so a long wave on a slow
6957        // filesystem moves the counters while it runs rather than when it lands.
6958        tally.flush(&result.scan);
6959    }
6960    result
6961}
6962
6963/// What one reconciliation listing has shown about its directory's control file.
6964///
6965/// A listing names the control by the spelling the directory stores. The exact name
6966/// `.gitignore` is the control wherever it is listed, so listing it proves a control
6967/// present even when its read then fails. A case variant is the control only where a
6968/// lookup of `.gitignore` resolves to it (`read_control_op`), so what listing one proves is
6969/// what that lookup returned: a control when it found one, nothing when it missed, and
6970/// nothing when the variant's own metadata could not be read and no lookup was made.
6971/// That last case leaves the rules unknown, which the walk error records
6972/// (`crate::control::unreadable_control`); the sweep removes rules nothing showed, so the
6973/// table ends as a cold walk's does, having read nothing there either.
6974#[derive(Default)]
6975struct ListedControl {
6976    /// The listing showed the directory's control, or failed to read it, so the closing
6977    /// sweep must not remove it.
6978    seen: bool,
6979    /// The listing showed a case variant of the control name, such as `.GITIGNORE`.
6980    variant_listed: bool,
6981    /// The last control a lookup of the canonical path found during this listing: a
6982    /// listed case variant's, or a narrowed walk's before the listing.
6983    ///
6984    /// A case-only rename on a case-insensitive volume (`.gitignore` to `.GITIGNORE`) leaves
6985    /// the old spelling's entry in the index, and the closing sweep removes it. Removing
6986    /// the entry at the canonical path drops the rules it governs
6987    /// (`Index::projected_controls`), yet those rules are now the variant's, so when this
6988    /// listing showed the variant the sweep restates this observation right after that
6989    /// removal ([`Self::restatement_after_removing`]).
6990    looked_up: Option<Op>,
6991}
6992
6993impl ListedControl {
6994    /// Note that `name` was listed, before its metadata or its control is read.
6995    fn listed(&mut self, name: &OsStr) {
6996        match crate::control::control_spelling(name) {
6997            Some(crate::control::ControlSpelling::Exact) => self.seen = true,
6998            Some(crate::control::ControlSpelling::Variant) => self.variant_listed = true,
6999            None => {}
7000        }
7001    }
7002
7003    /// Note what reading the control through the listed `name` returned. Only a case
7004    /// variant's read is a lookup; the exact name already counted when it was listed.
7005    fn read(&mut self, name: &OsStr, read: &Result<Option<Op>>) {
7006        if crate::control::control_spelling(name) == Some(crate::control::ControlSpelling::Variant)
7007        {
7008            self.lookup(read);
7009        }
7010    }
7011
7012    /// Note what a lookup of the directory's canonical control path returned.
7013    fn lookup(&mut self, lookup: &Result<Option<Op>>) {
7014        match lookup {
7015            Ok(Some(observed)) => {
7016                self.seen = true;
7017                self.looked_up = Some(observed.clone());
7018            }
7019            Ok(None) => {}
7020            Err(_) => self.seen = true,
7021        }
7022    }
7023
7024    /// The observation to push right after the sweep removes the entry `name`: what the
7025    /// lookup found, when `name` is the stale exact spelling and this listing showed a case
7026    /// variant in its place.
7027    ///
7028    /// Only a listed variant can hold rules the removal would drop. Without one, the exact
7029    /// file is simply gone: a narrowed walk's lookup found it before the listing and it was
7030    /// deleted in between, and the removal leaves the table as bare as the directory, as a
7031    /// cold walk would. Restating that lookup would keep rules for a file that no longer
7032    /// exists until the next pass. That window remains only where a case-sensitive
7033    /// directory stores a variant beside the exact file deleted in it, a variant that never
7034    /// held the rules there.
7035    fn restatement_after_removing(&self, name: &OsStr) -> Option<Op> {
7036        if name == crate::control::CONTROL_FILE_NAME && self.variant_listed {
7037            self.looked_up.clone()
7038        } else {
7039            None
7040        }
7041    }
7042}
7043
7044/// The expectation that guards the control observation a listed entry at `path` produced,
7045/// given the entry's own, `entry`.
7046///
7047/// A guard is the expectation of the path its operation names. That is the entry's path
7048/// for the exact name, but a case variant's observation names the canonical control path,
7049/// so only a variant pays for `expect` to read that path's expectation. Asked only once an
7050/// observation exists, so an ordinary entry never pays for it.
7051fn control_guard<T>(path: &Path, entry: T, expect: impl FnOnce(&Path) -> T) -> T {
7052    match crate::control::path_control_spelling(path) {
7053        Some(crate::control::ControlSpelling::Variant) => {
7054            expect(&crate::control::sibling_control_path(path))
7055        }
7056        _ => entry,
7057    }
7058}
7059
7060/// Push what reading the control through the entry at `path` returned: the observation,
7061/// guarded by the expectation of the path it names ([`control_guard`]), or, when the read
7062/// failed, the removal of rules the table holds there, since nothing verifies them now.
7063fn push_read_control(
7064    target: &ReconcileTarget<'_>,
7065    path: &Path,
7066    baseline: PathExpectation,
7067    read: Result<Option<Op>>,
7068    batch: &mut Vec<ObservationOp>,
7069    report: &mut ReconcileReport,
7070) -> Result<()> {
7071    match read {
7072        Ok(Some(control)) => {
7073            let guard = control_guard(path, Ok(baseline), |named| target.expectation(named))?;
7074            batch.push(ObservationOp::if_state(control, guard));
7075        }
7076        Ok(None) => {}
7077        Err(error) => {
7078            // The failed read was of the entry itself for the exact name, and of the
7079            // canonical path a case variant looked up: the canonical path either way.
7080            let control_path = crate::control::sibling_control_path(path);
7081            if target.has_control(&control_path)? {
7082                let guard = control_guard(path, Ok(baseline), |named| target.expectation(named))?;
7083                batch
7084                    .push(ObservationOp::if_state(Op::ControlRemove { path: control_path }, guard));
7085            }
7086            report.scan.errors.push(error);
7087        }
7088    }
7089    Ok(())
7090}
7091
7092/// Push what a reconcile root at `path` that vanished, or could not be observed, leaves of
7093/// its directory's control, when `path` spells the control name.
7094///
7095/// The exact name was the lookup's target, so its loss removes the rules the table holds,
7096/// as it always has. A case variant's loss says nothing by itself: another spelling may
7097/// still resolve, or the variant was never the control, so the canonical path is looked
7098/// up and its answer stands, a miss removing the rules.
7099fn push_lost_spelling_control(
7100    target: &ReconcileTarget<'_>,
7101    root: &Path,
7102    config: &ScanConfig,
7103    path: &Path,
7104    baseline: PathExpectation,
7105    batch: &mut Vec<ObservationOp>,
7106    report: &mut ReconcileReport,
7107) -> Result<()> {
7108    match crate::control::path_control_spelling(path) {
7109        Some(crate::control::ControlSpelling::Exact) => {
7110            if target.has_control(path)? {
7111                batch.push(ObservationOp::if_state(
7112                    Op::ControlRemove { path: path.to_path_buf() },
7113                    baseline,
7114                ));
7115            }
7116        }
7117        Some(crate::control::ControlSpelling::Variant) => {
7118            let control_path = crate::control::sibling_control_path(path);
7119            let read = read_directory_control_or_removal(config, root, &control_path);
7120            push_read_control(target, path, baseline, read, batch, report)?;
7121        }
7122        None => {}
7123    }
7124    Ok(())
7125}
7126
7127/// What a listed child gone at its stat removes: its entry, if the baseline holds one, and
7128/// its rules, if it is the directory's control file and the table holds them.
7129///
7130/// The stat's `NotFound` is positive evidence that both are gone, so neither waits for a
7131/// complete listing; only a name the listing never returned has to, because it may merely
7132/// be unread. Removing a retained control file's entry drops its rules too, but a
7133/// hidden-pruned one has no entry, and without its own removal one unreadable sibling would
7134/// leave its rules applied with no file behind them. Clears `had_control` once the rules
7135/// are removed, so the listing's closing removals do not repeat it.
7136fn vanished_child_removals(
7137    dir: &Path,
7138    name: &OsStr,
7139    entry_held: bool,
7140    had_control: &mut bool,
7141) -> [Option<Op>; 2] {
7142    let path = dir.join(name);
7143    let rules = (*had_control && name == crate::control::CONTROL_FILE_NAME).then(|| {
7144        *had_control = false;
7145        Op::ControlRemove { path: path.clone() }
7146    });
7147    [entry_held.then_some(Op::Remove { path }), rules]
7148}
7149
7150fn defer_reconcile_op(
7151    operation: Op,
7152    operations: &mut Vec<Op>,
7153    deferred_count: &std::sync::atomic::AtomicUsize,
7154    overflowed: &std::sync::atomic::AtomicBool,
7155    max_deferred_ops: usize,
7156) {
7157    let position = deferred_count.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
7158    if position < max_deferred_ops {
7159        operations.push(operation);
7160    } else {
7161        overflowed.store(true, std::sync::atomic::Ordering::Relaxed);
7162    }
7163}
7164
7165fn flush_direct_reconcile_batch(
7166    index: &mut Index,
7167    batch: &mut Vec<Op>,
7168    sink: &mut dyn FnMut(&Commit),
7169    stats: &mut ApplyStats,
7170    observations: &mut u64,
7171) -> Result<()> {
7172    if batch.is_empty() {
7173        return Ok(());
7174    }
7175    *observations = observations.saturating_add(u64::try_from(batch.len()).unwrap_or(u64::MAX));
7176    let outcome = index.apply(&Observation::new(std::mem::take(batch)))?;
7177    merge_apply_stats(stats, outcome.stats);
7178    if let Some(commit) = outcome.commit.as_ref() {
7179        sink(commit);
7180    }
7181    Ok(())
7182}
7183
7184/// Drain and reconcile every pending invalidation, collapsing nested requests.
7185///
7186/// An invalidation whose reconciliation comes back incomplete -- a subtree that could not
7187/// be read, or a conditional commit that lost a race -- is queued again, so the next call
7188/// retries it. That suits a caller that drains when it chooses. A caller that drains after
7189/// every event would re-walk an unreadable subtree each time, and belongs on
7190/// [`reconcile_pending_handle`], which settles it instead.
7191pub fn reconcile_pending(
7192    index: &mut Index,
7193    config: &ScanConfig,
7194    sink: &mut dyn FnMut(&Commit),
7195) -> Result<ReconcileReport> {
7196    let mut target = ReconcileTarget::Direct(index);
7197    reconcile_pending_target(&mut target, config, sink)
7198}
7199
7200/// Drain and reconcile invalidations on a shared index.
7201///
7202/// Unlike [`reconcile_pending`], a subtree that could not be read is not queued again.
7203/// `Watcher::apply_next` drains after every event, and a retry there re-walks the same
7204/// unreadable subtree on each unrelated one, for the life of the watch. The error is a
7205/// settled boundary instead: the subtree stays [`crate::Freshness::Partial`], the returned
7206/// report names the error once, and the watcher retains its cause as an issue. Only a lost
7207/// race -- a stale conditional commit -- is queued for the next call. Invalidate the subtree
7208/// again to retry it deliberately.
7209pub fn reconcile_pending_handle(
7210    handle: &IndexHandle,
7211    config: &ScanConfig,
7212    sink: &mut dyn FnMut(&Commit),
7213) -> Result<ReconcileReport> {
7214    let mut target = ReconcileTarget::Shared(handle);
7215    reconcile_pending_target(&mut target, config, sink)
7216}
7217
7218/// Drain and reconcile invalidations under an opened-root lifecycle and resource bound.
7219#[cfg(feature = "watch")]
7220pub(crate) fn reconcile_pending_handle_controlled(
7221    handle: &IndexHandle,
7222    config: &ScanConfig,
7223    control: &dyn ReconcileControl,
7224    sink: &mut dyn FnMut(&Commit),
7225) -> Result<ReconcileReport> {
7226    let mut target = ReconcileTarget::Controlled { handle, control };
7227    reconcile_pending_target(&mut target, config, sink)
7228}
7229
7230fn reconcile_pending_target(
7231    target: &mut ReconcileTarget<'_>,
7232    config: &ScanConfig,
7233    sink: &mut dyn FnMut(&Commit),
7234) -> Result<ReconcileReport> {
7235    config.validate_for_scope(target.scope()?)?;
7236    let roots = take_invalidation_roots(target)?;
7237    let mut combined = ReconcileReport::default();
7238    for (position, (root, reason)) in roots.iter().enumerate() {
7239        match reconcile_target(target, root, config, sink) {
7240            Ok(report) => {
7241                if target.retries_incomplete(&report) {
7242                    target.restore_pending_invalidations(vec![(root.clone(), *reason)])?;
7243                }
7244                merge_reconcile_report(&mut combined, report);
7245            }
7246            Err(error) => {
7247                target.restore_pending_invalidations(roots[position..].to_vec())?;
7248                return Err(error);
7249            }
7250        }
7251    }
7252    Ok(combined)
7253}
7254
7255fn take_invalidation_roots(
7256    target: &mut ReconcileTarget<'_>,
7257) -> Result<Vec<(PathBuf, crate::InvalidateReason)>> {
7258    let mut pending = target.take_pending_invalidations()?;
7259    pending.sort_by(|(left, _), (right, _)| {
7260        left.components().count().cmp(&right.components().count()).then_with(|| left.cmp(right))
7261    });
7262    let mut roots: Vec<(PathBuf, crate::InvalidateReason)> = Vec::new();
7263    for (path, reason) in pending {
7264        if roots.iter().any(|(root, _)| path.starts_with(root)) {
7265            continue;
7266        }
7267        roots.push((path, reason));
7268    }
7269
7270    Ok(roots)
7271}
7272
7273fn remove_known_children(
7274    target: &mut ReconcileTarget<'_>,
7275    path: &Path,
7276    config: &ScanConfig,
7277    batch: &mut Vec<ObservationOp>,
7278    sink: &mut dyn FnMut(&Commit),
7279    report: &mut ReconcileReport,
7280) -> Result<()> {
7281    for (name, baseline) in target.child_states(path)? {
7282        batch.push(ObservationOp::if_state(Op::Remove { path: path.join(name) }, baseline));
7283        if batch.len() >= config.batch_size.max(1) {
7284            flush_reconcile_batch(target, batch, sink, report)?;
7285        }
7286    }
7287    flush_reconcile_batch(target, batch, sink, report)
7288}
7289
7290fn push_reconcile_upsert(
7291    target: &ReconcileTarget<'_>,
7292    path: &Path,
7293    kind: EntryKind,
7294    attrs: Attrs,
7295    baseline: PathExpectation,
7296    batch: &mut Vec<ObservationOp>,
7297    report: &mut ReconcileReport,
7298) {
7299    // An exclusive Index borrow cannot race another index producer. If filesystem
7300    // metadata exactly matches the captured state, applying this upsert can only be a
7301    // no-op, so avoid allocating an owned op and walking the index again. Shared
7302    // reconciliation keeps the conditional observation so ABA arbitration remains
7303    // authoritative between its read and write lock boundaries.
7304    if target.direct_upsert_is_unchanged(baseline, kind, attrs) {
7305        report.observations = report.observations.saturating_add(1);
7306        report.apply.unchanged = report.apply.unchanged.saturating_add(1);
7307        return;
7308    }
7309    batch.push(ObservationOp::if_state(
7310        Op::Upsert { path: path.to_path_buf(), kind, attrs },
7311        baseline,
7312    ));
7313}
7314
7315fn flush_reconcile_batch(
7316    target: &mut ReconcileTarget<'_>,
7317    batch: &mut Vec<ObservationOp>,
7318    sink: &mut dyn FnMut(&Commit),
7319    report: &mut ReconcileReport,
7320) -> Result<()> {
7321    if batch.is_empty() {
7322        return Ok(());
7323    }
7324    report.observations =
7325        report.observations.saturating_add(u64::try_from(batch.len()).unwrap_or(u64::MAX));
7326    let started_at = report.reconcile_epoch.expect("reconciliation report has an owner");
7327    let outcome = target.apply(started_at, &Observation::from_ops(std::mem::take(batch)))?;
7328    merge_apply_stats(&mut report.apply, outcome.stats);
7329    if let Some(commit) = outcome.commit.as_ref() {
7330        sink(commit);
7331    }
7332    Ok(())
7333}
7334
7335fn merge_apply_stats(total: &mut ApplyStats, addition: ApplyStats) {
7336    total.inserted += addition.inserted;
7337    total.updated += addition.updated;
7338    total.removed += addition.removed;
7339    total.unchanged += addition.unchanged;
7340    total.invalidated += addition.invalidated;
7341    total.controls += addition.controls;
7342    total.reclassified += addition.reclassified;
7343    total.stale += addition.stale;
7344    total.resource_refused += addition.resource_refused;
7345}
7346
7347fn merge_reconcile_report(total: &mut ReconcileReport, addition: ReconcileReport) {
7348    total.retry_required |= addition.retry_required;
7349    total.scan.dirs_read += addition.scan.dirs_read;
7350    total.scan.entries += addition.scan.entries;
7351    total.scan.files_walked += addition.scan.files_walked;
7352    total.scan.bytes_walked = total.scan.bytes_walked.saturating_add(addition.scan.bytes_walked);
7353    total.scan.allocated_walked =
7354        total.scan.allocated_walked.saturating_add(addition.scan.allocated_walked);
7355    total.scan.errors.extend(addition.scan.errors);
7356    total.observations = total.observations.saturating_add(addition.observations);
7357    merge_apply_stats(&mut total.apply, addition.apply);
7358    total.listed_incomplete.extend(addition.listed_incomplete);
7359}
7360
7361fn should_descend(
7362    kind: EntryKind,
7363    attrs: Attrs,
7364    parent_depth: usize,
7365    root_dev: u64,
7366    config: &ScanConfig,
7367) -> bool {
7368    crate::admission::should_descend(
7369        kind,
7370        attrs,
7371        parent_depth,
7372        root_dev,
7373        config.max_depth,
7374        config.one_filesystem,
7375    )
7376}
7377
7378pub(crate) fn normalize_subtree(path: &Path) -> Result<PathBuf> {
7379    let mut normalized = PathBuf::new();
7380    for component in path.components() {
7381        match component {
7382            Component::Normal(part) => normalized.push(part),
7383            Component::CurDir => {}
7384            Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
7385                return Err(Error::PathEscapesRoot(path.to_path_buf()));
7386            }
7387        }
7388    }
7389    Ok(normalized)
7390}
7391
7392fn resolve_subtree_root(
7393    target: &ReconcileTarget<'_>,
7394    subtree: &Path,
7395    config: &ScanConfig,
7396) -> Result<PathBuf> {
7397    if subtree.as_os_str().is_empty() {
7398        return Ok(PathBuf::new());
7399    }
7400    let root = target.root_path()?;
7401    crate::counters::bump(|c| c.stats += 1);
7402    let Ok(root_metadata) = fs::symlink_metadata(&root) else {
7403        // The applying pass reports operational root failures as partial.
7404        return Ok(subtree.to_path_buf());
7405    };
7406    if !root_metadata.is_dir() {
7407        return Ok(subtree.to_path_buf());
7408    }
7409    let Ok(root_dev) = root_device(&root, &root_metadata) else {
7410        return Ok(subtree.to_path_buf());
7411    };
7412    let mut prefix = PathBuf::new();
7413    let mut components = subtree.components().peekable();
7414    while let Some(component) = components.next() {
7415        if components.peek().is_none() {
7416            break; // The boundary entry itself remains visible even when descent stops.
7417        }
7418        prefix.push(component.as_os_str());
7419        let (kind, attrs) = match observe_path(&root.join(&prefix)) {
7420            Ok(observed) => observed,
7421            Err(error)
7422                if matches!(
7423                    error.kind(),
7424                    std::io::ErrorKind::NotFound | std::io::ErrorKind::NotADirectory
7425                ) =>
7426            {
7427                return Ok(prefix);
7428            }
7429            Err(_) => break, // The applying pass records operational failures as partial.
7430        };
7431        if kind == EntryKind::Symlink {
7432            return Err(Error::SubtreeOutsideScanScope {
7433                path: subtree.to_path_buf(),
7434                scope: config.scope(),
7435            });
7436        }
7437        if kind != EntryKind::Dir {
7438            return Ok(prefix);
7439        }
7440        if config.one_filesystem && root_dev != 0 && attrs.dev != 0 && attrs.dev != root_dev {
7441            return Err(Error::SubtreeOutsideScanScope {
7442                path: subtree.to_path_buf(),
7443                scope: config.scope(),
7444            });
7445        }
7446    }
7447    Ok(subtree.to_path_buf())
7448}
7449
7450/// Read an entry's kind and roll-up attributes out of its metadata.
7451///
7452/// Exposed so the watch layer verifies entries exactly the way the walker records them —
7453/// two stat interpretations that could drift would show up as an index that disagrees
7454/// with itself depending on which producer last touched a path.
7455///
7456/// On Windows the observation comes from a fresh non-following handle, and `meta` is
7457/// what answers for an entry whose handle cannot be opened because it is locked or
7458/// access is denied — the same fallback std's `metadata` makes, with identity and change
7459/// time unavailable for that entry.
7460pub fn observe(path: &Path, meta: &fs::Metadata) -> std::io::Result<(EntryKind, Attrs)> {
7461    #[cfg(windows)]
7462    {
7463        windows_metadata::observe(path, || Ok(meta.clone()))
7464    }
7465    #[cfg(not(windows))]
7466    {
7467        Ok((kind_from(meta), attrs_from(path, meta)?))
7468    }
7469}
7470
7471pub(crate) fn observe_dir_entry(
7472    entry: &fs::DirEntry,
7473) -> std::io::Result<Option<(EntryKind, Attrs)>> {
7474    #[cfg(windows)]
7475    {
7476        crate::counters::bump(|c| c.stats += 1);
7477        #[cfg(test)]
7478        {
7479            let path = entry.path();
7480            if let Some(error) =
7481                walk_hook(&path).and_then(|hook| hook(WalkHookPoint::ChildMetadata(&path)))
7482            {
7483                return missing_as_none(Err(error));
7484            }
7485        }
7486        // The listing already holds the entry's enumeration data; it is read only when
7487        // the handle cannot be opened, so the ordinary path allocates nothing more.
7488        missing_as_none(windows_metadata::observe(&entry.path(), || entry.metadata()))
7489    }
7490    #[cfg(not(windows))]
7491    {
7492        #[cfg(test)]
7493        {
7494            let path = entry.path();
7495            if let Some(error) = child_metadata_hook(&path) {
7496                return missing_as_none(Err(error));
7497            }
7498        }
7499        #[cfg(all(target_os = "linux", target_env = "gnu"))]
7500        {
7501            // std's `DirEntry::metadata` is `statx` without `AT_NO_AUTOMOUNT` wherever
7502            // `statx` is served, and std keeps the listing's descriptor to itself, so the
7503            // reader's stat answers by path (fdu-d2fn). This is the route of a directory
7504            // the reader declined, so the whole-path resolution is paid rarely.
7505            if !linux_dents::statx_unavailable() {
7506                crate::counters::bump(|c| c.stats += 1);
7507                if let Some(observed) = linux_dents::stat_path(&entry.path()) {
7508                    return missing_as_none(observed);
7509                }
7510                // `statx` has just been found unavailable: std answers, with `fstatat`.
7511            }
7512        }
7513        let Some(meta) = missing_as_none(metadata_for_fingerprint(entry))? else {
7514            return Ok(None);
7515        };
7516        Ok(Some((kind_from(&meta), attrs_from(Path::new(""), &meta)?)))
7517    }
7518}
7519
7520/// The non-following observation of `path` itself, as [`observe`] reads it from
7521/// `symlink_metadata`, and on Linux one that never triggers an automount.
7522///
7523/// For a path a route holds as an entry and verifies by itself: a reconciliation's
7524/// subtree and the prefixes above it, a change the watch verifies, a control looked up
7525/// by name. The walk root is not one; [`root_device`] says why.
7526pub(crate) fn observe_path(path: &Path) -> std::io::Result<(EntryKind, Attrs)> {
7527    crate::counters::bump(|c| c.stats += 1);
7528    #[cfg(all(target_os = "linux", target_env = "gnu"))]
7529    if let Some(observed) = linux_dents::stat_path(path) {
7530        return observed;
7531    }
7532    let metadata = fs::symlink_metadata(path)?;
7533    observe(path, &metadata)
7534}
7535
7536#[cfg(not(windows))]
7537fn kind_from(meta: &fs::Metadata) -> EntryKind {
7538    let file_type = meta.file_type();
7539    if file_type.is_symlink() {
7540        EntryKind::Symlink
7541    } else if file_type.is_dir() {
7542        EntryKind::Dir
7543    } else if file_type.is_file() {
7544        EntryKind::File
7545    } else {
7546        EntryKind::Other
7547    }
7548}
7549
7550#[cfg(unix)]
7551#[allow(clippy::unnecessary_wraps)] // Windows observation is fallible; keep one call contract.
7552pub(crate) fn attrs_from(_path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
7553    use std::os::unix::fs::MetadataExt;
7554    Ok(Attrs {
7555        size: meta.size(),
7556        // st_blocks is in 512-byte units by POSIX convention regardless of the
7557        // filesystem's own block size.
7558        allocated: meta.blocks().saturating_mul(512),
7559        mtime_ns: compose_ns(meta.mtime(), meta.mtime_nsec()),
7560        ctime_ns: compose_ns(meta.ctime(), meta.ctime_nsec()),
7561        inode: meta.ino(),
7562        dev: meta.dev(),
7563    })
7564}
7565
7566#[cfg(unix)]
7567fn compose_ns(secs: i64, nanos: i64) -> i64 {
7568    secs.saturating_mul(1_000_000_000).saturating_add(nanos)
7569}
7570
7571// Windows observation goes through `windows_metadata::observe` on every route; only a
7572// test still derives attributes from metadata alone.
7573#[cfg(all(windows, test))]
7574pub(crate) fn attrs_from(path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
7575    windows_metadata::observe(path, || Ok(meta.clone())).map(|(_, attrs)| attrs)
7576}
7577
7578#[cfg(not(any(unix, windows)))]
7579#[allow(clippy::unnecessary_wraps)] // Windows observation is fallible; keep one call contract.
7580pub(crate) fn attrs_from(_path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
7581    let mtime_ns = meta.modified().map_or(0, system_time_ns);
7582    Ok(Attrs {
7583        size: meta.len(),
7584        // No allocated size without platform-specific calls; apparent size is the
7585        // honest fallback rather than a guess at block rounding.
7586        allocated: meta.len(),
7587        mtime_ns,
7588        // Windows has no ctime in the Unix sense. Leaving it zero means the fingerprint
7589        // degrades to size + mtime there, which is what every portable tool does.
7590        ctime_ns: 0,
7591        inode: 0,
7592        dev: 0,
7593    })
7594}
7595
7596/// The device a walk's root is on, which bounds a one-filesystem walk.
7597///
7598/// Only the device is needed, and on Windows it is read without demanding a consistent
7599/// observation of the root's times, which change whenever a child is created or removed.
7600/// The device that bounds a `one_filesystem` walk from `root`, whose non-following
7601/// metadata is `meta`.
7602///
7603/// The walk root is resolved. The user named it and the listing that follows opens it,
7604/// and on Linux an open mounts an unmounted autofs trigger where a non-following stat
7605/// need not: `symlink_metadata` does on glibc (std's `statx` without `AT_NO_AUTOMOUNT`)
7606/// and does not on musl (`fstatat`). A trigger's own device would then exclude every
7607/// entry the listing finds under it, so on Linux the device is read from an opened
7608/// descriptor on every route and under either libc, and `meta`'s device answers only
7609/// when the open fails, as the listing then fails the same way. Every other stat of the
7610/// tree never mounts ([`observe_dir_entry`], [`observe_path`], `linux_dents`).
7611pub(crate) fn root_device(root: &Path, meta: &fs::Metadata) -> std::io::Result<u64> {
7612    #[cfg(windows)]
7613    {
7614        let _ = meta;
7615        windows_metadata::volume_serial(root)
7616    }
7617    #[cfg(target_os = "linux")]
7618    {
7619        use std::os::unix::fs::OpenOptionsExt as _;
7620        let opened = fs::OpenOptions::new()
7621            .read(true)
7622            .custom_flags(libc::O_DIRECTORY | libc::O_NOFOLLOW)
7623            .open(root)
7624            .and_then(|directory| directory.metadata());
7625        attrs_from(root, opened.as_ref().unwrap_or(meta)).map(|attrs| attrs.dev)
7626    }
7627    #[cfg(not(any(windows, target_os = "linux")))]
7628    {
7629        attrs_from(root, meta).map(|attrs| attrs.dev)
7630    }
7631}
7632
7633pub(crate) fn attrs_from_file(file: &fs::File, meta: &fs::Metadata) -> std::io::Result<Attrs> {
7634    #[cfg(windows)]
7635    {
7636        let _ = meta;
7637        windows_metadata::attrs_from_file(file)
7638    }
7639    #[cfg(not(windows))]
7640    {
7641        let _ = file;
7642        attrs_from(Path::new(""), meta)
7643    }
7644}
7645
7646#[cfg(any(not(any(unix, windows)), test))]
7647fn system_time_ns(time: std::time::SystemTime) -> i64 {
7648    match time.duration_since(std::time::UNIX_EPOCH) {
7649        Ok(duration) => i64::try_from(duration.as_nanos()).unwrap_or(i64::MAX),
7650        Err(error) => {
7651            i64::try_from(error.duration().as_nanos()).map_or(i64::MIN, i64::saturating_neg)
7652        }
7653    }
7654}
7655
7656#[cfg(test)]
7657mod tests {
7658    use super::*;
7659    use std::fs::File;
7660    use std::io::Write;
7661
7662    fn write_file(path: &Path, contents: &[u8]) {
7663        if let Some(parent) = path.parent() {
7664            fs::create_dir_all(parent).expect("create parent");
7665        }
7666        let mut f = File::create(path).expect("create file");
7667        f.write_all(contents).expect("write");
7668    }
7669
7670    fn sample_tree() -> tempfile::TempDir {
7671        let dir = tempfile::tempdir().expect("tempdir");
7672        write_file(&dir.path().join("a.txt"), b"hello");
7673        write_file(&dir.path().join("src/main.rs"), b"fn main() {}");
7674        write_file(&dir.path().join("src/deep/nested.rs"), b"// nested");
7675        crate::test_support::settle_allocations(dir.path());
7676        dir
7677    }
7678
7679    /// A counter that silently reads zero is worse than a missing one, because a report
7680    /// full of zeroes invites the conclusion that the work did not happen.
7681    ///
7682    /// This has already gone wrong twice: once when the per-entry counter was added to
7683    /// the serial walk while the parallel walk went uninstrumented, and once when a
7684    /// clippy fix hoisted a `read_dir` out of a match scrutinee and took the counter
7685    /// with it. Both builds compiled, passed every other test, and reported zero. This
7686    /// asserts the relationships a real walk must satisfy, so the next such edit fails
7687    /// here instead of in a report someone believes.
7688    #[test]
7689    fn a_walk_moves_every_counter_it_should() {
7690        // Both walkers, because they are separate loops with separate call sites. The
7691        // first version of this test only exercised the parallel one, and deleting the
7692        // serial walker's counter still passed — a guard covering one path gives false
7693        // confidence about the other.
7694        let _serial = crate::counters::test_serial();
7695        crate::counters::enable(true);
7696        for threads in [Some(1), Some(4)] {
7697            let dir = sample_tree();
7698            let config = ScanConfig { threads, ..ScanConfig::default() };
7699
7700            // Deltas around the scan, not absolute totals. The counters are
7701            // process-global, so a test running beside this one can add to them — and
7702            // `test_serial` cannot prevent that, since it only serializes tests that
7703            // take it, not every test that happens to walk a tree.
7704            let before = crate::counters::snapshot();
7705            let report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
7706            crate::counters::flush_thread();
7707            let after = crate::counters::snapshot();
7708            let observed_entries = after.dir_entries - before.dir_entries;
7709            let observed_opens = after.dir_opens - before.dir_opens;
7710            let observed_stats = after.stats - before.stats;
7711
7712            // `>=` rather than `==`, and the direction is the whole point: concurrent
7713            // work can only inflate these, never deflate them. So a counter that is too
7714            // low means a path ran uninstrumented, which is the failure worth catching
7715            // and the one that has actually happened — the macOS bulk reader reported
7716            // zero opens against three real ones. Equality would catch double-counting
7717            // too, and would be flaky for it.
7718            assert!(
7719                observed_entries >= report.entries,
7720                "every enumerated entry is counted at {threads:?}: {observed_entries} < {}",
7721                report.entries
7722            );
7723            assert!(
7724                observed_opens >= report.dirs_read,
7725                "every directory open is counted at {threads:?}: {observed_opens} < {}",
7726                report.dirs_read
7727            );
7728            assert!(
7729                observed_stats >= report.entries,
7730                "every entry is stated at {threads:?}: {observed_stats} < {}",
7731                report.entries
7732            );
7733            #[cfg(any(target_os = "macos", all(target_os = "linux", target_env = "gnu")))]
7734            {
7735                let observed_enum = after.dir_enumeration_calls - before.dir_enumeration_calls;
7736                // The serial walker is the portable `read_dir` path, which cannot see
7737                // getdents multiplicity. Enumeration calls are a native-backend fact.
7738                if threads != Some(1) {
7739                    assert!(
7740                        observed_enum >= report.dirs_read,
7741                        "every successful native directory issues at least one enumeration \
7742                         call at {threads:?}: {observed_enum} < {}",
7743                        report.dirs_read
7744                    );
7745                }
7746            }
7747
7748            // Deliberately not asserted: `allocs` stays zero in a library test, because
7749            // allocation counting needs a binary to install `CountingAlloc` as its
7750            // global allocator and a test harness installs its own. The probe covers
7751            // that half; this covers the counters the library itself drives.
7752        }
7753        crate::counters::enable(false);
7754    }
7755
7756    #[test]
7757    fn summary_fold_skips_stat_on_directories_and_symlinks() {
7758        let _serial = crate::counters::test_serial();
7759        crate::counters::enable(true);
7760        let dir = tempfile::tempdir().expect("tempdir");
7761        fs::create_dir(dir.path().join("src")).expect("directory");
7762        write_file(&dir.path().join("a.txt"), b"hi");
7763        #[cfg(unix)]
7764        std::os::unix::fs::symlink("a.txt", dir.path().join("link")).expect("symlink");
7765        let config = ScanConfig { threads: Some(1), read_controls: false, ..ScanConfig::default() };
7766
7767        crate::counters::test_thread_reset();
7768        let scan_report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
7769        let scan_stats = crate::counters::test_thread_snapshot().stats;
7770
7771        crate::counters::test_thread_reset();
7772        let fold_report = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
7773        let fold_stats = crate::counters::test_thread_snapshot().stats;
7774        crate::counters::enable(false);
7775
7776        assert_eq!(fold_report.entries, scan_report.entries);
7777        assert_eq!(fold_report.files_walked, scan_report.files_walked);
7778        assert_eq!(fold_report.bytes_walked, scan_report.bytes_walked);
7779        // Windows observes every listed entry through a fresh handle on both paths, so the
7780        // fold performs exactly the retained walk's observations there; the skip is a
7781        // non-Windows saving.
7782        #[cfg(unix)]
7783        assert!(
7784            fold_stats < scan_stats,
7785            "fold {fold_stats} should skip directory/symlink stats versus scan {scan_stats}"
7786        );
7787        // The fold stats `a.txt` and, when the listing yields `src` or `link` before it,
7788        // the first of those, whose stat proves the directory searchable
7789        // ([`Searchability`]); the others take their kind from the listing.
7790        #[cfg(not(windows))]
7791        let proof = u64::from(first_listed(dir.path()) != "a.txt");
7792        #[cfg(unix)]
7793        assert_eq!(scan_stats.saturating_sub(fold_stats), 2 - proof);
7794        #[cfg(not(any(unix, windows)))]
7795        assert_eq!(scan_stats.saturating_sub(fold_stats), 1 - proof);
7796        #[cfg(windows)]
7797        assert_eq!(fold_stats, scan_stats);
7798    }
7799
7800    /// The name the filesystem lists first in `dir`, read rather than assumed: under the
7801    /// skip a listing stats its children until one stat succeeds ([`Searchability`]), so
7802    /// how many stats the fold saves depends on the enumeration order.
7803    #[cfg(not(windows))]
7804    fn first_listed(dir: &Path) -> std::ffi::OsString {
7805        fs::read_dir(dir).expect("listing").next().expect("an entry").expect("entry").file_name()
7806    }
7807
7808    #[cfg(unix)]
7809    #[test]
7810    fn summary_fold_still_stats_directories_when_bound_to_one_filesystem() {
7811        let _serial = crate::counters::test_serial();
7812        crate::counters::enable(true);
7813        let dir = tempfile::tempdir().expect("tempdir");
7814        fs::create_dir(dir.path().join("src")).expect("directory");
7815        write_file(&dir.path().join("a.txt"), b"hi");
7816        std::os::unix::fs::symlink("a.txt", dir.path().join("link")).expect("symlink");
7817        let config = ScanConfig {
7818            threads: Some(1),
7819            read_controls: false,
7820            one_filesystem: true,
7821            ..ScanConfig::default()
7822        };
7823
7824        crate::counters::test_thread_reset();
7825        let scan_report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
7826        let scan_stats = crate::counters::test_thread_snapshot().stats;
7827
7828        crate::counters::test_thread_reset();
7829        let fold_report = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
7830        let fold_stats = crate::counters::test_thread_snapshot().stats;
7831        crate::counters::enable(false);
7832
7833        assert_eq!(fold_report.entries, scan_report.entries);
7834        // `src` and `a.txt` are stated on both routes; `link` takes its kind from the
7835        // listing only once one of them has been stated before it, which proves the
7836        // directory searchable ([`Searchability`]).
7837        let proof = u64::from(first_listed(dir.path()) == "link");
7838        assert_eq!(scan_stats.saturating_sub(fold_stats), 1 - proof);
7839    }
7840
7841    #[test]
7842    fn summary_fold_reuses_cleared_recycled_batches() {
7843        // Four workers and a batch of three force StreamingEmission to send more than
7844        // once per worker on this tree. Without `recycled.clear()`, the next send
7845        // re-folds the previous ops and files/bytes/dirs double-count.
7846        const DIRS: usize = 16;
7847        const FILES_PER_DIR: usize = 40;
7848        let dir = tempfile::tempdir().expect("tempdir");
7849        let mut expected_bytes = 0u64;
7850        for directory in 0..DIRS {
7851            let child = dir.path().join(format!("d{directory:02}"));
7852            fs::create_dir(&child).expect("directory");
7853            for file in 0..FILES_PER_DIR {
7854                let size = directory * FILES_PER_DIR + file + 1;
7855                expected_bytes += size as u64;
7856                write_file(&child.join(format!("f{file:02}.dat")), &vec![b'x'; size]);
7857            }
7858        }
7859        let expected_files = (DIRS * FILES_PER_DIR) as u64;
7860        let expected_dirs = DIRS as u64;
7861        let expected_entries = expected_files + expected_dirs;
7862        let config = ScanConfig {
7863            threads: Some(4),
7864            batch_size: 3,
7865            read_controls: false,
7866            ..ScanConfig::default()
7867        };
7868        let mut files = 0u64;
7869        let mut bytes = 0u64;
7870        let mut dirs = 0u64;
7871        let mut ops = 0u64;
7872        let report = scan_summary_fold(dir.path(), &config, &mut |observed| {
7873            ops += 1;
7874            let Op::Upsert { kind, attrs, .. } = &observed.op else {
7875                return;
7876            };
7877            match kind {
7878                EntryKind::File => {
7879                    files += 1;
7880                    bytes += attrs.size;
7881                }
7882                EntryKind::Dir => dirs += 1,
7883                EntryKind::Symlink | EntryKind::Other => {}
7884            }
7885        })
7886        .expect("fold");
7887        assert_eq!(files, expected_files);
7888        assert_eq!(bytes, expected_bytes);
7889        assert_eq!(dirs, expected_dirs);
7890        assert_eq!(ops, report.entries);
7891        assert_eq!(report.entries, expected_entries);
7892        assert_eq!(report.files_walked, expected_files);
7893        assert_eq!(report.bytes_walked, expected_bytes);
7894    }
7895
7896    /// Publish one listing through `emission` and receive what the consumer would.
7897    fn publish_detached(
7898        emission: &mut DetachedEmission,
7899        directory: DetachedDirectory,
7900        sender: &std::sync::mpsc::Sender<WalkMessage>,
7901        receiver: &std::sync::mpsc::Receiver<WalkMessage>,
7902    ) -> (Vec<DetachedDirectory>, std::sync::mpsc::Sender<Vec<DetachedDirectory>>) {
7903        emission.finish_directory(directory);
7904        let mut send_ns = 0;
7905        assert!(emission.publish_before_discovery(true, sender, &mut send_ns, None));
7906        match receiver.try_recv().expect("a published chunk") {
7907            WalkMessage::DetachedDirectories { directories, recycle } => (directories, recycle),
7908            _ => panic!("the detached walker publishes only listings"),
7909        }
7910    }
7911
7912    /// H185: a folded index describes each directory once, by its own listing, so its
7913    /// directory and symlink entries carry the listing's default attributes, as the
7914    /// transient summary's do (H72); the full index keeps the parent's stat, which is the
7915    /// cache's freshness fingerprint. Under `--one-filesystem` descent reads each
7916    /// directory's device, so the folded walk keeps that stat too. macOS lists every
7917    /// child's attributes in bulk and Windows never takes the skip, so neither shows it.
7918    #[cfg(all(unix, not(target_os = "macos")))]
7919    #[test]
7920    fn a_folded_index_takes_directory_and_symlink_kinds_from_the_listing() {
7921        let root = tempfile::tempdir().expect("temp root");
7922        write_file(&root.path().join("dir/file.txt"), b"contents");
7923        write_file(&root.path().join("dir/nested/deep.txt"), b"more");
7924        write_file(&root.path().join("top.txt"), b"top");
7925        std::os::unix::fs::symlink("top.txt", root.path().join("link")).expect("symlink");
7926        let canonical = root.path().canonicalize().expect("canonical root");
7927        let retention = crate::execution::TreeRetention {
7928            largest_files: 100,
7929            size: crate::query::SizeMetric::Allocated,
7930        };
7931        let attrs_by_kind = |index: &Index| -> Vec<(EntryKind, bool)> {
7932            let mut seen = Vec::new();
7933            let mut stack = vec![(PathBuf::new(), crate::EntryId::ROOT)];
7934            while let Some((path, id)) = stack.pop() {
7935                let Some(children) = index.children_of(id) else { continue };
7936                for (name, child) in children {
7937                    let kind = index.kind_of(child).expect("live child");
7938                    let attrs = index.attrs_of(child).expect("live child");
7939                    seen.push((kind, *attrs == Attrs::default()));
7940                    if kind.is_dir() {
7941                        stack.push((path.join(name), child));
7942                    }
7943                }
7944            }
7945            seen.sort_by_key(|(kind, defaulted)| (*kind as u8, *defaulted));
7946            seen
7947        };
7948
7949        for threads in [1, 4] {
7950            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
7951            let (folded, _, _) =
7952                scan_into_folded_index(&canonical, &config, retention, false).expect("folded");
7953            let (full, _) = scan_into_index(&canonical, &config).expect("full");
7954            for (kind, defaulted) in attrs_by_kind(&folded) {
7955                assert_eq!(
7956                    defaulted,
7957                    matches!(kind, EntryKind::Dir | EntryKind::Symlink),
7958                    "{threads} workers: a folded {kind:?} takes its kind from the listing"
7959                );
7960            }
7961            assert!(
7962                attrs_by_kind(&full).iter().all(|(_, defaulted)| !defaulted),
7963                "{threads} workers: the full index stats every entry"
7964            );
7965
7966            let bound = ScanConfig { one_filesystem: true, ..config };
7967            let (folded, _, _) =
7968                scan_into_folded_index(&canonical, &bound, retention, false).expect("folded");
7969            for (kind, defaulted) in attrs_by_kind(&folded) {
7970                assert_eq!(
7971                    defaulted,
7972                    kind == EntryKind::Symlink,
7973                    "{threads} workers, one filesystem: directories keep their device"
7974                );
7975            }
7976        }
7977    }
7978
7979    #[test]
7980    fn detached_emission_reuses_returned_listings_as_fresh_ones() {
7981        // H159: the consumer hands drained listings back to the worker that allocated
7982        // them. A reused listing must be indistinguishable from a fresh one, including
7983        // after the consumer returned it unapplied, as it does after a build error.
7984        let (sender, receiver) = std::sync::mpsc::channel();
7985        let mut emission = DetachedEmission::new(false);
7986
7987        let mut skipped = emission.begin_directory(Path::new("a/much/longer/relative/path"));
7988        skipped.children.push(DetachedChild {
7989            name: OsString::from("stale.txt"),
7990            kind: EntryKind::File,
7991            attrs: Attrs::default(),
7992            position: 0,
7993        });
7994        skipped.control = Some(Op::ControlRemove { path: PathBuf::from("a/.gitignore") });
7995        let (directories, recycle) = publish_detached(&mut emission, skipped, &sender, &receiver);
7996        let returned_list = directories.as_ptr();
7997        let returned_children = directories[0].children.as_ptr();
7998        recycle.send(directories).expect("the worker still listens");
7999
8000        // Returned listings are taken back at the next publish, so this one is fresh.
8001        let mut large = emission.begin_directory(Path::new("large"));
8002        assert_eq!(large.children.capacity(), 0);
8003        large.children.reserve_exact(DETACHED_SPARE_CHILD_CAPACITY + 1);
8004        let (directories, recycle) = publish_detached(&mut emission, large, &sender, &receiver);
8005        recycle.send(directories).expect("the worker still listens");
8006
8007        let reused = emission.begin_directory(Path::new("b"));
8008        assert_eq!(reused.path.as_os_str(), OsStr::new("b"));
8009        assert!(reused.children.is_empty(), "a returned listing's children are dropped");
8010        assert!(reused.control.is_none(), "a returned listing's control is dropped");
8011        assert_eq!(reused.children.as_ptr(), returned_children, "the child buffer is reused");
8012        let (directories, _recycle) = publish_detached(&mut emission, reused, &sender, &receiver);
8013        assert_eq!(directories.as_ptr(), returned_list, "the published list is reused");
8014
8015        // The large listing came back past the retention bound and was freed instead.
8016        let fresh = emission.begin_directory(Path::new("c"));
8017        assert_eq!(fresh.path.as_os_str(), OsStr::new("c"));
8018        assert_eq!(fresh.children.capacity(), 0);
8019    }
8020
8021    /// A transient fold that reads `.gitignore` receives each directory's control, once,
8022    /// ahead of every entry in that directory, whatever the batch size, the worker count,
8023    /// or where `.gitignore` falls in the listing (fdu-1ovb). Its consumer classifies each
8024    /// entry as it arrives, so an entry folded before its directory's control would miss
8025    /// that control's rules. Batches smaller than a listing force the probe path; the
8026    /// default batch holds whole listings and moves the listed control.
8027    #[test]
8028    fn a_classifying_summary_fold_receives_each_control_ahead_of_its_entries() {
8029        const DIRS: usize = 12;
8030        const FILES_PER_DIR: usize = 30;
8031        let dir = tempfile::tempdir().expect("tempdir");
8032        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8033        for directory in 0..DIRS {
8034            let child = dir.path().join(format!("d{directory:02}"));
8035            write_file(&child.join(".gitignore"), b"*.tmp\n");
8036            for file in 0..FILES_PER_DIR {
8037                write_file(&child.join(format!("f{file:02}.dat")), b"x");
8038            }
8039            write_file(&child.join("nested/n.dat"), b"n");
8040        }
8041        let default_batch = ScanConfig::default().batch_size;
8042        for (threads, batch_size) in [
8043            (Some(1), 1),
8044            (Some(4), 1),
8045            (Some(4), 3),
8046            (None, 7),
8047            (Some(4), 16),
8048            (None, default_batch),
8049        ] {
8050            let config = ScanConfig { batch_size, threads, ..ScanConfig::default() };
8051            assert!(config.read_controls, "observation is the default");
8052            // Per directory: the positions of its control observations and of its entries.
8053            let mut seen: std::collections::BTreeMap<PathBuf, (Vec<usize>, Vec<usize>)> =
8054                std::collections::BTreeMap::new();
8055            let mut position = 0;
8056            scan_summary_fold(dir.path(), &config, &mut |observed| {
8057                let (path, control) = match &observed.op {
8058                    Op::Upsert { path, .. } => (path, false),
8059                    Op::ControlUpsert { path, .. } | Op::ControlRemove { path } => (path, true),
8060                    other => panic!("a cold walk emitted {other:?}"),
8061                };
8062                let directory = path.parent().unwrap_or_else(|| Path::new("")).to_path_buf();
8063                let (controls, entries) = seen.entry(directory).or_default();
8064                if control { controls } else { entries }.push(position);
8065                position += 1;
8066            })
8067            .expect("fold");
8068
8069            assert_eq!(seen.len(), 1 + 2 * DIRS, "the root, each child, and each nested");
8070            for (directory, (controls, entries)) in &seen {
8071                let label = format!("{directory:?} with {threads:?} workers, batch {batch_size}");
8072                let governed = directory.as_os_str().is_empty()
8073                    || directory.file_name().is_some_and(|name| name != "nested");
8074                assert_eq!(controls.len(), usize::from(governed), "{label}: one control read");
8075                if let (Some(last_control), Some(first_entry)) =
8076                    (controls.iter().max(), entries.iter().min())
8077                {
8078                    assert!(last_control < first_entry, "{label}: its control comes first");
8079                }
8080            }
8081        }
8082    }
8083
8084    /// A directory's control is what a lookup of `<dir>/.gitignore` resolves to, as git
8085    /// opens it (fdu-0w1b). A listed `.GITIGNORE` reads exactly what that lookup finds,
8086    /// named by the canonical path: its rules where the directory is case-insensitive, and
8087    /// nothing where it is not. The probe a transient fold makes is that same lookup.
8088    #[test]
8089    fn a_directory_control_is_what_a_lookup_of_its_canonical_path_resolves_to() {
8090        use crate::test_support::CaseLookups;
8091
8092        let dir = tempfile::tempdir().expect("tempdir");
8093        write_file(&dir.path().join("exact/.gitignore"), b"*.log\n");
8094        write_file(&dir.path().join("variant/.GITIGNORE"), b"*.tmp\n");
8095        fs::create_dir_all(dir.path().join("shape/.GitIgnore")).expect("a directory so named");
8096        write_file(&dir.path().join("absent/README"), b"no rules");
8097        let config = ScanConfig::default();
8098        let upsert = |path: &str, source: &[u8]| {
8099            Some(Op::ControlUpsert { path: PathBuf::from(path), source: source.to_vec() })
8100        };
8101
8102        for (lookups, insensitive) in CaseLookups::on_this_host(dir.path()) {
8103            let _lookups = lookups.install(dir.path());
8104            let label = format!("{lookups:?} lookups");
8105            let lookup = |directory: &str| {
8106                read_directory_control(
8107                    &config,
8108                    dir.path(),
8109                    &Path::new(directory).join(".gitignore"),
8110                )
8111                .expect("lookup")
8112            };
8113            let listed = |path: &str, kind| {
8114                let path = Path::new(path);
8115                let read = read_control_op(&config, dir.path(), path, kind).expect("listed read");
8116                let name = path.file_name().expect("a listed name");
8117                assert_eq!(
8118                    read_named_control_op(&config, dir.path(), path, name, kind).expect("named"),
8119                    read,
8120                    "{label}: a walker's read by the listed name decides as the path's (H180)"
8121                );
8122                read
8123            };
8124
8125            assert_eq!(lookup("exact"), upsert("exact/.gitignore", b"*.log\n"), "{label}");
8126            assert_eq!(listed("exact/.gitignore", EntryKind::File), lookup("exact"), "{label}");
8127            assert_eq!(
8128                lookup("variant"),
8129                if insensitive { upsert("variant/.gitignore", b"*.tmp\n") } else { None },
8130                "{label}"
8131            );
8132            assert_eq!(
8133                listed("variant/.GITIGNORE", EntryKind::File),
8134                lookup("variant"),
8135                "{label}: a listed variant reads what the lookup finds"
8136            );
8137            assert_eq!(
8138                listed("shape/.GitIgnore", EntryKind::Dir),
8139                insensitive.then(|| Op::ControlRemove { path: PathBuf::from("shape/.gitignore") }),
8140                "{label}: a directory is resolved to, and holds no rules"
8141            );
8142            assert_eq!(lookup("absent"), None, "{label}");
8143            assert_eq!(listed("absent/README", EntryKind::File), None, "{label}");
8144
8145            let blind = ScanConfig { read_controls: false, ..ScanConfig::default() };
8146            assert_eq!(
8147                read_control_op(
8148                    &blind,
8149                    dir.path(),
8150                    Path::new("variant/.GITIGNORE"),
8151                    EntryKind::File
8152                )
8153                .expect("read"),
8154                None,
8155                "{label}: a scan that reads no rules looks nothing up"
8156            );
8157            assert_eq!(
8158                read_named_control_op(
8159                    &blind,
8160                    dir.path(),
8161                    Path::new("exact/.gitignore"),
8162                    OsStr::new(".gitignore"),
8163                    EntryKind::File
8164                )
8165                .expect("read"),
8166                None,
8167                "{label}: nor does a read by the listed name"
8168            );
8169        }
8170    }
8171
8172    /// The walker's join makes [`Path::join`]'s path byte for byte, including under the
8173    /// root, whose relative directory is empty, and for a name longer than its directory
8174    /// (H180).
8175    #[test]
8176    fn a_listed_name_joins_as_path_join_does() {
8177        for (rel_dir, name) in [
8178            ("", "file"),
8179            ("", ".gitignore"),
8180            ("a", "b"),
8181            ("a/b", ".GITIGNORE"),
8182            ("node_modules/x", "a-name-longer-than-twice-its-directory.js"),
8183        ] {
8184            let joined = join_listed_name(Path::new(rel_dir), OsStr::new(name));
8185            assert_eq!(
8186                joined.as_os_str(),
8187                Path::new(rel_dir).join(name).as_os_str(),
8188                "{rel_dir:?} and {name:?}"
8189            );
8190        }
8191    }
8192
8193    /// Every entry's kind and ignored classification, and every retained control source:
8194    /// what a route's index says about a tree's `.gitignore` rules.
8195    type Classification = (BTreeMap<PathBuf, (EntryKind, Option<bool>)>, Vec<(PathBuf, Vec<u8>)>);
8196
8197    fn classification(index: &Index) -> Classification {
8198        let mut entries = BTreeMap::new();
8199        let mut pending = vec![PathBuf::new()];
8200        while let Some(directory) = pending.pop() {
8201            let children: Vec<(PathBuf, crate::EntryId)> = index
8202                .children(&directory)
8203                .into_iter()
8204                .flatten()
8205                .map(|(name, id)| (directory.join(name), id))
8206                .collect();
8207            for (path, id) in children {
8208                let kind = index.kind_of(id).expect("a live child");
8209                entries.insert(path.clone(), (kind, index.is_ignored(&path).expect("observed")));
8210                if kind.is_dir() {
8211                    pending.push(path);
8212                }
8213            }
8214        }
8215        let sources = index
8216            .controls()
8217            .expect("observed")
8218            .sources()
8219            .map(|(path, source)| (path, source.to_vec()))
8220            .collect();
8221        (entries, sources)
8222    }
8223
8224    /// A tree whose `up` directory holds its rules only in a case variant of the control
8225    /// name, beside a root `.gitignore` those rules partly override.
8226    fn case_variant_tree() -> tempfile::TempDir {
8227        let dir = tempfile::tempdir().expect("tempdir");
8228        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8229        write_file(&dir.path().join("a.log"), b"root rule");
8230        write_file(&dir.path().join("keep.txt"), b"kept");
8231        write_file(&dir.path().join("up/.GITIGNORE"), b"*.tmp\n!keep.log\nbuild/\n");
8232        write_file(&dir.path().join("up/x.tmp"), b"variant rule");
8233        write_file(&dir.path().join("up/keep.log"), b"negated by the variant");
8234        write_file(&dir.path().join("up/y.log"), b"root rule");
8235        write_file(&dir.path().join("up/build/out.bin"), b"variant rule");
8236        write_file(&dir.path().join("up/deep/z.tmp"), b"variant rule");
8237        for file in 0..12 {
8238            write_file(&dir.path().join(format!("up/f{file:02}.dat")), b"unignored");
8239        }
8240        dir
8241    }
8242
8243    /// What [`case_variant_tree`] classifies where its variant does and does not govern.
8244    fn assert_case_variant_classification(classified: &Classification, governs: bool, label: &str) {
8245        let (entries, sources) = classified;
8246        let ignored = |path: &str| entries.get(Path::new(path)).map(|(_, ignored)| *ignored);
8247        assert_eq!(ignored("a.log"), Some(Some(true)), "{label}");
8248        assert_eq!(ignored("up/y.log"), Some(Some(true)), "{label}");
8249        assert_eq!(ignored("up/x.tmp"), Some(Some(governs)), "{label}");
8250        assert_eq!(ignored("up/deep/z.tmp"), Some(Some(governs)), "{label}");
8251        assert_eq!(ignored("up/build"), Some(Some(governs)), "{label}");
8252        assert_eq!(ignored("up/build/out.bin"), Some(Some(governs)), "{label}");
8253        assert_eq!(ignored("up/keep.log"), Some(Some(!governs)), "{label}: negation");
8254        assert_eq!(ignored("up/f00.dat"), Some(Some(false)), "{label}");
8255        let governing: Vec<&Path> = sources.iter().map(|(path, _)| path.as_path()).collect();
8256        let expected: Vec<&Path> = if governs {
8257            vec![Path::new(".gitignore"), Path::new("up/.gitignore")]
8258        } else {
8259            vec![Path::new(".gitignore")]
8260        };
8261        assert_eq!(governing, expected, "{label}: rules are recorded by the canonical path");
8262    }
8263
8264    /// The detached builder, the retained scanner stream (serial and concurrent), and the
8265    /// narrowed-population walks classify a tree whose rules sit in `.GITIGNORE` alike, and
8266    /// apply those rules exactly where a lookup of `.gitignore` resolves to it (fdu-0w1b).
8267    /// Hidden pruning keeps the variant a control signal, as it keeps `.gitignore` one.
8268    #[test]
8269    fn every_cold_route_classifies_a_case_variant_control_alike() {
8270        use crate::test_support::CaseLookups;
8271
8272        let dir = case_variant_tree();
8273        let root = dir.path();
8274        let pruned = Some(std::sync::Arc::new(crate::HiddenPolicy::prune_hidden(Vec::<
8275            std::ffi::OsString,
8276        >::new())));
8277        for (lookups, governs) in CaseLookups::on_this_host(root) {
8278            let _lookups = lookups.install(root);
8279            let reference = ScanConfig { threads: Some(1), ..ScanConfig::default() };
8280            let (index, report) = scan_into_index(root, &reference).expect("detached scan");
8281            assert!(report.is_complete(), "{:?}", report.errors);
8282            let expected = classification(&index);
8283            assert_case_variant_classification(&expected, governs, &format!("{lookups:?}"));
8284
8285            for threads in [Some(1), Some(4)] {
8286                for batch_size in [ScanConfig::default().batch_size, 1, 3] {
8287                    let config = ScanConfig { batch_size, threads, ..ScanConfig::default() };
8288                    let label = format!("{lookups:?}, {threads:?} workers, batch {batch_size}");
8289                    let (detached, _) = scan_into_index(root, &config).expect("detached");
8290                    assert_eq!(classification(&detached), expected, "{label}: detached");
8291                    let (streamed, _) = scan_into_index_via_scanner(root, &config).expect("stream");
8292                    assert_eq!(classification(&streamed), expected, "{label}: scanner stream");
8293                }
8294            }
8295
8296            // Pruning hidden entries drops the variant's row, never its rules.
8297            let hidden = ScanConfig { hidden: pruned.clone(), ..ScanConfig::default() };
8298            let (without_hidden, _) = scan_into_index(root, &hidden).expect("hidden pruned");
8299            let (entries, sources) = classification(&without_hidden);
8300            assert_eq!(sources, expected.1, "{lookups:?}: hidden pruning keeps the rules");
8301            for (path, fact) in &entries {
8302                assert_eq!(expected.0.get(path), Some(fact), "{lookups:?}: {path:?}");
8303            }
8304            assert!(!entries.contains_key(Path::new("up/.GITIGNORE")), "{lookups:?}");
8305
8306            // A narrowed population reads each control before listing its directory, which
8307            // is the lookup itself; it keeps every spelling's row, as it keeps `.gitignore`.
8308            let spelled = |path: &Path| crate::control::path_control_spelling(path).is_some();
8309            for population in
8310                [crate::query::IgnoredEntries::Exclude, crate::query::IgnoredEntries::Only]
8311            {
8312                let config = ScanConfig { population, ..ScanConfig::default() };
8313                let (narrowed, report) = scan_into_index(root, &config).expect("narrowed");
8314                assert!(report.is_complete(), "{:?}", report.errors);
8315                let (entries, sources) = classification(&narrowed);
8316                let label = format!("{lookups:?}, {population:?}");
8317                assert_eq!(sources, expected.1, "{label}");
8318                let files =
8319                    |entries: &BTreeMap<PathBuf, (EntryKind, Option<bool>)>| -> Vec<PathBuf> {
8320                        entries
8321                            .iter()
8322                            .filter(|(_, (kind, _))| *kind == EntryKind::File)
8323                            .map(|(path, _)| path.clone())
8324                            .collect()
8325                    };
8326                let wanted: BTreeMap<PathBuf, (EntryKind, Option<bool>)> = expected
8327                    .0
8328                    .iter()
8329                    .filter(|(path, (_, ignored))| {
8330                        spelled(path)
8331                            || match population {
8332                                crate::query::IgnoredEntries::Exclude => *ignored == Some(false),
8333                                _ => *ignored == Some(true),
8334                            }
8335                    })
8336                    .map(|(path, fact)| (path.clone(), *fact))
8337                    .collect();
8338                assert_eq!(files(&entries), files(&wanted), "{label}: retained files");
8339            }
8340        }
8341    }
8342
8343    /// Reconciliation keeps a case-variant control exactly as a cold walk finds it through
8344    /// every change: a case-only rename of `.gitignore` to `.GITIGNORE`, an edit, a
8345    /// removal, and a refresh of the variant's own path after it is created and after it
8346    /// is gone. Serial and parallel reconciliation, and revalidation, agree with a cold
8347    /// walk after each (fdu-0w1b).
8348    ///
8349    /// The rename is the case a canonical path cannot follow on its own: the index still
8350    /// holds the old `.gitignore` entry, and removing an entry at the canonical path drops
8351    /// the rules it governs, which on a case-insensitive directory the variant still holds.
8352    #[test]
8353    fn reconciliation_follows_a_case_variant_control_through_every_change() {
8354        use crate::test_support::CaseLookups;
8355
8356        let probe = tempfile::tempdir().expect("tempdir");
8357        for (lookups, governs) in CaseLookups::on_this_host(probe.path()) {
8358            for threads in [Some(1), Some(4)] {
8359                let dir = tempfile::tempdir().expect("tempdir");
8360                let root = dir.path().canonicalize().expect("canonical root");
8361                let _lookups = lookups.install(&root);
8362                let config = ScanConfig { threads, batch_size: 3, ..ScanConfig::default() };
8363                write_file(&root.join(".gitignore"), b"*.log\n");
8364                write_file(&root.join("up/.gitignore"), b"*.tmp\n");
8365                write_file(&root.join("up/x.tmp"), b"governed");
8366                write_file(&root.join("up/y.bin"), b"governed after the edit");
8367                for file in 0..8 {
8368                    write_file(&root.join(format!("up/f{file}.dat")), b"unignored");
8369                }
8370                let (mut index, _) = scan_into_index(&root, &config).expect("cold scan");
8371                let cold = |label: &str| {
8372                    let (cold, report) = scan_into_index(&root, &config).expect("cold");
8373                    assert!(report.is_complete(), "{label}: {:?}", report.errors);
8374                    classification(&cold)
8375                };
8376                let reconciled = |index: &mut Index, label: &str| {
8377                    let report = reconcile(index, &config, &mut |_| {}).expect("reconcile");
8378                    assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
8379                    assert_eq!(report.apply.stale, 0, "{label}: no observation lost a race");
8380                };
8381                let revalidated = |snapshot: &Index, label: &str| {
8382                    let mut revalidated = snapshot.clone();
8383                    let mut observations = Vec::new();
8384                    revalidate(&revalidated, &config, &mut |observation| {
8385                        observations.push(observation);
8386                    })
8387                    .expect("revalidate");
8388                    for observation in &observations {
8389                        let outcome = revalidated.apply(observation).expect("apply");
8390                        assert_eq!(outcome.stats.stale, 0, "{label}: revalidation lost a race");
8391                    }
8392                    classification(&revalidated)
8393                };
8394                let governed = |label: &str| {
8395                    let (entries, _) = cold(label);
8396                    entries.get(Path::new("up/x.tmp")).map(|(_, ignored)| *ignored)
8397                };
8398
8399                let before_rename = index.clone();
8400                fs::rename(root.join("up/.gitignore"), root.join("up/.GITIGNORE")).expect("recase");
8401                let label = format!("{lookups:?}, {threads:?} workers: case-only rename");
8402                assert_eq!(governed(&label), Some(Some(governs)), "{label}");
8403                assert_eq!(
8404                    revalidated(&before_rename, &label),
8405                    cold(&label),
8406                    "{label}: revalidate"
8407                );
8408                reconciled(&mut index, &label);
8409                assert_eq!(classification(&index), cold(&label), "{label}");
8410
8411                write_file(&root.join("up/.GITIGNORE"), b"*.bin\n");
8412                let label = format!("{lookups:?}, {threads:?} workers: edit");
8413                reconciled(&mut index, &label);
8414                assert_eq!(classification(&index), cold(&label), "{label}");
8415
8416                fs::remove_file(root.join("up/.GITIGNORE")).expect("remove the variant");
8417                let label = format!("{lookups:?}, {threads:?} workers: removal");
8418                reconciled(&mut index, &label);
8419                assert_eq!(classification(&index), cold(&label), "{label}");
8420
8421                // A refresh of the variant's own path looks the directory's control up
8422                // whether the variant is there or gone.
8423                write_file(&root.join("up/.GITIGNORE"), b"*.tmp\n");
8424                for (step, label) in [
8425                    (None, format!("{lookups:?}, {threads:?} workers: refresh of a new variant")),
8426                    (
8427                        Some(()),
8428                        format!("{lookups:?}, {threads:?} workers: refresh of a removed variant"),
8429                    ),
8430                ] {
8431                    if step.is_some() {
8432                        fs::remove_file(root.join("up/.GITIGNORE")).expect("remove");
8433                    }
8434                    let report = reconcile_subtree(
8435                        &mut index,
8436                        Path::new("up/.GITIGNORE"),
8437                        &config,
8438                        &mut |_| {},
8439                    )
8440                    .expect("refresh");
8441                    assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
8442                    assert_eq!(classification(&index), cold(&label), "{label}");
8443                }
8444            }
8445        }
8446    }
8447
8448    /// Revalidate a copy of `index` and apply every observation it sends, none of which may
8449    /// lose a race.
8450    fn revalidated_copy(index: &Index, config: &ScanConfig, label: &str) -> Index {
8451        let mut revalidated = index.clone();
8452        let mut observations = Vec::new();
8453        let report = revalidate(&revalidated, config, &mut |observation| {
8454            observations.push(observation);
8455        })
8456        .expect("revalidate");
8457        assert!(report.is_complete(), "{label}: {:?}", report.errors);
8458        for observation in &observations {
8459            let outcome = revalidated.apply(observation).expect("apply");
8460            assert_eq!(outcome.stats.stale, 0, "{label}: revalidation lost a race");
8461        }
8462        revalidated
8463    }
8464
8465    /// A narrowed population's reconciliation keeps a case-variant control through a
8466    /// case-only rename, as a cold walk finds it. Its lookup before the listing is what
8467    /// finds the variant's rules, and the listing shows the variant, so the sweep restates
8468    /// the rules its removal of the stale `.gitignore` entry drops (fdu-0w1b).
8469    #[test]
8470    fn a_narrowed_reconciliation_follows_a_case_only_rename() {
8471        use crate::test_support::CaseLookups;
8472
8473        let probe = tempfile::tempdir().expect("tempdir");
8474        for (lookups, governs) in CaseLookups::on_this_host(probe.path()) {
8475            for population in
8476                [crate::query::IgnoredEntries::Exclude, crate::query::IgnoredEntries::Only]
8477            {
8478                let label = format!("{lookups:?}, {population:?}");
8479                let dir = tempfile::tempdir().expect("tempdir");
8480                let root = dir.path().canonicalize().expect("canonical root");
8481                let _lookups = lookups.install(&root);
8482                let config = ScanConfig { population, batch_size: 3, ..ScanConfig::default() };
8483                write_file(&root.join("up/.gitignore"), b"*.tmp\n");
8484                write_file(&root.join("up/x.tmp"), b"governed");
8485                for file in 0..4 {
8486                    write_file(&root.join(format!("up/f{file}.dat")), b"unignored");
8487                }
8488                let (mut index, _) = scan_into_index(&root, &config).expect("cold scan");
8489                fs::rename(root.join("up/.gitignore"), root.join("up/.GITIGNORE")).expect("recase");
8490                let (cold, report) = scan_into_index(&root, &config).expect("cold");
8491                assert!(report.is_complete(), "{label}: {:?}", report.errors);
8492                let cold = classification(&cold);
8493                assert_eq!(
8494                    cold.1.iter().any(|(path, _)| path == Path::new("up/.gitignore")),
8495                    governs,
8496                    "{label}: the variant governs exactly where the lookup resolves to it"
8497                );
8498
8499                let revalidated = revalidated_copy(&index, &config, &label);
8500                assert_eq!(classification(&revalidated), cold, "{label}: revalidate");
8501                let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
8502                assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
8503                assert_eq!(report.apply.stale, 0, "{label}: no observation lost a race");
8504                assert_eq!(classification(&index), cold, "{label}: reconcile");
8505            }
8506        }
8507    }
8508
8509    /// A narrowed population's reconciliation looks a directory's control up before
8510    /// listing it. When the `.gitignore` that lookup found is deleted before the listing,
8511    /// the sweep removes the stale entry, and the rules go with it: no case variant was
8512    /// listed to hold them, so nothing is restated, and the table is as bare as the
8513    /// directory. The next pass agrees with a cold walk.
8514    #[test]
8515    fn a_control_deleted_between_its_lookup_and_the_listing_leaves_no_rules() {
8516        for population in
8517            [crate::query::IgnoredEntries::Exclude, crate::query::IgnoredEntries::Only]
8518        {
8519            for revalidating in [true, false] {
8520                let label = format!("{population:?}, revalidating: {revalidating}");
8521                let dir = tempfile::tempdir().expect("tempdir");
8522                let root = dir.path().canonicalize().expect("canonical root");
8523                let config = ScanConfig { population, ..ScanConfig::default() };
8524                write_file(&root.join("up/.gitignore"), b"*.tmp\n");
8525                write_file(&root.join("up/x.tmp"), b"governed");
8526                write_file(&root.join("up/y.dat"), b"unignored");
8527                let control_path = Path::new("up/.gitignore");
8528                let (mut index, _) = scan_into_index(&root, &config).expect("cold scan");
8529                assert!(index.control_table().contains(control_path), "{label}");
8530
8531                let control = root.join(control_path);
8532                let deleted = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
8533                let hook_deleted = std::sync::Arc::clone(&deleted);
8534                let race = install_walk_hook(&root, move |point| {
8535                    if let WalkHookPoint::ControlLookup(looked_up) = point {
8536                        if looked_up == control && fs::remove_file(looked_up).is_ok() {
8537                            hook_deleted.store(true, std::sync::atomic::Ordering::SeqCst);
8538                        }
8539                    }
8540                    None
8541                });
8542                if revalidating {
8543                    index = revalidated_copy(&index, &config, &label);
8544                } else {
8545                    let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
8546                    assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
8547                }
8548                drop(race);
8549                assert!(
8550                    deleted.load(std::sync::atomic::Ordering::SeqCst),
8551                    "{label}: the lookup found the control before it was deleted"
8552                );
8553                assert_eq!(index.path_state(control_path), PathState::Absent, "{label}");
8554                assert!(
8555                    !index.control_table().contains(control_path),
8556                    "{label}: no rules remain for a file that is gone"
8557                );
8558
8559                let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
8560                assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
8561                let (cold, _) = scan_into_index(&root, &config).expect("cold");
8562                assert_eq!(classification(&index), classification(&cold), "{label}");
8563            }
8564        }
8565    }
8566
8567    /// The sweep restates what a lookup found only after removing the stale `.gitignore`
8568    /// entry of a listing that showed a case variant, the one spelling that can still hold
8569    /// the rules that removal drops. Without a listed variant the exact file is gone, and
8570    /// so are its rules.
8571    #[test]
8572    fn a_listing_restates_looked_up_rules_only_after_showing_a_case_variant() {
8573        let found =
8574            Op::ControlUpsert { path: PathBuf::from("up/.gitignore"), source: b"*.tmp\n".to_vec() };
8575        let exact = OsStr::new(".gitignore");
8576
8577        let mut deleted = ListedControl::default();
8578        deleted.lookup(&Ok(Some(found.clone())));
8579        deleted.listed(OsStr::new("x.tmp"));
8580        assert_eq!(deleted.restatement_after_removing(exact), None, "the exact file is gone");
8581
8582        let mut recased = ListedControl::default();
8583        recased.lookup(&Ok(Some(found.clone())));
8584        recased.listed(OsStr::new(".GITIGNORE"));
8585        assert_eq!(recased.restatement_after_removing(exact), Some(found.clone()));
8586        assert_eq!(
8587            recased.restatement_after_removing(OsStr::new(".GitIgnore")),
8588            None,
8589            "only removing the canonical entry drops rules"
8590        );
8591        assert_eq!(recased.restatement_after_removing(OsStr::new("x.tmp")), None);
8592
8593        // An included population looks the control up through the listed variant itself.
8594        let mut read = ListedControl::default();
8595        read.listed(OsStr::new(".GitIgnore"));
8596        read.read(OsStr::new(".GitIgnore"), &Ok(Some(found.clone())));
8597        assert_eq!(read.restatement_after_removing(exact), Some(found));
8598
8599        let mut missed = ListedControl::default();
8600        missed.lookup(&Ok(None));
8601        missed.listed(OsStr::new(".GITIGNORE"));
8602        assert_eq!(missed.restatement_after_removing(exact), None, "nothing was found");
8603    }
8604
8605    /// Where a directory is case-sensitive it can list `.gitignore` beside `.GITIGNORE`,
8606    /// and only the exact name governs, on every route and through every change: the
8607    /// variant's lookup resolves to the exact file, and removing the variant leaves the
8608    /// rules while removing the exact name takes them (fdu-0w1b).
8609    #[test]
8610    fn a_case_sensitive_directory_listing_both_spellings_takes_only_the_exact_name() {
8611        let dir = tempfile::tempdir().expect("tempdir");
8612        let root = dir.path().canonicalize().expect("canonical root");
8613        write_file(&root.join(".gitignore"), b"*.log\n");
8614        write_file(&root.join(".GITIGNORE"), b"*.tmp\n");
8615        let listed = fs::read_dir(&root).expect("list").count();
8616        if listed != 2 {
8617            eprintln!(
8618                "skipped: the temporary directory is case-insensitive, so it cannot hold both \
8619                 spellings"
8620            );
8621            return;
8622        }
8623        write_file(&root.join("x.log"), b"exact rule");
8624        write_file(&root.join("y.tmp"), b"variant, not a rule here");
8625        let ignored =
8626            |index: &Index, path: &str| index.is_ignored(Path::new(path)).expect("observed");
8627
8628        for threads in [Some(1), Some(4)] {
8629            let config = ScanConfig { threads, batch_size: 1, ..ScanConfig::default() };
8630            let (detached, _) = scan_into_index(&root, &config).expect("detached");
8631            let (streamed, _) = scan_into_index_via_scanner(&root, &config).expect("stream");
8632            for index in [&detached, &streamed] {
8633                assert_eq!(ignored(index, "x.log"), Some(true), "{threads:?}");
8634                assert_eq!(ignored(index, "y.tmp"), Some(false), "{threads:?}");
8635                assert_eq!(classification(index), classification(&detached), "{threads:?}");
8636            }
8637        }
8638
8639        let config = ScanConfig::default();
8640        let (mut index, _) = scan_into_index(&root, &config).expect("cold");
8641        fs::remove_file(root.join(".GITIGNORE")).expect("remove the variant");
8642        reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
8643        assert_eq!(ignored(&index, "x.log"), Some(true), "the exact name still governs");
8644        write_file(&root.join(".GITIGNORE"), b"*.tmp\n");
8645        fs::remove_file(root.join(".gitignore")).expect("remove the exact name");
8646        reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
8647        assert_eq!(ignored(&index, "x.log"), Some(false), "no rules remain");
8648        assert_eq!(ignored(&index, "y.tmp"), Some(false), "the variant never governs here");
8649        let (cold, _) = scan_into_index(&root, &config).expect("cold");
8650        assert_eq!(classification(&index), classification(&cold));
8651    }
8652
8653    /// A listing whose control was already probed does not read its `.gitignore` again:
8654    /// the entry is prepared with no control and, here, no error from a file that cannot
8655    /// be read, where reading it would have produced one.
8656    #[test]
8657    #[cfg(unix)]
8658    fn a_probed_listing_prepares_its_control_entry_without_reading_it() {
8659        use std::os::unix::fs::PermissionsExt;
8660        if !crate::test_support::require_permission_bits() {
8661            return;
8662        }
8663        let dir = tempfile::tempdir().expect("tempdir");
8664        let control = dir.path().join(".gitignore");
8665        write_file(&control, b"*.log\n");
8666        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("deny");
8667        let config = ScanConfig::default();
8668        let prepare = |read_control| {
8669            prepare_walk_entry_reading(
8670                dir.path(),
8671                Path::new(""),
8672                0,
8673                OsStr::new(".gitignore"),
8674                EntryKind::File,
8675                Attrs::default(),
8676                0,
8677                &config,
8678                read_control,
8679            )
8680            .expect("admitted")
8681        };
8682        let read = prepare(true);
8683        let skipped = prepare(false);
8684        fs::set_permissions(&control, fs::Permissions::from_mode(0o600)).expect("restore");
8685
8686        assert!(read.control.is_none() && read.control_error.is_some(), "the read was made");
8687        assert!(skipped.control.is_none() && skipped.control_error.is_none(), "no read");
8688        assert!(skipped.retained, "the entry is still a row");
8689    }
8690
8691    /// An automatic walk too short to fill its calibration window must say so.
8692    ///
8693    /// The failure this guards is quiet: such a walk runs on its initial pool, which is
8694    /// indistinguishable in the artifacts from a walk that measured the filesystem and
8695    /// chose to hold — unless the undecided case is recorded separately. Reading the
8696    /// first as the second is how a policy with no evidence behind it comes to look
8697    /// like a policy with evidence behind it.
8698    #[test]
8699    fn a_short_automatic_walk_records_an_undecided_policy() {
8700        let _serial = crate::counters::test_serial();
8701        let available = std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get);
8702        if automatic_worker_pool(available).calibration.is_none() {
8703            // A host reporting one processor has no reserve to unlock, so there is no
8704            // policy here to leave undecided.
8705            return;
8706        }
8707
8708        crate::counters::enable(true);
8709        let dir = sample_tree();
8710        let config = ScanConfig { threads: None, ..ScanConfig::default() };
8711        let before = crate::counters::snapshot();
8712        scan(dir.path(), &config, &mut |_| {}).expect("scan");
8713        crate::counters::flush_thread();
8714        let after = crate::counters::snapshot();
8715        crate::counters::enable(false);
8716
8717        // A strict increase, so a counter inflated by a test running beside this one
8718        // cannot turn the assertion into a false pass.
8719        assert!(
8720            after.adaptive_policy_undecided > before.adaptive_policy_undecided,
8721            "a three-file tree cannot fill a {ADAPTIVE_SCAN_CALIBRATION_ENTRIES}-entry window"
8722        );
8723    }
8724
8725    #[test]
8726    fn diagnostics_make_a_fixed_pool_and_backend_choice_explicit() {
8727        let dir = sample_tree();
8728        let config = ScanConfig { threads: Some(1), ..ScanConfig::default() };
8729
8730        let (report, diagnostics) =
8731            scan_with_diagnostics(dir.path(), &config, &mut |_| {}).expect("diagnostic scan");
8732
8733        assert_eq!(diagnostics.schema, SCAN_DIAGNOSTICS_SCHEMA);
8734        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Fixed);
8735        assert_eq!(diagnostics.worker_policy.initial_workers, 1);
8736        assert_eq!(diagnostics.worker_policy.maximum_workers, 1);
8737        assert_eq!(diagnostics.worker_policy.peak_active_workers, 1);
8738        assert!(diagnostics.worker_policy.windows.is_empty());
8739        assert!(!diagnostics.worker_policy.events_truncated);
8740        assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
8741        assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
8742        // On glibc the serial walk lists through the native reader too, and the Linux
8743        // backend fields count it: every attempt is a success or a fallback, and the
8744        // directories read are the native successes plus the portable reads.
8745        #[cfg(all(target_os = "linux", target_env = "gnu"))]
8746        {
8747            let backend = &diagnostics.backend;
8748            assert_eq!(backend.portable_attempts, backend.portable_directory_reads);
8749            assert!(
8750                backend.portable_directory_reads < report.dirs_read,
8751                "native listings are not portable reads: {backend:?}, {} read",
8752                report.dirs_read
8753            );
8754            let (Some(attempts), Some(successes), Some(fallbacks)) = (
8755                backend.linux_dents_attempts,
8756                backend.linux_dents_successes,
8757                backend.linux_dents_fallbacks,
8758            ) else {
8759                panic!("Linux native counts are present: {backend:?}");
8760            };
8761            assert_eq!(attempts, successes + fallbacks);
8762            assert!(successes > 0, "the serial walk lists natively: {backend:?}");
8763            assert_eq!(successes + backend.portable_directory_reads, report.dirs_read);
8764        }
8765        #[cfg(not(all(target_os = "linux", target_env = "gnu")))]
8766        assert_eq!(diagnostics.backend.portable_directory_reads, report.dirs_read);
8767
8768        #[cfg(target_os = "macos")]
8769        {
8770            assert_eq!(diagnostics.backend.macos_bulk_attempts, Some(0));
8771            assert_eq!(diagnostics.backend.macos_bulk_successes, Some(0));
8772            assert_eq!(diagnostics.backend.macos_bulk_fallbacks, Some(0));
8773            assert!(diagnostics.backend.unavailable_reason.is_none());
8774        }
8775        #[cfg(not(target_os = "macos"))]
8776        {
8777            assert_eq!(diagnostics.backend.macos_bulk_attempts, None);
8778            assert_eq!(diagnostics.backend.macos_bulk_successes, None);
8779            assert_eq!(diagnostics.backend.macos_bulk_fallbacks, None);
8780            assert_eq!(
8781                diagnostics.backend.unavailable_reason,
8782                Some("macOS bulk directory enumeration is unavailable on this platform")
8783            );
8784        }
8785
8786        // A parallel walk counts the same way, worker by worker.
8787        #[cfg(all(target_os = "linux", target_env = "gnu"))]
8788        {
8789            let config = ScanConfig { threads: Some(4), ..ScanConfig::default() };
8790            let (report, diagnostics) =
8791                scan_with_diagnostics(dir.path(), &config, &mut |_| {}).expect("diagnostic scan");
8792            let backend = &diagnostics.backend;
8793            assert!(report.is_complete(), "{:?}", report.errors);
8794            assert_eq!(backend.portable_attempts, backend.portable_directory_reads);
8795            assert!(
8796                backend.portable_directory_reads < report.dirs_read,
8797                "native listings are not portable reads: {backend:?}, {} read",
8798                report.dirs_read
8799            );
8800            assert_eq!(
8801                backend.unavailable_reason,
8802                Some("macOS bulk directory enumeration is unavailable on this platform")
8803            );
8804            let (Some(attempts), Some(successes), Some(fallbacks)) = (
8805                backend.linux_dents_attempts,
8806                backend.linux_dents_successes,
8807                backend.linux_dents_fallbacks,
8808            ) else {
8809                panic!("Linux native counts are present: {backend:?}");
8810            };
8811            assert_eq!(attempts, successes + fallbacks);
8812            assert!(successes > 0, "a parallel walk lists natively: {backend:?}");
8813            assert_eq!(successes + backend.portable_directory_reads, report.dirs_read);
8814        }
8815        #[cfg(not(all(target_os = "linux", target_env = "gnu")))]
8816        {
8817            assert_eq!(diagnostics.backend.linux_dents_attempts, None);
8818            assert_eq!(diagnostics.backend.linux_dents_successes, None);
8819            assert_eq!(diagnostics.backend.linux_dents_fallbacks, None);
8820        }
8821    }
8822
8823    #[test]
8824    fn diagnostics_fail_closed_when_an_automatic_window_is_incomplete() {
8825        let available = std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get);
8826        let pool = automatic_worker_pool(available);
8827        if pool.calibration.is_none() {
8828            return;
8829        }
8830        let dir = sample_tree();
8831        let config = ScanConfig { threads: None, ..ScanConfig::default() };
8832
8833        let (report, diagnostics) =
8834            scan_with_diagnostics(dir.path(), &config, &mut |_| {}).expect("diagnostic scan");
8835
8836        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Undecided);
8837        assert_eq!(diagnostics.worker_policy.available_parallelism, available);
8838        assert_eq!(diagnostics.worker_policy.initial_workers, pool.initial);
8839        assert_eq!(diagnostics.worker_policy.maximum_workers, pool.maximum);
8840        assert_eq!(
8841            diagnostics.worker_policy.calibration_window_entries,
8842            Some(ADAPTIVE_SCAN_CALIBRATION_ENTRIES)
8843        );
8844        assert_eq!(
8845            diagnostics.worker_policy.slow_threshold_ns_per_entry,
8846            Some(ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY)
8847        );
8848        assert_eq!(diagnostics.worker_policy.windows.len(), 1);
8849        let window = &diagnostics.worker_policy.windows[0];
8850        assert_eq!(window.sequence, 0);
8851        assert_eq!(window.start_entry_ordinal, 0);
8852        assert_eq!(window.end_entry_ordinal, report.entries);
8853        assert_eq!(window.observed_entries, report.entries);
8854        assert_eq!(window.decision, WorkerPolicyDecision::Undecided);
8855        assert!(window.end_entry_ordinal < ADAPTIVE_SCAN_CALIBRATION_ENTRIES);
8856        assert!(window.active_workers <= diagnostics.worker_policy.peak_active_workers);
8857        assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
8858        assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
8859        assert!(diagnostics.worker_policy.handoff_backlog_high_water >= 1);
8860    }
8861
8862    #[test]
8863    fn diagnostic_trace_is_bounded_and_marks_truncation() {
8864        let recorder = ScanDiagnosticsRecorder::new(
8865            WorkerPool::fixed(2),
8866            2,
8867            WorkerPolicyExperiment::ShippedOneShot,
8868        );
8869        for sequence in 0..=MAX_POLICY_TRACE_EVENTS {
8870            recorder.record_policy_window(PolicyWindowSnapshot {
8871                sequence: sequence as u64,
8872                start_entry_ordinal: sequence as u64,
8873                end_entry_ordinal: sequence as u64 + 1,
8874                observed_entries: 1,
8875                observed_chunks: 1,
8876                observed_work_ns: 10,
8877                ready_directories: 1,
8878                in_flight_directories: 1,
8879                active_workers: 1,
8880                handoff_backlog: 0,
8881                requested_workers: None,
8882                decision: WorkerPolicyDecision::Hold,
8883            });
8884        }
8885
8886        let diagnostics = recorder.finish();
8887        assert_eq!(diagnostics.worker_policy.windows.len(), MAX_POLICY_TRACE_EVENTS);
8888        assert!(diagnostics.worker_policy.events_truncated);
8889    }
8890
8891    #[test]
8892    fn diagnostic_trace_preserves_queue_order_when_recorders_arrive_out_of_order() {
8893        let pool =
8894            WorkerPool { initial: 2, maximum: 4, calibration: Some(WorkerCalibration::new(1, 1)) };
8895        let recorder =
8896            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::RepeatedWindows);
8897        let snapshot = |sequence, decision| PolicyWindowSnapshot {
8898            sequence,
8899            start_entry_ordinal: sequence,
8900            end_entry_ordinal: sequence + 1,
8901            observed_entries: 1,
8902            observed_chunks: 1,
8903            observed_work_ns: 1,
8904            ready_directories: 0,
8905            in_flight_directories: 0,
8906            active_workers: 1,
8907            handoff_backlog: 0,
8908            requested_workers: None,
8909            decision,
8910        };
8911
8912        recorder.record_policy_window(snapshot(1, WorkerPolicyDecision::HoldNoUsefulWork));
8913        recorder.record_policy_window(snapshot(0, WorkerPolicyDecision::Hold));
8914
8915        let diagnostics = recorder.finish();
8916        assert_eq!(
8917            diagnostics
8918                .worker_policy
8919                .windows
8920                .iter()
8921                .map(|window| window.sequence)
8922                .collect::<Vec<_>>(),
8923            vec![0, 1]
8924        );
8925        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::HeldNoUsefulWork);
8926    }
8927
8928    #[test]
8929    fn diagnostic_policy_aggregates_cross_check_runtime_counters() {
8930        let _serial = crate::counters::test_serial();
8931        let pool = WorkerPool {
8932            initial: 2,
8933            maximum: 4,
8934            calibration: Some(WorkerCalibration::new(17, 100)),
8935        };
8936        let recorder =
8937            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::ShippedOneShot);
8938
8939        crate::counters::enable(true);
8940        let before = crate::counters::snapshot();
8941        record_adaptive_calibration_chunk(Some(&recorder), 17, 2_100);
8942        record_adaptive_worker_expansion(Some(&recorder));
8943        crate::counters::flush_thread();
8944        let after = crate::counters::snapshot();
8945        crate::counters::enable(false);
8946
8947        recorder.record_policy_window(PolicyWindowSnapshot {
8948            sequence: 0,
8949            start_entry_ordinal: 0,
8950            end_entry_ordinal: 17,
8951            observed_entries: 17,
8952            observed_chunks: 1,
8953            observed_work_ns: 2_100,
8954            ready_directories: 2,
8955            in_flight_directories: 2,
8956            active_workers: 2,
8957            handoff_backlog: 0,
8958            requested_workers: Some(4),
8959            decision: WorkerPolicyDecision::ScaleUp,
8960        });
8961        let diagnostics = recorder.finish();
8962        let policy = diagnostics.worker_policy;
8963        assert_eq!(policy.calibration_chunks, 1);
8964        assert_eq!(policy.calibration_entries, 17);
8965        assert_eq!(policy.calibration_work_ns, 2_100);
8966        assert_eq!(policy.worker_expansions, 1);
8967        assert_eq!(policy.windows[0].observed_chunks, policy.calibration_chunks);
8968        assert_eq!(policy.windows[0].observed_entries, policy.calibration_entries);
8969        assert_eq!(policy.windows[0].observed_work_ns, policy.calibration_work_ns);
8970
8971        // Other tests can record while this process-global interval is enabled, so the
8972        // counter delta may be larger but must never be smaller than this run-scoped
8973        // trace. The shared helpers above make the two observations one event.
8974        assert!(
8975            after.adaptive_calibration_chunks - before.adaptive_calibration_chunks
8976                >= policy.calibration_chunks
8977        );
8978        assert!(
8979            after.adaptive_calibration_entries - before.adaptive_calibration_entries
8980                >= policy.calibration_entries
8981        );
8982        assert!(
8983            after.adaptive_calibration_work_us - before.adaptive_calibration_work_us
8984                >= policy.calibration_work_ns / 1_000
8985        );
8986        assert!(after.adaptive_scale_ups - before.adaptive_scale_ups >= policy.worker_expansions);
8987    }
8988
8989    #[test]
8990    fn diagnostic_index_scan_preserves_the_regular_result() {
8991        let dir = branching_tree();
8992        let config = ScanConfig { threads: Some(4), ..ScanConfig::default() };
8993        let (plain, plain_report) = scan_into_index(dir.path(), &config).expect("plain scan");
8994        let (diagnostic, diagnostic_report, diagnostics) =
8995            scan_into_index_with_diagnostics(dir.path(), &config).expect("diagnostic scan");
8996
8997        assert_eq!(index_fingerprint(&plain), index_fingerprint(&diagnostic));
8998        assert_eq!(plain_report.entries, diagnostic_report.entries);
8999        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Fixed);
9000    }
9001
9002    #[test]
9003    fn detached_bootstrap_matches_the_streaming_reducer_for_each_worker_count() {
9004        let dir = branching_tree();
9005        for threads in 1..=4 {
9006            let config = ScanConfig {
9007                read_controls: false,
9008                threads: Some(threads),
9009                ..ScanConfig::default()
9010            };
9011            let _ = detached_and_streaming_indexes(dir.path(), &config);
9012        }
9013    }
9014
9015    #[test]
9016    fn detached_control_bootstrap_matches_the_streaming_reducer_for_each_worker_count() {
9017        let dir = controlled_branching_tree();
9018        for threads in 1..=4 {
9019            let config =
9020                ScanConfig { read_controls: true, threads: Some(threads), ..ScanConfig::default() };
9021            let _ = detached_and_streaming_indexes(dir.path(), &config);
9022        }
9023    }
9024
9025    #[test]
9026    fn detached_bootstrap_preserves_the_exact_first_mutation() {
9027        let dir = branching_tree();
9028        let config = ScanConfig { read_controls: false, threads: Some(4), ..ScanConfig::default() };
9029        let (mut detached, mut streaming) = detached_and_streaming_indexes(dir.path(), &config);
9030        let created = dir.path().join("t3/m2/after-bootstrap.rs");
9031        write_file(&created, b"new fact");
9032        let attrs =
9033            attrs_from(&created, &fs::symlink_metadata(&created).expect("new file metadata"))
9034                .expect("observe new file");
9035        let observation = Observation::new(vec![Op::Upsert {
9036            path: PathBuf::from("t3/m2/after-bootstrap.rs"),
9037            kind: EntryKind::File,
9038            attrs,
9039        }]);
9040
9041        let detached_outcome = detached.apply(&observation).expect("detached mutation");
9042        let streaming_outcome = streaming.apply(&observation).expect("streaming mutation");
9043        assert_eq!(detached_outcome, streaming_outcome);
9044        assert_indexes_equal(&detached, &streaming);
9045    }
9046
9047    #[test]
9048    fn detached_control_bootstrap_preserves_the_exact_first_mutation() {
9049        let dir = controlled_branching_tree();
9050        let config = ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
9051        let (mut detached, mut streaming) = detached_and_streaming_indexes(dir.path(), &config);
9052        let observation = Observation::new(vec![Op::ControlUpsert {
9053            path: PathBuf::from(".gitignore"),
9054            source: b"leaf-2.dat\n".to_vec(),
9055        }]);
9056
9057        let detached_outcome = detached.apply(&observation).expect("detached control mutation");
9058        let streaming_outcome = streaming.apply(&observation).expect("streaming control mutation");
9059        assert_eq!(detached_outcome, streaming_outcome);
9060        assert_indexes_equal(&detached, &streaming);
9061    }
9062
9063    fn observed_coverage(index: &Index) -> crate::control::ControlObservation {
9064        match index.control_coverage() {
9065            crate::control::ControlCoverage::Observed(observation) => observation,
9066            crate::control::ControlCoverage::NotObserved => panic!("controls were observed"),
9067        }
9068    }
9069
9070    /// Both bootstrap lanes refuse a line over the limit and a file over the budget, and
9071    /// neither ends the scan or makes it partial. Both refusals are order-independent, so
9072    /// the lanes agree on exactly which files they refused.
9073    #[test]
9074    fn both_bootstrap_lanes_refuse_over_bound_controls_without_ending_the_scan() {
9075        let dir = tempfile::tempdir().expect("tempdir");
9076        let mut long_line = b"*.log\n".to_vec();
9077        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
9078        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
9079        write_file(&dir.path().join("guarded/kept.log"), b"guarded");
9080        write_file(
9081            &dir.path().join("huge/.gitignore"),
9082            &b"x\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
9083        );
9084        write_file(&dir.path().join("applied/.gitignore"), b"*.log\n");
9085        write_file(&dir.path().join("applied/dropped.log"), b"applied");
9086        let config = ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
9087
9088        let (detached, _) = detached_and_streaming_indexes(dir.path(), &config);
9089        let (_, report) = scan_into_index(dir.path(), &config).expect("scan");
9090
9091        assert!(report.is_complete(), "{:?}", report.errors);
9092        let coverage = observed_coverage(&detached);
9093        assert_eq!((coverage.applied, coverage.refused), (1, 2));
9094        assert_eq!(
9095            coverage.refusals,
9096            vec![
9097                crate::control::RefusedControl {
9098                    path: PathBuf::from("guarded/.gitignore"),
9099                    reason: crate::control::ControlRefusalReason::LineLimit,
9100                },
9101                crate::control::RefusedControl {
9102                    path: PathBuf::from("huge/.gitignore"),
9103                    reason: crate::control::ControlRefusalReason::Budget,
9104                },
9105            ]
9106        );
9107        assert_eq!(detached.is_ignored(Path::new("guarded/kept.log")).expect("observed"), None);
9108        assert_eq!(
9109            detached.is_ignored(Path::new("applied/dropped.log")).expect("observed"),
9110            Some(true)
9111        );
9112    }
9113
9114    /// The control counters attribute what a scan's control state cost: files read, sources
9115    /// refused, and sources that shared a retained content instead of parsing their own.
9116    ///
9117    /// Off by default and compiled in, like every counter, so the numbers a speed check
9118    /// reads come from the shipped path rather than an instrumented build.
9119    #[test]
9120    fn control_counters_attribute_reads_refusals_and_sharing() {
9121        let _serial = crate::counters::test_serial();
9122        let dir = tempfile::tempdir().expect("tempdir");
9123        let shared = b"*.log\n".to_vec();
9124        write_file(&dir.path().join(".gitignore"), &shared);
9125        write_file(&dir.path().join("twin/.gitignore"), &shared);
9126        let mut long_line = b"*.tmp\n".to_vec();
9127        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
9128        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
9129        let config = ScanConfig { read_controls: true, threads: Some(1), ..ScanConfig::default() };
9130
9131        crate::counters::enable(true);
9132        // Deltas around the scan rather than absolute totals, for the reason
9133        // `a_walk_moves_every_counter_it_should` gives: the counters are process-global,
9134        // `test_serial` only serializes the tests that take it, and every report in this
9135        // binary now reads `.gitignore` by default, so a test running beside this one can
9136        // add control reads of its own.
9137        let before = crate::counters::snapshot();
9138        let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
9139        crate::counters::flush_thread();
9140        let after = crate::counters::snapshot();
9141        crate::counters::enable(false);
9142
9143        assert!(report.is_complete(), "{:?}", report.errors);
9144        assert_eq!(observed_coverage(&index).refused, 1);
9145        // `>=` in the one direction concurrency can move them. A count that is too low
9146        // means a path ran uninstrumented, which is the defect worth catching; too high
9147        // is another test's tree, which is not.
9148        for (label, observed, expected) in [
9149            ("one read per .gitignore", after.control_reads - before.control_reads, 3),
9150            ("the line over the limit", after.control_refused - before.control_refused, 1),
9151            (
9152                "the twin shares one parsed content",
9153                after.control_sources_shared - before.control_sources_shared,
9154                1,
9155            ),
9156        ] {
9157            assert!(observed >= expected, "{label}: counted {observed}, expected {expected}");
9158        }
9159    }
9160
9161    /// Both limits are part of the scope, and each lifts only its own refusals: no budget
9162    /// still refuses a long line, and no line limit still refuses a file past the budget.
9163    #[test]
9164    fn each_control_limit_is_scope_and_lifts_only_its_own_refusals() {
9165        use crate::control::{ControlLimits, ControlRefusalReason, RefusedControl};
9166
9167        let with = |limits| ScanConfig { control_limits: limits, ..ScanConfig::default() };
9168        let defaults = ControlLimits::default();
9169        let default = ScanConfig::default();
9170        let no_budget = with(ControlLimits { budget: None, ..defaults });
9171        let no_line_limit = with(ControlLimits { line_limit: None, ..defaults });
9172        let configs = [
9173            default.clone(),
9174            with(ControlLimits { budget: Some(16 * 1024 * 1024), ..defaults }),
9175            no_budget.clone(),
9176            with(ControlLimits { line_limit: Some(64 * 1024), ..defaults }),
9177            no_line_limit.clone(),
9178            with(ControlLimits { budget: None, line_limit: None }),
9179            // The same values in each other's places are a different scope.
9180            with(ControlLimits { budget: defaults.line_limit, line_limit: defaults.budget }),
9181        ];
9182        let scopes: Vec<ScanScope> = configs.iter().map(ScanConfig::scope).collect();
9183        for (index, scope) in scopes.iter().enumerate() {
9184            assert!(scope.observes_controls());
9185            assert!(scopes[index + 1..].iter().all(|other| other != scope), "{scopes:?}");
9186        }
9187        for config in &configs {
9188            let blind = ScanConfig { read_controls: false, ..config.clone() };
9189            assert_eq!(blind.scope().ignore_rules_fingerprint, 0, "unobserved has one scope");
9190        }
9191
9192        let dir = tempfile::tempdir().expect("tempdir");
9193        let mut long_line = b"*.log\n".to_vec();
9194        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
9195        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
9196        write_file(&dir.path().join("guarded/dropped.log"), b"log");
9197        write_file(
9198            &dir.path().join("huge/.gitignore"),
9199            &b"x\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
9200        );
9201        let refused = |path: &str, reason| RefusedControl { path: PathBuf::from(path), reason };
9202        let (bounded, _) = scan_into_index(dir.path(), &default).expect("default scan");
9203        assert_eq!(observed_coverage(&bounded).refused, 2);
9204
9205        let (budget_lifted, _) = detached_and_streaming_indexes(dir.path(), &no_budget);
9206        let coverage = observed_coverage(&budget_lifted);
9207        assert_eq!(coverage.limits, no_budget.control_limits);
9208        assert_eq!(
9209            coverage.refusals,
9210            [refused("guarded/.gitignore", ControlRefusalReason::LineLimit)]
9211        );
9212        assert_eq!(
9213            budget_lifted.is_ignored(Path::new("guarded/dropped.log")).expect("observed"),
9214            None
9215        );
9216
9217        let (line_limit_lifted, _) = detached_and_streaming_indexes(dir.path(), &no_line_limit);
9218        let coverage = observed_coverage(&line_limit_lifted);
9219        assert_eq!(coverage.refusals, [refused("huge/.gitignore", ControlRefusalReason::Budget)]);
9220        assert_eq!(
9221            line_limit_lifted.is_ignored(Path::new("guarded/dropped.log")).expect("observed"),
9222            Some(true)
9223        );
9224        assert_eq!(line_limit_lifted.scope(), no_line_limit.scope());
9225    }
9226
9227    /// The synthetic tree that ended a cold scan (fdu-1onj): 1,105 directories, each with
9228    /// a distinct 510-byte `.gitignore` of short rules, plus one line over the limit. The
9229    /// scan completes with every size exact and names what it refused, on both lanes.
9230    #[test]
9231    fn a_tree_past_both_control_bounds_completes_with_exact_sizes() {
9232        const DIRECTORIES: usize = 1_105;
9233        let dir = tempfile::tempdir().expect("tempdir");
9234        for directory in 0..DIRECTORIES {
9235            let mut source = Vec::new();
9236            for line in 0..63 {
9237                source.extend(format!("p{directory:04}{line:02}\n").bytes());
9238            }
9239            source.extend(format!("q{directory:04}\n").bytes());
9240            assert_eq!(source.len(), 510);
9241            let root = dir.path().join(format!("d{directory:04}"));
9242            write_file(&root.join(".gitignore"), &source);
9243            write_file(&root.join("file.txt"), b"contents");
9244        }
9245        write_file(
9246            &dir.path().join("a-guard/.gitignore"),
9247            &vec![b'x'; crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1],
9248        );
9249        crate::test_support::settle_allocations(dir.path());
9250        let observing =
9251            ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
9252        let blind = ScanConfig { read_controls: false, ..observing.clone() };
9253
9254        let (unobserved, _) = scan_into_index(dir.path(), &blind).expect("controls-off scan");
9255        let canonical = dir.path().canonicalize().expect("canonical root");
9256        let lanes = [
9257            scan_into_index(dir.path(), &observing).expect("detached scan"),
9258            scan_into_index_via_scanner(&canonical, &observing).expect("streaming scan"),
9259        ];
9260        for (index, report) in &lanes {
9261            assert!(report.is_complete(), "{:?}", report.errors);
9262            assert_eq!(index.total(), unobserved.total(), "sizes do not depend on controls");
9263            let coverage = observed_coverage(index);
9264            assert!(coverage.refused > 1, "the budget refused sources: {coverage:?}");
9265            assert_eq!(
9266                coverage.applied + coverage.refused,
9267                u64::try_from(DIRECTORIES + 1).expect("small")
9268            );
9269            assert_eq!(coverage.refusals.len(), crate::MAX_RETAINED_ISSUES);
9270            assert!(!coverage.lists_every_refusal());
9271            assert_eq!(
9272                coverage.refusals[0],
9273                crate::control::RefusedControl {
9274                    path: PathBuf::from("a-guard/.gitignore"),
9275                    reason: crate::control::ControlRefusalReason::LineLimit,
9276                }
9277            );
9278            assert!(
9279                index.control_table().retained_cost() <= crate::control::DEFAULT_CONTROL_BUDGET
9280            );
9281        }
9282    }
9283
9284    #[test]
9285    fn fingerprint_metadata_observes_mutation_after_directory_enumeration() {
9286        let dir = tempfile::tempdir().expect("tempdir");
9287        let path = dir.path().join("changing.bin");
9288        write_file(&path, b"before");
9289        let entry = fs::read_dir(dir.path())
9290            .expect("read directory")
9291            .next()
9292            .expect("one entry")
9293            .expect("read entry");
9294
9295        write_file(&path, b"after mutation");
9296
9297        let metadata = metadata_for_fingerprint(&entry).expect("fresh metadata");
9298        assert_eq!(metadata.len(), b"after mutation".len() as u64);
9299    }
9300
9301    /// A tree wide and deep enough that workers genuinely interleave.
9302    ///
9303    /// A three-file fixture would pass every one of these tests with a broken queue,
9304    /// because one worker would finish before another started.
9305    fn branching_tree() -> tempfile::TempDir {
9306        let dir = tempfile::tempdir().expect("tempdir");
9307        for top in 0..12 {
9308            for middle in 0..6 {
9309                for leaf in 0..7 {
9310                    write_file(
9311                        &dir.path().join(format!("t{top}/m{middle}/leaf-{leaf}.dat")),
9312                        &vec![b'x'; leaf * 13],
9313                    );
9314                }
9315            }
9316            // A deep chain alongside the wide fan-out, so depth and width are both
9317            // exercised by the same walk.
9318            write_file(&dir.path().join(format!("t{top}/a/b/c/d/e/deep.txt")), b"deep");
9319        }
9320        crate::test_support::settle_allocations(dir.path());
9321        dir
9322    }
9323
9324    fn controlled_branching_tree() -> tempfile::TempDir {
9325        let dir = branching_tree();
9326        write_file(&dir.path().join(".gitignore"), b"leaf-1.dat\nt7/\n");
9327        write_file(&dir.path().join("t3/.gitignore"), b"!m2/leaf-1.dat\n*.tmp\n");
9328        write_file(&dir.path().join("t3/m2/generated.tmp"), b"ignored by nested control");
9329        write_file(&dir.path().join("t7/.gitignore"), b"!m0/leaf-1.dat\n");
9330        fs::create_dir_all(dir.path().join("t5/.gitignore")).expect("non-file control directory");
9331        write_file(&dir.path().join("t5/.gitignore/ordinary.txt"), b"ordinary child");
9332        crate::test_support::settle_allocations(dir.path());
9333        dir
9334    }
9335
9336    fn index_fingerprint(index: &Index) -> Vec<(PathBuf, EntryKind, Attrs)> {
9337        let mut entries: Vec<(PathBuf, EntryKind, Attrs)> = Vec::new();
9338        let mut queue = vec![PathBuf::new()];
9339        while let Some(path) = queue.pop() {
9340            let Some(children) = index.children(&path) else {
9341                continue;
9342            };
9343            let names: Vec<PathBuf> = children.map(|(name, _id)| path.join(name)).collect();
9344            for child_path in names {
9345                let kind = index.kind(&child_path).expect("child has a kind");
9346                let attrs = *index.attrs(&child_path).expect("child has attrs");
9347                entries.push((child_path.clone(), kind, attrs));
9348                if kind.is_dir() {
9349                    queue.push(child_path);
9350                }
9351            }
9352        }
9353        entries.sort_by(|left, right| left.0.cmp(&right.0));
9354        entries
9355    }
9356
9357    fn detached_and_streaming_indexes(root: &Path, config: &ScanConfig) -> (Index, Index) {
9358        let canonical = root.canonicalize().expect("canonical test root");
9359        let (streaming, streaming_report) =
9360            scan_into_index_via_scanner(&canonical, config).expect("streaming oracle");
9361        let (detached, detached_report) = scan_into_index(root, config).expect("detached scan");
9362
9363        assert_eq!(detached_report.dirs_read, streaming_report.dirs_read);
9364        assert_eq!(detached_report.entries, streaming_report.entries);
9365        assert_eq!(detached_report.files_walked, streaming_report.files_walked);
9366        assert_eq!(detached_report.bytes_walked, streaming_report.bytes_walked);
9367        assert_eq!(
9368            detached_report.errors.iter().map(ToString::to_string).collect::<Vec<_>>(),
9369            streaming_report.errors.iter().map(ToString::to_string).collect::<Vec<_>>()
9370        );
9371        assert_indexes_equal(&detached, &streaming);
9372        (detached, streaming)
9373    }
9374
9375    fn assert_indexes_equal(left: &Index, right: &Index) {
9376        assert_eq!(index_fingerprint(left), index_fingerprint(right));
9377        assert_eq!(left.total(), right.total());
9378        assert_eq!(left.partition_total().ok(), right.partition_total().ok());
9379        assert_eq!(left.scope(), right.scope());
9380        assert_eq!(left.freshness(), right.freshness());
9381        assert_eq!(left.state(), right.state());
9382        assert_eq!(left.clock(), right.clock());
9383        assert_eq!(left.len(), right.len());
9384        assert_eq!(left.issues(), right.issues());
9385        assert_eq!(left.observes_controls(), right.observes_controls());
9386        assert_eq!(left.control_coverage(), right.control_coverage());
9387        assert_eq!(
9388            left.control_table()
9389                .sources()
9390                .map(|(path, source)| (path, source.to_vec()))
9391                .collect::<Vec<_>>(),
9392            right
9393                .control_table()
9394                .sources()
9395                .map(|(path, source)| (path, source.to_vec()))
9396                .collect::<Vec<_>>()
9397        );
9398        for (path, _, _) in index_fingerprint(left) {
9399            assert_eq!(left.is_ignored(&path).ok(), right.is_ignored(&path).ok(), "{path:?}");
9400        }
9401    }
9402
9403    /// A small tree whose mutation crosses every structural reconciliation boundary.
9404    fn reconciliation_transition_tree() -> tempfile::TempDir {
9405        let dir = tempfile::tempdir().expect("tempdir");
9406        write_file(&dir.path().join("changed.txt"), b"before");
9407        write_file(&dir.path().join("removed.txt"), b"remove me");
9408        write_file(&dir.path().join("directory-to-file/old.rs"), b"old child");
9409        write_file(&dir.path().join("file-to-directory"), b"old file");
9410        write_file(&dir.path().join("removed-tree/nested/gone.md"), b"gone");
9411        write_file(&dir.path().join("stable/deep/kept.rs"), b"kept");
9412        crate::test_support::settle_allocations(dir.path());
9413        dir
9414    }
9415
9416    fn mutate_reconciliation_transition_tree(root: &Path) {
9417        write_file(&root.join("changed.txt"), b"after, with a distinct size");
9418        fs::remove_file(root.join("removed.txt")).expect("remove root file");
9419
9420        fs::remove_dir_all(root.join("directory-to-file")).expect("remove old directory");
9421        write_file(&root.join("directory-to-file"), b"replacement file");
9422
9423        fs::remove_file(root.join("file-to-directory")).expect("remove old file");
9424        write_file(&root.join("file-to-directory/new.txt"), b"replacement child");
9425
9426        fs::remove_dir_all(root.join("removed-tree")).expect("remove nested tree");
9427        write_file(&root.join("added-tree/nested/new.md"), b"new nested file");
9428        crate::test_support::settle_allocations(root);
9429    }
9430
9431    fn effective_ops(commits: &[Commit]) -> Vec<Op> {
9432        let mut operations: Vec<_> = commits
9433            .iter()
9434            .flat_map(|commit| commit.changes.iter())
9435            .filter_map(|change| match change {
9436                crate::EffectiveChange::Inserted { path, kind, attrs } => {
9437                    Some(Op::Upsert { path: path.clone(), kind: *kind, attrs: *attrs })
9438                }
9439                crate::EffectiveChange::Updated { path, kind, current, .. } => {
9440                    Some(Op::Upsert { path: path.clone(), kind: *kind, attrs: *current })
9441                }
9442                crate::EffectiveChange::Removed { path, .. } => {
9443                    Some(Op::Remove { path: path.clone() })
9444                }
9445                crate::EffectiveChange::Invalidated { path, reason } => {
9446                    Some(Op::InvalidateSubtree { path: path.clone(), reason: *reason })
9447                }
9448                crate::EffectiveChange::ControlUpdated { .. }
9449                | crate::EffectiveChange::ControlRefusalUpdated { .. }
9450                | crate::EffectiveChange::Reclassified { .. } => None,
9451            })
9452            .collect();
9453        operations.sort_by(|left, right| left.path().cmp(right.path()));
9454        operations
9455    }
9456
9457    fn commit_touches(commit: &Commit, path: &Path) -> bool {
9458        commit.changes.iter().any(|change| change.path() == path)
9459    }
9460
9461    #[test]
9462    fn parallel_and_serial_walks_produce_the_same_index() {
9463        let dir = branching_tree();
9464        let serial_config = ScanConfig { threads: Some(1), ..ScanConfig::default() };
9465        let (serial, serial_report) =
9466            scan_into_index(dir.path(), &serial_config).expect("serial scan");
9467        assert!(serial_report.is_complete());
9468
9469        for threads in [2_usize, 3, 8] {
9470            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
9471            let (parallel, report) = scan_into_index(dir.path(), &config).expect("parallel scan");
9472            assert!(report.is_complete(), "{threads} threads reported errors");
9473            assert_eq!(report.entries, serial_report.entries, "{threads} threads");
9474            assert_eq!(report.dirs_read, serial_report.dirs_read, "{threads} threads");
9475            assert_eq!(report.files_walked, serial_report.files_walked, "{threads} threads");
9476            assert_eq!(report.bytes_walked, serial_report.bytes_walked, "{threads} threads");
9477            // Public roll-ups carry extension names even though the internal merge path
9478            // uses ids whose assignment order differs between serial and parallel walks.
9479            let (serial_total, parallel_total) = (serial.total(), parallel.total());
9480            assert_eq!(
9481                (
9482                    parallel_total.files,
9483                    parallel_total.dirs,
9484                    parallel_total.bytes,
9485                    parallel_total.allocated,
9486                    parallel_total.newest_mtime_ns,
9487                ),
9488                (
9489                    serial_total.files,
9490                    serial_total.dirs,
9491                    serial_total.bytes,
9492                    serial_total.allocated,
9493                    serial_total.newest_mtime_ns,
9494                ),
9495                "{threads} threads roll-up"
9496            );
9497            assert_eq!(
9498                parallel_total.by_ext, serial_total.by_ext,
9499                "{threads} threads per-extension roll-up"
9500            );
9501            assert_eq!(
9502                index_fingerprint(&parallel),
9503                index_fingerprint(&serial),
9504                "{threads} threads produced a different index"
9505            );
9506        }
9507    }
9508
9509    #[test]
9510    fn parallel_walk_emits_every_entry_exactly_once() {
9511        let dir = branching_tree();
9512        let config = ScanConfig { threads: Some(4), batch_size: 16, ..ScanConfig::default() };
9513        let mut seen: BTreeMap<PathBuf, usize> = BTreeMap::new();
9514        let report = scan(dir.path(), &config, &mut |observation| {
9515            for op in &observation.ops {
9516                if let Op::Upsert { path, .. } = &op.op {
9517                    *seen.entry(path.clone()).or_default() += 1;
9518                }
9519            }
9520        })
9521        .expect("parallel scan");
9522
9523        assert!(report.is_complete());
9524        assert_eq!(seen.len() as u64, report.entries, "entry count disagrees with the report");
9525        let duplicated: Vec<_> =
9526            seen.iter().filter(|(_path, count)| **count != 1).map(|(path, _)| path).collect();
9527        assert!(duplicated.is_empty(), "paths emitted more than once: {duplicated:?}");
9528    }
9529
9530    #[test]
9531    fn parallel_walk_honours_max_depth() {
9532        let dir = branching_tree();
9533        for threads in [1_usize, 4] {
9534            let config =
9535                ScanConfig { threads: Some(threads), max_depth: Some(2), ..ScanConfig::default() };
9536            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
9537            assert!(report.is_complete());
9538            for (path, _kind, _attrs) in index_fingerprint(&index) {
9539                assert!(
9540                    path.components().count() <= 2,
9541                    "{threads} threads kept {path:?} past the depth limit"
9542                );
9543            }
9544        }
9545    }
9546
9547    #[test]
9548    fn scan_order_never_changes_the_resulting_index() {
9549        let dir = branching_tree();
9550        let depth_first =
9551            ScanConfig { order: ScanOrder::DepthFirst, threads: Some(1), ..ScanConfig::default() };
9552        let (expected, expected_report) =
9553            scan_into_index(dir.path(), &depth_first).expect("depth-first scan");
9554
9555        for (order, threads) in
9556            [(ScanOrder::BreadthFirst, 1), (ScanOrder::BreadthFirst, 4), (ScanOrder::DepthFirst, 4)]
9557        {
9558            let config = ScanConfig { order, threads: Some(threads), ..ScanConfig::default() };
9559            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
9560            assert_eq!(report.entries, expected_report.entries, "{order:?}/{threads}");
9561            assert_eq!(report.dirs_read, expected_report.dirs_read, "{order:?}/{threads}");
9562            // Public roll-ups resolve internal ids, so their named maps are stable even
9563            // when traversal order changes id assignment.
9564            let (totals, expected_totals) = (index.total(), expected.total());
9565            assert_eq!(
9566                (totals.files, totals.dirs, totals.bytes, totals.allocated),
9567                (
9568                    expected_totals.files,
9569                    expected_totals.dirs,
9570                    expected_totals.bytes,
9571                    expected_totals.allocated
9572                ),
9573                "{order:?}/{threads} roll-up"
9574            );
9575            assert_eq!(
9576                totals.newest_mtime_ns, expected_totals.newest_mtime_ns,
9577                "{order:?}/{threads} newest mtime"
9578            );
9579            assert_eq!(
9580                totals.by_ext, expected_totals.by_ext,
9581                "{order:?}/{threads} extension tallies"
9582            );
9583            assert_eq!(
9584                index_fingerprint(&index),
9585                index_fingerprint(&expected),
9586                "{order:?}/{threads} produced a different index"
9587            );
9588        }
9589    }
9590
9591    #[test]
9592    fn a_single_worker_breadth_first_walk_is_strictly_level_ordered() {
9593        // The strict guarantee, which holds only with one worker. With several, the
9594        // queue is ordered but the claims are not: a fast worker can enqueue and claim
9595        // depth d+2 while a slow worker still holds depth d+1. See
9596        // `breadth_first_starts_every_top_level_subtree_early` for the property the
9597        // default configuration actually provides, which is the one consumers rely on.
9598        let dir = branching_tree();
9599        let config = ScanConfig {
9600            order: ScanOrder::BreadthFirst,
9601            threads: Some(1),
9602            batch_size: 1,
9603            ..ScanConfig::default()
9604        };
9605        let mut depths_in_order: Vec<usize> = Vec::new();
9606        scan(dir.path(), &config, &mut |observation| {
9607            for op in &observation.ops {
9608                if let Op::Upsert { path, kind, .. } = &op.op {
9609                    if kind.is_dir() {
9610                        depths_in_order.push(path.components().count());
9611                    }
9612                }
9613            }
9614        })
9615        .expect("scan");
9616
9617        assert!(depths_in_order.len() > 10, "fixture should have many directories");
9618        assert!(
9619            depths_in_order.windows(2).all(|pair| pair[0] <= pair[1]),
9620            "directory depths were not non-decreasing: {depths_in_order:?}"
9621        );
9622    }
9623
9624    /// How many of the fixture's twelve top-level subtrees have received any file by
9625    /// the time half the files have been emitted.
9626    ///
9627    /// This is the product metric — "is a mid-scan ranking meaningful?" — rather than
9628    /// first-touch, which cannot distinguish the orders at all: reading the root
9629    /// enumerates all twelve children at once either way. What a ranking needs is that
9630    /// the subtrees grow *together*.
9631    fn subtrees_started_at_halfway(order: ScanOrder, threads: usize, dir: &Path) -> usize {
9632        let config =
9633            ScanConfig { order, batch_size: 1, threads: Some(threads), ..ScanConfig::default() };
9634
9635        let mut files: Vec<PathBuf> = Vec::new();
9636        scan(dir, &config, &mut |observation| {
9637            for op in &observation.ops {
9638                if let Op::Upsert { path, kind, .. } = &op.op {
9639                    if !kind.is_dir() {
9640                        files.push(path.clone());
9641                    }
9642                }
9643            }
9644        })
9645        .expect("scan");
9646
9647        let halfway = files.len() / 2;
9648        let mut started: BTreeSet<PathBuf> = BTreeSet::new();
9649        for path in files.iter().take(halfway) {
9650            if let Some(top) = path.components().next() {
9651                started.insert(PathBuf::from(top.as_os_str()));
9652            }
9653        }
9654        started.len()
9655    }
9656
9657    #[test]
9658    fn a_parallel_walk_accounts_for_where_its_time_went() {
9659        // The attribution identity: every named cause is a disjoint slice of worker
9660        // wall time, so the parts can never exceed the whole, and the counters that
9661        // amortization depends on are actually incremented. This is the instrument
9662        // the scheduler experiments will read; if it drifts, they measure noise.
9663        let dir = branching_tree();
9664        let config = ScanConfig { threads: Some(4), batch_size: 64, ..ScanConfig::default() };
9665        let report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
9666        let a = report.attribution;
9667
9668        assert!(a.claims > 0, "a parallel walk claims chunks: {a:?}");
9669        assert!(a.work_ns > 0, "reading directories takes time: {a:?}");
9670        assert!(a.wall_ns > 0);
9671        // claim() locks at least once per successful claim, and release() locks once
9672        // per claim cycle too.
9673        assert!(a.lock_ops >= a.claims * 2, "lock ops out of step with claims: {a:?}");
9674        assert!(
9675            a.accounted_ns() <= a.wall_ns,
9676            "attributed slices are disjoint intervals inside worker wall: {a:?}"
9677        );
9678    }
9679
9680    #[test]
9681    fn a_serial_walk_has_no_coordination_to_attribute() {
9682        // Serial semantics: wall is the loop, "send" is the inline sink (the consumer
9683        // actually running), work is the rest — and the coordination counters stay
9684        // zero because there is no queue lock and no channel.
9685        let dir = branching_tree();
9686        let config = ScanConfig { threads: Some(1), batch_size: 64, ..ScanConfig::default() };
9687        let mut observations = 0usize;
9688        let report = scan(dir.path(), &config, &mut |_| observations += 1).expect("scan");
9689        let a = report.attribution;
9690
9691        assert!(observations > 0, "the sink ran, so send_ns measured something real");
9692        assert!(a.work_ns > 0 && a.wall_ns >= a.work_ns);
9693        assert_eq!(
9694            (a.claims, a.lock_ops, a.lock_contended, a.starved_ns, a.lock_wait_ns),
9695            (0, 0, 0, 0, 0),
9696            "no queue, no lock, nothing to wait on: {a:?}"
9697        );
9698    }
9699
9700    /// Twelve top-level subtrees, each a branching tree several levels deep.
9701    ///
9702    /// Branching matters: an earlier fixture gave every level exactly one child, which
9703    /// pinned the frontier at twelve directories and made both orders behave
9704    /// identically — a LIFO cannot dive when there is nothing to dive into. With two
9705    /// children per level, depth-first pushes siblings and immediately descends into
9706    /// the last one, which is the behaviour that leaves other subtrees behind.
9707    ///
9708    /// It is also deliberately uniform. A version using one deep spur beside shallow
9709    /// siblings made the result depend on whether `readdir` returned the spur early:
9710    /// it passed on APFS and failed on ext4.
9711    fn deep_forest() -> tempfile::TempDir {
9712        let dir = tempfile::tempdir().expect("tempdir");
9713        for top in 0..12 {
9714            let mut level: Vec<PathBuf> = vec![dir.path().join(format!("t{top}"))];
9715            for _ in 0..5 {
9716                let mut next = Vec::new();
9717                for parent in &level {
9718                    for child in 0..2 {
9719                        let path = parent.join(format!("c{child}"));
9720                        for file in 0..3 {
9721                            write_file(&path.join(format!("f{file}.dat")), b"xxxxxxxxxx");
9722                        }
9723                        next.push(path);
9724                    }
9725                }
9726                level = next;
9727            }
9728        }
9729        dir
9730    }
9731
9732    /// Files accumulated by the *least advanced* top-level subtree in the first
9733    /// quarter of the walk.
9734    ///
9735    /// Counting subtrees merely *started* cannot discriminate on a tree whose root
9736    /// fans out twelve ways: every scheduler touches all twelve immediately, because
9737    /// reading the root enumerates them. What differs is whether they then advance
9738    /// together, so the question is how far behind the laggard is.
9739    fn leanest_subtree_early(order: ScanOrder, threads: usize, dir: &Path) -> usize {
9740        let config =
9741            ScanConfig { order, batch_size: 1, threads: Some(threads), ..ScanConfig::default() };
9742        let mut files: Vec<PathBuf> = Vec::new();
9743        scan(dir, &config, &mut |observation| {
9744            for op in &observation.ops {
9745                if let Op::Upsert { path, kind, .. } = &op.op {
9746                    if !kind.is_dir() {
9747                        files.push(path.clone());
9748                    }
9749                }
9750            }
9751        })
9752        .expect("scan");
9753
9754        let quarter = files.len() / 4;
9755        let mut per_top: BTreeMap<PathBuf, usize> = BTreeMap::new();
9756        for path in files.iter().take(quarter) {
9757            if let Some(top) = path.components().next() {
9758                *per_top.entry(PathBuf::from(top.as_os_str())).or_default() += 1;
9759            }
9760        }
9761        (0..12)
9762            .map(|top| per_top.get(&PathBuf::from(format!("t{top}"))).copied().unwrap_or(0))
9763            .min()
9764            .unwrap_or(0)
9765    }
9766
9767    #[test]
9768    fn deep_subtrees_do_not_delay_their_siblings() {
9769        // The orientation property, and the reason breadth-first is the default: when
9770        // every top-level subtree is deep, depth-first pours its early effort down
9771        // whichever ones it picked up and leaves the rest at zero, while the region
9772        // scheduler advances all twelve together. A user watching the top level fill
9773        // in sees a meaningful ranking in the first case and a misleading one in the
9774        // second.
9775        //
9776        // Asserted at one worker only, and that bound is deliberate. This metric reads
9777        // *emission* order, and under several workers emission reflects which worker
9778        // finished first as much as which region was claimed — so it varies with core
9779        // count. Measured on a six-core machine the margin is wide (33-37 files against
9780        // 6); on a CI runner with fewer cores both orders can report zero. That makes it
9781        // a benchmark-grade observation, recorded in exp-013, not a unit-test assertion.
9782        //
9783        // The scheduling property itself *is* asserted deterministically, against the
9784        // queue rather than through a walk, by
9785        // `the_region_scheduler_spreads_workers_over_distinct_subtrees`.
9786        let dir = deep_forest();
9787        let breadth = leanest_subtree_early(ScanOrder::BreadthFirst, 1, dir.path());
9788        let depth = leanest_subtree_early(ScanOrder::DepthFirst, 1, dir.path());
9789        assert!(
9790            breadth > depth,
9791            "breadth-first should leave its least advanced top-level subtree further \
9792             along: {breadth} files against {depth}"
9793        );
9794    }
9795
9796    #[test]
9797    fn the_region_scheduler_spreads_workers_over_distinct_subtrees() {
9798        // The scheduler invariant, checked directly on the queue rather than through a
9799        // walk: consecutive claims by *different* workers must land in different
9800        // regions while several regions have work. This is what the round-robin ready
9801        // ring buys, and it is the thing a global FIFO could not promise.
9802        let queue = DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, None, None);
9803        let mut timing = WalkAttribution::default();
9804
9805        // Bootstrap: drain the root, then seed four top-level regions.
9806        let mut claimed = Vec::new();
9807        let root = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
9808        claimed.clear();
9809        queue.extend(
9810            (0..4).map(|top| (PathBuf::from(format!("t{top}")), 1, RegionId::UNASSIGNED)),
9811            &mut timing,
9812        );
9813        assert!(root.release(0, 0, &mut timing).is_none());
9814
9815        // Four workers with no affinity must each be handed a different region. The
9816        // claims are held for the whole loop, as four concurrent workers would hold
9817        // them, because releasing between them would let one worker take every region.
9818        let mut regions = BTreeSet::new();
9819        let mut held = Vec::new();
9820        for _ in 0..4 {
9821            let mut claimed = Vec::new();
9822            held.push(queue.claim(&mut claimed, &mut timing).expect("a region has work"));
9823            regions.insert(claimed[0].2.0);
9824            assert_eq!(claimed.len(), 1, "one directory per region so far");
9825        }
9826        assert_eq!(regions.len(), 4, "each claim took a distinct region: {regions:?}");
9827    }
9828
9829    #[test]
9830    fn breadth_first_spreads_early_work_across_top_level_subtrees() {
9831        // The justification for making breadth-first the default: at the halfway point
9832        // more of the tree's top-level subtrees have started filling, so a consumer
9833        // ranking by size mid-scan is comparing partial values rather than a mix of
9834        // final values and zeros.
9835        //
9836        // Pinned with one worker, where the ordering guarantee is strict and the result
9837        // is deterministic. The multi-worker case is deliberately NOT asserted here:
9838        // measured on this fixture the advantage disappears under the default worker
9839        // count (both orders start 7-8 subtrees, run to run), because emission order is
9840        // then dominated by worker scheduling rather than by queue order. That is a
9841        // real limitation of the current design, recorded in the plan and tracked
9842        // rather than papered over with a test tuned until it passed.
9843        let dir = branching_tree();
9844        let breadth = subtrees_started_at_halfway(ScanOrder::BreadthFirst, 1, dir.path());
9845        let depth = subtrees_started_at_halfway(ScanOrder::DepthFirst, 1, dir.path());
9846
9847        assert!(
9848            breadth > depth,
9849            "breadth-first should have more top-level subtrees underway at the halfway \
9850             point, but started {breadth} against depth-first's {depth}"
9851        );
9852    }
9853
9854    #[test]
9855    fn scan_order_does_not_change_the_cache_scope() {
9856        // Order is operational, like the worker count: it changes when observations
9857        // appear, never which ones, so it must not be able to invalidate a snapshot.
9858        let breadth = ScanConfig { order: ScanOrder::BreadthFirst, ..ScanConfig::default() };
9859        let depth = ScanConfig { order: ScanOrder::DepthFirst, ..ScanConfig::default() };
9860        assert_eq!(breadth.scope(), depth.scope());
9861    }
9862
9863    #[test]
9864    fn worker_threads_are_bounded_and_never_zero() {
9865        let zero = ScanConfig { threads: Some(0), ..ScanConfig::default() };
9866        assert_eq!(zero.worker_threads(), 1, "zero threads must fall back to the serial walk");
9867        let absurd = ScanConfig { threads: Some(usize::MAX), ..ScanConfig::default() };
9868        assert_eq!(absurd.worker_threads(), MAX_SCAN_THREADS);
9869        // The automatic choice is capped well below what a caller may request, because
9870        // the measured knee is far below the core count on a large machine.
9871        let automatic = ScanConfig { threads: None, ..ScanConfig::default() };
9872        assert!((1..=DEFAULT_SCAN_THREADS_CAP).contains(&automatic.worker_threads()));
9873    }
9874
9875    #[test]
9876    fn automatic_worker_pool_keeps_a_conservative_start_and_bounded_reserve() {
9877        assert_eq!(automatic_worker_pool(1), WorkerPool::fixed(1));
9878        assert_eq!(
9879            automatic_worker_pool(4),
9880            WorkerPool {
9881                initial: 4,
9882                maximum: 8,
9883                calibration: Some(WorkerCalibration::new(
9884                    ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
9885                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
9886                )),
9887            }
9888        );
9889        assert_eq!(
9890            automatic_worker_pool(10),
9891            WorkerPool {
9892                initial: DEFAULT_SCAN_THREADS_CAP,
9893                maximum: ADAPTIVE_SCAN_THREADS_CAP,
9894                calibration: Some(WorkerCalibration::new(
9895                    ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
9896                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
9897                )),
9898            }
9899        );
9900    }
9901
9902    #[test]
9903    fn an_abandoned_claim_does_not_strand_the_other_workers() {
9904        // The liveness property behind `DirectoryClaim`. A worker that stops mid-chunk
9905        // — consumer gone, or a panic unwinding through the directory read — still owes
9906        // the queue its claim, and `claim` parks everyone else until `outstanding`
9907        // reaches zero. Before the claim was an RAII guard both of those exits skipped
9908        // the release, and every remaining worker waited on the condvar forever while
9909        // the scoped join waited on them.
9910        let queue = std::sync::Arc::new(DirectoryQueue::new(
9911            (PathBuf::new(), 0),
9912            ScanOrder::BreadthFirst,
9913            None,
9914            None,
9915        ));
9916        let mut timing = WalkAttribution::default();
9917        let mut claimed = Vec::new();
9918
9919        // One worker takes the root and abandons it without publishing anything.
9920        drop(queue.claim(&mut claimed, &mut timing).expect("root is claimable"));
9921
9922        // A second worker must now be told the walk is over rather than parking.
9923        let waiter = queue.clone();
9924        let (done, finished) = std::sync::mpsc::sync_channel(1);
9925        std::thread::spawn(move || {
9926            let mut timing = WalkAttribution::default();
9927            let mut claimed = Vec::new();
9928            let outcome = waiter.claim(&mut claimed, &mut timing).is_some();
9929            done.send(outcome).expect("publish the claim outcome");
9930        });
9931
9932        assert_eq!(
9933            finished.recv_timeout(std::time::Duration::from_secs(5)),
9934            Ok(false),
9935            "the queue must report the walk finished instead of parking the worker"
9936        );
9937    }
9938
9939    #[test]
9940    fn automatic_queue_activates_its_reserve_only_for_slow_initial_work() {
9941        let slow = WorkerCalibration::new(3, 10);
9942        let queue =
9943            DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, Some(slow), None);
9944        let mut timing = WalkAttribution::default();
9945        let mut claimed = Vec::new();
9946
9947        let claim = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
9948        queue.extend([(PathBuf::from("child"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
9949        assert!(claim.release(2, 20, &mut timing).is_none());
9950
9951        claimed.clear();
9952        let claim = queue.claim(&mut claimed, &mut timing).expect("child is claimable");
9953        queue.extend([(PathBuf::from("grandchild"), 2, claimed[0].2)].into_iter(), &mut timing);
9954        assert_eq!(claim.release(1, 10, &mut timing), Some(2));
9955
9956        let fast = WorkerCalibration::new(3, 11);
9957        let queue =
9958            DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, Some(fast), None);
9959        let mut timing = WalkAttribution::default();
9960        let mut claimed = Vec::new();
9961        let claim = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
9962        assert!(claim.release(3, 30, &mut timing).is_none());
9963        assert!(queue.lock().controller.is_none(), "calibration decides only once");
9964    }
9965
9966    #[test]
9967    fn repeated_windows_reconsider_a_late_slow_phase_in_entry_order() {
9968        let calibration = WorkerCalibration::new(4, 10);
9969        let mut controller =
9970            WorkerController::new(calibration, WorkerPolicyExperiment::RepeatedWindows);
9971
9972        let fast = controller.observe(4, 20).expect("first complete window");
9973        let slow = controller.observe(4, 80).expect("second complete window");
9974
9975        assert!(!fast.slow);
9976        assert!(slow.slow);
9977        assert_eq!((fast.start_entry_ordinal, fast.end_entry_ordinal), (0, 4));
9978        assert_eq!((slow.start_entry_ordinal, slow.end_entry_ordinal), (4, 8));
9979    }
9980
9981    #[test]
9982    fn shipped_trace_retains_post_decision_windows_without_changing_policy() {
9983        let calibration = WorkerCalibration::new(2, 10);
9984        let pool = WorkerPool { initial: 2, maximum: 4, calibration: Some(calibration) };
9985        let recorder =
9986            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::ShippedOneShot);
9987        let queue = DirectoryQueue::new_with_policy(
9988            (PathBuf::new(), 0),
9989            ScanOrder::BreadthFirst,
9990            Some(calibration),
9991            Some(recorder.clone()),
9992            pool.initial,
9993            pool.maximum,
9994            WorkerPolicyExperiment::ShippedOneShot,
9995        );
9996        let mut timing = WalkAttribution::default();
9997        let mut claimed = Vec::new();
9998
9999        let first = queue.claim(&mut claimed, &mut timing).expect("first window");
10000        queue.extend([(PathBuf::from("late"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
10001        assert_eq!(first.release(2, 10, &mut timing), None, "fast prefix holds");
10002
10003        claimed.clear();
10004        let late = queue.claim(&mut claimed, &mut timing).expect("late phase");
10005        queue.extend([(PathBuf::from("tail"), 2, RegionId::UNASSIGNED)].into_iter(), &mut timing);
10006        assert_eq!(late.release(2, 40, &mut timing), None, "shadow cannot scale");
10007
10008        let diagnostics = recorder.finish();
10009        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Held);
10010        assert_eq!(
10011            diagnostics
10012                .worker_policy
10013                .windows
10014                .iter()
10015                .map(|window| window.decision)
10016                .collect::<Vec<_>>(),
10017            vec![WorkerPolicyDecision::Hold, WorkerPolicyDecision::ObserveSlow]
10018        );
10019    }
10020
10021    #[test]
10022    fn staged_controller_requires_a_useful_frontier_then_stays_bounded() {
10023        let calibration = WorkerCalibration::new(1, 10);
10024        let pool = WorkerPool { initial: 2, maximum: 8, calibration: Some(calibration) };
10025        let recorder =
10026            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::StagedGatedWindows);
10027        let queue = DirectoryQueue::new_with_policy(
10028            (PathBuf::new(), 0),
10029            ScanOrder::BreadthFirst,
10030            Some(calibration),
10031            Some(recorder.clone()),
10032            pool.initial,
10033            pool.maximum,
10034            WorkerPolicyExperiment::StagedGatedWindows,
10035        );
10036        let mut timing = WalkAttribution::default();
10037        let mut claimed = Vec::new();
10038
10039        let root = queue.claim(&mut claimed, &mut timing).expect("root");
10040        queue.extend([(PathBuf::from("narrow"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
10041        assert_eq!(root.release(1, 20, &mut timing), None);
10042
10043        claimed.clear();
10044        let narrow = queue.claim(&mut claimed, &mut timing).expect("narrow child");
10045        queue.extend(
10046            (0..9).map(|index| (PathBuf::from(format!("wide-{index}")), 2, RegionId::UNASSIGNED)),
10047            &mut timing,
10048        );
10049        assert_eq!(narrow.release(1, 20, &mut timing), Some(4));
10050
10051        claimed.clear();
10052        let wide = queue.claim(&mut claimed, &mut timing).expect("wide claim");
10053        queue.extend(
10054            (0..9).map(|index| (PathBuf::from(format!("wider-{index}")), 3, RegionId::UNASSIGNED)),
10055            &mut timing,
10056        );
10057        assert_eq!(wide.release(1, 20, &mut timing), Some(8));
10058
10059        let diagnostics = recorder.finish();
10060        let decisions: Vec<_> = diagnostics
10061            .worker_policy
10062            .windows
10063            .iter()
10064            .map(|window| (window.decision, window.requested_workers))
10065            .collect();
10066        assert_eq!(
10067            decisions,
10068            vec![
10069                (WorkerPolicyDecision::HoldInsufficientFrontier, None),
10070                (WorkerPolicyDecision::ScaleUp, Some(4)),
10071                (WorkerPolicyDecision::ScaleUp, Some(8)),
10072            ]
10073        );
10074        assert!(
10075            diagnostics
10076                .worker_policy
10077                .windows
10078                .windows(2)
10079                .all(|pair| pair[0].end_entry_ordinal <= pair[1].start_entry_ordinal)
10080        );
10081        assert!(diagnostics.worker_policy.windows.iter().all(|window| {
10082            window.requested_workers.is_none_or(|workers| workers <= pool.maximum)
10083        }));
10084    }
10085
10086    #[test]
10087    fn staged_controller_does_not_add_producers_to_a_delayed_handoff() {
10088        let calibration = WorkerCalibration::new(1, 10);
10089        let pool = WorkerPool { initial: 2, maximum: 8, calibration: Some(calibration) };
10090        let recorder =
10091            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::StagedGatedWindows);
10092        recorder.handoff_sent();
10093        recorder.handoff_sent();
10094        let queue = DirectoryQueue::new_with_policy(
10095            (PathBuf::new(), 0),
10096            ScanOrder::BreadthFirst,
10097            Some(calibration),
10098            Some(recorder.clone()),
10099            pool.initial,
10100            pool.maximum,
10101            WorkerPolicyExperiment::StagedGatedWindows,
10102        );
10103        let mut timing = WalkAttribution::default();
10104        let mut claimed = Vec::new();
10105        let claim = queue.claim(&mut claimed, &mut timing).expect("root");
10106        queue.extend(
10107            (0..9).map(|index| (PathBuf::from(format!("ready-{index}")), 1, RegionId::UNASSIGNED)),
10108            &mut timing,
10109        );
10110
10111        assert_eq!(claim.release(1, 20, &mut timing), None);
10112        recorder.handoff_received();
10113        recorder.handoff_received();
10114        let diagnostics = recorder.finish();
10115        assert_eq!(
10116            diagnostics.worker_policy.windows[0].decision,
10117            WorkerPolicyDecision::HoldHandoffBacklog
10118        );
10119    }
10120
10121    #[test]
10122    fn candidate_retains_post_expansion_shadow_history() {
10123        let calibration = WorkerCalibration::new(1, 10);
10124        let pool = WorkerPool { initial: 2, maximum: 4, calibration: Some(calibration) };
10125        let recorder =
10126            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::RepeatedWindows);
10127        let queue = DirectoryQueue::new_with_policy(
10128            (PathBuf::new(), 0),
10129            ScanOrder::BreadthFirst,
10130            Some(calibration),
10131            Some(recorder.clone()),
10132            pool.initial,
10133            pool.maximum,
10134            WorkerPolicyExperiment::RepeatedWindows,
10135        );
10136        let mut timing = WalkAttribution::default();
10137        let mut claimed = Vec::new();
10138
10139        let slow = queue.claim(&mut claimed, &mut timing).expect("slow prefix");
10140        queue.extend(
10141            (0..4).map(|index| (PathBuf::from(format!("fast-{index}")), 1, RegionId::UNASSIGNED)),
10142            &mut timing,
10143        );
10144        assert_eq!(slow.release(1, 20, &mut timing), Some(4));
10145
10146        claimed.clear();
10147        let fast = queue.claim(&mut claimed, &mut timing).expect("fast suffix");
10148        queue.extend([(PathBuf::from("tail"), 2, RegionId::UNASSIGNED)].into_iter(), &mut timing);
10149        assert_eq!(fast.release(1, 1, &mut timing), None);
10150
10151        let diagnostics = recorder.finish();
10152        assert_eq!(
10153            diagnostics
10154                .worker_policy
10155                .windows
10156                .iter()
10157                .map(|window| window.decision)
10158                .collect::<Vec<_>>(),
10159            vec![WorkerPolicyDecision::ScaleUp, WorkerPolicyDecision::ObserveFast]
10160        );
10161    }
10162
10163    #[test]
10164    fn every_experimental_controller_preserves_exactness_and_shutdown() {
10165        let dir = branching_tree();
10166        let serial = ScanConfig { threads: Some(1), ..ScanConfig::default() };
10167        let (reference, _) = scan_into_index(dir.path(), &serial).expect("serial reference");
10168        let automatic = ScanConfig { threads: None, ..ScanConfig::default() };
10169
10170        for policy in [
10171            WorkerPolicyExperiment::ShippedOneShot,
10172            WorkerPolicyExperiment::RepeatedWindows,
10173            WorkerPolicyExperiment::StagedGatedWindows,
10174        ] {
10175            let (index, report, diagnostics) =
10176                scan_into_index_with_policy_diagnostics(dir.path(), &automatic, policy)
10177                    .expect("candidate scan finishes");
10178            assert!(report.is_complete(), "{policy:?}: {:?}", report.errors);
10179            assert_eq!(index_fingerprint(&reference), index_fingerprint(&index), "{policy:?}");
10180            assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
10181            assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
10182            assert_eq!(diagnostics.worker_policy.handoff_backlog_at_finish, 0);
10183            assert!(
10184                diagnostics.worker_policy.workers_spawned
10185                    <= diagnostics.worker_policy.maximum_workers
10186            );
10187        }
10188    }
10189
10190    /// A deterministic model of the automatic worker policy under *completion* order.
10191    ///
10192    /// The scaling decision is driven by chunk releases, and chunks complete in whatever
10193    /// order the filesystem and the workers produce them — not in traversal order. On a
10194    /// homogeneous tree that distinction is invisible, because every prefix looks like
10195    /// every other. On a heterogeneous one it decides the answer.
10196    ///
10197    /// These tests exist because the alternative is a stopwatch on a real tree, which
10198    /// measures one host on one day and cannot separate a policy defect from ambient
10199    /// noise. Replaying an explicit completion order through the shipped calibration
10200    /// isolates the policy exactly, and does so identically on every platform.
10201    ///
10202    /// They characterize behavior; they do not endorse a replacement. Which controller
10203    /// is *faster* is a question only the held-out Apple Silicon/APFS matrix can answer.
10204    mod completion_order {
10205        use super::{ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, WorkerCalibration};
10206
10207        /// One chunk release: entries observed and worker time spent observing them.
10208        #[derive(Clone, Copy, Debug)]
10209        struct Chunk {
10210            entries: u64,
10211            work_ns: u64,
10212        }
10213
10214        impl Chunk {
10215            /// A run of `entries` entries costing `per_entry_ns` each.
10216            const fn at(entries: u64, per_entry_ns: u64) -> Self {
10217                Self { entries, work_ns: entries.saturating_mul(per_entry_ns) }
10218            }
10219        }
10220
10221        /// What a policy concluded over one completion order.
10222        #[derive(Debug, PartialEq, Eq)]
10223        enum Outcome {
10224            /// The policy found the filesystem slow and expanded the reserve.
10225            ScaledUp { after_chunks: usize },
10226            /// The policy found the filesystem fast and held the initial pool.
10227            Held { after_chunks: usize },
10228            /// The walk ended before the policy observed enough to conclude anything.
10229            ///
10230            /// Distinct from [`Outcome::Held`] on purpose: nothing was measured, so a
10231            /// held pool here is an absence of evidence rather than a decision.
10232            Undecided,
10233        }
10234
10235        /// Mean cost per entry over a whole trace, which is what the threshold *means*.
10236        fn whole_trace_ns_per_entry(trace: &[Chunk]) -> u64 {
10237            let entries: u64 = trace.iter().map(|chunk| chunk.entries).sum();
10238            let work_ns: u64 = trace.iter().map(|chunk| chunk.work_ns).sum();
10239            assert!(entries > 0, "a trace must observe entries");
10240            work_ns / entries
10241        }
10242
10243        /// Replay a completion order through the *shipped* calibration.
10244        ///
10245        /// This drives [`WorkerCalibration::observe`] itself rather than restating its
10246        /// arithmetic, so the model cannot quietly drift from the policy it is evidence
10247        /// about. The loop mirrors `DirectoryQueue::release`: fold each chunk in, and
10248        /// stop at the first one that produces a verdict.
10249        fn shipped(window: u64, threshold_ns: u64, trace: &[Chunk]) -> Outcome {
10250            let mut calibration = WorkerCalibration::new(window, threshold_ns);
10251            for (index, chunk) in trace.iter().enumerate() {
10252                if let Some(slow) = calibration.observe(chunk.entries, chunk.work_ns) {
10253                    let after_chunks = index + 1;
10254                    return if slow {
10255                        Outcome::ScaledUp { after_chunks }
10256                    } else {
10257                        Outcome::Held { after_chunks }
10258                    };
10259                }
10260            }
10261            Outcome::Undecided
10262        }
10263
10264        /// Entries per chunk in the traces below. Four fill the 16,384-entry window.
10265        const CHUNK: u64 = 4_096;
10266        /// A shallow, cache-warm phase: metadata already resident.
10267        const FAST: Chunk = Chunk::at(CHUNK, 2_000);
10268        /// A deep, cold phase: the latency-bound regime the reserve exists to hide.
10269        const SLOW: Chunk = Chunk::at(CHUNK, 90_000);
10270
10271        #[test]
10272        fn completion_order_alone_flips_the_shipped_decision() {
10273            // The defect, stated as an experiment: hold the *tree* constant and vary
10274            // only the order its chunks complete in. Both traces contain the same four
10275            // fast and four slow chunks, so they describe the same filesystem work.
10276            let fast_phase_first = [FAST, FAST, FAST, FAST, SLOW, SLOW, SLOW, SLOW];
10277            let interleaved = [SLOW, FAST, SLOW, FAST, SLOW, FAST, SLOW, FAST];
10278
10279            let window = 4 * CHUNK;
10280            let threshold = ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY;
10281
10282            // Whole-walk truth is identical, and by the policy's own threshold both
10283            // walks are latency-bound: 46 µs per entry against a 30 µs trigger.
10284            let truth = whole_trace_ns_per_entry(&fast_phase_first);
10285            assert_eq!(truth, whole_trace_ns_per_entry(&interleaved));
10286            assert!(
10287                truth >= threshold,
10288                "both traces are slow walks by the shipped threshold: {truth} < {threshold}"
10289            );
10290
10291            // Yet the decision depends entirely on which chunks happened to finish
10292            // first. One walk hides latency; the other runs the whole slow phase on the
10293            // starting pool, having concluded from an unrepresentative prefix.
10294            assert_eq!(
10295                shipped(window, threshold, &fast_phase_first),
10296                Outcome::Held { after_chunks: 4 },
10297                "a fast prefix holds the pool for a walk that is slow overall"
10298            );
10299            assert_eq!(
10300                shipped(window, threshold, &interleaved),
10301                Outcome::ScaledUp { after_chunks: 4 },
10302                "the same tree scales up when its slow chunks land in the window"
10303            );
10304        }
10305
10306        #[test]
10307        fn a_slow_phase_after_the_window_is_never_reconsidered() {
10308            // The heterogeneous-tree case from the field report. A small fast region
10309            // fills the window, and everything after it is slow — but the calibration
10310            // is already gone, so no amount of later evidence can reopen the decision.
10311            let mut trace = vec![FAST; 4];
10312            trace.extend(std::iter::repeat_n(SLOW, 400));
10313
10314            let observed = whole_trace_ns_per_entry(&trace);
10315            assert!(
10316                observed >= 2 * ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
10317                "the walk is overwhelmingly latency-bound: {observed} ns per entry"
10318            );
10319
10320            assert_eq!(
10321                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
10322                Outcome::Held { after_chunks: 4 },
10323                "1% of the walk decided the worker policy for the other 99%"
10324            );
10325        }
10326
10327        #[test]
10328        fn slow_in_flight_work_is_censored_by_fast_completions() {
10329            // Four slow chunks have already been claimed, but their filesystem calls
10330            // remain in flight while four cache-warm chunks complete. Completion-order
10331            // calibration cannot see owed work: the fast completions close the window
10332            // and permanently hold before any slow claim returns.
10333            let completed_before_slow_returns = [FAST, FAST, FAST, FAST];
10334            assert_eq!(
10335                shipped(
10336                    4 * CHUNK,
10337                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
10338                    &completed_before_slow_returns,
10339                ),
10340                Outcome::Held { after_chunks: 4 }
10341            );
10342
10343            let mut eventual_completions = completed_before_slow_returns.to_vec();
10344            eventual_completions.extend([SLOW, SLOW, SLOW, SLOW]);
10345            assert!(
10346                whole_trace_ns_per_entry(&eventual_completions)
10347                    >= ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY
10348            );
10349            assert_eq!(
10350                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &eventual_completions,),
10351                Outcome::Held { after_chunks: 4 }
10352            );
10353        }
10354
10355        #[test]
10356        fn a_slow_prefix_can_scale_a_walk_that_is_fast_overall() {
10357            // The mirror-image error. A one-way expansion reacts correctly to the
10358            // prefix by its local threshold, but the prefix is under 1% of this walk
10359            // and the whole trace is firmly in the fast regime. A repeated trigger
10360            // alone cannot undo an expansion; staged growth limits exposure but does
10361            // not make reversible parking unnecessary.
10362            let mut trace = vec![SLOW; 4];
10363            trace.extend(std::iter::repeat_n(FAST, 400));
10364            let observed = whole_trace_ns_per_entry(&trace);
10365            assert!(observed < ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY);
10366            assert_eq!(
10367                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
10368                Outcome::ScaledUp { after_chunks: 4 }
10369            );
10370            assert_eq!(
10371                sliding(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
10372                Outcome::ScaledUp { after_chunks: 4 }
10373            );
10374        }
10375
10376        #[test]
10377        fn a_walk_shorter_than_the_window_decides_nothing() {
10378            // Fails closed rather than reporting a held pool: a walk this short never
10379            // observed enough to have an opinion, and an artifact that recorded `Held`
10380            // would claim a measurement that was never taken.
10381            let trace = [FAST, SLOW];
10382            assert_eq!(
10383                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
10384                Outcome::Undecided
10385            );
10386        }
10387
10388        /// A screening-only candidate: a window that slides instead of closing once.
10389        ///
10390        /// Present as *evidence about a design*, not as a proposed change. It keeps the
10391        /// shipped trigger and pool bounds and alters only when the question is asked,
10392        /// which is the narrowest edit that could address the order sensitivity above.
10393        /// Whether it is faster on a real tree is unmeasured here and unmeasurable in a
10394        /// virtualized non-APFS environment; selecting it would need the held-out Apple
10395        /// Silicon matrix that this workstream has not yet been able to run.
10396        struct SlidingWindow {
10397            window_entries: u64,
10398            threshold_ns: u64,
10399            recent: std::collections::VecDeque<Chunk>,
10400            entries: u64,
10401            work_ns: u64,
10402        }
10403
10404        impl SlidingWindow {
10405            fn new(window_entries: u64, threshold_ns: u64) -> Self {
10406                Self {
10407                    window_entries,
10408                    threshold_ns,
10409                    recent: std::collections::VecDeque::new(),
10410                    entries: 0,
10411                    work_ns: 0,
10412                }
10413            }
10414
10415            /// Fold in a chunk and re-ask the question over the trailing window.
10416            fn observe(&mut self, chunk: Chunk) -> Option<bool> {
10417                self.recent.push_back(chunk);
10418                self.entries = self.entries.saturating_add(chunk.entries);
10419                self.work_ns = self.work_ns.saturating_add(chunk.work_ns);
10420
10421                // Drop from the front while the window stays full without the oldest
10422                // chunk, so the answer describes recent work rather than the whole walk.
10423                while let Some(oldest) = self.recent.front().copied() {
10424                    if self.entries - oldest.entries < self.window_entries {
10425                        break;
10426                    }
10427                    self.recent.pop_front();
10428                    self.entries -= oldest.entries;
10429                    self.work_ns -= oldest.work_ns;
10430                }
10431
10432                (self.entries >= self.window_entries)
10433                    .then(|| self.work_ns / self.entries >= self.threshold_ns)
10434            }
10435        }
10436
10437        /// Replay a completion order through the candidate, stopping at its first
10438        /// scale-up. The shipped pool only grows, so a later verdict cannot undo one.
10439        fn sliding(window: u64, threshold_ns: u64, trace: &[Chunk]) -> Outcome {
10440            let mut policy = SlidingWindow::new(window, threshold_ns);
10441            let mut decided = None;
10442            for (index, chunk) in trace.iter().enumerate() {
10443                if let Some(slow) = policy.observe(*chunk) {
10444                    let after_chunks = index + 1;
10445                    if slow {
10446                        return Outcome::ScaledUp { after_chunks };
10447                    }
10448                    decided.get_or_insert(Outcome::Held { after_chunks });
10449                }
10450            }
10451            decided.unwrap_or(Outcome::Undecided)
10452        }
10453
10454        #[test]
10455        fn screening_a_sliding_window_against_the_order_sensitivity() {
10456            let window = 4 * CHUNK;
10457            let threshold = ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY;
10458
10459            // The pair that splits the shipped policy reaches one answer here, and it
10460            // is the answer the whole-trace mean supports in both orders.
10461            let fast_phase_first = [FAST, FAST, FAST, FAST, SLOW, SLOW, SLOW, SLOW];
10462            let interleaved = [SLOW, FAST, SLOW, FAST, SLOW, FAST, SLOW, FAST];
10463            // Both reach the same verdict; they differ only in how long the fast prefix
10464            // delays it, which is the behavior a trailing window is supposed to have.
10465            assert_eq!(
10466                sliding(window, threshold, &fast_phase_first),
10467                Outcome::ScaledUp { after_chunks: 6 }
10468            );
10469            assert_eq!(
10470                sliding(window, threshold, &interleaved),
10471                Outcome::ScaledUp { after_chunks: 4 }
10472            );
10473
10474            // And the late slow phase is reached rather than missed: two slow chunks
10475            // after the window closes are enough to pull the trailing mean over.
10476            let mut late = vec![FAST; 4];
10477            late.extend(std::iter::repeat_n(SLOW, 400));
10478            assert_eq!(sliding(window, threshold, &late), Outcome::ScaledUp { after_chunks: 6 });
10479
10480            // A genuinely fast tree must still hold the pool: the candidate has to keep
10481            // the property the shipped policy gets right, or it is not a candidate.
10482            let uniformly_fast = vec![FAST; 40];
10483            assert_eq!(
10484                sliding(window, threshold, &uniformly_fast),
10485                Outcome::Held { after_chunks: 4 }
10486            );
10487
10488            // A short walk still decides nothing, for the same reason as above.
10489            assert_eq!(sliding(window, threshold, &[FAST, SLOW]), Outcome::Undecided);
10490        }
10491    }
10492
10493    #[test]
10494    fn thread_count_does_not_change_the_cache_scope() {
10495        // Threads are an operational choice. If they leaked into the scope, changing
10496        // the pool size would invalidate every snapshot on disk.
10497        let serial = ScanConfig { threads: Some(1), ..ScanConfig::default() };
10498        let parallel = ScanConfig { threads: Some(8), ..ScanConfig::default() };
10499        assert_eq!(serial.scope(), parallel.scope());
10500    }
10501
10502    #[test]
10503    fn scan_populates_an_index_end_to_end() {
10504        let dir = sample_tree();
10505        let (index, report) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10506
10507        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10508        let total = index.total();
10509        assert_eq!(total.files, 3);
10510        assert_eq!(total.dirs, 2);
10511        assert_eq!(total.bytes, 5 + 12 + 9);
10512        assert_eq!(total.by_ext[".rs"].files, 2);
10513        assert_eq!(total.by_ext[".txt"].files, 1);
10514
10515        let src = index.rollup(Path::new("src")).expect("src");
10516        assert_eq!(src.files, 2);
10517        assert_eq!(src.dirs, 1);
10518    }
10519
10520    #[test]
10521    fn cold_scan_routes_control_sources_through_both_walkers() {
10522        let dir = tempfile::tempdir().expect("tempdir");
10523        write_file(&dir.path().join(".gitignore"), b"*.log\n");
10524        write_file(&dir.path().join("debug.log"), b"ignored");
10525        write_file(&dir.path().join("keep.rs"), b"visible");
10526
10527        for threads in [1, 4] {
10528            let config =
10529                ScanConfig { read_controls: true, threads: Some(threads), ..ScanConfig::default() };
10530            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
10531
10532            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10533            assert!(
10534                index
10535                    .controls()
10536                    .expect("control state observed")
10537                    .source_is(Path::new(".gitignore"), b"*.log\n")
10538            );
10539            assert_eq!(
10540                index.is_ignored(Path::new("debug.log")).expect("control state observed"),
10541                Some(true)
10542            );
10543            assert_eq!(
10544                index.is_ignored(Path::new("keep.rs")).expect("control state observed"),
10545                Some(false)
10546            );
10547            let partitions = index.partition_total().expect("control state observed");
10548            assert_eq!(partitions.all.files, 3);
10549            assert_eq!(partitions.unignored.files, 2);
10550        }
10551    }
10552
10553    #[cfg(unix)]
10554    #[test]
10555    fn raced_fifo_control_source_is_rejected_without_blocking() {
10556        let dir = tempfile::tempdir().expect("tempdir");
10557        let control = dir.path().join(".gitignore");
10558        let status = match std::process::Command::new("mkfifo").arg(&control).status() {
10559            Ok(status) => status,
10560            Err(error) if error.kind() == std::io::ErrorKind::NotFound => return,
10561            Err(error) => panic!("create fifo: {error}"),
10562        };
10563        assert!(status.success(), "mkfifo exited with {status}");
10564
10565        let root = dir.path().to_path_buf();
10566        let (sender, receiver) = std::sync::mpsc::channel();
10567        std::thread::spawn(move || {
10568            let result = read_control_op_unconditional(
10569                &root,
10570                Path::new(".gitignore"),
10571                EntryKind::File,
10572                Some(crate::control::DEFAULT_CONTROL_BUDGET),
10573            );
10574            sender.send(result).ok();
10575        });
10576        let result = receiver
10577            .recv_timeout(std::time::Duration::from_secs(1))
10578            .expect("a raced FIFO must not block the scan worker")
10579            .expect("the non-regular replacement is a normal control removal");
10580
10581        assert!(matches!(result, Some(Op::ControlRemove { .. })));
10582    }
10583
10584    #[test]
10585    fn hidden_admission_keeps_exact_allowlist_and_control_signals_only() {
10586        let dir = tempfile::tempdir().expect("tempdir");
10587        write_file(&dir.path().join(".gitignore"), b"*.log\n");
10588        write_file(&dir.path().join("debug.log"), b"ignored");
10589        write_file(&dir.path().join(".secret/token"), b"hidden");
10590        write_file(&dir.path().join(".github/workflows/check.yml"), b"visible");
10591        let hidden = std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([".github"]));
10592
10593        for threads in [1, 4] {
10594            let config = ScanConfig {
10595                hidden: Some(std::sync::Arc::clone(&hidden)),
10596                threads: Some(threads),
10597                read_controls: true,
10598                ..ScanConfig::default()
10599            };
10600            let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
10601
10602            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10603            assert!(index.lookup(Path::new(".gitignore")).is_none());
10604            assert!(index.lookup(Path::new(".secret")).is_none());
10605            assert!(index.lookup(Path::new(".secret/token")).is_none());
10606            assert!(index.lookup(Path::new(".github/workflows/check.yml")).is_some());
10607            assert!(
10608                index
10609                    .controls()
10610                    .expect("control state observed")
10611                    .source_is(Path::new(".gitignore"), b"*.log\n")
10612            );
10613            assert_eq!(
10614                index.is_ignored(Path::new("debug.log")).expect("control state observed"),
10615                Some(true)
10616            );
10617
10618            fs::remove_file(dir.path().join(".gitignore")).expect("remove control");
10619            if threads > 1 {
10620                fs::create_dir(dir.path().join(".gitignore")).expect("replace with directory");
10621            }
10622            let reconciled = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
10623            assert!(reconciled.is_complete());
10624            assert!(index.controls().expect("control state observed").is_empty());
10625            if threads > 1 {
10626                fs::remove_dir(dir.path().join(".gitignore")).expect("remove directory");
10627            }
10628            write_file(&dir.path().join(".gitignore"), b"*.log\n");
10629        }
10630    }
10631
10632    #[test]
10633    fn excluded_ignored_directory_is_not_enumerated_and_rule_edits_reconcile_it() {
10634        let root = tempfile::tempdir().expect("root");
10635        write_file(&root.path().join(".gitignore"), b"target/\n");
10636        write_file(&root.path().join("target/deep/file.rs"), b"code");
10637        write_file(&root.path().join("keep.rs"), b"kept");
10638        let config = ScanConfig {
10639            population: crate::query::IgnoredEntries::Exclude,
10640            threads: Some(4),
10641            ..ScanConfig::default()
10642        };
10643
10644        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
10645        assert!(cold.is_complete(), "{:?}", cold.errors);
10646        assert_eq!(cold.dirs_read, 1, "ignored target was not opened");
10647        assert!(index.lookup(Path::new("target")).is_none());
10648        assert!(index.lookup(Path::new("keep.rs")).is_some());
10649
10650        write_file(&root.path().join(".gitignore"), b"");
10651        let exposed = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile exposure");
10652        assert!(exposed.is_complete(), "{:?}", exposed.scan.errors);
10653        assert!(index.lookup(Path::new("target/deep/file.rs")).is_some());
10654        assert!(exposed.scan.dirs_read >= 3);
10655
10656        write_file(&root.path().join(".gitignore"), b"target/\n");
10657        let excluded = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile exclusion");
10658        assert!(excluded.is_complete(), "{:?}", excluded.scan.errors);
10659        assert!(index.lookup(Path::new("target")).is_none());
10660        assert_eq!(excluded.scan.dirs_read, 1, "ignored target was not reopened");
10661    }
10662
10663    #[test]
10664    fn excluded_subtree_refresh_keeps_ignored_file_and_directory_out_of_scope() {
10665        let root = tempfile::tempdir().expect("root");
10666        write_file(&root.path().join(".gitignore"), b"target/\n*.log\n");
10667        write_file(&root.path().join("target/deep/file.rs"), b"code");
10668        write_file(&root.path().join("debug.log"), b"ignored");
10669        let config = ScanConfig {
10670            population: crate::query::IgnoredEntries::Exclude,
10671            ..ScanConfig::default()
10672        };
10673        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
10674        assert!(cold.is_complete());
10675        assert!(index.lookup(Path::new("target")).is_none());
10676        assert!(index.lookup(Path::new("debug.log")).is_none());
10677
10678        for path in ["target", "debug.log"] {
10679            let refresh = reconcile_subtree(&mut index, Path::new(path), &config, &mut |_| {})
10680                .expect("subtree refresh");
10681            assert!(refresh.is_complete(), "{path}: {:?}", refresh.scan.errors);
10682            assert!(index.lookup(Path::new(path)).is_none(), "{path} is outside scope");
10683        }
10684        assert_eq!(index.total().dirs, 0);
10685    }
10686
10687    #[test]
10688    fn handled_subtree_refresh_recovers_pruned_ancestry_after_control_edit() {
10689        let root = tempfile::tempdir().expect("root");
10690        write_file(&root.path().join(".gitignore"), b"target/\n");
10691        write_file(&root.path().join("target/deep/file.rs"), b"code");
10692        let config = ScanConfig {
10693            population: crate::query::IgnoredEntries::Exclude,
10694            ..ScanConfig::default()
10695        };
10696        let (index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
10697        assert!(cold.is_complete());
10698        let handle = IndexHandle::new(index);
10699
10700        let hidden = reconcile_subtree_handle(
10701            &handle,
10702            Path::new("target/deep/file.rs"),
10703            &config,
10704            &mut |_| {},
10705        )
10706        .expect("refresh pruned descendant");
10707        assert!(hidden.is_complete());
10708        assert!(
10709            !handle
10710                .read_with(|index| index.lookup(Path::new("target")).is_some())
10711                .expect("read after hidden refresh")
10712        );
10713
10714        write_file(&root.path().join(".gitignore"), b"");
10715        let exposed = reconcile_subtree_handle(&handle, Path::new("target"), &config, &mut |_| {})
10716            .expect("refresh changed control");
10717        assert!(exposed.is_complete());
10718        assert!(
10719            handle
10720                .read_with(|index| index.lookup(Path::new("target/deep/file.rs")).is_some())
10721                .expect("read after control recovery")
10722        );
10723
10724        write_file(&root.path().join(".gitignore"), b"target/\n");
10725        let hidden_again =
10726            reconcile_subtree_handle(&handle, Path::new("target"), &config, &mut |_| {})
10727                .expect("refresh restored control");
10728        assert!(hidden_again.is_complete());
10729        assert!(
10730            !handle
10731                .read_with(|index| index.lookup(Path::new("target")).is_some())
10732                .expect("read after restored control")
10733        );
10734    }
10735
10736    #[cfg(unix)]
10737    #[test]
10738    fn targeted_refresh_keeps_unknown_population_below_unreadable_ancestor_control() {
10739        use std::os::unix::fs::PermissionsExt;
10740
10741        if !crate::test_support::require_permission_bits() {
10742            return;
10743        }
10744        let root = tempfile::tempdir().expect("root");
10745        let control = root.path().join(".gitignore");
10746        write_file(&control, b"*.log\n");
10747        write_file(&root.path().join("a/keep.rs"), b"unknown membership");
10748        let config =
10749            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
10750        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("unreadable");
10751        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
10752        assert!(!cold.is_complete());
10753        assert!(index.lookup(Path::new("a/keep.rs")).is_some());
10754        assert_eq!(index.ignored_classification(Path::new("a/keep.rs")), None);
10755
10756        let refreshed = reconcile_subtree(&mut index, Path::new("a/keep.rs"), &config, &mut |_| {});
10757        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore control");
10758        let refreshed = refreshed.expect("targeted refresh");
10759        assert!(!refreshed.is_complete(), "unreadable governing control was not visited");
10760        assert!(index.lookup(Path::new("a/keep.rs")).is_some(), "unknown must stay retained");
10761        assert_eq!(index.ignored_classification(Path::new("a/keep.rs")), None);
10762    }
10763
10764    #[test]
10765    fn exclusion_honors_nested_negation_and_refused_rule_changes() {
10766        let root = tempfile::tempdir().expect("root");
10767        write_file(&root.path().join("nested/.gitignore"), b"*.log\n!keep.log\n");
10768        write_file(&root.path().join("nested/keep.log"), b"negated");
10769        write_file(&root.path().join("nested/drop.log"), b"ignored");
10770        let config = ScanConfig {
10771            population: crate::query::IgnoredEntries::Exclude,
10772            control_limits: crate::control::ControlLimits {
10773                line_limit: Some(20),
10774                ..crate::control::ControlLimits::default()
10775            },
10776            ..ScanConfig::default()
10777        };
10778        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold");
10779        assert!(cold.is_complete(), "{:?}", cold.errors);
10780        assert!(index.lookup(Path::new("nested/keep.log")).is_some());
10781        assert!(index.lookup(Path::new("nested/drop.log")).is_none());
10782
10783        write_file(&root.path().join("nested/.gitignore"), b"this-line-is-over-the-limit\n");
10784        let changed =
10785            reconcile(&mut index, &config, &mut |_| {}).expect("reconcile refused source");
10786        assert!(changed.is_complete(), "{:?}", changed.scan.errors);
10787        assert!(index.lookup(Path::new("nested/drop.log")).is_some(), "unknown cannot be pruned");
10788        assert_eq!(index.ignored_classification(Path::new("nested/drop.log")), None);
10789    }
10790
10791    #[test]
10792    fn only_population_skips_nonignored_content_candidates_and_refusals_are_unknown() {
10793        let root = tempfile::tempdir().expect("root");
10794        write_file(&root.path().join(".gitignore"), b"*.log\n");
10795        write_file(&root.path().join("keep.rs"), b"code");
10796        write_file(&root.path().join("debug.log"), b"ignored");
10797        write_file(&root.path().join("nested/keep.rs"), b"kept");
10798        write_file(&root.path().join("nested/debug.log"), b"ignored below nonignored dir");
10799        let only =
10800            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
10801        let (index, report) = scan_into_index(root.path(), &only).expect("scan");
10802        assert!(report.is_complete());
10803        assert!(index.lookup(Path::new("keep.rs")).is_none());
10804        assert!(index.lookup(Path::new("nested")).is_some());
10805        assert!(index.lookup(Path::new("nested/keep.rs")).is_none());
10806        assert!(index.lookup(Path::new("nested/debug.log")).is_some());
10807        let candidates = index.analysis_candidates(crate::content::AnalysisSet::NONE.with_lines());
10808        assert_eq!(candidates.len(), 2);
10809        assert!(
10810            candidates.iter().any(|candidate| candidate.relative_path == Path::new("debug.log"))
10811        );
10812        assert!(
10813            candidates
10814                .iter()
10815                .any(|candidate| candidate.relative_path == Path::new("nested/debug.log"))
10816        );
10817
10818        let refused = ScanConfig {
10819            population: crate::query::IgnoredEntries::Exclude,
10820            control_limits: crate::control::ControlLimits {
10821                line_limit: Some(1),
10822                ..crate::control::ControlLimits::default()
10823            },
10824            ..ScanConfig::default()
10825        };
10826        let (index, report) = scan_into_index(root.path(), &refused).expect("refused scan");
10827        assert!(report.is_complete());
10828        assert!(index.lookup(Path::new("debug.log")).is_some(), "unknown is not pruned");
10829        assert_eq!(index.ignored_classification(Path::new("debug.log")), None);
10830        assert!(
10831            index.analysis_candidates(crate::content::AnalysisSet::NONE.with_lines()).is_empty()
10832        );
10833    }
10834
10835    #[test]
10836    fn only_population_content_is_complete_with_retained_control_file() {
10837        let root = tempfile::tempdir().expect("root");
10838        write_file(&root.path().join(".gitignore"), b"vendor/\n");
10839        write_file(&root.path().join("main.rs"), b"fn main() {}\n");
10840        write_file(&root.path().join("vendor/lib.rs"), b"fn lib() {}\n");
10841        let config =
10842            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
10843        let (mut index, scan) = scan_into_index(root.path(), &config).expect("scan");
10844        assert!(scan.is_complete());
10845        let profile = crate::content::AnalysisSet::NONE.with_lines().with_code();
10846        let analyzed = crate::content::analyze_index(
10847            &mut index,
10848            crate::content::AnalysisRequest {
10849                profile,
10850                ..crate::content::AnalysisRequest::default()
10851            },
10852        );
10853        assert!(analyzed.is_complete(), "{analyzed:?}");
10854        assert_eq!(analyzed.candidates, 1);
10855        assert!(!index.content_has_pending(profile));
10856    }
10857
10858    #[test]
10859    fn only_population_reconciles_rule_changes_without_losing_traversal() {
10860        let root = tempfile::tempdir().expect("root");
10861        write_file(&root.path().join("nested/.gitignore"), b"*.log\n");
10862        write_file(&root.path().join("nested/first.log"), b"first");
10863        write_file(&root.path().join("nested/second.txt"), b"second");
10864        let config =
10865            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
10866        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold");
10867        assert!(cold.is_complete());
10868        assert!(index.lookup(Path::new("nested")).is_some());
10869        assert!(index.lookup(Path::new("nested/first.log")).is_some());
10870        assert!(index.lookup(Path::new("nested/second.txt")).is_none());
10871
10872        write_file(&root.path().join("nested/.gitignore"), b"*.txt\n");
10873        let changed = reconcile(&mut index, &config, &mut |_| {}).expect("rule change");
10874        assert!(changed.is_complete(), "{:?}", changed.scan.errors);
10875        assert!(index.lookup(Path::new("nested/first.log")).is_none());
10876        assert!(index.lookup(Path::new("nested/second.txt")).is_some());
10877    }
10878
10879    #[cfg(unix)]
10880    #[test]
10881    fn excluded_special_objects_never_enter_cold_or_reconciled_facts() {
10882        use std::os::unix::net::UnixListener;
10883
10884        let dir = tempfile::tempdir().expect("tempdir");
10885        let socket_path = dir.path().join("service.sock");
10886        let _listener = UnixListener::bind(&socket_path).expect("bind socket");
10887        write_file(&dir.path().join("replacement"), b"ordinary");
10888        crate::test_support::settle_allocations(dir.path());
10889        let (kept, kept_report) =
10890            scan_into_index(dir.path(), &ScanConfig::default()).expect("default scan");
10891        assert!(kept_report.is_complete());
10892        assert_eq!(kept.kind(Path::new("service.sock")), Some(EntryKind::Other));
10893
10894        let serial_config =
10895            ScanConfig { exclude_special: true, threads: Some(1), ..ScanConfig::default() };
10896        let parallel_config =
10897            ScanConfig { exclude_special: true, threads: Some(4), ..ScanConfig::default() };
10898        let (mut serial, serial_report) =
10899            scan_into_index(dir.path(), &serial_config).expect("serial scan");
10900        let (mut parallel, parallel_report) =
10901            scan_into_index(dir.path(), &parallel_config).expect("parallel scan");
10902
10903        for (index, report) in [(&serial, &serial_report), (&parallel, &parallel_report)] {
10904            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10905            assert!(index.lookup(Path::new("service.sock")).is_none());
10906            assert!(index.lookup(Path::new("replacement")).is_some());
10907        }
10908
10909        fs::remove_file(dir.path().join("replacement")).expect("remove file");
10910        let _replacement =
10911            UnixListener::bind(dir.path().join("replacement")).expect("bind replacement socket");
10912        let serial_reconciled =
10913            reconcile(&mut serial, &serial_config, &mut |_| {}).expect("serial reconcile");
10914        let parallel_reconciled =
10915            reconcile(&mut parallel, &parallel_config, &mut |_| {}).expect("parallel reconcile");
10916
10917        assert!(serial_reconciled.is_complete());
10918        assert!(parallel_reconciled.is_complete());
10919        assert!(serial.lookup(Path::new("replacement")).is_none());
10920        assert!(parallel.lookup(Path::new("replacement")).is_none());
10921        assert_eq!(index_fingerprint(&serial), index_fingerprint(&parallel));
10922    }
10923
10924    #[test]
10925    fn control_sources_respect_a_single_operation_batch_bound() {
10926        let dir = tempfile::tempdir().expect("tempdir");
10927        write_file(&dir.path().join(".gitignore"), b"*.log\n");
10928        write_file(&dir.path().join("debug.log"), b"ignored");
10929
10930        for threads in [1, 4] {
10931            let config = ScanConfig {
10932                read_controls: true,
10933                threads: Some(threads),
10934                batch_size: 1,
10935                ..ScanConfig::default()
10936            };
10937            let mut largest = 0;
10938            let report = scan(dir.path(), &config, &mut |observation| {
10939                largest = largest.max(observation.len());
10940            })
10941            .expect("scan");
10942
10943            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10944            assert_eq!(largest, 1);
10945        }
10946    }
10947
10948    #[test]
10949    fn cold_scan_matches_the_metabrowser_nested_control_fixture() {
10950        let dir = tempfile::tempdir().expect("tempdir");
10951        write_file(&dir.path().join(".gitignore"), b"node_modules/\n*.pyc\n");
10952        write_file(&dir.path().join("src/app.py"), b"x");
10953        write_file(&dir.path().join("src/thing.pyc"), b"x");
10954        write_file(&dir.path().join("src/generated/.gitignore"), b"*.gen\n");
10955        write_file(&dir.path().join("src/generated/out.gen"), b"x");
10956        write_file(&dir.path().join("node_modules/.gitignore"), b"!keep-me.py\n");
10957        write_file(&dir.path().join("node_modules/keep-me.py"), b"x");
10958
10959        let (index, report) = scan_into_index(
10960            dir.path(),
10961            &ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() },
10962        )
10963        .expect("scan fixture");
10964
10965        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10966        assert_eq!(
10967            index.is_ignored(Path::new("src/app.py")).expect("control state observed"),
10968            Some(false)
10969        );
10970        assert_eq!(
10971            index.is_ignored(Path::new("src/thing.pyc")).expect("control state observed"),
10972            Some(true)
10973        );
10974        assert_eq!(
10975            index.is_ignored(Path::new("src/generated")).expect("control state observed"),
10976            Some(false)
10977        );
10978        assert_eq!(
10979            index.is_ignored(Path::new("src/generated/out.gen")).expect("control state observed"),
10980            Some(true)
10981        );
10982        assert_eq!(
10983            index.is_ignored(Path::new("node_modules")).expect("control state observed"),
10984            Some(true)
10985        );
10986        assert_eq!(
10987            index.is_ignored(Path::new("node_modules/keep-me.py")).expect("control state observed"),
10988            Some(true)
10989        );
10990    }
10991
10992    #[test]
10993    fn reconciliation_observes_same_metadata_control_edits_and_last_deletion() {
10994        let dir = tempfile::tempdir().expect("tempdir");
10995        write_file(&dir.path().join(".gitignore"), b"*.log\n");
10996        write_file(&dir.path().join("debug.log"), b"ignored");
10997        let config = ScanConfig { read_controls: true, threads: Some(1), ..ScanConfig::default() };
10998        let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
10999        assert!(report.is_complete());
11000        assert_eq!(
11001            index.is_ignored(Path::new("debug.log")).expect("control state observed"),
11002            Some(true)
11003        );
11004
11005        // Same-length content proves control identity is not inferred from stat-tier
11006        // metadata, which can remain unchanged on coarse filesystems.
11007        write_file(&dir.path().join(".gitignore"), b"*.tmp\n");
11008        let edited = reconcile(&mut index, &config, &mut |_| {}).expect("edit reconcile");
11009        assert!(edited.is_complete());
11010        assert_eq!(edited.apply.controls, 1);
11011        assert_eq!(edited.apply.reclassified, 1);
11012        assert!(
11013            index
11014                .controls()
11015                .expect("control state observed")
11016                .source_is(Path::new(".gitignore"), b"*.tmp\n")
11017        );
11018        assert_eq!(
11019            index.is_ignored(Path::new("debug.log")).expect("control state observed"),
11020            Some(false)
11021        );
11022
11023        fs::remove_file(dir.path().join(".gitignore")).expect("remove control");
11024        let removed = reconcile(&mut index, &config, &mut |_| {}).expect("remove reconcile");
11025        assert!(removed.is_complete());
11026        assert_eq!(removed.apply.controls, 1);
11027        assert!(index.controls().expect("control state observed").is_empty());
11028        let partitions = index.partition_total().expect("control state observed");
11029        assert_eq!(partitions.all, partitions.unignored);
11030    }
11031
11032    #[cfg(unix)]
11033    #[test]
11034    fn directory_entry_metadata_does_not_follow_symlinks() {
11035        use std::os::unix::fs::symlink;
11036
11037        let root = tempfile::tempdir().expect("root");
11038        let outside = tempfile::tempdir().expect("outside");
11039        write_file(&outside.path().join("must-not-be-scanned.txt"), b"outside");
11040        symlink(outside.path(), root.path().join("link")).expect("symlink");
11041
11042        let (index, report) = scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
11043
11044        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
11045        assert_eq!(index.kind(Path::new("link")), Some(EntryKind::Symlink));
11046        assert!(index.lookup(Path::new("link/must-not-be-scanned.txt")).is_none());
11047        assert_eq!(index.total().files, 0);
11048    }
11049
11050    #[test]
11051    fn cold_scan_establishes_a_baseline_without_change_history() {
11052        let dir = sample_tree();
11053        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11054
11055        assert_eq!(index.clock(), crate::Clock::ZERO);
11056        assert!(index.since(crate::Clock::ZERO).commits.is_empty());
11057    }
11058
11059    #[test]
11060    fn max_depth_stops_descent() {
11061        let dir = sample_tree();
11062        let config = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
11063        let (index, _) = scan_into_index(dir.path(), &config).expect("scan");
11064
11065        assert!(index.lookup(Path::new("src")).is_some());
11066        assert!(index.lookup(Path::new("src/main.rs")).is_none());
11067    }
11068
11069    #[test]
11070    fn zero_max_depth_keeps_only_the_index_root() {
11071        let dir = sample_tree();
11072        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
11073        let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
11074
11075        assert!(index.is_empty());
11076        assert_eq!(report.entries, 0);
11077        assert_eq!(report.dirs_read, 0);
11078    }
11079
11080    #[test]
11081    fn direct_scan_records_the_canonical_root() {
11082        let dir = sample_tree();
11083        let aliased = dir.path().join(".");
11084        let (index, _) = scan_into_index(&aliased, &ScanConfig::default()).expect("scan");
11085
11086        assert_eq!(index.root_path(), dir.path().canonicalize().expect("canonical root"));
11087    }
11088
11089    #[test]
11090    fn unsupported_symlink_following_is_rejected_on_cold_and_warm_paths() {
11091        let dir = sample_tree();
11092        let unsupported = ScanConfig { follow_symlinks: true, ..ScanConfig::default() };
11093        assert!(matches!(
11094            scan_into_index(dir.path(), &unsupported),
11095            Err(Error::UnsupportedScanConfig(_))
11096        ));
11097
11098        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11099        assert!(matches!(
11100            revalidate(&index, &unsupported, &mut |_| {}),
11101            Err(Error::UnsupportedScanConfig(_))
11102        ));
11103    }
11104
11105    #[test]
11106    fn revalidation_uses_the_same_depth_boundary_as_cold_scan() {
11107        let dir = sample_tree();
11108        let config = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
11109        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
11110        write_file(&dir.path().join("src/added-after-scan.txt"), b"new");
11111
11112        let mut observations = Vec::new();
11113        revalidate(&index, &config, &mut |observation| observations.push(observation))
11114            .expect("revalidate");
11115        for observation in &observations {
11116            index.apply_ok(observation);
11117        }
11118
11119        assert!(index.lookup(Path::new("src/added-after-scan.txt")).is_none());
11120    }
11121
11122    #[test]
11123    fn zero_depth_revalidation_prunes_cached_root_children() {
11124        let dir = tempfile::tempdir().expect("tempdir");
11125        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
11126        let mut index = Index::new_with_scope(dir.path(), config.scope());
11127        index.apply_baseline_ok(&Observation::new(vec![Op::Upsert {
11128            path: PathBuf::from("stale.txt"),
11129            kind: EntryKind::File,
11130            attrs: Attrs::default(),
11131        }]));
11132
11133        let mut observations = Vec::new();
11134        let report = revalidate(&index, &config, &mut |observation| {
11135            observations.push(observation);
11136        })
11137        .expect("revalidate");
11138        for observation in &observations {
11139            index.apply_ok(observation);
11140        }
11141
11142        assert!(index.is_empty());
11143        assert_eq!(report.dirs_read, 0);
11144    }
11145
11146    #[test]
11147    fn zero_depth_applying_reconciliation_prunes_cached_root_children() {
11148        let dir = tempfile::tempdir().expect("tempdir");
11149        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
11150        let mut index = Index::new_with_scope(dir.path(), config.scope());
11151        index.apply_baseline_ok(&Observation::new(vec![Op::Upsert {
11152            path: PathBuf::from("stale.txt"),
11153            kind: EntryKind::File,
11154            attrs: Attrs::default(),
11155        }]));
11156
11157        let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
11158
11159        assert!(index.is_empty());
11160        assert_eq!(report.scan.dirs_read, 0);
11161    }
11162
11163    #[test]
11164    fn filesystem_boundary_is_part_of_the_shared_descent_policy() {
11165        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
11166        let attrs = Attrs { dev: 22, ..Attrs::default() };
11167        assert!(!should_descend(EntryKind::Dir, attrs, 0, 11, &config));
11168        assert!(should_descend(EntryKind::Dir, Attrs { dev: 11, ..attrs }, 0, 11, &config,));
11169    }
11170
11171    /// A cold scan's index records its own pass start, the stamp a snapshot of it writes:
11172    /// never earlier than an instant taken before the scan, so it is not a stale or zero
11173    /// stamp, and never later than one taken after it. The builder constructs the index,
11174    /// and so takes the stamp, before the walk begins.
11175    #[test]
11176    fn a_cold_scan_stamps_its_own_pass_start() {
11177        let nanos = || {
11178            i64::try_from(
11179                std::time::SystemTime::now()
11180                    .duration_since(std::time::UNIX_EPOCH)
11181                    .expect("after the epoch")
11182                    .as_nanos(),
11183            )
11184            .expect("nanoseconds")
11185        };
11186        let dir = sample_tree();
11187        let before = nanos();
11188        let (index, report) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11189        let after = nanos();
11190        assert!(report.is_complete() && report.entries > 0, "{report:?}");
11191        let stamp = index.writing_pass_started_at_ns();
11192        assert!(before <= stamp && stamp <= after, "{before} <= {stamp} <= {after}");
11193    }
11194
11195    #[test]
11196    fn scanning_a_file_is_an_error_not_a_panic() {
11197        let dir = sample_tree();
11198        let err = scan_into_index(&dir.path().join("a.txt"), &ScanConfig::default());
11199        assert!(err.is_err());
11200    }
11201
11202    #[test]
11203    fn deltas_arrive_in_batches_of_the_configured_size() {
11204        let dir = tempfile::tempdir().expect("tempdir");
11205        for i in 0..25 {
11206            write_file(&dir.path().join(format!("f{i}.txt")), b"x");
11207        }
11208        let config = ScanConfig { batch_size: 10, ..ScanConfig::default() };
11209        let mut sizes = Vec::new();
11210        scan(dir.path(), &config, &mut |d| sizes.push(d.len())).expect("scan");
11211
11212        assert!(sizes.len() >= 3, "expected several batches, got {sizes:?}");
11213        assert!(sizes.iter().all(|&n| n <= 10));
11214        assert_eq!(sizes.iter().sum::<usize>(), 25);
11215    }
11216
11217    #[test]
11218    fn invalid_batch_sizes_are_rejected_before_allocation() {
11219        let zero = ScanConfig { batch_size: 0, ..ScanConfig::default() };
11220        let unbounded = ScanConfig { batch_size: usize::MAX, ..ScanConfig::default() };
11221
11222        assert!(matches!(zero.validate(), Err(Error::UnsupportedScanConfig(_))));
11223        assert!(matches!(unbounded.validate(), Err(Error::UnsupportedScanConfig(_))));
11224    }
11225
11226    #[test]
11227    fn reconciliation_scope_budget_publishes_before_returning_retry() {
11228        let directory = tempfile::tempdir().expect("root");
11229        let config = ScanConfig::default();
11230        let (index, _) = scan_into_index(directory.path(), &config).expect("scan root-only tree");
11231        let handle = IndexHandle::new(index);
11232        let mut started = false;
11233        let mut published_partial = false;
11234        let report = reconcile_handle(&handle, &config, &mut |commit| {
11235            if !started {
11236                started = true;
11237                // No entries are added: distinct absent children must not grow history
11238                // for the paused root pass without bound.
11239                for child in ["missing-a", "missing-b", "missing-c"] {
11240                    let nested = reconcile_subtree_handle(&handle, Path::new(child), &config, &mut |_| {}).expect("newer absent scope");
11241                    assert!(nested.is_complete());
11242                }
11243            }
11244            if commit.state.iter().any(|state| matches!(state,
11245                crate::StateTransition::IndexState { current, .. }
11246                    if current.coverage == crate::Coverage::Partial(crate::CoverageReason::Inaccessible))) {
11247                assert_eq!(handle.read_with(Index::state).expect("coherent state").coverage,
11248                    crate::Coverage::Partial(crate::CoverageReason::Inaccessible));
11249                published_partial = true;
11250            }
11251        }).expect("interrupted pass returns retryable report");
11252        assert!(!report.is_complete());
11253        assert!(report.retry_required);
11254        assert!(published_partial, "the transition precedes the caller's retry result");
11255        let recovered = reconcile_handle(&handle, &config, &mut |_| {}).expect("retry");
11256        assert!(recovered.is_complete());
11257        assert_eq!(
11258            handle.read_with(Index::state).expect("recovered state").coverage,
11259            crate::Coverage::Complete
11260        );
11261    }
11262
11263    #[test]
11264    fn stale_arbitration_keeps_a_reconciliation_incomplete() {
11265        let report = ReconcileReport {
11266            scan: ScanReport::default(),
11267            apply: ApplyStats { stale: 1, ..ApplyStats::default() },
11268            observations: 1,
11269            ..ReconcileReport::default()
11270        };
11271
11272        assert!(!report.is_complete());
11273    }
11274
11275    #[test]
11276    fn portable_system_time_conversion_preserves_pre_epoch_values() {
11277        let before_epoch = std::time::UNIX_EPOCH
11278            // Windows timestamps have 100 ns granularity, so use a duration that every
11279            // supported platform can represent without rounding back to the epoch.
11280            .checked_sub(std::time::Duration::from_secs(1))
11281            .expect("represent pre-epoch fixture");
11282
11283        assert_eq!(system_time_ns(before_epoch), -1_000_000_000);
11284        assert_eq!(system_time_ns(std::time::UNIX_EPOCH), 0);
11285    }
11286
11287    #[cfg(not(unix))]
11288    #[test]
11289    fn one_filesystem_fails_when_device_identity_is_unavailable() {
11290        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
11291
11292        assert!(matches!(config.validate(), Err(Error::UnsupportedScanConfig(_))));
11293    }
11294
11295    #[test]
11296    fn revalidate_is_a_no_op_against_an_unchanged_tree() {
11297        let dir = sample_tree();
11298        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11299        let before = index.total();
11300
11301        let mut deltas = Vec::new();
11302        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
11303        let mut unchanged = 0;
11304        for delta in &deltas {
11305            unchanged += index.apply_ok(delta).unchanged;
11306        }
11307
11308        assert_eq!(unchanged, 5, "3 files + 2 dirs all already known");
11309        assert_eq!(index.total(), before);
11310    }
11311
11312    #[cfg(windows)]
11313    #[test]
11314    fn windows_reconcile_detects_same_size_rewrite_with_preserved_mtime() {
11315        let root = tempfile::tempdir().expect("tempdir");
11316        let path = root.path().join("same.txt");
11317        write_file(&path, b"first");
11318        let modified = fs::metadata(&path).expect("metadata").modified().expect("mtime");
11319        let (mut index, _) =
11320            scan_into_index(root.path(), &ScanConfig::default()).expect("initial scan");
11321        let before = *index.attrs(Path::new("same.txt")).expect("initial attrs");
11322
11323        // NTFS stamps change time from the system clock, which advances in ticks of up to
11324        // 15.625 ms, so a rewrite stamped in the same tick as the first write is
11325        // indistinguishable from it. Wait for the clock to leave that tick rather than for
11326        // a fixed interval; the precise clock `SystemTime` reads runs at most one tick ahead.
11327        let stamped = std::time::UNIX_EPOCH
11328            + std::time::Duration::from_nanos(
11329                u64::try_from(before.ctime_ns).expect("change time after the epoch"),
11330            );
11331        while std::time::SystemTime::now() <= stamped + std::time::Duration::from_millis(20) {
11332            std::thread::sleep(std::time::Duration::from_millis(5));
11333        }
11334        write_file(&path, b"other");
11335        File::options()
11336            .write(true)
11337            .open(&path)
11338            .expect("open rewritten file")
11339            .set_times(std::fs::FileTimes::new().set_modified(modified))
11340            .expect("restore mtime");
11341        reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
11342
11343        let after = *index.attrs(Path::new("same.txt")).expect("rewritten attrs");
11344        assert_eq!((after.size, after.mtime_ns), (before.size, before.mtime_ns));
11345        assert_ne!(after.ctime_ns, before.ctime_ns, "change time detects the rewrite");
11346        assert_ne!(after.fingerprint(), before.fingerprint());
11347    }
11348
11349    #[cfg(windows)]
11350    #[test]
11351    fn windows_reconcile_detects_path_identity_replacement() {
11352        let root = tempfile::tempdir().expect("tempdir");
11353        let path = root.path().join("replace.txt");
11354        let displaced = root.path().join("displaced.txt");
11355        write_file(&path, b"first");
11356        let modified = fs::metadata(&path).expect("metadata").modified().expect("mtime");
11357        let (mut index, _) =
11358            scan_into_index(root.path(), &ScanConfig::default()).expect("initial scan");
11359        let before = *index.attrs(Path::new("replace.txt")).expect("initial attrs");
11360
11361        fs::rename(&path, &displaced).expect("retain old file identity");
11362        write_file(&path, b"other");
11363        File::options()
11364            .write(true)
11365            .open(&path)
11366            .expect("open replacement")
11367            .set_times(std::fs::FileTimes::new().set_modified(modified))
11368            .expect("restore mtime");
11369        reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
11370
11371        let after = *index.attrs(Path::new("replace.txt")).expect("replacement attrs");
11372        assert_eq!((after.size, after.mtime_ns), (before.size, before.mtime_ns));
11373        assert_ne!(
11374            (after.dev, after.inode),
11375            (before.dev, before.inode),
11376            "volume serial and file index identify the replacement"
11377        );
11378        assert_ne!(after.fingerprint(), before.fingerprint());
11379    }
11380
11381    #[test]
11382    fn direct_reconciliation_counts_unchanged_entries_and_publishes_state_commits() {
11383        let dir = sample_tree();
11384        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11385        let before_total = index.total();
11386        let before_clock = index.clock();
11387        let mut commits = Vec::new();
11388
11389        let report = reconcile(&mut index, &ScanConfig::default(), &mut |commit| {
11390            commits.push(commit.clone());
11391        })
11392        .expect("reconcile");
11393
11394        assert!(report.is_complete());
11395        assert_eq!(report.apply.unchanged, 5, "3 files + 2 dirs all already known");
11396        assert_eq!(commits.len(), 2);
11397        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
11398        assert_eq!(
11399            index.clock(),
11400            crate::Clock(before_clock.0 + 2),
11401            "start and finish are state commits"
11402        );
11403        let commits = index.since(before_clock).commits;
11404        assert_eq!(commits.len(), 2);
11405        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
11406        assert_eq!(index.total(), before_total);
11407    }
11408
11409    #[test]
11410    fn parallel_and_serial_reconciliation_produce_the_same_index() {
11411        let dir = sample_tree();
11412        let portable_config =
11413            ScanConfig { threads: Some(1), batch_size: 2, ..ScanConfig::default() };
11414        let (mut portable, _) =
11415            scan_into_index(dir.path(), &portable_config).expect("portable baseline");
11416        let mut bulk = portable.clone();
11417
11418        fs::remove_file(dir.path().join("a.txt")).expect("remove file");
11419        write_file(&dir.path().join("src/main.rs"), b"fn main() { much longer }");
11420        write_file(&dir.path().join("src/added.md"), b"new file");
11421        crate::test_support::settle_allocations(dir.path());
11422
11423        let portable_report = reconcile(&mut portable, &portable_config, &mut |_| {})
11424            .expect("portable reconciliation");
11425        let bulk_config = ScanConfig { threads: Some(2), ..portable_config };
11426        let bulk_report =
11427            reconcile(&mut bulk, &bulk_config, &mut |_| {}).expect("bulk reconciliation");
11428
11429        assert!(portable_report.is_complete());
11430        assert!(bulk_report.is_complete());
11431        assert_eq!(bulk_report.scan.attribution, WalkAttribution::default());
11432        assert_eq!(bulk_report.scan.entries, portable_report.scan.entries);
11433        assert_eq!(bulk_report.scan.dirs_read, portable_report.scan.dirs_read);
11434        assert_eq!(bulk_report.apply, portable_report.apply);
11435        assert_eq!(index_fingerprint(&bulk), index_fingerprint(&portable));
11436        assert_eq!(bulk.total(), portable.total());
11437    }
11438
11439    #[test]
11440    fn parallel_reconciliation_workers_publish_directory_counters() {
11441        let _serial = crate::counters::test_serial();
11442        let dir = sample_tree();
11443        let baseline = ScanConfig { threads: Some(1), ..ScanConfig::default() };
11444        let (mut index, scan) = scan_into_index(dir.path(), &baseline).expect("baseline scan");
11445        assert!(scan.is_complete());
11446
11447        crate::counters::enable(true);
11448        crate::counters::reset();
11449        let config = ScanConfig { threads: Some(4), ..baseline };
11450        let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconciliation");
11451        crate::counters::flush_thread();
11452        let counts = crate::counters::snapshot();
11453        crate::counters::reset();
11454        crate::counters::enable(false);
11455
11456        assert!(report.is_complete());
11457        assert!(
11458            counts.dir_opens >= report.scan.dirs_read,
11459            "parallel worker directory opens were folded: {counts:?}, report={report:?}"
11460        );
11461    }
11462
11463    #[test]
11464    fn parallel_reconciliation_matches_serial_across_structural_transitions() {
11465        for max_depth in [None, Some(1), Some(2)] {
11466            for order in [ScanOrder::BreadthFirst, ScanOrder::DepthFirst] {
11467                let dir = reconciliation_transition_tree();
11468                let reference_config = ScanConfig {
11469                    order,
11470                    max_depth,
11471                    threads: Some(1),
11472                    batch_size: 2,
11473                    ..ScanConfig::default()
11474                };
11475                let (baseline, baseline_report) =
11476                    scan_into_index(dir.path(), &reference_config).expect("baseline scan");
11477                assert!(baseline_report.is_complete());
11478                mutate_reconciliation_transition_tree(dir.path());
11479
11480                let mut serial = baseline.clone();
11481                let mut serial_commits = Vec::new();
11482                let serial_report = reconcile(&mut serial, &reference_config, &mut |commit| {
11483                    serial_commits.push(commit.clone());
11484                })
11485                .expect("serial reconciliation");
11486                let (fresh, fresh_report) =
11487                    scan_into_index(dir.path(), &reference_config).expect("fresh oracle");
11488                assert!(serial_report.is_complete(), "serial {order:?}/{max_depth:?}");
11489                assert!(fresh_report.is_complete(), "fresh {order:?}/{max_depth:?}");
11490                assert_eq!(
11491                    index_fingerprint(&serial),
11492                    index_fingerprint(&fresh),
11493                    "serial did not converge to a fresh scan for {order:?}/{max_depth:?}"
11494                );
11495
11496                for workers in [2, 4] {
11497                    let mut parallel = baseline.clone();
11498                    let config = ScanConfig { threads: Some(workers), ..reference_config.clone() };
11499                    let mut parallel_commits = Vec::new();
11500                    let report = reconcile(&mut parallel, &config, &mut |commit| {
11501                        parallel_commits.push(commit.clone());
11502                    })
11503                    .expect("parallel reconciliation");
11504                    let context = format!("{order:?}/{max_depth:?}/{workers} workers");
11505
11506                    assert!(report.is_complete(), "{context}: unexpected partial report");
11507                    assert_eq!(report.scan.entries, serial_report.scan.entries, "{context}");
11508                    assert_eq!(report.scan.dirs_read, serial_report.scan.dirs_read, "{context}");
11509                    assert_eq!(report.apply, serial_report.apply, "{context}");
11510                    assert_eq!(
11511                        effective_ops(&parallel_commits),
11512                        effective_ops(&serial_commits),
11513                        "{context}: effective delta differs"
11514                    );
11515                    assert_eq!(
11516                        index_fingerprint(&parallel),
11517                        index_fingerprint(&serial),
11518                        "{context}: final index differs"
11519                    );
11520                    let (parallel_total, serial_total) = (parallel.total(), serial.total());
11521                    assert_eq!(
11522                        (
11523                            parallel_total.files,
11524                            parallel_total.dirs,
11525                            parallel_total.bytes,
11526                            parallel_total.allocated,
11527                            parallel_total.newest_mtime_ns,
11528                        ),
11529                        (
11530                            serial_total.files,
11531                            serial_total.dirs,
11532                            serial_total.bytes,
11533                            serial_total.allocated,
11534                            serial_total.newest_mtime_ns,
11535                        ),
11536                        "{context}: roll-up differs"
11537                    );
11538                    assert_eq!(
11539                        parallel_total.by_ext, serial_total.by_ext,
11540                        "{context}: extension roll-up differs"
11541                    );
11542                }
11543            }
11544        }
11545    }
11546
11547    #[cfg(unix)]
11548    #[test]
11549    fn revalidation_metadata_errors_do_not_delete_enumerated_entries() {
11550        use std::os::unix::fs::PermissionsExt;
11551
11552        if !crate::test_support::require_permission_bits() {
11553            return;
11554        }
11555
11556        let dir = sample_tree();
11557        let config = ScanConfig::default();
11558        let (mut index, baseline_report) =
11559            scan_into_index(dir.path(), &config).expect("baseline scan");
11560        assert!(baseline_report.is_complete());
11561        let before = index_fingerprint(&index);
11562        let original_permissions = fs::metadata(dir.path()).expect("root metadata").permissions();
11563
11564        fs::set_permissions(dir.path(), fs::Permissions::from_mode(0o400))
11565            .expect("remove search permission");
11566        let mut observations = Vec::new();
11567        let outcome = revalidate(&index, &config, &mut |observation| {
11568            observations.push(observation);
11569        });
11570        fs::set_permissions(dir.path(), original_permissions).expect("restore permissions");
11571
11572        let report = outcome.expect("operational metadata errors are a partial report");
11573        assert!(!report.errors.is_empty(), "the fixture did not induce metadata errors");
11574        for observation in &observations {
11575            index.apply_ok(observation);
11576        }
11577        assert_eq!(index_fingerprint(&index), before);
11578        assert!(index.attrs(Path::new("a.txt")).is_some(), "existing entry was removed");
11579    }
11580
11581    #[cfg(unix)]
11582    #[test]
11583    fn reconciliation_metadata_errors_drop_unverified_entries_like_a_cold_scan() {
11584        use std::os::unix::fs::PermissionsExt;
11585
11586        if !crate::test_support::require_permission_bits() {
11587            return;
11588        }
11589
11590        // macOS's parallel path may satisfy the whole directory through
11591        // getattrlistbulk even without search permission. One worker pins the portable
11592        // fallback there; Linux also exercises the parallel portable worker.
11593        let worker_counts = if cfg!(target_os = "macos") { vec![1] } else { vec![1, 2] };
11594        for workers in worker_counts {
11595            let dir = sample_tree();
11596            let config =
11597                ScanConfig { threads: Some(workers), batch_size: 2, ..ScanConfig::default() };
11598            let (mut index, baseline_report) =
11599                scan_into_index(dir.path(), &config).expect("baseline scan");
11600            assert!(baseline_report.is_complete());
11601            let before = index_fingerprint(&index);
11602            let original_permissions =
11603                fs::metadata(dir.path()).expect("root metadata").permissions();
11604
11605            // Reading names requires read permission; looking up their metadata also
11606            // requires search permission. This makes enumeration succeed and each
11607            // metadata lookup fail, the boundary where an encountered name used to be
11608            // misclassified as a deletion.
11609            fs::set_permissions(dir.path(), fs::Permissions::from_mode(0o400))
11610                .expect("remove search permission");
11611            let outcome = reconcile(&mut index, &config, &mut |_| {});
11612            let cold = scan_into_index(dir.path(), &config);
11613            fs::set_permissions(dir.path(), original_permissions).expect("restore permissions");
11614
11615            let report = outcome.expect("operational metadata errors are a partial report");
11616            let (cold, cold_report) = cold.expect("cold partial scan");
11617            assert!(!report.scan.errors.is_empty(), "the fixture did not induce metadata errors");
11618            assert!(!report.is_complete());
11619            assert!(!cold_report.is_complete());
11620            assert_eq!(index_fingerprint(&index), index_fingerprint(&cold));
11621            assert!(
11622                index_fingerprint(&index).is_empty(),
11623                "neither warm nor cold may retain attributes it could not verify"
11624            );
11625            assert_eq!(index.directory_complete(Path::new("")), Some(false));
11626            assert_eq!(cold.directory_complete(Path::new("")), Some(false));
11627            assert!(!before.is_empty(), "the fixture began with retained facts");
11628        }
11629    }
11630
11631    #[cfg(unix)]
11632    #[test]
11633    fn failed_listing_withdraws_retained_completeness_and_recovers() {
11634        use std::os::unix::fs::PermissionsExt;
11635
11636        if !crate::test_support::require_permission_bits() {
11637            return;
11638        }
11639        for workers in [1, 2] {
11640            let dir = tempfile::tempdir().expect("root");
11641            write_file(&dir.path().join("ancestor/blocked/unknown.txt"), b"unknown");
11642            write_file(&dir.path().join("healthy/known.txt"), b"known");
11643            let config = ScanConfig { threads: Some(workers), ..ScanConfig::default() };
11644            let (mut warm, baseline) = scan_into_index(dir.path(), &config).expect("baseline");
11645            assert!(baseline.is_complete());
11646            let blocked = dir.path().join("ancestor/blocked");
11647            fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny listing");
11648            let probe = fs::read_dir(&blocked);
11649            let mut commits = Vec::new();
11650            let refreshed =
11651                reconcile(&mut warm, &config, &mut |commit| commits.push(commit.clone()));
11652            let cold = scan_into_index(dir.path(), &config);
11653            fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore");
11654            assert_eq!(
11655                probe.expect_err("real denied listing").kind(),
11656                std::io::ErrorKind::PermissionDenied
11657            );
11658            assert!(!refreshed.expect("partial refresh").is_complete());
11659            let (cold, report) = cold.expect("partial cold scan");
11660            assert!(!report.is_complete());
11661            for index in [&warm, &cold] {
11662                for path in ["", "ancestor", "healthy"] {
11663                    assert_eq!(
11664                        index.directory_complete(Path::new(path)),
11665                        Some(true),
11666                        "workers={workers}: {path}"
11667                    );
11668                }
11669                assert_eq!(index.directory_complete(Path::new("ancestor/blocked")), Some(false));
11670                assert_eq!(index.freshness_at(Path::new("ancestor")), crate::Freshness::Partial);
11671            }
11672            assert!(commits.iter().flat_map(|commit| &commit.state).any(|state| matches!(state,
11673                crate::StateTransition::IndexState { current, .. } if current.coverage != crate::Coverage::Complete
11674            )), "failure is published");
11675            assert!(reconcile(&mut warm, &config, &mut |_| {}).expect("recovery").is_complete());
11676            assert_eq!(warm.directory_complete(Path::new("ancestor/blocked")), Some(true));
11677            assert_eq!(warm.state().coverage, crate::Coverage::Complete);
11678        }
11679    }
11680
11681    #[test]
11682    fn unreadable_directory_warm_answer_matches_cold_verified_tree() {
11683        for workers in [1, 2] {
11684            let dir = tempfile::tempdir().expect("tempdir");
11685            write_file(&dir.path().join("blocked/old.txt"), b"old");
11686            write_file(&dir.path().join("verified.txt"), b"verified");
11687            crate::test_support::settle_allocations(dir.path());
11688            let config = ScanConfig { threads: Some(workers), ..ScanConfig::default() };
11689            let (mut warm, baseline) = scan_into_index(dir.path(), &config).expect("baseline");
11690            assert!(baseline.is_complete());
11691
11692            let blocked = dir.path().join("blocked").canonicalize().expect("blocked path");
11693            let hook = install_child_metadata_hook(dir.path(), move |path| {
11694                (path == blocked)
11695                    .then(|| std::io::Error::from(std::io::ErrorKind::PermissionDenied))
11696            });
11697            let warm_report = reconcile(&mut warm, &config, &mut |_| {}).expect("warm partial");
11698            let (cold, cold_report) = scan_into_index(dir.path(), &config).expect("cold partial");
11699            drop(hook);
11700
11701            assert!(!warm_report.is_complete(), "workers={workers}");
11702            assert!(!cold_report.is_complete(), "workers={workers}");
11703            assert!(warm.lookup(Path::new("blocked")).is_none(), "workers={workers}");
11704            assert!(warm.lookup(Path::new("blocked/old.txt")).is_none(), "workers={workers}");
11705            assert_eq!(index_fingerprint(&warm), index_fingerprint(&cold), "workers={workers}");
11706            assert!(warm.lookup(Path::new("verified.txt")).is_some(), "workers={workers}");
11707        }
11708    }
11709
11710    #[test]
11711    fn deferred_change_overflow_retries_without_applying_a_partial_wave() {
11712        let dir = sample_tree();
11713        let config = ScanConfig { threads: Some(2), batch_size: 2, ..ScanConfig::default() };
11714        let (mut index, _) = scan_into_index(dir.path(), &config).expect("baseline");
11715        let before = index_fingerprint(&index);
11716
11717        fs::remove_file(dir.path().join("a.txt")).expect("remove file");
11718        write_file(&dir.path().join("added.md"), b"new file");
11719        write_file(&dir.path().join("src/main.rs"), b"fn main() { much longer }");
11720        crate::test_support::settle_allocations(dir.path());
11721
11722        let root = index.root_path().to_path_buf();
11723        let root_meta = {
11724            crate::counters::bump(|c| c.stats += 1);
11725            fs::symlink_metadata(&root)
11726        }
11727        .expect("root metadata");
11728        let mut commits = Vec::new();
11729        let outcome = reconcile_direct_parallel(
11730            &mut index,
11731            &root,
11732            root_device(&root, &root_meta).expect("root device"),
11733            &config,
11734            1,
11735            &mut |commit| commits.push(commit.clone()),
11736        )
11737        .expect("parallel attempt");
11738        let DirectParallelOutcome::RetrySerial { prefix, remaining } = outcome else {
11739            panic!("the deliberately tiny deferred budget must trigger the retry");
11740        };
11741
11742        assert_eq!(prefix.apply, ApplyStats::default());
11743        assert_eq!(remaining, VecDeque::from([(PathBuf::new(), 0)]));
11744        assert!(commits.is_empty());
11745        assert_eq!(index_fingerprint(&index), before);
11746
11747        let serial = ScanConfig { threads: Some(1), ..config };
11748        let report = reconcile(&mut index, &serial, &mut |_| {}).expect("serial retry");
11749        let (expected, expected_report) = scan_into_index(dir.path(), &serial).expect("oracle");
11750        assert!(report.is_complete());
11751        assert!(expected_report.is_complete());
11752        assert_eq!(index_fingerprint(&index), index_fingerprint(&expected));
11753        assert_eq!(index.total().files, expected.total().files);
11754        assert_eq!(index.total().dirs, expected.total().dirs);
11755        assert_eq!(index.total().bytes, expected.total().bytes);
11756        assert_eq!(index.total().allocated, expected.total().allocated);
11757        assert_eq!(index.total().newest_mtime_ns, expected.total().newest_mtime_ns);
11758        assert_eq!(index.total().by_ext, expected.total().by_ext);
11759    }
11760
11761    #[test]
11762    fn late_overflow_resumes_without_double_counting_completed_waves() {
11763        let dir = tempfile::tempdir().expect("tempdir");
11764        // The root wave discovers more than one full wave of directories. A change in
11765        // the second wave then forces the serial fallback only after the first wave's
11766        // unchanged entries have already been counted.
11767        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
11768            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), b"unchanged");
11769        }
11770        let parallel = ScanConfig { threads: Some(2), ..ScanConfig::default() };
11771        let (baseline, _) = scan_into_index(dir.path(), &parallel).expect("baseline");
11772        let mut candidate = baseline.clone();
11773        let mut serial_oracle = baseline;
11774
11775        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
11776            write_file(
11777                &dir.path().join(format!("d{directory:04}/file.txt")),
11778                b"changed after the first wave",
11779            );
11780        }
11781        crate::test_support::settle_allocations(dir.path());
11782
11783        let candidate_report = reconcile_target_inner(
11784            &mut ReconcileTarget::Direct(&mut candidate),
11785            Path::new(""),
11786            0,
11787            &parallel,
11788            0,
11789            &mut |_| {},
11790        )
11791        .expect("late-overflow reconciliation");
11792        let serial = ScanConfig { threads: Some(1), ..parallel };
11793        let oracle_report =
11794            reconcile(&mut serial_oracle, &serial, &mut |_| {}).expect("serial oracle");
11795
11796        assert_eq!(candidate_report.apply, oracle_report.apply);
11797        assert_eq!(candidate_report.scan.entries, oracle_report.scan.entries);
11798        assert_eq!(candidate_report.scan.dirs_read, oracle_report.scan.dirs_read);
11799        assert_eq!(index_fingerprint(&candidate), index_fingerprint(&serial_oracle));
11800    }
11801
11802    /// The four counts a walk reports, in the order [`crate::ProgressSnapshot`] shows them.
11803    fn walked(report: &ScanReport) -> (u64, u64, u64, u64) {
11804        (report.dirs_read, report.files_walked, report.bytes_walked, report.allocated_walked)
11805    }
11806
11807    fn reported(progress: &crate::Progress) -> (u64, u64, u64, u64) {
11808        let snapshot = progress.snapshot();
11809        (snapshot.directories, snapshot.files, snapshot.bytes, snapshot.allocated)
11810    }
11811
11812    /// Each walker is a separate loop with its own reporting sites, so each is checked:
11813    /// the detached cold walk, the streaming walk, the transient summary fold, the
11814    /// reference revalidation, exclusive reconciliation serial and in parallel waves,
11815    /// and shared-handle reconciliation. The tree is wide enough that every parallel
11816    /// walker claims several chunks and the small batch size fills several batches, so a
11817    /// walker that reported only its final state would still fail on the counts a
11818    /// mid-walk chunk added twice or not at all.
11819    #[test]
11820    fn every_walker_reports_exactly_what_its_report_counts() {
11821        let dir = tempfile::tempdir().expect("tempdir");
11822        for directory in 0..12 {
11823            for file in 0..5 {
11824                write_file(
11825                    &dir.path().join(format!("d{directory}/f{file}.txt")),
11826                    &vec![b'x'; directory * 5 + file + 1],
11827                );
11828            }
11829        }
11830        for threads in [1, 4] {
11831            let context = format!("threads={threads}");
11832            let progress = crate::Progress::new();
11833            let cold_config = ScanConfig {
11834                threads: Some(threads),
11835                batch_size: 4,
11836                progress: Some(progress.clone()),
11837                ..ScanConfig::default()
11838            };
11839            let (mut index, cold) = scan_into_index(dir.path(), &cold_config).expect("cold scan");
11840            assert_eq!(cold.dirs_read, 13, "{context}: the root and twelve children");
11841            assert_eq!(
11842                progress.snapshot().phase,
11843                crate::ProgressPhase::Indexing,
11844                "{context}: the detached walk ends by assembling the index"
11845            );
11846            assert_eq!(reported(&progress), walked(&cold), "{context}: detached cold walk");
11847
11848            let progress = crate::Progress::new();
11849            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
11850            let streamed = scan(dir.path(), &config, &mut |_| {}).expect("streaming scan");
11851            assert_eq!(walked(&streamed), walked(&cold), "{context}");
11852            assert_eq!(reported(&progress), walked(&streamed), "{context}: streaming walk");
11853
11854            let progress = crate::Progress::new();
11855            let config = ScanConfig {
11856                read_controls: false,
11857                progress: Some(progress.clone()),
11858                ..cold_config.clone()
11859            };
11860            let folded = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
11861            assert_eq!(walked(&folded), walked(&cold), "{context}");
11862            assert_eq!(reported(&progress), walked(&folded), "{context}: summary fold");
11863
11864            let progress = crate::Progress::new();
11865            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
11866            let revalidated = revalidate(&index, &config, &mut |_| {}).expect("revalidate");
11867            assert_eq!(progress.snapshot().phase, crate::ProgressPhase::Revalidating, "{context}");
11868            assert_eq!(reported(&progress), walked(&revalidated), "{context}: revalidate");
11869
11870            // Changes, so reconciliation defers and applies operations rather than
11871            // discarding every entry as unchanged.
11872            write_file(&dir.path().join(format!("d0/new{threads}.txt")), b"added");
11873            fs::remove_file(dir.path().join(format!("d1/f{}.txt", threads - 1))).expect("remove");
11874            write_file(&dir.path().join("d2/f0.txt"), &vec![b'y'; 40 + threads]);
11875            // An independent walk of the changed tree. The handle and the reconcile's own
11876            // report both come from the walker's counts, so agreeing with each other
11877            // would not show that the walker counted anything; agreeing with this does.
11878            let (_, fresh) = scan_into_index(dir.path(), &ScanConfig::default()).expect("fresh");
11879            let progress = crate::Progress::new();
11880            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
11881            let reconciled = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
11882            assert!(reconciled.apply.mutated(), "{context}: the changes were applied");
11883            assert_eq!(progress.snapshot().phase, crate::ProgressPhase::Revalidating, "{context}");
11884            assert_eq!(reported(&progress), walked(&reconciled.scan), "{context}: reconcile");
11885            assert_eq!(walked(&reconciled.scan), walked(&fresh), "{context}: the whole tree");
11886
11887            let handle = crate::IndexHandle::new(index);
11888            let progress = crate::Progress::new();
11889            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config };
11890            let shared = reconcile_handle(&handle, &config, &mut |_| {}).expect("shared");
11891            assert_eq!(reported(&progress), walked(&shared.scan), "{context}: shared handle");
11892        }
11893    }
11894
11895    /// A wave that overflows its deferred-operation budget is thrown away and rewalked
11896    /// serially. The report counts each directory once, as the logical pass does, and
11897    /// progress counts the wave's reads both times, as the filesystem did them: work
11898    /// done, not the answer. This is the one walker relation that is not equality, and
11899    /// the difference is exactly the rewalked wave.
11900    #[test]
11901    fn progress_counts_a_rewalked_wave_twice_where_the_report_counts_it_once() {
11902        let dir = tempfile::tempdir().expect("tempdir");
11903        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
11904            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), b"unchanged");
11905        }
11906        // A file in the wave that completes, so the serial rewalk has counts of that wave
11907        // to carry forward as already added rather than add again.
11908        write_file(&dir.path().join("root.txt"), b"counted by the wave that completes");
11909        let parallel = ScanConfig { threads: Some(2), ..ScanConfig::default() };
11910        let (mut index, _) = scan_into_index(dir.path(), &parallel).expect("baseline");
11911        // Larger than any filesystem stores inline in the inode, so each copy occupies
11912        // blocks of its own wherever the test runs.
11913        let changed = &vec![b'c'; 8_193];
11914        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
11915            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), changed);
11916        }
11917        crate::test_support::settle_allocations(dir.path());
11918
11919        let progress = crate::Progress::new();
11920        let observed = ScanConfig { progress: Some(progress.clone()), ..parallel };
11921        let report = reconcile_target_inner(
11922            &mut ReconcileTarget::Direct(&mut index),
11923            Path::new(""),
11924            0,
11925            &observed,
11926            0,
11927            &mut |_| {},
11928        )
11929        .expect("late-overflow reconciliation");
11930
11931        // The root wave changes nothing and completes; the second wave holds exactly
11932        // one full wave of changed directories, overflows, and is rewalked with the one
11933        // directory the wave left behind.
11934        let rewalked = u64::try_from(RECONCILE_WAVE_DIRECTORIES).expect("fits");
11935        let snapshot = progress.snapshot();
11936        assert_eq!(snapshot.directories, report.scan.dirs_read + rewalked);
11937        assert_eq!(snapshot.files, report.scan.files_walked + rewalked);
11938        assert_eq!(
11939            snapshot.bytes,
11940            report.scan.bytes_walked + rewalked * u64::try_from(changed.len()).expect("fits")
11941        );
11942        let allocated = index.attrs(Path::new("d0000/file.txt")).expect("indexed").allocated;
11943        assert!(allocated > 0, "a file with content occupies blocks");
11944        assert_eq!(snapshot.allocated, report.scan.allocated_walked + rewalked * allocated);
11945    }
11946
11947    /// Several invalidated roots are reconciled one at a time and their reports summed,
11948    /// so every walked count has to survive the sum, allocated bytes included.
11949    #[test]
11950    fn a_multi_root_reconcile_sums_every_walked_count() {
11951        let dir = tempfile::tempdir().expect("tempdir");
11952        for directory in ["a", "b", "c"] {
11953            for file in 0..3 {
11954                write_file(&dir.path().join(format!("{directory}/f{file}.txt")), b"before");
11955            }
11956        }
11957        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11958        for directory in ["a", "b"] {
11959            write_file(&dir.path().join(format!("{directory}/f0.txt")), &vec![b'x'; 5_000]);
11960        }
11961        index.apply_ok(&Observation::new(
11962            ["a", "b"]
11963                .into_iter()
11964                .map(|directory| Op::InvalidateSubtree {
11965                    path: PathBuf::from(directory),
11966                    reason: crate::InvalidateReason::Requested,
11967                })
11968                .collect(),
11969        ));
11970        let report =
11971            reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
11972
11973        let attrs: Vec<Attrs> = ["a", "b"]
11974            .into_iter()
11975            .flat_map(|directory| (0..3).map(move |file| format!("{directory}/f{file}.txt")))
11976            .map(|path| *index.attrs(Path::new(&path)).expect("indexed"))
11977            .collect();
11978        assert_eq!(report.scan.files_walked, 6, "the two roots' files, and not c's");
11979        assert_eq!(report.scan.bytes_walked, attrs.iter().map(|attrs| attrs.size).sum::<u64>());
11980        assert_eq!(
11981            report.scan.allocated_walked,
11982            attrs.iter().map(|attrs| attrs.allocated).sum::<u64>()
11983        );
11984    }
11985
11986    #[test]
11987    fn shared_reconciliation_retains_conditional_no_op_arbitration() {
11988        let dir = sample_tree();
11989        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11990        let handle = crate::IndexHandle::new(index);
11991        let before_clock = handle.clock().expect("clock");
11992        let mut commits = Vec::new();
11993
11994        let report = reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
11995            commits.push(commit.clone());
11996        })
11997        .expect("reconcile");
11998
11999        assert!(report.is_complete());
12000        assert_eq!(report.apply.unchanged, 5, "3 files + 2 dirs all already known");
12001        assert_eq!(commits.len(), 2);
12002        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
12003        assert_eq!(
12004            handle.clock().expect("clock"),
12005            crate::Clock(before_clock.0 + 2),
12006            "start and finish are state commits"
12007        );
12008        let commits = handle.since(before_clock).expect("state commits").commits;
12009        assert_eq!(commits.len(), 2);
12010        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
12011    }
12012
12013    #[test]
12014    fn revalidate_detects_additions_edits_and_deletions() {
12015        let dir = sample_tree();
12016        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12017
12018        fs::remove_file(dir.path().join("a.txt")).expect("remove");
12019        write_file(&dir.path().join("src/main.rs"), b"fn main() { longer }");
12020        write_file(&dir.path().join("added.md"), b"new");
12021
12022        let mut deltas = Vec::new();
12023        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
12024        let mut stats = crate::index::ApplyStats::default();
12025        for delta in &deltas {
12026            let s = index.apply_ok(delta);
12027            stats.inserted += s.inserted;
12028            stats.updated += s.updated;
12029            stats.removed += s.removed;
12030        }
12031
12032        assert_eq!(stats.inserted, 1, "added.md");
12033        assert_eq!(stats.updated, 1, "main.rs grew");
12034        assert_eq!(stats.removed, 1, "a.txt is gone");
12035
12036        let total = index.total();
12037        assert_eq!(total.files, 3);
12038        assert_eq!(total.bytes, 20 + 9 + 3);
12039        assert!(!total.by_ext.contains_key(".txt"));
12040        assert_eq!(total.by_ext[".md"].files, 1);
12041    }
12042
12043    #[test]
12044    fn revalidate_removes_a_whole_vanished_directory() {
12045        let dir = sample_tree();
12046        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12047        fs::remove_dir_all(dir.path().join("src")).expect("remove dir");
12048
12049        let mut deltas = Vec::new();
12050        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
12051        for delta in &deltas {
12052            index.apply_ok(delta);
12053        }
12054
12055        let total = index.total();
12056        assert_eq!(total.files, 1);
12057        assert_eq!(total.dirs, 0);
12058        assert!(index.lookup(Path::new("src")).is_none());
12059    }
12060
12061    #[test]
12062    fn pending_invalidation_reconciles_the_requested_subtree() {
12063        let dir = sample_tree();
12064        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12065        write_file(&dir.path().join("src/added.rs"), b"new");
12066        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
12067            path: PathBuf::from("src"),
12068            reason: crate::InvalidateReason::Requested,
12069        }]));
12070        assert_eq!(index.freshness_at(Path::new("src")), crate::Freshness::Stale);
12071
12072        let mut applied = Vec::new();
12073        let report = reconcile_pending(&mut index, &ScanConfig::default(), &mut |delta| {
12074            applied.push(delta.clone());
12075        })
12076        .expect("reconcile pending");
12077
12078        assert!(report.is_complete());
12079        assert!(index.lookup(Path::new("src/added.rs")).is_some());
12080        assert_eq!(index.freshness_at(Path::new("src")), crate::Freshness::Fresh);
12081        assert!(index.take_pending_invalidations().is_empty());
12082        assert!(applied.iter().any(|commit| commit_touches(commit, Path::new("src/added.rs"))));
12083    }
12084
12085    /// A retained `.gitignore` reconciled as the root of its own walk re-reads its rules.
12086    /// A file does not descend, so the subtree-root branch was the only place that could
12087    /// read them, and it did not: the table kept `*.log` while the pass reported complete.
12088    #[test]
12089    fn reconciling_a_retained_control_file_as_the_subtree_root_rereads_its_rules() {
12090        let dir = tempfile::tempdir().expect("tempdir");
12091        write_file(&dir.path().join(".gitignore"), b"*.log\n");
12092        write_file(&dir.path().join("a.log"), b"log");
12093        let config = ScanConfig { read_controls: true, ..ScanConfig::default() };
12094        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
12095        assert_eq!(
12096            index.is_ignored(Path::new("a.log")).expect("control state observed"),
12097            Some(true)
12098        );
12099
12100        write_file(&dir.path().join(".gitignore"), b"# nothing is ignored now\n");
12101        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
12102            path: PathBuf::from(".gitignore"),
12103            reason: crate::InvalidateReason::Requested,
12104        }]));
12105        let report = reconcile_pending(&mut index, &config, &mut |_| {}).expect("reconcile");
12106
12107        assert!(report.is_complete(), "{:?}", report.scan.errors);
12108        assert_eq!(index.freshness_at(Path::new(".gitignore")), crate::Freshness::Fresh);
12109        assert_eq!(
12110            index.is_ignored(Path::new("a.log")).expect("control state observed"),
12111            Some(false)
12112        );
12113    }
12114
12115    /// A control file the pass cannot verify contributes neither stale rules nor an entry.
12116    #[cfg(unix)]
12117    #[test]
12118    fn reconciling_an_unreadable_control_file_root_drops_its_rules_and_stays_partial() {
12119        use std::os::unix::fs::PermissionsExt;
12120
12121        if !crate::test_support::require_permission_bits() {
12122            return;
12123        }
12124
12125        let dir = tempfile::tempdir().expect("tempdir");
12126        let control = dir.path().join(".gitignore");
12127        write_file(&control, b"*.log\n");
12128        write_file(&dir.path().join("a.log"), b"log");
12129        let config = ScanConfig { read_controls: true, ..ScanConfig::default() };
12130        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
12131
12132        write_file(&control, b"# rewritten, then made unreadable\n");
12133        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("chmod");
12134        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
12135            path: PathBuf::from(".gitignore"),
12136            reason: crate::InvalidateReason::Requested,
12137        }]));
12138        let report = reconcile_pending(&mut index, &config, &mut |_| {});
12139        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore");
12140        let report = report.expect("reconcile");
12141
12142        assert!(!report.is_complete());
12143        assert_eq!(report.scan.errors.len(), 1, "{:?}", report.scan.errors);
12144        assert_eq!(index.freshness_at(Path::new(".gitignore")), crate::Freshness::Partial);
12145        assert!(
12146            !index.controls().expect("control state observed").contains(Path::new(".gitignore"))
12147        );
12148        assert_eq!(index.is_ignored(Path::new("a.log")).expect("control state observed"), None);
12149    }
12150
12151    #[cfg(unix)]
12152    #[test]
12153    fn unreadable_control_keeps_new_excluded_file_unknown_and_out_of_analysis() {
12154        use std::os::unix::fs::PermissionsExt;
12155
12156        if !crate::test_support::require_permission_bits() {
12157            return;
12158        }
12159        let dir = tempfile::tempdir().expect("tempdir");
12160        let control = dir.path().join(".gitignore");
12161        write_file(&control, b"*.log\n");
12162        write_file(&dir.path().join("keep.rs"), b"code");
12163        let config = ScanConfig {
12164            population: crate::query::IgnoredEntries::Exclude,
12165            ..ScanConfig::default()
12166        };
12167        let (mut index, cold) = scan_into_index(dir.path(), &config).expect("cold");
12168        assert!(cold.is_complete());
12169        write_file(&dir.path().join("debug.log"), b"must not analyze");
12170        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("chmod");
12171        let report = reconcile(&mut index, &config, &mut |_| {});
12172        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore");
12173        let report = report.expect("reconcile");
12174        assert!(!report.is_complete());
12175        assert!(index.lookup(Path::new("debug.log")).is_some(), "unknown is retained");
12176        assert_eq!(index.ignored_classification(Path::new("debug.log")), None);
12177        assert!(!index.ignored_classification_complete_below(Path::new("")));
12178        let candidates = index.analysis_candidates(crate::content::AnalysisSet::NONE.with_lines());
12179        assert!(
12180            candidates.iter().all(|candidate| candidate.relative_path != Path::new("debug.log"))
12181        );
12182        let repaired =
12183            reconcile(&mut index, &config, &mut |_| {}).expect("reconcile repaired control");
12184        assert!(repaired.is_complete(), "{:?}", repaired.scan.errors);
12185        assert!(index.ignored_classification_complete_below(Path::new("")));
12186        assert!(index.lookup(Path::new("debug.log")).is_none(), "known ignored file is pruned");
12187    }
12188
12189    #[test]
12190    fn handle_reconciliation_publishes_after_each_delta_is_applied() {
12191        let dir = sample_tree();
12192        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12193        let handle = crate::IndexHandle::new(index);
12194        let reader = handle.clone();
12195        write_file(&dir.path().join("added.md"), b"new");
12196
12197        let mut observed_after_apply = false;
12198        reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
12199            if commit_touches(commit, Path::new("added.md")) {
12200                observed_after_apply =
12201                    reader.kind(Path::new("added.md")).expect("query index").is_some();
12202            }
12203        })
12204        .expect("reconcile handle");
12205
12206        assert!(observed_after_apply);
12207    }
12208
12209    /// Delete `name` under `root` from inside its own metadata lookup, after the listing
12210    /// returned it, on whichever thread performs the lookup.
12211    fn delete_between_listing_and_stat(root: &Path, name: &'static str) -> WalkHookGuard {
12212        install_child_metadata_hook(root, move |path| {
12213            delete_if_named(path, name);
12214            None
12215        })
12216    }
12217
12218    /// As [`delete_between_listing_and_stat`], and every reconciliation listing under `root`
12219    /// also ends in an error, so none of them is complete.
12220    fn delete_between_listing_and_stat_in_a_failing_listing(
12221        root: &Path,
12222        name: &'static str,
12223    ) -> WalkHookGuard {
12224        install_walk_hook(root, move |point| match point {
12225            WalkHookPoint::ChildMetadata(path) => {
12226                delete_if_named(path, name);
12227                None
12228            }
12229            WalkHookPoint::ListingEnd => Some(std::io::Error::other("injected listing error")),
12230            WalkHookPoint::ControlLookup(_) => None,
12231        })
12232    }
12233
12234    fn delete_if_named(path: &Path, name: &str) {
12235        if path.file_name() == Some(OsStr::new(name)) {
12236            fs::remove_file(path).expect("delete between listing and stat");
12237        }
12238    }
12239
12240    /// A name the listing returned that is gone by the time it is stat'd was deleted, on
12241    /// the serial path and on the parallel waves a full-root pass takes by default. The
12242    /// walk removes it; recorded as an error, it would settle as a phantom entry with
12243    /// permanent partial freshness.
12244    #[test]
12245    fn a_child_deleted_between_listing_and_stat_is_removed_rather_than_reported() {
12246        for threads in [1, 4] {
12247            let dir = tempfile::tempdir().expect("tempdir");
12248            write_file(&dir.path().join("keep.txt"), b"keep");
12249            write_file(&dir.path().join("gone.txt"), b"gone");
12250            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
12251            let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
12252            assert!(index.lookup(Path::new("gone.txt")).is_some());
12253
12254            index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
12255                path: PathBuf::new(),
12256                reason: crate::InvalidateReason::Requested,
12257            }]));
12258            let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
12259            let report = reconcile_pending(&mut index, &config, &mut |_| {});
12260            drop(hook);
12261            let report = report.expect("reconcile");
12262
12263            assert!(report.is_complete(), "threads {threads}: {:?}", report.scan.errors);
12264            assert_eq!(report.apply.removed, 1, "threads {threads}");
12265            assert!(index.lookup(Path::new("gone.txt")).is_none(), "threads {threads}");
12266            assert!(index.lookup(Path::new("keep.txt")).is_some(), "threads {threads}");
12267            assert_eq!(index.freshness(), crate::Freshness::Fresh, "threads {threads}");
12268        }
12269    }
12270
12271    /// A vanished control file takes its rules with it on both reconcile paths, even when the
12272    /// rest of its listing fails: the stat's `NotFound` is the evidence. A retained file's
12273    /// rules go with its entry's removal; a hidden-pruned one has no entry to remove, so
12274    /// without its own removal its rules would go on ignoring its siblings.
12275    #[test]
12276    fn a_control_file_deleted_between_listing_and_stat_takes_its_rules_with_it() {
12277        for prune_hidden in [false, true] {
12278            for (threads, listing_fails) in [(1, false), (4, false), (1, true), (4, true)] {
12279                let case = format!(
12280                    "prune hidden {prune_hidden}, threads {threads}, listing fails {listing_fails}"
12281                );
12282                let dir = tempfile::tempdir().expect("tempdir");
12283                write_file(&dir.path().join(".gitignore"), b"*.log\n");
12284                write_file(&dir.path().join("a.log"), b"log");
12285                let config = ScanConfig {
12286                    read_controls: true,
12287                    threads: Some(threads),
12288                    hidden: prune_hidden
12289                        .then(|| std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([""; 0]))),
12290                    ..ScanConfig::default()
12291                };
12292                let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
12293                assert_eq!(
12294                    index.is_ignored(Path::new("a.log")).expect("control state observed"),
12295                    Some(true),
12296                    "{case}"
12297                );
12298
12299                index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
12300                    path: PathBuf::new(),
12301                    reason: crate::InvalidateReason::Requested,
12302                }]));
12303                let hook = if listing_fails {
12304                    delete_between_listing_and_stat_in_a_failing_listing(dir.path(), ".gitignore")
12305                } else {
12306                    delete_between_listing_and_stat(dir.path(), ".gitignore")
12307                };
12308                let report = reconcile_pending(&mut index, &config, &mut |_| {});
12309                drop(hook);
12310                let report = report.expect("reconcile");
12311
12312                assert_eq!(
12313                    report.is_complete(),
12314                    !listing_fails,
12315                    "{case}: {:?}",
12316                    report.scan.errors
12317                );
12318                assert!(index.lookup(Path::new(".gitignore")).is_none(), "{case}");
12319                assert!(
12320                    !index
12321                        .controls()
12322                        .expect("control state observed")
12323                        .contains(Path::new(".gitignore")),
12324                    "{case}"
12325                );
12326                assert_eq!(
12327                    index.is_ignored(Path::new("a.log")).expect("control state observed"),
12328                    Some(false),
12329                    "{case}"
12330                );
12331            }
12332        }
12333    }
12334
12335    /// A cold walk records a name gone by its stat as it records a name the listing never
12336    /// returned: not at all, and without an error that would make the walk partial.
12337    #[test]
12338    fn a_cold_walk_omits_a_child_deleted_between_listing_and_stat() {
12339        for threads in [1, 4] {
12340            let dir = tempfile::tempdir().expect("tempdir");
12341            write_file(&dir.path().join("keep.txt"), b"keep");
12342            write_file(&dir.path().join("gone.txt"), b"gone");
12343            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
12344
12345            let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
12346            let scanned = scan_into_index_via_scanner(dir.path(), &config);
12347            drop(hook);
12348            let (index, report) = scanned.expect("scan");
12349
12350            assert!(report.is_complete(), "threads {threads}: {:?}", report.errors);
12351            assert!(index.lookup(Path::new("gone.txt")).is_none(), "threads {threads}");
12352            assert!(index.lookup(Path::new("keep.txt")).is_some(), "threads {threads}");
12353        }
12354    }
12355
12356    /// Revalidation emits the removal a reconciliation would apply.
12357    #[test]
12358    fn revalidation_removes_a_child_deleted_between_listing_and_stat() {
12359        let dir = tempfile::tempdir().expect("tempdir");
12360        write_file(&dir.path().join("keep.txt"), b"keep");
12361        write_file(&dir.path().join("gone.txt"), b"gone");
12362        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12363
12364        let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
12365        let mut observations = Vec::new();
12366        let report = revalidate(&index, &ScanConfig::default(), &mut |observation| {
12367            observations.push(observation);
12368        });
12369        drop(hook);
12370        let report = report.expect("revalidate");
12371        for observation in &observations {
12372            index.apply_ok(observation);
12373        }
12374
12375        assert!(report.is_complete(), "{:?}", report.errors);
12376        assert!(index.lookup(Path::new("gone.txt")).is_none());
12377        assert!(index.lookup(Path::new("keep.txt")).is_some());
12378    }
12379
12380    /// Revalidation emits the same rules removal when the rest of the listing fails.
12381    #[test]
12382    fn revalidation_removes_the_rules_of_a_control_file_deleted_in_a_failing_listing() {
12383        for prune_hidden in [false, true] {
12384            let dir = tempfile::tempdir().expect("tempdir");
12385            write_file(&dir.path().join(".gitignore"), b"*.log\n");
12386            write_file(&dir.path().join("a.log"), b"log");
12387            let config = ScanConfig {
12388                read_controls: true,
12389                hidden: prune_hidden
12390                    .then(|| std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([""; 0]))),
12391                ..ScanConfig::default()
12392            };
12393            let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
12394            assert_eq!(
12395                index.is_ignored(Path::new("a.log")).expect("control state observed"),
12396                Some(true)
12397            );
12398
12399            let hook =
12400                delete_between_listing_and_stat_in_a_failing_listing(dir.path(), ".gitignore");
12401            let mut observations = Vec::new();
12402            let report = revalidate(&index, &config, &mut |observation| {
12403                observations.push(observation);
12404            });
12405            drop(hook);
12406            let report = report.expect("revalidate");
12407            for observation in &observations {
12408                index.apply_ok(observation);
12409            }
12410
12411            assert!(!report.is_complete(), "prune hidden {prune_hidden}");
12412            assert!(index.lookup(Path::new(".gitignore")).is_none(), "prune hidden {prune_hidden}");
12413            assert!(
12414                !index
12415                    .controls()
12416                    .expect("control state observed")
12417                    .contains(Path::new(".gitignore")),
12418                "prune hidden {prune_hidden}"
12419            );
12420            assert_eq!(
12421                index.is_ignored(Path::new("a.log")).expect("control state observed"),
12422                Some(false),
12423                "prune hidden {prune_hidden}"
12424            );
12425        }
12426    }
12427
12428    #[test]
12429    fn reconciliation_does_not_clear_a_newer_invalidation() {
12430        let dir = sample_tree();
12431        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12432        let handle = crate::IndexHandle::new(index);
12433        write_file(&dir.path().join("added.md"), b"new");
12434
12435        let invalidator = handle.clone();
12436        let mut saw_reconciling = false;
12437        reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
12438            if commit_touches(commit, Path::new("added.md")) {
12439                saw_reconciling =
12440                    invalidator.freshness().expect("query") == crate::Freshness::Reconciling;
12441                invalidator
12442                    .apply(&Observation::new(vec![Op::InvalidateSubtree {
12443                        path: PathBuf::new(),
12444                        reason: crate::InvalidateReason::WatchOverflow,
12445                    }]))
12446                    .expect("new invalidation");
12447            }
12448        })
12449        .expect("reconcile handle");
12450
12451        assert!(saw_reconciling);
12452        assert_eq!(handle.freshness().expect("query"), crate::Freshness::Stale);
12453    }
12454
12455    #[test]
12456    fn failed_reconciliation_marks_the_scope_partial() {
12457        let dir = sample_tree();
12458        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12459        fs::remove_dir_all(dir.path()).expect("remove root");
12460
12461        assert!(reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).is_err());
12462        assert_eq!(index.freshness(), crate::Freshness::Partial);
12463    }
12464
12465    #[test]
12466    fn successful_subtree_retry_restores_complete_root_coverage() {
12467        let dir = tempfile::tempdir().expect("tempdir");
12468        write_file(&dir.path().join("blocked/known.txt"), b"known");
12469        let config = ScanConfig::default();
12470        let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
12471        assert!(report.is_complete());
12472        let blocked = dir.path().join("blocked");
12473        let fault = install_walk_hook(&blocked, |_| {
12474            Some(std::io::Error::new(
12475                std::io::ErrorKind::PermissionDenied,
12476                "deterministic subtree refusal",
12477            ))
12478        });
12479
12480        let failed = reconcile_subtree(&mut index, Path::new("blocked"), &config, &mut |_| {})
12481            .expect("partial");
12482        assert!(!failed.scan.is_complete());
12483        assert_eq!(
12484            index.state().coverage,
12485            crate::Coverage::Partial(crate::CoverageReason::Inaccessible)
12486        );
12487        drop(fault);
12488
12489        let recovered = reconcile_subtree(&mut index, Path::new("blocked"), &config, &mut |_| {})
12490            .expect("retry");
12491
12492        assert!(recovered.scan.is_complete());
12493        assert_eq!(index.state().coverage, crate::Coverage::Complete);
12494    }
12495
12496    #[test]
12497    fn failed_pending_reconciliation_remains_queued_for_retry() {
12498        let dir = tempfile::tempdir().expect("tempdir");
12499        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12500        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
12501            path: PathBuf::new(),
12502            reason: crate::InvalidateReason::Requested,
12503        }]));
12504        fs::remove_dir_all(dir.path()).expect("remove root");
12505
12506        assert!(reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {}).is_err());
12507        assert_eq!(
12508            index.take_pending_invalidations(),
12509            vec![(PathBuf::new(), crate::InvalidateReason::Requested)]
12510        );
12511        assert_eq!(index.freshness(), crate::Freshness::Partial);
12512    }
12513
12514    #[cfg(unix)]
12515    #[test]
12516    fn partial_cold_scan_keeps_verified_siblings_complete() {
12517        use std::os::unix::fs::PermissionsExt;
12518        if !crate::test_support::require_permission_bits() {
12519            return;
12520        }
12521        let root = tempfile::tempdir().expect("root");
12522        write_file(&root.path().join("blocked/unknown.txt"), b"unread");
12523        write_file(&root.path().join("healthy/nested/known.txt"), b"known");
12524        let blocked = root.path().join("blocked");
12525        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
12526        let scan = ScanConfig::default();
12527        let detached = scan_into_index(root.path(), &scan);
12528        let streamed = scan_into_index_via_scanner(root.path(), &scan);
12529        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
12530        for result in [detached, streamed] {
12531            let (index, report) = result.expect("partial scan still returns its facts");
12532            assert!(!report.is_complete(), "permission fixture must fail the blocked listing");
12533            assert!(
12534                index
12535                    .issues()
12536                    .iter()
12537                    .any(|issue| issue.path.as_deref() == Some(Path::new("blocked")))
12538            );
12539            assert_eq!(index.freshness_at(Path::new("")), crate::Freshness::Partial);
12540            assert_eq!(index.directory_complete(Path::new("")), Some(true));
12541            assert_eq!(index.directory_complete(Path::new("blocked")), Some(false));
12542            assert_eq!(index.freshness_at(Path::new("blocked")), crate::Freshness::Partial);
12543            assert!(index.lookup(Path::new("blocked/unknown.txt")).is_none());
12544            for sibling in ["healthy", "healthy/nested"] {
12545                assert_eq!(index.directory_complete(Path::new(sibling)), Some(true), "{sibling}");
12546                assert_eq!(
12547                    index.freshness_at(Path::new(sibling)),
12548                    crate::Freshness::Fresh,
12549                    "{sibling}"
12550                );
12551            }
12552            assert!(index.lookup(Path::new("healthy/nested/known.txt")).is_some());
12553            assert!(
12554                !crate::stored_state::entries_writable(&index),
12555                "partial root cannot persist metadata"
12556            );
12557        }
12558    }
12559
12560    #[cfg(unix)]
12561    #[test]
12562    fn partial_pending_reconciliation_remains_queued_for_retry() {
12563        use std::os::unix::fs::PermissionsExt;
12564
12565        if !crate::test_support::require_permission_bits() {
12566            return;
12567        }
12568        let dir = tempfile::tempdir().expect("tempdir");
12569        write_file(&dir.path().join("blocked/known.txt"), b"known");
12570        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12571        let blocked = dir.path().join("blocked");
12572        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
12573        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
12574            path: PathBuf::from("blocked"),
12575            reason: crate::InvalidateReason::VerificationFailed,
12576        }]));
12577
12578        let report = reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {})
12579            .expect("permission failure is a partial report");
12580        let pending = index.take_pending_invalidations();
12581        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
12582        assert!(!report.is_complete(), "permission fixture must make reconciliation partial");
12583        assert_eq!(
12584            pending,
12585            vec![(PathBuf::from("blocked"), crate::InvalidateReason::VerificationFailed)]
12586        );
12587        assert_eq!(index.freshness_at(Path::new("blocked")), crate::Freshness::Partial);
12588    }
12589
12590    /// The shared API settles an unreadable subtree instead of queueing it again.
12591    ///
12592    /// Its per-event driver, `Watcher::apply_next`, drains after every event, so a retry
12593    /// re-walked the same unreadable subtree on each unrelated event, forever. The subtree
12594    /// stays partial and the report still names the error, once.
12595    #[cfg(unix)]
12596    #[test]
12597    fn partial_shared_pending_reconciliation_settles_instead_of_retrying() {
12598        use std::os::unix::fs::PermissionsExt;
12599
12600        if !crate::test_support::require_permission_bits() {
12601            return;
12602        }
12603        let dir = tempfile::tempdir().expect("tempdir");
12604        write_file(&dir.path().join("blocked/known.txt"), b"known");
12605        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12606        let handle = crate::IndexHandle::new(index);
12607        let blocked = dir.path().join("blocked");
12608        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
12609        handle
12610            .apply(&Observation::new(vec![Op::InvalidateSubtree {
12611                path: PathBuf::from("blocked"),
12612                reason: crate::InvalidateReason::VerificationFailed,
12613            }]))
12614            .expect("invalidate");
12615
12616        let report = reconcile_pending_handle(&handle, &ScanConfig::default(), &mut |_| {})
12617            .expect("permission failure is a partial report");
12618        let pending = handle.take_pending_invalidations().expect("pending");
12619        let freshness = handle.freshness_at(Path::new("blocked")).expect("freshness");
12620        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
12621        assert!(!report.is_complete(), "permission fixture must make reconciliation partial");
12622        assert!(pending.is_empty(), "{pending:?}");
12623        assert_eq!(freshness, crate::Freshness::Partial);
12624        assert!(!report.scan.errors.is_empty());
12625    }
12626
12627    #[test]
12628    fn pending_scope_mismatch_does_not_drain_the_retry_queue() {
12629        let dir = tempfile::tempdir().expect("tempdir");
12630        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
12631        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
12632        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
12633            path: PathBuf::new(),
12634            reason: crate::InvalidateReason::Requested,
12635        }]));
12636
12637        let error = reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {})
12638            .expect_err("mismatched scope must fail");
12639
12640        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
12641        assert_eq!(
12642            index.take_pending_invalidations(),
12643            vec![(PathBuf::new(), crate::InvalidateReason::Requested)]
12644        );
12645    }
12646
12647    #[test]
12648    fn reconciliation_rejects_a_scope_mismatch_before_mutating() {
12649        let dir = tempfile::tempdir().expect("tempdir");
12650        write_file(&dir.path().join("deep/nested.txt"), b"nested");
12651        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
12652        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
12653        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
12654
12655        let error = reconcile(&mut index, &ScanConfig::default(), &mut |_| {})
12656            .expect_err("mismatched scope must fail");
12657
12658        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
12659        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
12660        assert_eq!(index.freshness(), crate::Freshness::Fresh);
12661    }
12662
12663    #[test]
12664    fn subtree_reconciliation_rejects_a_path_beyond_the_depth_scope() {
12665        let dir = tempfile::tempdir().expect("tempdir");
12666        write_file(&dir.path().join("deep/nested.txt"), b"nested");
12667        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
12668        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
12669
12670        let result =
12671            reconcile_subtree(&mut index, Path::new("deep/nested.txt"), &shallow, &mut |_| {});
12672
12673        assert!(matches!(result, Err(Error::SubtreeOutsideScanScope { .. })));
12674        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
12675        assert_eq!(index.freshness(), crate::Freshness::Fresh);
12676    }
12677
12678    #[cfg(unix)]
12679    #[test]
12680    fn subtree_reconciliation_does_not_follow_an_ancestor_symlink() {
12681        use std::os::unix::fs::symlink;
12682
12683        let root = tempfile::tempdir().expect("root");
12684        let outside = tempfile::tempdir().expect("outside");
12685        write_file(&outside.path().join("secret.txt"), b"secret");
12686        symlink(outside.path(), root.path().join("link")).expect("symlink");
12687        let config = ScanConfig::default();
12688        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
12689
12690        let result =
12691            reconcile_subtree(&mut index, Path::new("link/secret.txt"), &config, &mut |_| {});
12692
12693        assert!(matches!(result, Err(Error::SubtreeOutsideScanScope { .. })));
12694        assert!(index.lookup(Path::new("link/secret.txt")).is_none());
12695        assert_eq!(index.freshness(), crate::Freshness::Fresh);
12696    }
12697
12698    #[test]
12699    fn subtree_reconciliation_widens_to_a_non_directory_ancestor() {
12700        let root = tempfile::tempdir().expect("root");
12701        write_file(&root.path().join("parent/child.txt"), b"old");
12702        let config = ScanConfig::default();
12703        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
12704        fs::remove_dir_all(root.path().join("parent")).expect("remove directory");
12705        write_file(&root.path().join("parent"), b"replacement");
12706
12707        let report =
12708            reconcile_subtree(&mut index, Path::new("parent/child.txt"), &config, &mut |_| {})
12709                .expect("reconcile widened ancestor");
12710
12711        assert!(report.is_complete());
12712        assert_eq!(index.kind(Path::new("parent")), Some(EntryKind::File));
12713        assert!(index.lookup(Path::new("parent/child.txt")).is_none());
12714        assert_eq!(index.freshness(), crate::Freshness::Fresh);
12715    }
12716
12717    #[test]
12718    fn subtree_reconciliation_widens_to_a_missing_ancestor() {
12719        let root = tempfile::tempdir().expect("root");
12720        write_file(&root.path().join("parent/child.txt"), b"old");
12721        let config = ScanConfig::default();
12722        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
12723        fs::remove_dir_all(root.path().join("parent")).expect("remove directory");
12724
12725        let report =
12726            reconcile_subtree(&mut index, Path::new("parent/child.txt"), &config, &mut |_| {})
12727                .expect("reconcile widened ancestor");
12728
12729        assert!(report.is_complete());
12730        assert!(index.lookup(Path::new("parent")).is_none());
12731        assert_eq!(index.freshness(), crate::Freshness::Fresh);
12732    }
12733
12734    #[test]
12735    fn observation_only_revalidation_rejects_a_scope_mismatch() {
12736        let dir = tempfile::tempdir().expect("tempdir");
12737        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
12738        let (index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
12739        let mut observations = Vec::new();
12740
12741        let error = revalidate(&index, &ScanConfig::default(), &mut |observation| {
12742            observations.push(observation);
12743        })
12744        .expect_err("mismatched scope must fail");
12745
12746        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
12747        assert!(observations.is_empty());
12748    }
12749
12750    #[cfg(unix)]
12751    #[test]
12752    fn a_new_filesystem_boundary_prunes_cached_descendants() {
12753        use std::os::unix::fs::MetadataExt;
12754
12755        let root = Path::new("/");
12756        let root_dev = {
12757            crate::counters::bump(|c| c.stats += 1);
12758            fs::symlink_metadata(root)
12759        }
12760        .expect("stat root")
12761        .dev();
12762        let Some(mount) = [Path::new("/dev"), Path::new("/proc"), Path::new("/sys")]
12763            .into_iter()
12764            .find(|candidate| {
12765                fs::symlink_metadata(candidate)
12766                    .is_ok_and(|metadata| metadata.is_dir() && metadata.dev() != root_dev)
12767            })
12768        else {
12769            return; // This host exposes no convenient cross-device directory.
12770        };
12771        let relative = mount.strip_prefix(root).expect("mount is below root");
12772        let stale_child = relative.join(".fdu-stale-snapshot-entry");
12773        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
12774        let mount_meta = fs::symlink_metadata(mount).expect("stat mount");
12775        let mut index = Index::new_with_scope(root, config.scope());
12776        index.apply_baseline_ok(&Observation::new(vec![
12777            Op::Upsert {
12778                path: relative.to_path_buf(),
12779                kind: EntryKind::Dir,
12780                attrs: attrs_from(mount, &mount_meta).expect("mount attrs"),
12781            },
12782            Op::Upsert {
12783                path: stale_child.clone(),
12784                kind: EntryKind::File,
12785                attrs: Attrs { size: 10, allocated: 10, ..Attrs::default() },
12786            },
12787        ]));
12788
12789        let error = reconcile_subtree(&mut index, &stale_child, &config, &mut |_| {})
12790            .expect_err("a descendant below the mount boundary is outside scope");
12791        assert!(matches!(error, Error::SubtreeOutsideScanScope { .. }));
12792        assert!(index.lookup(&stale_child).is_some());
12793
12794        reconcile_subtree(&mut index, relative, &config, &mut |_| {}).expect("reconcile mount");
12795
12796        assert!(index.lookup(relative).is_some(), "the mount point itself stays visible");
12797        assert!(index.lookup(&stale_child).is_none(), "out-of-scope descendants are pruned");
12798    }
12799
12800    #[test]
12801    fn subtree_reconciliation_rejects_paths_outside_the_root() {
12802        let dir = sample_tree();
12803        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12804
12805        assert!(matches!(
12806            reconcile_subtree(
12807                &mut index,
12808                Path::new("../outside"),
12809                &ScanConfig::default(),
12810                &mut |_| {},
12811            ),
12812            Err(Error::PathEscapesRoot(_))
12813        ));
12814        assert_eq!(index.freshness(), crate::Freshness::Fresh);
12815    }
12816
12817    #[test]
12818    fn normalized_walk_errors_keep_index_and_one_shot_status_in_lockstep() {
12819        let root = Path::new("/root");
12820        let mut order: Vec<_> = (0..66).rev().collect();
12821        order.push(65);
12822        let mut errors = order
12823            .into_iter()
12824            .map(|number| {
12825                Error::io(
12826                    root.join(format!("file-{number:02}")),
12827                    std::io::Error::new(std::io::ErrorKind::PermissionDenied, "denied"),
12828                )
12829            })
12830            .collect::<Vec<_>>();
12831
12832        normalize_walk_errors(root, &mut errors);
12833        assert_eq!(errors.len(), 66, "the repeated cause is removed once");
12834
12835        let mut index = crate::Index::new(root);
12836        index.record_walk_errors(&mut errors);
12837        let status = crate::query::TreeStatus::of_walk(
12838            root,
12839            &mut ScanReport { errors, ..ScanReport::default() },
12840        );
12841
12842        assert_eq!(status.errors, index.issues());
12843        assert_eq!(status.errors_omitted, index.state().issues.omitted);
12844        assert_eq!(status.errors.len(), crate::MAX_RETAINED_ISSUES);
12845        assert_eq!(status.errors_omitted, 2);
12846    }
12847
12848    #[cfg(target_os = "linux")]
12849    #[test]
12850    fn scan_and_revalidate_keep_non_utf8_names_distinct() {
12851        use std::ffi::OsString;
12852        use std::os::unix::ffi::OsStringExt;
12853
12854        let dir = tempfile::tempdir().expect("tempdir");
12855        let first = PathBuf::from(OsString::from_vec(vec![b'n', 0x80]));
12856        let second = PathBuf::from(OsString::from_vec(vec![b'n', 0x81]));
12857        write_file(&dir.path().join(&first), b"a");
12858        write_file(&dir.path().join(&second), b"bb");
12859
12860        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12861        assert_eq!(index.total().files, 2);
12862        assert_eq!(index.total().bytes, 3);
12863
12864        fs::remove_file(dir.path().join(&first)).expect("remove first");
12865        let mut observations = Vec::new();
12866        revalidate(&index, &ScanConfig::default(), &mut |observation| {
12867            observations.push(observation);
12868        })
12869        .expect("revalidate");
12870        for observation in &observations {
12871            index.apply_ok(observation);
12872        }
12873        assert!(index.lookup(&first).is_none());
12874        assert!(index.lookup(&second).is_some());
12875        assert_eq!(index.total().bytes, 2);
12876    }
12877}