Skip to main content

fdu_core/
scan.rs

1//! The scan layer: walking a tree, producing observations, and applying reconciliation.
2//!
3//! Public scans emit upsert observations, and a revalidation sweep is the diff between
4//! what the index believes and what the filesystem says. Both speak the same
5//! [`Observation`] vocabulary as the watch layer. A detached one-shot index may consume
6//! equivalent parent-first directory groups privately because no observer can see its
7//! construction; every later mutation still crosses the shared observation boundary.
8//!
9//! # Status
10//!
11//! The serial walk is the portable `read_dir` plus non-following metadata reference.
12//! Parallel scans use the same path on most platforms; on macOS they first try a
13//! measured `getattrlistbulk` backend that returns directory entries and stat-tier
14//! metadata together. Unsupported filesystems, malformed results, mount points, and
15//! firmlinks fail closed to the portable path for the complete containing directory.
16//! Every backend produces the same [`Observation`] contract.
17
18use std::collections::{BTreeMap, BTreeSet, VecDeque};
19use std::ffi::{OsStr, OsString};
20use std::fmt::Write as _;
21use std::fs;
22use std::io::Read as _;
23use std::path::{Component, Path, PathBuf};
24
25use crate::ApplyStats;
26use crate::engine_contract::{
27    Attrs, Commit, EntryKind, Error, Observation, ObservationOp, Op, PathExpectation, PathState,
28    Result, ScanScope,
29};
30use crate::index::{
31    DetachedIndexBuilder, Index, IndexHandle, ReconcileErrors, ReconcileFinish,
32    collect_child_expectations,
33};
34use crate::query::ScopeAxis;
35use crate::stored_state::{ControlTierIdentity, EntryScope, EntryTierIdentity, SnapshotIdentity};
36
37// Keep the FFI exception at the platform boundary. The rest of the engine, including
38// every consumer of these observations, remains under the workspace's unsafe-code
39// denial.
40#[cfg(target_os = "macos")]
41#[allow(unsafe_code)]
42mod macos_bulk;
43
44#[cfg(windows)]
45#[allow(unsafe_code)]
46mod windows_metadata;
47
48/// How many ops accumulate before an observation is handed to the sink.
49///
50/// Batching matters for more than syscall economy: consumers coalesce per path within a
51/// batch and stat once per batch, and a live UI wants partial results while a large tree
52/// is still being walked rather than one delta at the end.
53const DEFAULT_BATCH_SIZE: usize = crate::platform_tuning::tuning().batch_size.get();
54
55/// Largest producer batch accepted before work must be published incrementally.
56pub const MAX_SCAN_BATCH_SIZE: usize = 64 * 1024;
57
58/// Most changed paths an exclusive parallel reconciliation may defer before applying.
59///
60/// Workers compare against one immutable index image, so mutations wait until the wave
61/// joins. Bounding that change set keeps a churned tree from turning the fast unchanged
62/// path into an unbounded allocation; overflow discards the wave and retries through
63/// the incremental serial reconciler.
64const MAX_DEFERRED_RECONCILE_OPS: usize = MAX_SCAN_BATCH_SIZE;
65
66/// Directories compared against one immutable index baseline before changes are applied.
67///
68/// The wave is large enough to amortize scoped worker creation and small enough that a
69/// changed tree publishes progress throughout a long reconciliation.
70const RECONCILE_WAVE_DIRECTORIES: usize =
71    crate::platform_tuning::tuning().reconcile_wave_directories.get();
72
73/// Identity of the fixed stat-tier reducer set.
74const REDUCERS_FINGERPRINT: u64 = 1;
75
76/// The order directories are visited in.
77///
78/// This changes *when* observations are produced, never *which* ones: both orders
79/// visit every entry exactly once and leave an identical index behind. It therefore
80/// stays out of [`ScanScope`] and cannot invalidate a cache, exactly like the worker
81/// count.
82///
83/// The choice only matters to a consumer that reads the index while the walk is still
84/// running, and there it matters a great deal.
85///
86/// # Strength of the guarantee
87///
88/// **These are scheduling preferences, not strict orders, whenever more than one worker
89/// is running** — which is the default.
90///
91/// The queue is ordered, but the *claims* are not. Workers take directories from the
92/// shared queue in the policy's order; a worker that finishes early can enqueue its
93/// children and another worker can claim them while a slower worker still holds
94/// unfinished work from a shallower level. Nothing releases a level barrier, because
95/// a barrier would idle every fast worker at each level boundary and give back most of
96/// the parallel producer's win.
97///
98/// So:
99///
100/// - With `threads: Some(1)`, [`ScanOrder::BreadthFirst`] is strict: no directory is
101///   read before one closer to the root.
102/// - With several workers it is *shallow-first*: shallow work is always preferred when
103///   a worker chooses, and deeper observations can still interleave.
104///
105/// That weaker property is what the browser use case actually needs — every top-level
106/// subtree starts filling early, so a mid-scan ranking is meaningful — and it is the
107/// property the tests pin. A caller that needs strict level order must ask for one
108/// worker and pay for it.
109#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
110pub enum ScanOrder {
111    /// Shallow directories before deep ones.
112    ///
113    /// The default, because it is the order whose partial results mean something.
114    /// Roll-ups are maintained per directory as the walk proceeds, so a consumer that
115    /// looks mid-scan sees top-level totals grow together — bars fill, rankings
116    /// converge — instead of one subtree finishing while its siblings read zero.
117    /// Interrupting early leaves a usefully complete picture of the top of the tree.
118    ///
119    /// Under several workers this is a preference rather than a guarantee; see the
120    /// type-level note above.
121    ///
122    /// Note that totals only grow *while an additive walk is running*. Monotonicity
123    /// comes from the producer being additive, not from the order — the order decides
124    /// which subtrees get to grow early.
125    #[default]
126    BreadthFirst,
127    /// One subtree toward completion before starting the next.
128    ///
129    /// Lower peak memory, since the frontier is bounded by depth rather than by the
130    /// width of a level, and better locality within a subtree. The cost is that
131    /// partial results are actively misleading: one child of the root approaches its
132    /// final total while its siblings read zero, so anything ranking by size mid-scan
133    /// ranks confidently and wrongly. Correct for a caller that only reads the
134    /// finished index and wants the smallest footprint.
135    ///
136    /// Under several workers this too is a preference: several subtrees will be in
137    /// flight at once, one per worker.
138    DepthFirst,
139}
140
141/// Knobs for a scan.
142#[derive(Clone, Debug)]
143// Four booleans, each an independent admission or observation switch with its own
144// semantic-scope consequence, not an enum in disguise: any combination is legal and
145// means what its fields say. The lint suspects flag-soup states; this is a config
146// surface whose fields are documented one by one.
147#[allow(clippy::struct_excessive_bools)]
148pub struct ScanConfig {
149    /// Maximum relative entry depth to retain. Zero keeps only the index root and `None`
150    /// means unlimited.
151    pub max_depth: Option<usize>,
152    /// Ops per emitted observation. Must be between one and [`MAX_SCAN_BATCH_SIZE`].
153    pub batch_size: usize,
154    /// Follow symlinks to directories. Off by default: following them turns a tree walk
155    /// into a graph walk with cycles, and every surveyed tool defaults to off.
156    pub follow_symlinks: bool,
157    /// Stay on the filesystem the root lives on.
158    pub one_filesystem: bool,
159    /// Hidden-component admission, or `None` to retain every component.
160    pub hidden: Option<std::sync::Arc<crate::admission::HiddenPolicy>>,
161    /// Exclude filesystem objects other than files, directories, and symlinks.
162    pub exclude_special: bool,
163    /// Directory-reading worker threads.
164    ///
165    /// A tree walk is a pile of independent, latency-bound directory reads, so it
166    /// scales with threads far better than most work does. One means the serial
167    /// walker, which stays the reference implementation and the thing every result is
168    /// checked against. [`None`] asks for a bounded default derived from the
169    /// machine's available parallelism. The automatic pool starts conservatively and
170    /// unlocks more latency-hiding workers only when initial chunk timing identifies a
171    /// slow filesystem path.
172    ///
173    /// This is an operational knob, not a semantic one: it changes how fast the same
174    /// observations are produced, never which observations they are. That is why it
175    /// stays out of [`ScanScope`] and cannot invalidate a cache.
176    pub threads: Option<usize>,
177    /// The order directories are visited in. See [`ScanOrder`].
178    pub order: ScanOrder,
179    /// File-type rules to classify against, or `None` for the ones compiled into fdu.
180    ///
181    /// Unlike [`Self::threads`] this *is* semantic: a different taxonomy classifies the
182    /// same tree differently, which is why its fingerprint rides in [`ScanScope`] and a
183    /// change to it invalidates a snapshot. Shared rather than owned because a scan
184    /// clones its config per wave and a registry is read-only once built.
185    pub types: Option<std::sync::Arc<crate::classify::TypeRegistry>>,
186    /// Observe `.gitignore` control files and retain ignore classification.
187    ///
188    /// On by default on every surface: an [`Index`] from [`crate::open`] or a scan keeps
189    /// the exact control state it exposes and a watch maintains -- which entries are
190    /// ignored, and the ignored and unignored partitions of every roll-up -- and a one-shot
191    /// report from [`crate::prepare_report`] shows the ignored share of every row
192    /// (fdu-elnn). It costs a read of every `.gitignore` in the tree. A file past the
193    /// [`Self::control_limits`] is refused and named in [`Index::control_coverage`] rather
194    /// than ending the scan; a file that cannot be read is an error at its path, which
195    /// makes the result partial.
196    ///
197    /// Off, the scan performs no control-file I/O and retains no control table, and that is
198    /// stamped into [`ScanScope`], so an index-returning call never serves a snapshot taken
199    /// one way as the other. An [`Index`] built that way answers [`Index::is_ignored`],
200    /// [`Index::controls`], and the partition accessors with
201    /// [`crate::Error::ControlStateNotObserved`], never with "not ignored", and refuses
202    /// control input; a report's rows carry no ignored share, and a selection by ignored
203    /// state is refused. The command line spells it `--no-gitignore`.
204    ///
205    /// An opened root ([`crate::OpenedIndex`]) always observes control state, because its
206    /// ignored and unignored partitions are part of what it serves.
207    pub read_controls: bool,
208    /// The budget and the line limit `.gitignore` files are applied under, each a size or
209    /// unbounded. See [`crate::control::ControlLimits`].
210    ///
211    /// A source that would take the table past the budget, or that has a line longer than
212    /// the line limit, is refused: its rules do not apply, the scan continues with every
213    /// size exact, and [`Index::control_coverage`] names it and the limit that fired. The
214    /// command line spells these `--gitignore-budget SIZE|all` and
215    /// `--gitignore-line-limit SIZE|all`; the Python API spells them `control_budget` and
216    /// `control_line_limit`.
217    ///
218    /// Semantic, like [`Self::read_controls`]: the limits decide which rules apply, so both
219    /// are part of [`ScanScope`] and a snapshot taken under other limits is not reused.
220    /// Ignored when control state is not observed.
221    pub control_limits: crate::control::ControlLimits,
222    /// Where to report how much of the walk has been done, or `None` to report nothing.
223    ///
224    /// An observer rather than a knob: it changes neither which observations a walk
225    /// produces nor how it produces them, so it is no part of [`ScanScope`] or of any
226    /// snapshot identity, and two configs that differ only here are the same scan.
227    /// Honoured by every walker in this module -- the cold scans, the summary fold,
228    /// [`revalidate`], and each `reconcile` entry point -- which enter
229    /// [`ProgressPhase::Scanning`](crate::ProgressPhase) or
230    /// [`ProgressPhase::Revalidating`](crate::ProgressPhase) and add their counts once
231    /// per chunk of directories, never per entry. See [`crate::Progress`] for what the
232    /// counts mean and what holds when a walk returns.
233    pub progress: Option<crate::Progress>,
234}
235
236impl Default for ScanConfig {
237    fn default() -> Self {
238        Self {
239            max_depth: None,
240            batch_size: DEFAULT_BATCH_SIZE,
241            follow_symlinks: false,
242            one_filesystem: false,
243            hidden: None,
244            exclude_special: false,
245            threads: None,
246            order: ScanOrder::default(),
247            types: None,
248            read_controls: crate::query::Request::DEFAULTS.read_controls,
249            control_limits: crate::query::Request::DEFAULTS.control_limits,
250            progress: None,
251        }
252    }
253}
254
255/// Why watching cannot narrow its scan scope, said once for every surface.
256///
257/// The CLI used to carry this guidance and the library carried "requires event-scope
258/// filtering", which names the implementation rather than the caller's next move -- so a
259/// library caller hitting the same wall got jargon and the CLI user got help. Two
260/// messages for one rule also drift, and the parity harness could not tell they were the
261/// same rule.
262///
263/// The knobs are named by the calling surface: `--scan-depth` on the command line,
264/// `max_depth` through the API. Everything else is identical, so the harness can verify
265/// mechanically that both surfaces state the same rule.
266pub const WATCH_SCOPE_GUIDANCE: &str = concat!(
267    "watching requires full scope and cannot be combined with max_depth or one_filesystem: ",
268    "a watcher cannot filter backend events against a narrowed boundary. Selection such as ",
269    "depth, include, and modified_since does work while watching, because it filters the ",
270    "retained index rather than narrowing the scan"
271);
272
273impl ScanConfig {
274    /// Classify this scan with `types` and include their derived identity in its scope.
275    #[must_use]
276    pub fn with_types(mut self, types: std::sync::Arc<crate::classify::TypeRegistry>) -> Self {
277        self.types = Some(types);
278        self
279    }
280
281    /// The file-type rules in effect: the supplied registry, or the compiled default.
282    pub fn types(&self) -> &crate::classify::TypeRegistry {
283        match &self.types {
284            Some(types) => types,
285            None => crate::classify::TypeRegistry::compiled(),
286        }
287    }
288
289    /// Share the file-type rules with an index that retains them.
290    pub(crate) fn types_shared(&self) -> std::sync::Arc<crate::classify::TypeRegistry> {
291        self.types
292            .as_ref()
293            .map_or_else(crate::classify::TypeRegistry::compiled_shared, std::sync::Arc::clone)
294    }
295
296    /// Hidden-component policy in effect.
297    pub fn hidden(&self) -> &crate::admission::HiddenPolicy {
298        self.hidden.as_deref().unwrap_or_else(|| crate::admission::HiddenPolicy::keep_all())
299    }
300
301    /// Semantic cache identity, excluding operational batching choices.
302    ///
303    /// Composed from [`Self::snapshot_identity`], so the scope an index records and the
304    /// tier identities a snapshot of it carries are one value in two shapes.
305    ///
306    /// No longer `const`: the type-rule fingerprint is now a property of the registry in
307    /// effect rather than a compiled-in constant, which is the whole point of letting a
308    /// caller supply one. A snapshot taken under different rules must not be reused.
309    pub fn scope(&self) -> ScanScope {
310        self.snapshot_identity().scan_scope()
311    }
312
313    /// Which entries this scan retains, the part of its scope no `.gitignore` setting
314    /// changes.
315    pub fn entry_scope(&self) -> EntryScope {
316        EntryScope {
317            max_depth: self.max_depth,
318            follow_symlinks: self.follow_symlinks,
319            one_filesystem: self.one_filesystem,
320            hidden_fingerprint: self.hidden().fingerprint(),
321            exclude_special: self.exclude_special,
322        }
323    }
324
325    /// Whether this scan observes `.gitignore` control state, and under which limits.
326    ///
327    /// The limits are part of the identity only when control state is observed: a scan
328    /// that reads no control file applies none, whatever [`Self::control_limits`] says.
329    pub fn control_identity(&self) -> ControlTierIdentity {
330        if self.read_controls {
331            ControlTierIdentity::Observed { limits: self.control_limits }
332        } else {
333            ControlTierIdentity::NotObserved
334        }
335    }
336
337    /// The identity of every tier a snapshot of this scan holds.
338    pub fn snapshot_identity(&self) -> SnapshotIdentity {
339        SnapshotIdentity {
340            entries: EntryTierIdentity {
341                engine: crate::snapshot::engine_fingerprint(),
342                scope: self.entry_scope(),
343                type_rules_fingerprint: self.types().fingerprint(),
344                reducers_fingerprint: REDUCERS_FINGERPRINT,
345            },
346            controls: self.control_identity(),
347        }
348    }
349
350    /// Resolve [`Self::threads`] to the workers active when a scan begins.
351    #[cfg(any(target_os = "macos", test))]
352    fn worker_threads(&self) -> usize {
353        self.worker_pool().initial
354    }
355
356    /// Resolve the worker count for immutable-baseline reconciliation waves.
357    fn reconciliation_worker_threads(&self) -> usize {
358        match self.threads {
359            Some(threads) => threads.clamp(1, MAX_SCAN_THREADS),
360            None => std::thread::available_parallelism()
361                .map_or(1, std::num::NonZero::get)
362                .clamp(1, DEFAULT_RECONCILE_THREADS_CAP),
363        }
364    }
365
366    /// Resolve the initial and maximum worker counts for one scan.
367    #[cfg(any(target_os = "macos", test))]
368    fn worker_pool(&self) -> WorkerPool {
369        self.worker_pool_for(std::thread::available_parallelism().map_or(1, std::num::NonZero::get))
370    }
371
372    /// Resolve the worker pool from one captured operating-system parallelism value.
373    fn worker_pool_for(&self, available_parallelism: usize) -> WorkerPool {
374        match self.threads {
375            Some(threads) => WorkerPool::fixed(threads.clamp(1, MAX_SCAN_THREADS)),
376            None => automatic_worker_pool(available_parallelism),
377        }
378    }
379
380    /// The scope axis this build cannot honour, if any.
381    ///
382    /// The one statement of the capability rule, so it is asked rather than restated.
383    /// [`Request::validate`](crate::query::Request::validate) asks it before any stored
384    /// state is read, which is what makes a scope this build cannot honour refuse the same
385    /// way on every route, every cache policy, and both surfaces; [`Self::validate`] asks
386    /// it for the engine-internal callers -- a bound root, a raw scan, an observation --
387    /// that never carry a request.
388    pub(crate) const fn unsupported_axis(&self) -> Option<ScopeAxis> {
389        if self.follow_symlinks {
390            return Some(ScopeAxis::FollowSymlinks);
391        }
392        #[cfg(not(unix))]
393        if self.one_filesystem {
394            return Some(ScopeAxis::OneFilesystem);
395        }
396        None
397    }
398
399    pub(crate) fn validate(&self) -> Result<()> {
400        if self.batch_size == 0 || self.batch_size > MAX_SCAN_BATCH_SIZE {
401            return Err(Error::UnsupportedScanConfig(
402                "batch_size must be nonzero and no greater than MAX_SCAN_BATCH_SIZE",
403            ));
404        }
405        if let Some(axis) = self.unsupported_axis() {
406            return Err(Error::UnsupportedScanConfig(axis.reason()));
407        }
408        Ok(())
409    }
410
411    pub(crate) fn validate_for_scope(&self, indexed: ScanScope) -> Result<()> {
412        self.validate()?;
413        let requested = self.scope();
414        if indexed != requested {
415            return Err(Error::ScanScopeMismatch { indexed, requested });
416        }
417        Ok(())
418    }
419
420    /// Scope equality, plus the boundary a watcher cannot filter its backend's events
421    /// against.
422    ///
423    /// The rule belongs to the request model, which refuses a watch of a narrowed scope
424    /// before anything is opened ([`RequestError::WatchScope`](crate::query::RequestError));
425    /// this is the same rule where a watcher is bound without a request -- an opened root
426    /// that observes, and each batch the adapter applies -- and it renders the one
427    /// guidance string the model renders.
428    #[cfg(feature = "watch")]
429    pub(crate) fn validate_for_watch_scope(&self, indexed: ScanScope) -> Result<()> {
430        self.validate_for_scope(indexed)?;
431        if self.max_depth.is_some() || self.one_filesystem {
432            return Err(Error::UnsupportedScanConfig(WATCH_SCOPE_GUIDANCE));
433        }
434        Ok(())
435    }
436}
437
438impl Default for ScanScope {
439    fn default() -> Self {
440        ScanConfig::default().scope()
441    }
442}
443
444/// What a scan did, including the errors it walked past.
445///
446/// Unreadable directories are skipped rather than aborting the scan — a permission-denied
447/// subdirectory should not cost you the other 499,000 files — but they are reported
448/// rather than swallowed, so a caller can tell a complete answer from a partial one.
449#[derive(Debug, Default)]
450pub struct ScanReport {
451    /// Directories successfully listed.
452    pub dirs_read: u64,
453    /// Entries observed, directories included.
454    pub entries: u64,
455    /// Regular files whose metadata was observed.
456    pub files_walked: u64,
457    /// Apparent bytes represented by the regular files whose metadata was observed.
458    pub bytes_walked: u64,
459    /// Allocated bytes of those files: what the default size metric counts, and what a
460    /// sparse disk image or a clone makes far smaller than their apparent bytes.
461    pub allocated_walked: u64,
462    /// Paths that could not be read, with the reason.
463    pub errors: Vec<Error>,
464    /// Where the walk's time went, summed across workers.
465    pub attribution: WalkAttribution,
466}
467
468impl ScanReport {
469    /// True when every directory in scope was read successfully.
470    pub fn is_complete(&self) -> bool {
471        self.errors.is_empty()
472    }
473
474    /// Fold one worker's share of a parallel walk into the whole-walk report.
475    fn absorb(&mut self, other: Self) {
476        self.dirs_read += other.dirs_read;
477        self.entries += other.entries;
478        self.files_walked += other.files_walked;
479        self.bytes_walked += other.bytes_walked;
480        self.allocated_walked += other.allocated_walked;
481        self.errors.extend(other.errors);
482        self.attribution.absorb(other.attribution);
483    }
484
485    /// Record one successfully stated directory entry.
486    fn observe(&mut self, kind: EntryKind, attrs: Attrs) {
487        self.entries += 1;
488        if kind == EntryKind::File {
489            self.files_walked += 1;
490            self.bytes_walked += attrs.size;
491            self.allocated_walked += attrs.allocated;
492        }
493    }
494}
495
496/// One walker's running share of the progress counters.
497///
498/// Each walker keeps its own [`ScanReport`]; this remembers how much of that report it
499/// has already added to the shared [`crate::Progress`] cells, so each addition is the
500/// difference since the last. It lives on the worker's stack beside the report rather
501/// than inside it, so a report absorbed into another never carries a stale baseline.
502struct ProgressTally<'a> {
503    progress: Option<&'a crate::Progress>,
504    directories: u64,
505    files: u64,
506    bytes: u64,
507    allocated: u64,
508}
509
510impl<'a> ProgressTally<'a> {
511    const fn new(progress: Option<&'a crate::Progress>) -> Self {
512        Self { progress, directories: 0, files: 0, bytes: 0, allocated: 0 }
513    }
514
515    /// Add what `report` has counted since the last call.
516    ///
517    /// The one `Option` check is the whole cost when no handle is attached. Called once
518    /// per chunk of directories a walker hands over, never per entry.
519    fn flush(&mut self, report: &ScanReport) {
520        let Some(progress) = self.progress else { return };
521        let directories = report.dirs_read - self.directories;
522        let files = report.files_walked - self.files;
523        let bytes = report.bytes_walked - self.bytes;
524        let allocated = report.allocated_walked - self.allocated;
525        if directories != 0 || files != 0 || bytes != 0 || allocated != 0 {
526            progress.add_walked(directories, files, bytes, allocated);
527            self.directories = report.dirs_read;
528            self.files = report.files_walked;
529            self.bytes = report.bytes_walked;
530            self.allocated = report.allocated_walked;
531        }
532    }
533
534    /// Treat everything `report` holds as already added.
535    ///
536    /// For a walker that continues a report whose counts other workers added themselves.
537    fn skip_to(&mut self, report: &ScanReport) {
538        self.directories = report.dirs_read;
539        self.files = report.files_walked;
540        self.bytes = report.bytes_walked;
541        self.allocated = report.allocated_walked;
542    }
543}
544
545/// Normalize filesystem failures before one of the bounded status collectors retains them.
546///
547/// A walk may encounter the same inaccessible path from several worker paths. The report is
548/// already the full, transient set for this pass, so sorting and deduplicating it here avoids
549/// allocating or formatting a second unbounded set solely to decide which 64 details survive.
550/// I/O causes are keyed by their native root-relative path and the issue category, exactly the
551/// cause identity retained by an index. Other engine failures are left distinct: walker errors
552/// are I/O failures, and treating arbitrary engine errors as equivalent without constructing
553/// their bounded issue representation would lose information.
554pub(crate) fn normalize_walk_errors(root: &Path, errors: &mut Vec<Error>) {
555    errors.sort_by(|left, right| match (left, right) {
556        (
557            Error::Io { path: left_path, source: left_source },
558            Error::Io { path: right_path, source: right_source },
559        ) => left_path
560            .strip_prefix(root)
561            .unwrap_or(left_path)
562            .cmp(right_path.strip_prefix(root).unwrap_or(right_path))
563            .then_with(|| {
564                walk_issue_kind_rank(left_source).cmp(&walk_issue_kind_rank(right_source))
565            }),
566        (Error::Io { .. }, _) => std::cmp::Ordering::Less,
567        (_, Error::Io { .. }) => std::cmp::Ordering::Greater,
568        _ => std::cmp::Ordering::Equal,
569    });
570    errors.dedup_by(|right, left| match (left, right) {
571        (
572            Error::Io { path: left_path, source: left_source },
573            Error::Io { path: right_path, source: right_source },
574        ) => {
575            left_path.strip_prefix(root).unwrap_or(left_path)
576                == right_path.strip_prefix(root).unwrap_or(right_path)
577                && walk_issue_kind_rank(left_source) == walk_issue_kind_rank(right_source)
578        }
579        _ => false,
580    });
581}
582
583fn walk_issue_kind_rank(error: &std::io::Error) -> u8 {
584    match error.kind() {
585        std::io::ErrorKind::PermissionDenied => 0,
586        std::io::ErrorKind::NotFound => 1,
587        std::io::ErrorKind::InvalidData | std::io::ErrorKind::InvalidInput => 2,
588        _ => 5,
589    }
590}
591
592/// Schema carried by [`ScanDiagnostics`].
593///
594/// Diagnostics are an opt-in measurement contract rather than stable human output.
595/// Consumers must reject an unknown schema instead of guessing that fields retained
596/// their meaning.
597pub const SCAN_DIAGNOSTICS_SCHEMA: &str = "fdu-scan-diagnostics-v1";
598
599/// Maximum policy-window records retained by one diagnostic scan.
600///
601/// The bound is on controller evaluations, not filesystem entries. A controller that
602/// needs more history must mark the artifact truncated; claim-grade consumers reject
603/// that artifact rather than silently analyzing an incomplete policy history.
604const MAX_POLICY_TRACE_EVENTS: usize = 256;
605
606/// Opt-in, run-scoped evidence about a filesystem scan.
607///
608/// Obtain this through [`scan_with_diagnostics`] or
609/// [`scan_into_index_with_diagnostics`]. Keeping it out of [`ScanReport`] preserves the
610/// existing scan API and keeps ordinary callers off the measurement path entirely.
611#[derive(Clone, Debug, PartialEq, Eq)]
612pub struct ScanDiagnostics {
613    /// Version of this diagnostic contract.
614    pub schema: &'static str,
615    /// Automatic worker-controller history and queue state.
616    pub worker_policy: WorkerPolicyDiagnostics,
617    /// Directory-enumeration backends used by this run.
618    pub backend: ScanBackendDiagnostics,
619}
620
621/// Repository-only controller variants used by the performance evidence probe.
622///
623/// These variants are not selected by [`scan`] or [`scan_with_diagnostics`]; both keep
624/// the shipped one-shot policy. The explicit experimental APIs make candidate behavior
625/// measurable without hiding a production change behind an environment variable.
626#[doc(hidden)]
627#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
628pub enum WorkerPolicyExperiment {
629    /// The production controller: one prefix window and at most one expansion.
630    #[default]
631    ShippedOneShot,
632    /// Re-evaluate independent windows until a slow phase requests the full reserve.
633    RepeatedWindows,
634    /// Re-evaluate independent windows, gate on useful frontier/handoff backlog, and grow
635    /// the pool in stages.
636    StagedGatedWindows,
637}
638
639/// Final state of the automatic worker controller.
640#[derive(Clone, Copy, Debug, PartialEq, Eq)]
641pub enum WorkerPolicyOutcome {
642    /// The scan did no walking, for example because `max_depth` was zero.
643    NotRun,
644    /// A fixed pool had no adaptive decision to make.
645    Fixed,
646    /// The walk ended before an adaptive window became observable.
647    Undecided,
648    /// The controller measured a window and retained the initial pool.
649    Held,
650    /// The controller requested and activated the reserve workers.
651    ScaledUp,
652    /// Slow work was observed only after no useful queued or in-flight work remained.
653    HeldNoUsefulWork,
654}
655
656/// One controller evaluation over a half-open range of completed entry ordinals.
657#[derive(Clone, Debug, PartialEq, Eq)]
658pub struct WorkerPolicyWindow {
659    /// Monotonic record number within this scan.
660    pub sequence: u64,
661    /// First completed-entry ordinal represented by this window, inclusive.
662    pub start_entry_ordinal: u64,
663    /// Ordinal immediately after the last represented entry.
664    pub end_entry_ordinal: u64,
665    /// Entries contributing to the service-time signal.
666    pub observed_entries: u64,
667    /// Completed directory claims contributing to the service-time signal.
668    pub observed_chunks: u64,
669    /// Worker time contributing to the service-time signal.
670    pub observed_work_ns: u64,
671    /// Derived service time, or null when no entry made the signal observable.
672    pub work_ns_per_entry: Option<u64>,
673    /// Why `work_ns_per_entry` is null.
674    pub work_ns_per_entry_unavailable_reason: Option<&'static str>,
675    /// Directories ready to claim when the controller evaluated the window.
676    pub ready_directories: usize,
677    /// Claimed directories still being processed at that point.
678    pub in_flight_directories: usize,
679    /// Live worker threads at that point, including workers waiting for a claim.
680    pub active_workers: usize,
681    /// Observation batches sent but not yet received by the consumer.
682    pub handoff_backlog: usize,
683    /// Worker target requested by a scale decision.
684    pub requested_workers: Option<usize>,
685    /// What the controller concluded from this window.
686    pub decision: WorkerPolicyDecision,
687}
688
689/// Decision represented by a [`WorkerPolicyWindow`].
690#[derive(Clone, Copy, Debug, PartialEq, Eq)]
691pub enum WorkerPolicyDecision {
692    /// The walk ended before the window could support a decision.
693    Undecided,
694    /// The observed window retained the current pool.
695    Hold,
696    /// The observed window activated reserve workers.
697    ScaleUp,
698    /// The trigger fired after all useful work had drained.
699    HoldNoUsefulWork,
700    /// A complete window held because reserve workers had no useful frontier to claim.
701    HoldInsufficientFrontier,
702    /// A complete window held because the unbounded handoff backlog was already high.
703    HoldHandoffBacklog,
704    /// A post-decision observation window remained below the slow threshold.
705    ObserveFast,
706    /// A post-decision observation window met the slow threshold.
707    ObserveSlow,
708    /// A trailing partial window carried no new terminal decision.
709    Incomplete,
710    /// A trailing partial post-decision observation carried no policy decision.
711    ObserveIncomplete,
712}
713
714/// Worker-controller configuration, trace, and terminal queue state.
715#[derive(Clone, Debug, PartialEq, Eq)]
716pub struct WorkerPolicyDiagnostics {
717    /// Controller variant exercised by this scan.
718    pub controller: &'static str,
719    /// Parallelism reported by the operating system when the scan began.
720    pub available_parallelism: usize,
721    /// Workers in the pool before any adaptive decision.
722    pub initial_workers: usize,
723    /// Hard maximum workers this scan could activate.
724    pub maximum_workers: usize,
725    /// Entry target for an adaptive window, or null for a fixed pool.
726    pub calibration_window_entries: Option<u64>,
727    /// Slow-service trigger, or null for a fixed pool.
728    pub slow_threshold_ns_per_entry: Option<u64>,
729    /// Directory chunks folded into live controller windows.
730    pub calibration_chunks: u64,
731    /// Entries folded into live controller windows.
732    pub calibration_entries: u64,
733    /// Worker time folded into live controller windows.
734    pub calibration_work_ns: u64,
735    /// Expansion messages that caused the consumer to create more workers.
736    pub worker_expansions: u64,
737    /// Terminal policy outcome.
738    pub outcome: WorkerPolicyOutcome,
739    /// Explanation when no adaptive evaluation exists.
740    pub outcome_reason: Option<&'static str>,
741    /// Total worker threads created during the walk.
742    pub workers_spawned: usize,
743    /// Maximum simultaneously live worker threads, including workers waiting for work.
744    pub peak_active_workers: usize,
745    /// Ready directories at scan completion; a complete walk must leave zero.
746    pub ready_directories_at_finish: usize,
747    /// In-flight directories at scan completion; a complete walk must leave zero.
748    pub in_flight_directories_at_finish: usize,
749    /// Observation batches outstanding at scan completion.
750    pub handoff_backlog_at_finish: usize,
751    /// Maximum outstanding observation batches during the scan.
752    pub handoff_backlog_high_water: usize,
753    /// Bounded controller history.
754    pub windows: Vec<WorkerPolicyWindow>,
755    /// True when controller history exceeded the 256-event diagnostic bound.
756    pub events_truncated: bool,
757}
758
759/// Directory enumeration backends used by one scan.
760#[derive(Clone, Debug, PartialEq, Eq)]
761pub struct ScanBackendDiagnostics {
762    /// Portable `read_dir` calls attempted.
763    pub portable_attempts: u64,
764    /// Portable directory listings completed successfully.
765    pub portable_directory_reads: u64,
766    /// macOS bulk enumeration attempts, or null off macOS.
767    pub macos_bulk_attempts: Option<u64>,
768    /// Successful macOS bulk listings, or null off macOS.
769    pub macos_bulk_successes: Option<u64>,
770    /// Bulk attempts that fell back to portable enumeration, or null off macOS.
771    pub macos_bulk_fallbacks: Option<u64>,
772    /// Why macOS fields are null.
773    pub unavailable_reason: Option<&'static str>,
774}
775
776impl ScanDiagnostics {
777    /// Serialize this versioned diagnostic contract as compact JSON.
778    ///
779    /// This deliberately lives beside the contract instead of in a benchmark binary:
780    /// claim-grade installed-command measurements and the repository probe must emit
781    /// byte-for-byte equivalent evidence without adding a serialization dependency to
782    /// the core crate.
783    pub fn to_json(&self) -> String {
784        let policy = &self.worker_policy;
785        let backend = &self.backend;
786        let mut windows = String::from("[");
787        for (index, window) in policy.windows.iter().enumerate() {
788            if index > 0 {
789                windows.push(',');
790            }
791            let _ = write!(
792                windows,
793                concat!(
794                    "{{\"active_workers\":{},\"decision\":\"{}\",",
795                    "\"end_entry_ordinal\":{},\"handoff_backlog\":{},",
796                    "\"in_flight_directories\":{},\"observed_chunks\":{},",
797                    "\"observed_entries\":{},",
798                    "\"observed_work_ns\":{},\"ready_directories\":{},",
799                    "\"requested_workers\":{},\"sequence\":{},\"start_entry_ordinal\":{},",
800                    "\"work_ns_per_entry\":{},",
801                    "\"work_ns_per_entry_unavailable_reason\":{}}}"
802                ),
803                window.active_workers,
804                worker_policy_decision_name(window.decision),
805                window.end_entry_ordinal,
806                window.handoff_backlog,
807                window.in_flight_directories,
808                window.observed_chunks,
809                window.observed_entries,
810                window.observed_work_ns,
811                window.ready_directories,
812                json_optional_usize(window.requested_workers),
813                window.sequence,
814                window.start_entry_ordinal,
815                json_optional_u64(window.work_ns_per_entry),
816                json_optional_string(window.work_ns_per_entry_unavailable_reason),
817            );
818        }
819        windows.push(']');
820        format!(
821            concat!(
822                "{{\"backend\":{{\"macos_bulk_attempts\":{},",
823                "\"macos_bulk_fallbacks\":{},\"macos_bulk_successes\":{},",
824                "\"portable_attempts\":{},\"portable_directory_reads\":{},",
825                "\"unavailable_reason\":{}}},",
826                "\"schema\":\"{}\",\"worker_policy\":{{",
827                "\"available_parallelism\":{},\"calibration_chunks\":{},",
828                "\"calibration_entries\":{},\"calibration_window_entries\":{},",
829                "\"calibration_work_ns\":{},",
830                "\"controller\":\"{}\",",
831                "\"events_truncated\":{},\"handoff_backlog_at_finish\":{},",
832                "\"handoff_backlog_high_water\":{},\"in_flight_directories_at_finish\":{},",
833                "\"initial_workers\":{},\"maximum_workers\":{},\"outcome\":\"{}\",",
834                "\"outcome_reason\":{},\"peak_active_workers\":{},",
835                "\"ready_directories_at_finish\":{},\"slow_threshold_ns_per_entry\":{},",
836                "\"windows\":{},\"worker_expansions\":{},\"workers_spawned\":{}}}}}"
837            ),
838            json_optional_u64(backend.macos_bulk_attempts),
839            json_optional_u64(backend.macos_bulk_fallbacks),
840            json_optional_u64(backend.macos_bulk_successes),
841            backend.portable_attempts,
842            backend.portable_directory_reads,
843            json_optional_string(backend.unavailable_reason),
844            self.schema,
845            policy.available_parallelism,
846            policy.calibration_chunks,
847            policy.calibration_entries,
848            json_optional_u64(policy.calibration_window_entries),
849            policy.calibration_work_ns,
850            policy.controller,
851            policy.events_truncated,
852            policy.handoff_backlog_at_finish,
853            policy.handoff_backlog_high_water,
854            policy.in_flight_directories_at_finish,
855            policy.initial_workers,
856            policy.maximum_workers,
857            worker_policy_outcome_name(policy.outcome),
858            json_optional_string(policy.outcome_reason),
859            policy.peak_active_workers,
860            policy.ready_directories_at_finish,
861            json_optional_u64(policy.slow_threshold_ns_per_entry),
862            windows,
863            policy.worker_expansions,
864            policy.workers_spawned,
865        )
866    }
867}
868
869const fn worker_policy_outcome_name(value: WorkerPolicyOutcome) -> &'static str {
870    match value {
871        WorkerPolicyOutcome::NotRun => "not_run",
872        WorkerPolicyOutcome::Fixed => "fixed",
873        WorkerPolicyOutcome::Undecided => "undecided",
874        WorkerPolicyOutcome::Held => "held",
875        WorkerPolicyOutcome::ScaledUp => "scaled_up",
876        WorkerPolicyOutcome::HeldNoUsefulWork => "held_no_useful_work",
877    }
878}
879
880const fn worker_policy_decision_name(value: WorkerPolicyDecision) -> &'static str {
881    match value {
882        WorkerPolicyDecision::Undecided => "undecided",
883        WorkerPolicyDecision::Hold => "hold",
884        WorkerPolicyDecision::ScaleUp => "scale_up",
885        WorkerPolicyDecision::HoldNoUsefulWork => "hold_no_useful_work",
886        WorkerPolicyDecision::HoldInsufficientFrontier => "hold_insufficient_frontier",
887        WorkerPolicyDecision::HoldHandoffBacklog => "hold_handoff_backlog",
888        WorkerPolicyDecision::ObserveFast => "observe_fast",
889        WorkerPolicyDecision::ObserveSlow => "observe_slow",
890        WorkerPolicyDecision::Incomplete => "incomplete",
891        WorkerPolicyDecision::ObserveIncomplete => "observe_incomplete",
892    }
893}
894
895fn json_optional_string(value: Option<&str>) -> String {
896    value.map_or_else(|| "null".into(), |value| format!("\"{}\"", json_escape(value)))
897}
898
899fn json_optional_u64(value: Option<u64>) -> String {
900    value.map_or_else(|| "null".into(), |value| value.to_string())
901}
902
903fn json_optional_usize(value: Option<usize>) -> String {
904    value.map_or_else(|| "null".into(), |value| value.to_string())
905}
906
907fn json_escape(value: &str) -> String {
908    let mut escaped = String::new();
909    for character in value.chars() {
910        match character {
911            '"' => escaped.push_str("\\\""),
912            '\\' => escaped.push_str("\\\\"),
913            '\u{08}' => escaped.push_str("\\b"),
914            '\u{0c}' => escaped.push_str("\\f"),
915            '\n' => escaped.push_str("\\n"),
916            '\r' => escaped.push_str("\\r"),
917            '\t' => escaped.push_str("\\t"),
918            character if character <= '\u{1f}' => {
919                let _ = write!(escaped, "\\u{:04x}", u32::from(character));
920            }
921            character => escaped.push(character),
922        }
923    }
924    escaped
925}
926
927/// Where a walk's time went, so "blocked" is never one undifferentiated number.
928///
929/// The performance loop's standing question is whether a walk is bound by disk I/O,
930/// by CPU, or by coordination, and process-level counters cannot answer it: user and
931/// system time say how much CPU was burned, but a fused "blocked" number cannot say
932/// whether workers were waiting on the filesystem, on the queue lock, or on nothing
933/// at all because the queue was empty. These counters split that out at the source.
934///
935/// Everything is measured in *chunks*, never per file: one timing pair per claimed
936/// run of directories, per contended lock, per batch handoff. On the 60k-entry
937/// reference tree that is a few thousand `Instant` reads against hundreds of
938/// milliseconds of walking — the instrumentation follows the same amortization rule
939/// it exists to verify.
940///
941/// In a parallel walk the fields sum over workers, so `wall_ns` is worker-seconds
942/// (it can exceed the scan's wall clock) and every other duration is a disjoint
943/// slice of it: `work_ns + starved_ns + lock_wait_ns + send_ns <= wall_ns`, with the
944/// remainder being uninstrumented odds and ends (uncontended lock ops, loop
945/// bookkeeping). A serial walk fills only `wall_ns`, `work_ns`, and `send_ns` —
946/// there is no coordination to attribute.
947#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
948pub struct WalkAttribution {
949    /// Total time workers spent in the walk loop, summed across workers.
950    pub wall_ns: u64,
951    /// Reading directories and stating entries — the real work, syscalls plus the
952    /// compute between them. Separating disk from CPU *within* this span needs the
953    /// process-level user/system counters alongside; per-syscall timing would break
954    /// the chunk-amortization rule.
955    pub work_ns: u64,
956    /// Waiting on the queue's condvar because no work was available. Starvation:
957    /// either the frontier is momentarily narrower than the worker pool, or the walk
958    /// is ending.
959    pub starved_ns: u64,
960    /// Waiting to acquire the queue lock when another worker held it. This is the
961    /// contention the shared-queue design bets stays negligible; now it is measured
962    /// instead of argued.
963    pub lock_wait_ns: u64,
964    /// Handing observation batches to the consumer: the channel send in a parallel
965    /// walk, the inline sink call — which is the consumer actually running — in a
966    /// serial one.
967    pub send_ns: u64,
968    /// Chunks of directories claimed from the queue.
969    pub claims: u64,
970    /// Queue lock acquisitions, contended or not.
971    pub lock_ops: u64,
972    /// Lock acquisitions that found the lock already held.
973    pub lock_contended: u64,
974}
975
976impl WalkAttribution {
977    /// Fold one worker's counters into the whole-walk totals.
978    fn absorb(&mut self, other: Self) {
979        self.wall_ns += other.wall_ns;
980        self.work_ns += other.work_ns;
981        self.starved_ns += other.starved_ns;
982        self.lock_wait_ns += other.lock_wait_ns;
983        self.send_ns += other.send_ns;
984        self.claims += other.claims;
985        self.lock_ops += other.lock_ops;
986        self.lock_contended += other.lock_contended;
987    }
988
989    /// Time attributed to a named cause, as opposed to `wall_ns`'s total.
990    pub fn accounted_ns(&self) -> u64 {
991        self.work_ns + self.starved_ns + self.lock_wait_ns + self.send_ns
992    }
993}
994
995/// Filesystem and index effects from an applying reconciliation pass.
996#[derive(Debug, Default)]
997pub struct ReconcileReport {
998    /// Filesystem walk effects and partial errors.
999    ///
1000    /// [`ScanReport::attribution`] remains zero for reconciliation because neither the
1001    /// serial nor parallel path has complete, comparable instrumentation yet. Zero
1002    /// means "not measured" here, not "no work".
1003    pub scan: ScanReport,
1004    /// Index arbitration and mutation effects.
1005    pub apply: ApplyStats,
1006    /// Exact producer operations considered, including no-op controls that do not
1007    /// increment an effect counter or create a commit.
1008    pub(crate) observations: u64,
1009    /// Directories this pass listed in full, with no error inside them, that the index did
1010    /// not yet hold as complete.
1011    ///
1012    /// The closing commit records each one's child set as authoritative, as discovery's
1013    /// own listing commit does, whether or not the rest of the pass completed: one transient
1014    /// child error
1015    /// elsewhere used to keep every directory the pass listed incomplete, and a directory
1016    /// first listed by such a pass stayed `Unknown { Building }` under a complete root.
1017    pub(crate) listed_incomplete: Vec<PathBuf>,
1018    /// Ownership epoch for conditional reconciliation batches.
1019    reconcile_epoch: Option<u64>,
1020    /// Retry after bounded verification evidence was superseded.
1021    retry_required: bool,
1022}
1023
1024impl ReconcileReport {
1025    /// True when the filesystem walk was complete and no conditional observation lost
1026    /// a race with another producer.
1027    pub fn is_complete(&self) -> bool {
1028        self.scan.is_complete()
1029            && self.apply.stale == 0
1030            && self.apply.resource_refused == 0
1031            && !self.retry_required
1032    }
1033
1034    /// True when a newer verification retired this pass's bounded evidence before it
1035    /// closed, so its scope is published partial and must be walked again.
1036    pub(crate) const fn retry_required(&self) -> bool {
1037        self.retry_required
1038    }
1039
1040    /// The directories whose listings this pass can vouch for, taken out of the report.
1041    ///
1042    /// None when a conditional commit lost a race or was refused: a child of any listed
1043    /// directory may then be missing from the index until the retry that race earns, and
1044    /// the retry records completeness for what it lists.
1045    pub(crate) fn take_recordable_completeness(&mut self) -> Vec<PathBuf> {
1046        let listed = std::mem::take(&mut self.listed_incomplete);
1047        if self.apply.stale > 0 || self.apply.resource_refused > 0 { Vec::new() } else { listed }
1048    }
1049}
1050
1051enum ReconcileTarget<'a> {
1052    Direct(&'a mut Index),
1053    Shared(&'a IndexHandle),
1054    Controlled { handle: &'a IndexHandle, control: &'a dyn ReconcileControl },
1055}
1056
1057/// Lifecycle checkpoints used by an owned long-running reconciliation.
1058///
1059/// The ordinary one-shot APIs use no controller. An [`crate::OpenedIndex`] supplies one
1060/// so close can stop a refresh before another write, and deterministic tests can pause
1061/// after filesystem verification but before conditional arbitration.
1062pub(crate) trait ReconcileControl {
1063    /// Fail when the owning operation may no longer publish state.
1064    fn check_active(&self) -> Result<()>;
1065
1066    /// Boundary after filesystem verification and before a conditional fact commit.
1067    fn before_conditional_commit(&self) -> Result<()>;
1068
1069    /// Atomic file-retention limit shared with every producer for this opened root.
1070    fn max_files(&self) -> Option<u64>;
1071}
1072
1073impl ReconcileTarget<'_> {
1074    fn scope(&self) -> Result<ScanScope> {
1075        match self {
1076            Self::Direct(index) => Ok(index.scope()),
1077            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.scope(),
1078        }
1079    }
1080
1081    fn root_path(&self) -> Result<PathBuf> {
1082        match self {
1083            Self::Direct(index) => Ok(index.root_path().to_path_buf()),
1084            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.root_path(),
1085        }
1086    }
1087
1088    fn expectation(&self, path: &Path) -> Result<PathExpectation> {
1089        match self {
1090            Self::Direct(index) => Ok(index.expectation(path)),
1091            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.expectation(path),
1092        }
1093    }
1094
1095    fn child_states(&self, path: &Path) -> Result<BTreeMap<OsString, PathExpectation>> {
1096        match self {
1097            Self::Direct(index) => Ok(collect_child_expectations(index, path)),
1098            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.child_states(path),
1099        }
1100    }
1101
1102    /// Child baselines for one directory listing, and whether a complete listing of it
1103    /// would be news to the index's directory completeness.
1104    ///
1105    /// A directory whose upsert has not been flushed yet is not held at all and counts as
1106    /// incomplete.
1107    fn listing_baseline(&self, path: &Path) -> Result<(BTreeMap<OsString, PathExpectation>, bool)> {
1108        match self {
1109            Self::Direct(index) => Ok((
1110                collect_child_expectations(index, path),
1111                index.directory_complete(path) != Some(true),
1112            )),
1113            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.listing_baseline(path),
1114        }
1115    }
1116
1117    fn has_control(&self, path: &Path) -> Result<bool> {
1118        match self {
1119            Self::Direct(index) => Ok(index.control_table().contains(path)),
1120            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.has_control(path),
1121        }
1122    }
1123
1124    fn apply(&mut self, started_at: u64, observation: &Observation) -> Result<crate::ApplyOutcome> {
1125        match self {
1126            Self::Direct(index) => index.apply(observation),
1127            Self::Shared(handle) => handle.apply_reconcile(started_at, observation),
1128            Self::Controlled { handle, control } => {
1129                control.before_conditional_commit()?;
1130                handle.apply_opened_reconcile(started_at, observation, control.max_files())
1131            }
1132        }
1133    }
1134
1135    fn direct_upsert_is_unchanged(
1136        &self,
1137        baseline: PathExpectation,
1138        kind: EntryKind,
1139        attrs: Attrs,
1140    ) -> bool {
1141        matches!(self, Self::Direct(_)) && baseline.state == (PathState::Present { kind, attrs })
1142    }
1143
1144    fn take_pending_invalidations(&mut self) -> Result<Vec<(PathBuf, crate::InvalidateReason)>> {
1145        match self {
1146            Self::Direct(index) => Ok(index.take_pending_invalidations()),
1147            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1148                handle.take_pending_invalidations()
1149            }
1150        }
1151    }
1152
1153    fn restore_pending_invalidations(
1154        &mut self,
1155        invalidations: Vec<(PathBuf, crate::InvalidateReason)>,
1156    ) -> Result<()> {
1157        match self {
1158            Self::Direct(index) => index.restore_pending_invalidations(invalidations),
1159            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1160                handle.restore_pending_invalidations(invalidations)?;
1161            }
1162        }
1163        Ok(())
1164    }
1165
1166    /// Whether an invalidation whose reconciliation came back incomplete is queued again.
1167    ///
1168    /// A caller of the one-shot API owns its index exclusively and drains the queue when it
1169    /// chooses, so an unreadable subtree stays queued for it to retry. The shared API and an
1170    /// opened root are drained after every observed event -- by `Watcher::apply_next` and by
1171    /// the opened root's observer -- where that retry is a full walk of the same unreadable
1172    /// subtree per unrelated event, for the life of the session. There only a lost race is
1173    /// worth retrying: a stale conditional commit, or one the budget refused. A scan error
1174    /// is a settled boundary: the subtree stays partial, as it does at the observation
1175    /// handoff, and the report names the error once.
1176    fn retries_incomplete(&self, report: &ReconcileReport) -> bool {
1177        match self {
1178            Self::Direct(_) => !report.is_complete(),
1179            Self::Shared(_) | Self::Controlled { .. } => {
1180                report.apply.stale > 0 || report.apply.resource_refused > 0 || report.retry_required
1181            }
1182        }
1183    }
1184
1185    fn begin_reconcile(&mut self, path: &Path) -> Result<(u64, Option<Commit>)> {
1186        match self {
1187            Self::Direct(index) => index.begin_reconcile(path),
1188            Self::Shared(handle) => handle.begin_reconcile(path),
1189            Self::Controlled { handle, control } => {
1190                control.check_active()?;
1191                handle.begin_reconcile(path)
1192            }
1193        }
1194    }
1195
1196    fn finish_reconcile(
1197        &mut self,
1198        path: &Path,
1199        started_at: u64,
1200        complete: bool,
1201        listed_incomplete: &[PathBuf],
1202        failed_paths: &[PathBuf],
1203        errors: ReconcileErrors<'_>,
1204    ) -> Result<ReconcileFinish> {
1205        match self {
1206            Self::Direct(index) => index.finish_reconcile(
1207                path,
1208                started_at,
1209                complete,
1210                listed_incomplete,
1211                failed_paths,
1212                errors,
1213            ),
1214            Self::Shared(handle) => handle.finish_reconcile(
1215                path,
1216                started_at,
1217                complete,
1218                listed_incomplete,
1219                failed_paths,
1220                errors,
1221            ),
1222            Self::Controlled { handle, control } => {
1223                control.check_active()?;
1224                handle.finish_reconcile(
1225                    path,
1226                    started_at,
1227                    complete,
1228                    listed_incomplete,
1229                    failed_paths,
1230                    errors,
1231                )
1232            }
1233        }
1234    }
1235}
1236
1237#[cfg(unix)]
1238pub(crate) fn metadata_for_fingerprint(entry: &fs::DirEntry) -> std::io::Result<fs::Metadata> {
1239    crate::counters::bump(|c| c.stats += 1);
1240    entry.metadata()
1241}
1242
1243#[cfg(any(all(windows, test), not(any(unix, windows))))]
1244pub(crate) fn metadata_for_fingerprint(entry: &fs::DirEntry) -> std::io::Result<fs::Metadata> {
1245    crate::counters::bump(|c| c.stats += 1);
1246    // Windows serves DirEntry metadata from directory-enumeration data, which the
1247    // platform permits to be stale. Fingerprints need a fresh non-following query.
1248    fs::symlink_metadata(entry.path())
1249}
1250
1251#[cfg(test)]
1252type WalkHook = std::sync::Arc<dyn Fn(WalkHookPoint<'_>) -> Option<std::io::Error> + Send + Sync>;
1253
1254/// Where a test hook runs in a listing walk.
1255#[cfg(test)]
1256#[derive(Clone, Copy, Debug)]
1257pub(crate) enum WalkHookPoint<'a> {
1258    /// Before the metadata lookup of the listed child at this absolute path; an error
1259    /// stands in for the lookup's.
1260    ChildMetadata(&'a Path),
1261    /// After a reconciliation's listing of a directory returns its last entry; an error is
1262    /// read as one more listing item, which leaves the listing incomplete.
1263    ListingEnd,
1264}
1265
1266/// Hooks run at each [`WalkHookPoint`], each for the paths under its root.
1267///
1268/// Process-wide, because a parallel walk looks children up on its worker threads; keyed
1269/// by root, because tests run in parallel and each walks its own temporary directory.
1270#[cfg(test)]
1271static WALK_HOOKS: std::sync::RwLock<Vec<(Vec<PathBuf>, WalkHook)>> =
1272    std::sync::RwLock::new(Vec::new());
1273
1274/// Removes its hook from [`WALK_HOOKS`] when dropped.
1275#[cfg(test)]
1276#[must_use = "the hook is removed as soon as the guard is dropped"]
1277pub(crate) struct WalkHookGuard(WalkHook);
1278
1279#[cfg(test)]
1280impl Drop for WalkHookGuard {
1281    fn drop(&mut self) {
1282        WALK_HOOKS
1283            .write()
1284            .unwrap_or_else(std::sync::PoisonError::into_inner)
1285            .retain(|(_, hook)| !std::sync::Arc::ptr_eq(hook, &self.0));
1286    }
1287}
1288
1289/// Run `hook` at every [`WalkHookPoint`] under `root`, on any thread, until the guard drops.
1290///
1291/// The hook may also change the tree before it returns. `root` matches as given and
1292/// canonical, since an opened root and a detached scan walk the canonical path.
1293#[cfg(test)]
1294pub(crate) fn install_walk_hook(
1295    root: &Path,
1296    hook: impl Fn(WalkHookPoint<'_>) -> Option<std::io::Error> + Send + Sync + 'static,
1297) -> WalkHookGuard {
1298    let mut roots = vec![root.to_path_buf()];
1299    if let Ok(canonical) = root.canonicalize() {
1300        roots.push(canonical);
1301    }
1302    let hook: WalkHook = std::sync::Arc::new(hook);
1303    WALK_HOOKS
1304        .write()
1305        .unwrap_or_else(std::sync::PoisonError::into_inner)
1306        .push((roots, std::sync::Arc::clone(&hook)));
1307    WalkHookGuard(hook)
1308}
1309
1310/// Run `hook` before every listed child's metadata lookup under `root`, with the child's
1311/// absolute path, until the guard drops.
1312#[cfg(test)]
1313pub(crate) fn install_child_metadata_hook(
1314    root: &Path,
1315    hook: impl Fn(&Path) -> Option<std::io::Error> + Send + Sync + 'static,
1316) -> WalkHookGuard {
1317    install_walk_hook(root, move |point| match point {
1318        WalkHookPoint::ChildMetadata(path) => hook(path),
1319        WalkHookPoint::ListingEnd => None,
1320    })
1321}
1322
1323/// The hook installed for a root containing `path`, if any.
1324#[cfg(test)]
1325fn walk_hook(path: &Path) -> Option<WalkHook> {
1326    WALK_HOOKS
1327        .read()
1328        .unwrap_or_else(std::sync::PoisonError::into_inner)
1329        .iter()
1330        .find(|(roots, _)| roots.iter().any(|root| path.starts_with(root)))
1331        .map(|(_, hook)| std::sync::Arc::clone(hook))
1332}
1333
1334/// A reconciliation's listing of `dir`, followed by any error a test hook injects.
1335///
1336/// Callers bind the result to `listing` and iterate it as `for … in listing`, because the
1337/// admission audit (`scripts/check-admission-sites.mjs`) counts routed listing loops by that
1338/// shape. Keep the binding when editing a call site; dropping it silently removes the loop
1339/// from the audit, whose expected count would then look too high rather than wrong.
1340#[cfg(test)]
1341fn reconcile_listing(
1342    listing: fs::ReadDir,
1343    dir: &Path,
1344) -> impl Iterator<Item = std::io::Result<fs::DirEntry>> {
1345    let injected = walk_hook(dir).and_then(|hook| hook(WalkHookPoint::ListingEnd));
1346    listing.chain(injected.map(Err))
1347}
1348
1349/// A reconciliation's listing of `dir`.
1350#[cfg(not(test))]
1351fn reconcile_listing(listing: fs::ReadDir, _dir: &Path) -> fs::ReadDir {
1352    listing
1353}
1354
1355/// Metadata for one entry a directory listing returned, or `None` when it is gone.
1356///
1357/// `NotFound` for a name the listing just returned means the entry was deleted in
1358/// between, and every walk records it as it records a name the listing never returned: a
1359/// cold walk has nothing to record, and a reconciliation removes what its baseline held.
1360/// Reported as an error, it would make a walk over a tree being cleaned partial, and in a
1361/// reconciliation it would settle as a phantom entry with permanent partial freshness.
1362/// Any other error means the entry is present but unreadable.
1363#[cfg(not(windows))]
1364pub(crate) fn listed_child_metadata(entry: &fs::DirEntry) -> std::io::Result<Option<fs::Metadata>> {
1365    #[cfg(test)]
1366    {
1367        let path = entry.path();
1368        if let Some(error) =
1369            walk_hook(&path).and_then(|hook| hook(WalkHookPoint::ChildMetadata(&path)))
1370        {
1371            return missing_as_none(Err(error));
1372        }
1373    }
1374    missing_as_none(metadata_for_fingerprint(entry))
1375}
1376
1377/// Kind and attributes for one listed child.
1378///
1379/// The transient summary fold counts directories and ignores symlink attributes, so a
1380/// listing `file_type` (`d_type` on Linux) is enough for those kinds when the walk is
1381/// not bound to one filesystem. Files and specials still need a metadata lookup for
1382/// size, allocated bytes, and mtime. `one_filesystem` still stats directories because
1383/// descent compares `attrs.dev` to the root device, and `dev == 0` would otherwise
1384/// cross a mount.
1385///
1386/// Where `d_type` is `DT_UNKNOWN` (XFS without `ftype`, some FUSE/NFS mounts, older
1387/// ext3), std's `file_type` performs the non-following stat itself. The skip is a
1388/// no-op there, and the `stats` counter does not see that fallback.
1389///
1390/// Windows never takes the skip: its observation contract reads every listed entry
1391/// through a fresh non-following handle ([`observe_dir_entry`]), so the transient fold
1392/// there performs exactly the observations the retained walk performs.
1393fn listed_child_kind_and_attrs(
1394    entry: &fs::DirEntry,
1395    skip_dir_symlink_stat: bool,
1396    one_filesystem: bool,
1397) -> std::io::Result<Option<(EntryKind, Attrs)>> {
1398    #[cfg(not(windows))]
1399    {
1400        if skip_dir_symlink_stat {
1401            if let Ok(file_type) = entry.file_type() {
1402                if file_type.is_dir() && !one_filesystem {
1403                    return Ok(Some((EntryKind::Dir, Attrs::default())));
1404                }
1405                if file_type.is_symlink() {
1406                    return Ok(Some((EntryKind::Symlink, Attrs::default())));
1407                }
1408            }
1409        }
1410    }
1411    #[cfg(windows)]
1412    {
1413        let _ = (skip_dir_symlink_stat, one_filesystem);
1414    }
1415    observe_dir_entry(entry)
1416}
1417
1418fn missing_as_none<T>(lookup: std::io::Result<T>) -> std::io::Result<Option<T>> {
1419    match lookup {
1420        Ok(metadata) => Ok(Some(metadata)),
1421        Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(None),
1422        Err(error) => Err(error),
1423    }
1424}
1425
1426/// Whether a test hook observes lookups or listings under `path`, which a bulk read would
1427/// not make.
1428#[cfg(all(test, target_os = "macos"))]
1429fn walk_hook_covers(path: &Path) -> bool {
1430    walk_hook(path).is_some()
1431}
1432
1433#[cfg(all(not(test), target_os = "macos"))]
1434const fn walk_hook_covers(_path: &Path) -> bool {
1435    false
1436}
1437
1438/// Owned output from the filesystem walker before it crosses a public mutation boundary.
1439///
1440/// Only the scan and opened-discovery producers construct this type. Their admission,
1441/// depth, filesystem, and symlink checks have already selected every operation, and the
1442/// index consumes the owned paths while proving their parent identities under its write
1443/// boundary. Public scan callers receive an [`Observation`] instead and therefore keep
1444/// the full public normalization and atomic-validation contract.
1445#[derive(Debug)]
1446pub(crate) struct ScannerBatch {
1447    ops: Vec<ObservationOp>,
1448    /// When set, the consumer must return `ops` through this sender instead of dropping
1449    /// them. Workers allocate the `PathBuf`s; returning the drained vec lets glibc free
1450    /// those arenas on the producing thread. The public [`scan`] path leaves this unset.
1451    recycle: Option<std::sync::mpsc::Sender<Vec<ObservationOp>>>,
1452}
1453
1454impl ScannerBatch {
1455    pub(crate) const fn new(ops: Vec<ObservationOp>) -> Self {
1456        Self { ops, recycle: None }
1457    }
1458
1459    fn with_recycle(self, recycle: std::sync::mpsc::Sender<Vec<ObservationOp>>) -> Self {
1460        Self { recycle: Some(recycle), ..self }
1461    }
1462
1463    #[cfg(test)]
1464    pub(crate) fn from_ops(ops: Vec<Op>) -> Self {
1465        Self { ops: ops.into_iter().map(ObservationOp::unconditional).collect(), recycle: None }
1466    }
1467
1468    pub(crate) fn len(&self) -> usize {
1469        self.ops.len()
1470    }
1471
1472    pub(crate) fn ops(&self) -> &[ObservationOp] {
1473        &self.ops
1474    }
1475
1476    pub(crate) fn into_ops(self) -> Vec<ObservationOp> {
1477        self.ops
1478    }
1479
1480    fn into_observation(self) -> Observation {
1481        Observation::from_ops(self.ops)
1482    }
1483
1484    fn recycle(self) {
1485        if let Some(recycle) = self.recycle {
1486            let _ = recycle.send(self.ops);
1487        }
1488    }
1489}
1490
1491/// One direct child retained by the private detached cold-bootstrap builder.
1492///
1493/// The worker owns the component once. Unlike [`ScannerBatch`], this record does not
1494/// manufacture a full relative path or a public observation for every entry.
1495#[derive(Debug)]
1496pub(crate) struct DetachedChild {
1497    pub(crate) name: OsString,
1498    pub(crate) kind: EntryKind,
1499    pub(crate) attrs: Attrs,
1500    /// Enumeration order within the listing. An enumerator can repeat a name while its
1501    /// directory is modified, and the builder keeps the later observation, as a
1502    /// streaming re-upsert does.
1503    pub(crate) position: u32,
1504}
1505
1506/// One directory listing retained by a worker for detached bootstrap consolidation.
1507///
1508/// `path` is paid once per directory. Its children remain grouped exactly as the
1509/// filesystem enumerator produced them, so consolidation resolves the parent once and
1510/// never reconstructs a child path for nondirectories. A fixed control is retained
1511/// separately so the consumer can install the directory's complete control state
1512/// before it classifies any sibling or makes descendants visible.
1513#[derive(Debug)]
1514pub(crate) struct DetachedDirectory {
1515    pub(crate) path: PathBuf,
1516    pub(crate) children: Vec<DetachedChild>,
1517    pub(crate) control: Option<Op>,
1518}
1519
1520/// Walk `root` and emit observations describing everything found.
1521pub fn scan(
1522    root: &Path,
1523    config: &ScanConfig,
1524    sink: &mut dyn FnMut(Observation),
1525) -> Result<ScanReport> {
1526    let mut public_sink = |batch: ScannerBatch| sink(batch.into_observation());
1527    let (mut report, _diagnostics) = scan_internal(
1528        root,
1529        config,
1530        &mut public_sink,
1531        false,
1532        WorkerPolicyExperiment::ShippedOneShot,
1533        SinkMode::Retained,
1534    )?;
1535    normalize_walk_errors(root, &mut report.errors);
1536    Ok(report)
1537}
1538
1539/// Walk `root` for the transient summary tier, folding each op without retaining it.
1540///
1541/// The public [`scan`] path hands each batch to the caller as an [`Observation`], so
1542/// worker-allocated `PathBuf`s are freed on the consumer thread. This path returns
1543/// drained batches to the producing worker so each arena is allocated and freed on one
1544/// thread. Tallies must match [`scan`], and so must the normalized error set: the
1545/// summary report's status is built from these errors exactly as a retained walk's is.
1546pub(crate) fn scan_summary_fold(
1547    root: &Path,
1548    config: &ScanConfig,
1549    fold: &mut dyn FnMut(&ObservationOp),
1550) -> Result<ScanReport> {
1551    let mut sink = |batch: ScannerBatch| {
1552        for op in batch.ops() {
1553            fold(op);
1554        }
1555        batch.recycle();
1556    };
1557    let (mut report, _diagnostics) = scan_internal(
1558        root,
1559        config,
1560        &mut sink,
1561        false,
1562        WorkerPolicyExperiment::ShippedOneShot,
1563        SinkMode::TransientFold,
1564    )?;
1565    normalize_walk_errors(root, &mut report.errors);
1566    Ok(report)
1567}
1568
1569/// [`scan_summary_fold`] plus the diagnostic trace [`scan_with_diagnostics`] collects.
1570pub(crate) fn scan_summary_fold_with_diagnostics(
1571    root: &Path,
1572    config: &ScanConfig,
1573    fold: &mut dyn FnMut(&ObservationOp),
1574) -> Result<(ScanReport, ScanDiagnostics)> {
1575    let mut sink = |batch: ScannerBatch| {
1576        for op in batch.ops() {
1577            fold(op);
1578        }
1579        batch.recycle();
1580    };
1581    let (mut report, diagnostics) = scan_internal(
1582        root,
1583        config,
1584        &mut sink,
1585        true,
1586        WorkerPolicyExperiment::ShippedOneShot,
1587        SinkMode::TransientFold,
1588    )?;
1589    normalize_walk_errors(root, &mut report.errors);
1590    Ok((report, diagnostics.expect("diagnostic scan creates a recorder")))
1591}
1592
1593/// Walk `root`, emitting observations and a bounded run-scoped diagnostic trace.
1594///
1595/// This is the measurement counterpart to [`scan`]. It produces the same observation
1596/// stream and report while recording controller and backend evidence that ordinary
1597/// scans intentionally do not collect.
1598pub fn scan_with_diagnostics(
1599    root: &Path,
1600    config: &ScanConfig,
1601    sink: &mut dyn FnMut(Observation),
1602) -> Result<(ScanReport, ScanDiagnostics)> {
1603    scan_with_policy_diagnostics(root, config, sink, WorkerPolicyExperiment::ShippedOneShot)
1604}
1605
1606/// Exercise a repository-only worker-controller candidate and retain its trace.
1607#[doc(hidden)]
1608pub fn scan_with_policy_diagnostics(
1609    root: &Path,
1610    config: &ScanConfig,
1611    sink: &mut dyn FnMut(Observation),
1612    policy: WorkerPolicyExperiment,
1613) -> Result<(ScanReport, ScanDiagnostics)> {
1614    let mut public_sink = |batch: ScannerBatch| sink(batch.into_observation());
1615    let (mut report, diagnostics) =
1616        scan_internal(root, config, &mut public_sink, true, policy, SinkMode::Retained)?;
1617    normalize_walk_errors(root, &mut report.errors);
1618    Ok((report, diagnostics.expect("diagnostic scan creates a recorder")))
1619}
1620
1621/// What the caller does with each batch of observations.
1622///
1623/// Two measured keeps hang off this one concept, and both were measured on the
1624/// transient fold alone: returning drained batches to the producing worker (H147,
1625/// exp-151) and taking directory and symlink kind from the listing without a stat
1626/// (H72, exp-153). They are named here as properties of the mode rather than passed as
1627/// one flag under one of their names, so a measurement on another platform can move
1628/// one without silently moving the other.
1629#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1630enum SinkMode {
1631    /// The consumer keeps the observations: the public [`scan`] and the index.
1632    Retained,
1633    /// The consumer folds each batch and drops it: the transient summary tier.
1634    TransientFold,
1635}
1636
1637impl SinkMode {
1638    /// Drained batches go back to the worker that allocated them (H147).
1639    fn recycles_batches(self) -> bool {
1640        self == Self::TransientFold
1641    }
1642
1643    /// Directory and symlink kind come from the listing without a stat (H72).
1644    fn skips_dir_symlink_stat(self) -> bool {
1645        self == Self::TransientFold
1646    }
1647}
1648
1649fn scan_internal(
1650    root: &Path,
1651    config: &ScanConfig,
1652    sink: &mut dyn FnMut(ScannerBatch),
1653    collect_diagnostics: bool,
1654    policy: WorkerPolicyExperiment,
1655    sink_mode: SinkMode,
1656) -> Result<(ScanReport, Option<ScanDiagnostics>)> {
1657    config.validate()?;
1658    if let Some(progress) = &config.progress {
1659        progress.enter(crate::ProgressPhase::Scanning);
1660    }
1661    let root_meta = {
1662        crate::counters::bump(|c| c.stats += 1);
1663        fs::symlink_metadata(root)
1664    }
1665    .map_err(|e| Error::io(root, e))?;
1666    if !root_meta.is_dir() {
1667        return Err(Error::io(
1668            root,
1669            std::io::Error::new(std::io::ErrorKind::NotADirectory, "scan root is not a directory"),
1670        ));
1671    }
1672    let root_dev = root_device(root, &root_meta).map_err(|error| Error::io(root, error))?;
1673    let available_parallelism =
1674        std::thread::available_parallelism().map_or(1, std::num::NonZero::get);
1675    let pool = config.worker_pool_for(available_parallelism);
1676    let diagnostics = collect_diagnostics
1677        .then(|| ScanDiagnosticsRecorder::new(pool, available_parallelism, policy));
1678
1679    if config.max_depth != Some(0) && pool.initial > 1 {
1680        let report = scan_concurrent(
1681            root,
1682            config,
1683            root_dev,
1684            sink,
1685            pool,
1686            diagnostics.as_ref(),
1687            policy,
1688            sink_mode,
1689        );
1690        return Ok((report, diagnostics.as_ref().map(|value| value.finish())));
1691    }
1692
1693    let mut report = ScanReport::default();
1694    if config.max_depth == Some(0) {
1695        if let Some(diagnostics) = &diagnostics {
1696            diagnostics.mark_not_run();
1697            diagnostics.record_queue_finish(0, 0);
1698        }
1699        return Ok((report, diagnostics.as_ref().map(|value| value.finish())));
1700    }
1701    let worker_guard = diagnostics.as_ref().map(ScanDiagnosticsRecorder::worker_guard);
1702    let walk_started = std::time::Instant::now();
1703    let mut batch: Vec<ObservationOp> = Vec::with_capacity(config.batch_size);
1704    let mut queue: VecDeque<(PathBuf, usize)> = VecDeque::from(vec![(PathBuf::new(), 0)]);
1705    let mut tally = ProgressTally::new(config.progress.as_ref());
1706    // Every batch leaves through here, so the batch is where the serial walk reports
1707    // its progress: the handoff the consumer already pays for, never the entry.
1708    let mut emit = |ops: Vec<ObservationOp>, report: &mut ScanReport| {
1709        let send_started = std::time::Instant::now();
1710        sink(ScannerBatch::new(ops));
1711        report.attribution.send_ns += elapsed_ns(send_started);
1712        tally.flush(report);
1713    };
1714
1715    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
1716        let abs_dir = root.join(&rel_dir);
1717        crate::counters::bump(|c| c.dir_opens += 1);
1718        if let Some(diagnostics) = &diagnostics {
1719            diagnostics.portable_attempted();
1720        }
1721        let listing = match fs::read_dir(&abs_dir) {
1722            Ok(listing) => {
1723                if let Some(diagnostics) = &diagnostics {
1724                    diagnostics.portable_succeeded();
1725                }
1726                listing
1727            }
1728            Err(e) => {
1729                report.errors.push(Error::io(abs_dir, e));
1730                continue;
1731            }
1732        };
1733        report.dirs_read += 1;
1734
1735        for item in listing {
1736            let item = match item {
1737                Ok(item) => item,
1738                Err(e) => {
1739                    report.errors.push(Error::io(&abs_dir, e));
1740                    continue;
1741                }
1742            };
1743            crate::counters::bump(|c| c.dir_entries += 1);
1744            let name = item.file_name();
1745            let rel_path = rel_dir.join(&name);
1746            let (kind, attrs) = match listed_child_kind_and_attrs(
1747                &item,
1748                sink_mode.skips_dir_symlink_stat(),
1749                config.one_filesystem,
1750            ) {
1751                Ok(Some(observed)) => observed,
1752                Ok(None) => continue,
1753                Err(error) => {
1754                    report.errors.push(Error::io(item.path(), error));
1755                    continue;
1756                }
1757            };
1758            let disposition =
1759                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
1760            if disposition == crate::admission::Disposition::Reject {
1761                continue;
1762            }
1763            let control = match read_control_op(config, root, &rel_path, kind) {
1764                Ok(control) => control,
1765                Err(error) => {
1766                    report.errors.push(error);
1767                    None
1768                }
1769            };
1770            if disposition == crate::admission::Disposition::ControlOnly {
1771                if let Some(control) = control {
1772                    batch.push(ObservationOp::unconditional(control));
1773                    if batch.len() >= config.batch_size {
1774                        emit(std::mem::take(&mut batch), &mut report);
1775                        batch.reserve(config.batch_size);
1776                    }
1777                }
1778                continue;
1779            }
1780            report.observe(kind, attrs);
1781            batch.push(ObservationOp::unconditional(Op::Upsert {
1782                path: rel_path.clone(),
1783                kind,
1784                attrs,
1785            }));
1786            if batch.len() >= config.batch_size {
1787                emit(std::mem::take(&mut batch), &mut report);
1788                batch.reserve(config.batch_size);
1789            }
1790            if let Some(control) = control {
1791                batch.push(ObservationOp::unconditional(control));
1792                if batch.len() >= config.batch_size {
1793                    emit(std::mem::take(&mut batch), &mut report);
1794                    batch.reserve(config.batch_size);
1795                }
1796            }
1797
1798            if should_descend(kind, attrs, depth, root_dev, config) {
1799                queue.push_back((rel_path, depth + 1));
1800            }
1801        }
1802    }
1803
1804    if !batch.is_empty() {
1805        emit(batch, &mut report);
1806    }
1807    // A walk whose last directories filled no batch has counted them and sent nothing.
1808    tally.flush(&report);
1809    // A serial walk has no coordination to attribute: wall is the loop, "send" is the
1810    // inline sink — which is the consumer actually running — and work is the rest.
1811    report.attribution.wall_ns = elapsed_ns(walk_started);
1812    report.attribution.work_ns =
1813        report.attribution.wall_ns.saturating_sub(report.attribution.send_ns);
1814    drop(worker_guard);
1815    if let Some(diagnostics) = &diagnostics {
1816        diagnostics.record_queue_finish(0, 0);
1817    }
1818    Ok((report, diagnostics.as_ref().map(|value| value.finish())))
1819}
1820
1821/// Take the next directory in the configured order.
1822///
1823/// Both orders push to the back; only the end they are taken from differs, which is
1824/// what keeps this a one-line policy rather than two walkers.
1825fn take_next(queue: &mut VecDeque<(PathBuf, usize)>, order: ScanOrder) -> Option<(PathBuf, usize)> {
1826    match order {
1827        ScanOrder::BreadthFirst => queue.pop_front(),
1828        ScanOrder::DepthFirst => queue.pop_back(),
1829    }
1830}
1831
1832/// Largest worker pool a caller may ask for explicitly.
1833///
1834/// Well past anything measured to help. It exists so a caller that computes a thread
1835/// count from something silly cannot spawn thousands of threads.
1836const MAX_SCAN_THREADS: usize = 32;
1837
1838/// Ceiling on the workers active at the start of an automatic scan.
1839///
1840/// Measured, not guessed. On a 10-core machine walking a 60k-entry `node_modules`
1841/// tree, wall time fell 37% at two workers and 50% at four, then stopped improving:
1842/// six matched four within noise and eight was 4% worse than four. The walk becomes
1843/// bound by the single index consumer, so past this point extra workers buy queue
1844/// contention and efficiency-core scheduling rather than throughput. See
1845/// `docs/project/reports/report-2026-08-10-fdu-performance-experiments.md`.
1846///
1847/// That measurement was taken on macOS, and every constant in this group now reads its
1848/// value from [`crate::platform_tuning`], which records per platform whether the number
1849/// was measured there or inherited. On Linux these are inherited.
1850const DEFAULT_SCAN_THREADS_CAP: usize = crate::platform_tuning::tuning().scan_threads_cap.get();
1851
1852/// Ceiling on automatic workers for an immutable-baseline reconciliation wave.
1853///
1854/// Reconciliation reads both filesystem and index state. Its measured knee arrives
1855/// before the cold producer's because additional metadata calls amplify kernel work
1856/// after the index comparisons already saturate the performance cores.
1857const DEFAULT_RECONCILE_THREADS_CAP: usize =
1858    crate::platform_tuning::tuning().reconcile_threads_cap.get();
1859
1860/// Ceiling an automatic scan may unlock after it establishes that the tree is large.
1861///
1862/// Sixteen was the knee on the 720k-entry cache-pressure corpus in exp-015. Thirty-two
1863/// did not improve on it and spent substantially more worker time waiting at the end.
1864const ADAPTIVE_SCAN_THREADS_CAP: usize =
1865    crate::platform_tuning::tuning().adaptive_scan_threads_cap.get();
1866
1867/// Maximum reserve depth relative to the host's reported parallelism.
1868const ADAPTIVE_SCAN_PARALLELISM_MULTIPLIER: usize =
1869    crate::platform_tuning::tuning().adaptive_scan_parallelism_multiplier.get();
1870
1871/// Entries used to calibrate the initial workers' filesystem service time.
1872const ADAPTIVE_SCAN_CALIBRATION_ENTRIES: u64 =
1873    crate::platform_tuning::tuning().adaptive_scan_calibration_entries.get();
1874
1875/// Average worker time per observed entry that identifies a latency-bound scan.
1876///
1877/// Whole-run attribution separated the measured regimes: roughly 18 microseconds on
1878/// the 60k tree, 22 on the 120k boundary, and 42 or more on the 720k cache-pressure
1879/// tree. Thirty leaves margin between them. The calibration uses the same chunk timing
1880/// already collected for attribution, so it adds no per-entry clock reads.
1881///
1882/// Those are APFS regimes. The Linux warm floor is about 1.5 µs per entry, twenty times
1883/// below this threshold, so the trigger may never fire there — which is exactly the kind
1884/// of inherited constant [`crate::platform_tuning`] exists to make visible (H84).
1885const ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY: u64 =
1886    crate::platform_tuning::tuning().adaptive_scan_slow_work_ns_per_entry.get();
1887
1888#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1889struct WorkerPool {
1890    initial: usize,
1891    maximum: usize,
1892    calibration: Option<WorkerCalibration>,
1893}
1894
1895#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1896struct WorkerCalibration {
1897    minimum_entries: u64,
1898    slow_work_ns_per_entry: u64,
1899    entries: u64,
1900    work_ns: u64,
1901    chunks: u64,
1902}
1903
1904#[derive(Clone, Copy, Debug)]
1905struct PolicyWindowSnapshot {
1906    sequence: u64,
1907    start_entry_ordinal: u64,
1908    end_entry_ordinal: u64,
1909    observed_entries: u64,
1910    observed_chunks: u64,
1911    observed_work_ns: u64,
1912    ready_directories: usize,
1913    in_flight_directories: usize,
1914    active_workers: usize,
1915    handoff_backlog: usize,
1916    requested_workers: Option<usize>,
1917    decision: WorkerPolicyDecision,
1918}
1919
1920struct PolicyTraceState {
1921    outcome: WorkerPolicyOutcome,
1922    outcome_reason: Option<&'static str>,
1923    outcome_sequence: Option<u64>,
1924    windows: Vec<WorkerPolicyWindow>,
1925    events_truncated: bool,
1926    ready_directories_at_finish: usize,
1927    in_flight_directories_at_finish: usize,
1928}
1929
1930/// Shared state used only by the opt-in diagnostic scan APIs.
1931///
1932/// Normal scans pass no recorder and therefore never touch these atomics or locks. The
1933/// trace mutex is deliberately separate from the directory queue: recording a policy
1934/// window may add diagnostic cost, but it cannot alter the queue's synchronization or
1935/// the controller's decision.
1936struct ScanDiagnosticsRecorder {
1937    available_parallelism: usize,
1938    pool: WorkerPool,
1939    policy: WorkerPolicyExperiment,
1940    trace: std::sync::Mutex<PolicyTraceState>,
1941    workers_spawned: std::sync::atomic::AtomicUsize,
1942    active_workers: std::sync::atomic::AtomicUsize,
1943    peak_active_workers: std::sync::atomic::AtomicUsize,
1944    handoff_backlog: std::sync::atomic::AtomicUsize,
1945    handoff_backlog_high_water: std::sync::atomic::AtomicUsize,
1946    calibration_chunks: std::sync::atomic::AtomicU64,
1947    calibration_entries: std::sync::atomic::AtomicU64,
1948    calibration_work_ns: std::sync::atomic::AtomicU64,
1949    worker_expansions: std::sync::atomic::AtomicU64,
1950    portable_attempts: std::sync::atomic::AtomicU64,
1951    portable_successes: std::sync::atomic::AtomicU64,
1952    #[cfg(target_os = "macos")]
1953    macos_bulk_attempts: std::sync::atomic::AtomicU64,
1954    #[cfg(target_os = "macos")]
1955    macos_bulk_successes: std::sync::atomic::AtomicU64,
1956    #[cfg(target_os = "macos")]
1957    macos_bulk_fallbacks: std::sync::atomic::AtomicU64,
1958}
1959
1960impl ScanDiagnosticsRecorder {
1961    fn new(
1962        pool: WorkerPool,
1963        available_parallelism: usize,
1964        policy: WorkerPolicyExperiment,
1965    ) -> std::sync::Arc<Self> {
1966        let (outcome, outcome_reason) = if pool.calibration.is_some() {
1967            (
1968                WorkerPolicyOutcome::Undecided,
1969                Some("the adaptive calibration window has not completed"),
1970            )
1971        } else {
1972            (WorkerPolicyOutcome::Fixed, Some("this worker pool has no adaptive reserve"))
1973        };
1974        std::sync::Arc::new(Self {
1975            available_parallelism,
1976            pool,
1977            policy,
1978            trace: std::sync::Mutex::new(PolicyTraceState {
1979                outcome,
1980                outcome_reason,
1981                outcome_sequence: None,
1982                windows: Vec::new(),
1983                events_truncated: false,
1984                ready_directories_at_finish: 0,
1985                in_flight_directories_at_finish: 0,
1986            }),
1987            workers_spawned: std::sync::atomic::AtomicUsize::new(0),
1988            active_workers: std::sync::atomic::AtomicUsize::new(0),
1989            peak_active_workers: std::sync::atomic::AtomicUsize::new(0),
1990            handoff_backlog: std::sync::atomic::AtomicUsize::new(0),
1991            handoff_backlog_high_water: std::sync::atomic::AtomicUsize::new(0),
1992            calibration_chunks: std::sync::atomic::AtomicU64::new(0),
1993            calibration_entries: std::sync::atomic::AtomicU64::new(0),
1994            calibration_work_ns: std::sync::atomic::AtomicU64::new(0),
1995            worker_expansions: std::sync::atomic::AtomicU64::new(0),
1996            portable_attempts: std::sync::atomic::AtomicU64::new(0),
1997            portable_successes: std::sync::atomic::AtomicU64::new(0),
1998            #[cfg(target_os = "macos")]
1999            macos_bulk_attempts: std::sync::atomic::AtomicU64::new(0),
2000            #[cfg(target_os = "macos")]
2001            macos_bulk_successes: std::sync::atomic::AtomicU64::new(0),
2002            #[cfg(target_os = "macos")]
2003            macos_bulk_fallbacks: std::sync::atomic::AtomicU64::new(0),
2004        })
2005    }
2006
2007    fn worker_guard(self: &std::sync::Arc<Self>) -> ScanWorkerGuard {
2008        let active = self
2009            .active_workers
2010            .fetch_add(1, std::sync::atomic::Ordering::Relaxed)
2011            .saturating_add(1);
2012        self.workers_spawned.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2013        atomic_update_max(&self.peak_active_workers, active);
2014        ScanWorkerGuard { recorder: self.clone() }
2015    }
2016
2017    fn record_policy_window(&self, snapshot: PolicyWindowSnapshot) {
2018        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2019        let sequence = snapshot.sequence;
2020        let supersedes = trace.outcome_sequence.is_none_or(|current| sequence >= current);
2021        match snapshot.decision {
2022            WorkerPolicyDecision::Undecided if supersedes => {
2023                trace.outcome = WorkerPolicyOutcome::Undecided;
2024                trace.outcome_reason =
2025                    Some("the walk ended before the adaptive calibration window completed");
2026                trace.outcome_sequence = Some(sequence);
2027            }
2028            WorkerPolicyDecision::Hold
2029                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2030            {
2031                trace.outcome = WorkerPolicyOutcome::Held;
2032                trace.outcome_reason = None;
2033                trace.outcome_sequence = Some(sequence);
2034            }
2035            WorkerPolicyDecision::ScaleUp => {
2036                trace.outcome = WorkerPolicyOutcome::ScaledUp;
2037                trace.outcome_reason = None;
2038                trace.outcome_sequence = Some(sequence);
2039            }
2040            WorkerPolicyDecision::HoldNoUsefulWork
2041                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2042            {
2043                trace.outcome = WorkerPolicyOutcome::HeldNoUsefulWork;
2044                trace.outcome_reason =
2045                    Some("the slow trigger fired only after ready and in-flight work had drained");
2046                trace.outcome_sequence = Some(sequence);
2047            }
2048            WorkerPolicyDecision::HoldInsufficientFrontier
2049                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2050            {
2051                trace.outcome = WorkerPolicyOutcome::Held;
2052                trace.outcome_reason =
2053                    Some("the observed frontier could not use additional workers");
2054                trace.outcome_sequence = Some(sequence);
2055            }
2056            WorkerPolicyDecision::HoldHandoffBacklog
2057                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2058            {
2059                trace.outcome = WorkerPolicyOutcome::Held;
2060                trace.outcome_reason =
2061                    Some("the observation handoff backlog was already at the controller limit");
2062                trace.outcome_sequence = Some(sequence);
2063            }
2064            WorkerPolicyDecision::Incomplete
2065                if supersedes && trace.outcome == WorkerPolicyOutcome::Undecided =>
2066            {
2067                trace.outcome_reason =
2068                    Some("the walk ended before any adaptive calibration window completed");
2069                trace.outcome_sequence = Some(sequence);
2070            }
2071            _ => {}
2072        }
2073        if sequence >= MAX_POLICY_TRACE_EVENTS as u64 {
2074            trace.events_truncated = true;
2075            return;
2076        }
2077        let work_ns_per_entry = (snapshot.observed_entries > 0)
2078            .then(|| snapshot.observed_work_ns / snapshot.observed_entries);
2079        trace.windows.push(WorkerPolicyWindow {
2080            sequence,
2081            start_entry_ordinal: snapshot.start_entry_ordinal,
2082            end_entry_ordinal: snapshot.end_entry_ordinal,
2083            observed_entries: snapshot.observed_entries,
2084            observed_chunks: snapshot.observed_chunks,
2085            observed_work_ns: snapshot.observed_work_ns,
2086            work_ns_per_entry,
2087            work_ns_per_entry_unavailable_reason: work_ns_per_entry
2088                .is_none()
2089                .then_some("the window observed no entries"),
2090            ready_directories: snapshot.ready_directories,
2091            in_flight_directories: snapshot.in_flight_directories,
2092            active_workers: snapshot.active_workers,
2093            handoff_backlog: snapshot.handoff_backlog,
2094            requested_workers: snapshot.requested_workers,
2095            decision: snapshot.decision,
2096        });
2097    }
2098
2099    fn mark_not_run(&self) {
2100        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2101        trace.outcome = WorkerPolicyOutcome::NotRun;
2102        trace.outcome_reason = Some("max_depth zero requested no directory walk");
2103    }
2104
2105    fn record_queue_finish(&self, ready_directories: usize, in_flight_directories: usize) {
2106        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2107        trace.ready_directories_at_finish = ready_directories;
2108        trace.in_flight_directories_at_finish = in_flight_directories;
2109    }
2110
2111    fn handoff_sent(&self) {
2112        let backlog = self
2113            .handoff_backlog
2114            .fetch_add(1, std::sync::atomic::Ordering::Relaxed)
2115            .saturating_add(1);
2116        atomic_update_max(&self.handoff_backlog_high_water, backlog);
2117    }
2118
2119    fn handoff_received(&self) {
2120        let previous = self.handoff_backlog.fetch_sub(1, std::sync::atomic::Ordering::Relaxed);
2121        debug_assert!(previous > 0, "received handoff must have been sent");
2122    }
2123
2124    fn calibration_chunk(&self, entries: u64, work_ns: u64) {
2125        self.calibration_chunks.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2126        self.calibration_entries.fetch_add(entries, std::sync::atomic::Ordering::Relaxed);
2127        self.calibration_work_ns.fetch_add(work_ns, std::sync::atomic::Ordering::Relaxed);
2128    }
2129
2130    fn worker_expanded(&self) {
2131        self.worker_expansions.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2132    }
2133
2134    fn portable_attempted(&self) {
2135        self.portable_attempts.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2136    }
2137
2138    fn portable_succeeded(&self) {
2139        self.portable_successes.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2140    }
2141
2142    #[cfg(target_os = "macos")]
2143    fn macos_bulk_attempted(&self) {
2144        self.macos_bulk_attempts.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2145    }
2146
2147    #[cfg(target_os = "macos")]
2148    fn macos_bulk_succeeded(&self) {
2149        self.macos_bulk_successes.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2150    }
2151
2152    #[cfg(target_os = "macos")]
2153    fn macos_bulk_fell_back(&self) {
2154        self.macos_bulk_fallbacks.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2155    }
2156
2157    fn finish(&self) -> ScanDiagnostics {
2158        let trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2159        let calibration = self.pool.calibration;
2160        let backend = ScanBackendDiagnostics {
2161            portable_attempts: self.portable_attempts.load(std::sync::atomic::Ordering::Relaxed),
2162            portable_directory_reads: self
2163                .portable_successes
2164                .load(std::sync::atomic::Ordering::Relaxed),
2165            #[cfg(target_os = "macos")]
2166            macos_bulk_attempts: Some(
2167                self.macos_bulk_attempts.load(std::sync::atomic::Ordering::Relaxed),
2168            ),
2169            #[cfg(not(target_os = "macos"))]
2170            macos_bulk_attempts: None,
2171            #[cfg(target_os = "macos")]
2172            macos_bulk_successes: Some(
2173                self.macos_bulk_successes.load(std::sync::atomic::Ordering::Relaxed),
2174            ),
2175            #[cfg(not(target_os = "macos"))]
2176            macos_bulk_successes: None,
2177            #[cfg(target_os = "macos")]
2178            macos_bulk_fallbacks: Some(
2179                self.macos_bulk_fallbacks.load(std::sync::atomic::Ordering::Relaxed),
2180            ),
2181            #[cfg(not(target_os = "macos"))]
2182            macos_bulk_fallbacks: None,
2183            #[cfg(target_os = "macos")]
2184            unavailable_reason: None,
2185            #[cfg(not(target_os = "macos"))]
2186            unavailable_reason: Some(
2187                "macOS bulk directory enumeration is unavailable on this platform",
2188            ),
2189        };
2190        ScanDiagnostics {
2191            schema: SCAN_DIAGNOSTICS_SCHEMA,
2192            worker_policy: WorkerPolicyDiagnostics {
2193                controller: worker_policy_experiment_name(self.policy),
2194                available_parallelism: self.available_parallelism,
2195                initial_workers: self.pool.initial,
2196                maximum_workers: self.pool.maximum,
2197                calibration_window_entries: calibration.map(|value| value.minimum_entries),
2198                slow_threshold_ns_per_entry: calibration.map(|value| value.slow_work_ns_per_entry),
2199                calibration_chunks: self
2200                    .calibration_chunks
2201                    .load(std::sync::atomic::Ordering::Relaxed),
2202                calibration_entries: self
2203                    .calibration_entries
2204                    .load(std::sync::atomic::Ordering::Relaxed),
2205                calibration_work_ns: self
2206                    .calibration_work_ns
2207                    .load(std::sync::atomic::Ordering::Relaxed),
2208                worker_expansions: self
2209                    .worker_expansions
2210                    .load(std::sync::atomic::Ordering::Relaxed),
2211                outcome: trace.outcome,
2212                outcome_reason: trace.outcome_reason,
2213                workers_spawned: self.workers_spawned.load(std::sync::atomic::Ordering::Relaxed),
2214                peak_active_workers: self
2215                    .peak_active_workers
2216                    .load(std::sync::atomic::Ordering::Relaxed),
2217                ready_directories_at_finish: trace.ready_directories_at_finish,
2218                in_flight_directories_at_finish: trace.in_flight_directories_at_finish,
2219                handoff_backlog_at_finish: self
2220                    .handoff_backlog
2221                    .load(std::sync::atomic::Ordering::Relaxed),
2222                handoff_backlog_high_water: self
2223                    .handoff_backlog_high_water
2224                    .load(std::sync::atomic::Ordering::Relaxed),
2225                windows: {
2226                    let mut windows = trace.windows.clone();
2227                    windows.sort_by_key(|window| window.sequence);
2228                    windows
2229                },
2230                events_truncated: trace.events_truncated,
2231            },
2232            backend,
2233        }
2234    }
2235}
2236
2237const fn worker_policy_experiment_name(value: WorkerPolicyExperiment) -> &'static str {
2238    match value {
2239        WorkerPolicyExperiment::ShippedOneShot => "shipped_one_shot",
2240        WorkerPolicyExperiment::RepeatedWindows => "repeated_windows",
2241        WorkerPolicyExperiment::StagedGatedWindows => "staged_gated_windows",
2242    }
2243}
2244
2245struct ScanWorkerGuard {
2246    recorder: std::sync::Arc<ScanDiagnosticsRecorder>,
2247}
2248
2249impl Drop for ScanWorkerGuard {
2250    fn drop(&mut self) {
2251        let previous =
2252            self.recorder.active_workers.fetch_sub(1, std::sync::atomic::Ordering::Relaxed);
2253        debug_assert!(previous > 0, "worker guard must balance worker start");
2254    }
2255}
2256
2257fn atomic_update_max(target: &std::sync::atomic::AtomicUsize, value: usize) {
2258    let mut observed = target.load(std::sync::atomic::Ordering::Relaxed);
2259    while value > observed {
2260        match target.compare_exchange_weak(
2261            observed,
2262            value,
2263            std::sync::atomic::Ordering::Relaxed,
2264            std::sync::atomic::Ordering::Relaxed,
2265        ) {
2266            Ok(_) => break,
2267            Err(actual) => observed = actual,
2268        }
2269    }
2270}
2271
2272enum WalkMessage {
2273    Batch(ScannerBatch),
2274    DetachedDirectories(Vec<DetachedDirectory>),
2275    ScaleUp { sender: std::sync::mpsc::Sender<Self>, target_workers: usize },
2276}
2277
2278impl WorkerPool {
2279    const fn fixed(workers: usize) -> Self {
2280        Self { initial: workers, maximum: workers, calibration: None }
2281    }
2282}
2283
2284impl WorkerCalibration {
2285    const fn new(minimum_entries: u64, slow_work_ns_per_entry: u64) -> Self {
2286        Self { minimum_entries, slow_work_ns_per_entry, entries: 0, work_ns: 0, chunks: 0 }
2287    }
2288
2289    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<bool> {
2290        self.chunks = self.chunks.saturating_add(1);
2291        self.entries = self.entries.saturating_add(entries);
2292        self.work_ns = self.work_ns.saturating_add(work_ns);
2293        (self.entries >= self.minimum_entries)
2294            .then(|| self.work_ns / self.entries >= self.slow_work_ns_per_entry)
2295    }
2296}
2297
2298#[derive(Clone, Copy, Debug)]
2299struct CalibrationWindow {
2300    start_entry_ordinal: u64,
2301    end_entry_ordinal: u64,
2302    entries: u64,
2303    chunks: u64,
2304    work_ns: u64,
2305    slow: bool,
2306}
2307
2308#[derive(Debug)]
2309struct RepeatedCalibration {
2310    minimum_entries: u64,
2311    slow_work_ns_per_entry: u64,
2312    window_start: u64,
2313    entries: u64,
2314    chunks: u64,
2315    work_ns: u64,
2316    completed_windows: u64,
2317}
2318
2319impl RepeatedCalibration {
2320    const fn new(calibration: WorkerCalibration) -> Self {
2321        Self {
2322            minimum_entries: calibration.minimum_entries,
2323            slow_work_ns_per_entry: calibration.slow_work_ns_per_entry,
2324            window_start: 0,
2325            entries: 0,
2326            chunks: 0,
2327            work_ns: 0,
2328            completed_windows: 0,
2329        }
2330    }
2331
2332    const fn starting_at(calibration: WorkerCalibration, window_start: u64) -> Self {
2333        let mut repeated = Self::new(calibration);
2334        repeated.window_start = window_start;
2335        repeated
2336    }
2337
2338    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<CalibrationWindow> {
2339        self.chunks = self.chunks.saturating_add(1);
2340        self.entries = self.entries.saturating_add(entries);
2341        self.work_ns = self.work_ns.saturating_add(work_ns);
2342        if self.entries < self.minimum_entries {
2343            return None;
2344        }
2345        let end_entry_ordinal = self.window_start.saturating_add(self.entries);
2346        let window = CalibrationWindow {
2347            start_entry_ordinal: self.window_start,
2348            end_entry_ordinal,
2349            entries: self.entries,
2350            chunks: self.chunks,
2351            work_ns: self.work_ns,
2352            slow: self.work_ns / self.entries >= self.slow_work_ns_per_entry,
2353        };
2354        self.window_start = end_entry_ordinal;
2355        self.entries = 0;
2356        self.chunks = 0;
2357        self.work_ns = 0;
2358        self.completed_windows = self.completed_windows.saturating_add(1);
2359        Some(window)
2360    }
2361}
2362
2363#[derive(Debug)]
2364enum WorkerController {
2365    OneShot(WorkerCalibration),
2366    Repeated { calibration: RepeatedCalibration, staged_gated: bool },
2367}
2368
2369impl WorkerController {
2370    fn new(calibration: WorkerCalibration, policy: WorkerPolicyExperiment) -> Self {
2371        match policy {
2372            WorkerPolicyExperiment::ShippedOneShot => Self::OneShot(calibration),
2373            WorkerPolicyExperiment::RepeatedWindows => Self::Repeated {
2374                calibration: RepeatedCalibration::new(calibration),
2375                staged_gated: false,
2376            },
2377            WorkerPolicyExperiment::StagedGatedWindows => Self::Repeated {
2378                calibration: RepeatedCalibration::new(calibration),
2379                staged_gated: true,
2380            },
2381        }
2382    }
2383
2384    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<CalibrationWindow> {
2385        match self {
2386            Self::OneShot(calibration) => {
2387                let slow = calibration.observe(entries, work_ns)?;
2388                Some(CalibrationWindow {
2389                    start_entry_ordinal: 0,
2390                    end_entry_ordinal: calibration.entries,
2391                    entries: calibration.entries,
2392                    chunks: calibration.chunks,
2393                    work_ns: calibration.work_ns,
2394                    slow,
2395                })
2396            }
2397            Self::Repeated { calibration, .. } => calibration.observe(entries, work_ns),
2398        }
2399    }
2400
2401    fn partial_window(&self) -> (CalibrationWindow, WorkerPolicyDecision) {
2402        match self {
2403            Self::OneShot(calibration) => (
2404                CalibrationWindow {
2405                    start_entry_ordinal: 0,
2406                    end_entry_ordinal: calibration.entries,
2407                    entries: calibration.entries,
2408                    chunks: calibration.chunks,
2409                    work_ns: calibration.work_ns,
2410                    slow: false,
2411                },
2412                WorkerPolicyDecision::Undecided,
2413            ),
2414            Self::Repeated { calibration, .. } => (
2415                CalibrationWindow {
2416                    start_entry_ordinal: calibration.window_start,
2417                    end_entry_ordinal: calibration.window_start.saturating_add(calibration.entries),
2418                    entries: calibration.entries,
2419                    chunks: calibration.chunks,
2420                    work_ns: calibration.work_ns,
2421                    slow: false,
2422                },
2423                if calibration.completed_windows == 0 {
2424                    WorkerPolicyDecision::Undecided
2425                } else {
2426                    WorkerPolicyDecision::Incomplete
2427                },
2428            ),
2429        }
2430    }
2431
2432    const fn is_staged_gated(&self) -> bool {
2433        matches!(self, Self::Repeated { staged_gated: true, .. })
2434    }
2435
2436    const fn is_one_shot(&self) -> bool {
2437        matches!(self, Self::OneShot(_))
2438    }
2439
2440    const fn calibration_spec(&self) -> WorkerCalibration {
2441        match self {
2442            Self::OneShot(calibration) => WorkerCalibration::new(
2443                calibration.minimum_entries,
2444                calibration.slow_work_ns_per_entry,
2445            ),
2446            Self::Repeated { calibration, .. } => WorkerCalibration::new(
2447                calibration.minimum_entries,
2448                calibration.slow_work_ns_per_entry,
2449            ),
2450        }
2451    }
2452}
2453
2454fn automatic_worker_pool(available: usize) -> WorkerPool {
2455    let initial = available.clamp(1, DEFAULT_SCAN_THREADS_CAP);
2456    // Preserve the serial fallback when the platform cannot report more than one
2457    // available processor. There is no measured basis for inventing parallelism there.
2458    if initial == 1 {
2459        return WorkerPool::fixed(1);
2460    }
2461    let maximum = available
2462        .saturating_mul(ADAPTIVE_SCAN_PARALLELISM_MULTIPLIER)
2463        .clamp(initial, ADAPTIVE_SCAN_THREADS_CAP);
2464    WorkerPool {
2465        initial,
2466        maximum,
2467        calibration: (maximum > initial).then_some(WorkerCalibration::new(
2468            ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
2469            ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
2470        )),
2471    }
2472}
2473
2474/// Directories handed to a worker in one go.
2475///
2476/// Popping one directory at a time makes the queue lock the bottleneck on a wide,
2477/// shallow tree; taking a small run amortizes the lock without letting one worker
2478/// starve the others by hoarding the queue.
2479const DIR_CLAIM: usize = 4;
2480
2481/// A parallel directory walk that produces exactly the observations the serial walk does.
2482///
2483/// The shape is deliberate. Workers read directories and *produce* observations; they
2484/// never touch an index. A single consumer — the caller's sink, on this thread —
2485/// applies them. That keeps the crate's one mutation contract intact: parallelism is a
2486/// property of the producer, and the index still sees one ordered stream of observations.
2487///
2488/// Ordering across independent subtrees is not fixed, but a directory observation is
2489/// published before that directory becomes claimable. The index therefore sees a
2490/// parent-first causal stream without imposing a global level barrier or serializing
2491/// filesystem work. The resulting index is byte-identical to the serial walker's,
2492/// which the benchmark harness re-proves on every trial by comparing engine digests
2493/// against an independent oracle.
2494#[allow(clippy::too_many_arguments)]
2495fn scan_concurrent(
2496    root: &Path,
2497    config: &ScanConfig,
2498    root_dev: u64,
2499    sink: &mut dyn FnMut(ScannerBatch),
2500    pool: WorkerPool,
2501    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2502    policy: WorkerPolicyExperiment,
2503    sink_mode: SinkMode,
2504) -> ScanReport {
2505    let mut consume = |message| match message {
2506        WalkMessage::Batch(batch) => {
2507            if let Some(diagnostics) = diagnostics {
2508                diagnostics.handoff_received();
2509            }
2510            sink(batch);
2511        }
2512        WalkMessage::DetachedDirectories(_) => {
2513            unreachable!("the streaming walker never publishes detached directories")
2514        }
2515        WalkMessage::ScaleUp { .. } => {
2516            unreachable!("the shared runner consumes scale-up messages")
2517        }
2518    };
2519    run_concurrent_walk(
2520        root,
2521        config,
2522        root_dev,
2523        pool,
2524        diagnostics,
2525        policy,
2526        match sink_mode {
2527            SinkMode::Retained => walk_worker,
2528            SinkMode::TransientFold => walk_worker_transient_fold,
2529        },
2530        &mut consume,
2531    )
2532}
2533
2534/// Parallel cold walk for a detached index that has no streaming consumer.
2535///
2536/// Workers publish directory-shaped facts before making their children claimable. The
2537/// caller consumes those groups into a private builder while filesystem work continues,
2538/// preserving parent-first causality and pipeline overlap without sending one full path
2539/// or public observation per entry.
2540fn scan_concurrent_detached(
2541    root: &Path,
2542    config: &ScanConfig,
2543    root_dev: u64,
2544    pool: WorkerPool,
2545    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2546    policy: WorkerPolicyExperiment,
2547) -> Result<(ScanReport, DetachedIndexBuilder)> {
2548    let mut builder = DetachedIndexBuilder::new(root, config.scope(), config.types_shared())
2549        .with_control_limits(config.control_limits);
2550    let mut build_error = None;
2551    let output = {
2552        let mut consume = |message| match message {
2553            WalkMessage::Batch(_) => {
2554                unreachable!("the detached walker never publishes scanner batches")
2555            }
2556            WalkMessage::DetachedDirectories(directories) => {
2557                if let Some(diagnostics) = diagnostics {
2558                    diagnostics.handoff_received();
2559                }
2560                if build_error.is_none() {
2561                    for directory in directories {
2562                        if let Err(error) = builder.push_directory(directory) {
2563                            build_error = Some(error);
2564                            break;
2565                        }
2566                    }
2567                }
2568            }
2569            WalkMessage::ScaleUp { .. } => {
2570                unreachable!("the shared runner consumes scale-up messages")
2571            }
2572        };
2573        run_concurrent_walk(
2574            root,
2575            config,
2576            root_dev,
2577            pool,
2578            diagnostics,
2579            policy,
2580            walk_detached_worker,
2581            &mut consume,
2582        )
2583    };
2584    if let Some(error) = build_error {
2585        return Err(error);
2586    }
2587    Ok((output, builder))
2588}
2589
2590type WalkWorker = fn(
2591    &Path,
2592    &ScanConfig,
2593    u64,
2594    &DirectoryQueue,
2595    &std::sync::mpsc::Sender<WalkMessage>,
2596    Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2597) -> ScanReport;
2598
2599/// Run the shared pool, scaling controller, diagnostics, and report reduction.
2600///
2601/// Streaming and detached scans differ only in their worker emission and main-thread
2602/// consumer. Keeping orchestration here prevents fixes to termination, diagnostics, or
2603/// panic handling from diverging between the two cold paths.
2604#[allow(clippy::too_many_arguments)]
2605fn run_concurrent_walk<C>(
2606    root: &Path,
2607    config: &ScanConfig,
2608    root_dev: u64,
2609    pool: WorkerPool,
2610    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2611    policy: WorkerPolicyExperiment,
2612    worker: WalkWorker,
2613    consume: &mut C,
2614) -> ScanReport
2615where
2616    C: FnMut(WalkMessage),
2617{
2618    let diagnostics = diagnostics.cloned();
2619    let queue = DirectoryQueue::new_with_policy(
2620        (PathBuf::new(), 0),
2621        config.order,
2622        pool.calibration,
2623        diagnostics.clone(),
2624        pool.initial,
2625        pool.maximum,
2626        policy,
2627    );
2628    let (sender, receiver) = std::sync::mpsc::channel::<WalkMessage>();
2629
2630    let mut report = std::thread::scope(|scope| {
2631        let mut handles: Vec<_> = (0..pool.initial)
2632            .map(|_| {
2633                let sender = sender.clone();
2634                let queue = &queue;
2635                let diagnostics = diagnostics.clone();
2636                scope.spawn(move || {
2637                    worker(root, config, root_dev, queue, &sender, diagnostics.as_ref())
2638                })
2639            })
2640            .collect();
2641        // The loop below ends when every sender is gone, so this one must go first.
2642        drop(sender);
2643
2644        let mut spawned_workers = pool.initial;
2645        for message in receiver {
2646            match message {
2647                WalkMessage::ScaleUp { sender, target_workers }
2648                    if target_workers > spawned_workers =>
2649                {
2650                    let target_workers = target_workers.min(pool.maximum);
2651                    record_adaptive_worker_expansion(diagnostics.as_ref());
2652                    for _ in spawned_workers..target_workers {
2653                        let sender = sender.clone();
2654                        let queue = &queue;
2655                        let diagnostics = diagnostics.clone();
2656                        handles.push(scope.spawn(move || {
2657                            worker(root, config, root_dev, queue, &sender, diagnostics.as_ref())
2658                        }));
2659                    }
2660                    spawned_workers = target_workers;
2661                }
2662                WalkMessage::ScaleUp { .. } => {}
2663                output => consume(output),
2664            }
2665        }
2666
2667        // A walk that ends before its calibration window fills never observed enough to
2668        // decide anything. That is an *unobservable* policy, not a decision to hold the
2669        // initial pool, and an artifact that conflated the two would report a held pool
2670        // as if the walk had measured one and chosen it.
2671        let queue_finish = {
2672            let mut state = queue.lock();
2673            let mut trailing_window = None;
2674            if let Some(controller) = &state.controller {
2675                let (window, decision) = controller.partial_window();
2676                if decision == WorkerPolicyDecision::Undecided {
2677                    crate::counters::bump(|counts| {
2678                        counts.adaptive_policy_undecided =
2679                            counts.adaptive_policy_undecided.saturating_add(1);
2680                    });
2681                }
2682                if diagnostics.is_some() {
2683                    let sequence = state.allocate_policy_sequence();
2684                    trailing_window = Some(PolicyWindowSnapshot {
2685                        sequence,
2686                        start_entry_ordinal: window.start_entry_ordinal,
2687                        end_entry_ordinal: window.end_entry_ordinal,
2688                        observed_entries: window.entries,
2689                        observed_chunks: window.chunks,
2690                        observed_work_ns: window.work_ns,
2691                        ready_directories: state.ready_directories,
2692                        in_flight_directories: state.in_flight_directories,
2693                        active_workers: diagnostics.as_ref().map_or(0, |diagnostics| {
2694                            diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
2695                        }),
2696                        handoff_backlog: diagnostics.as_ref().map_or(0, |diagnostics| {
2697                            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
2698                        }),
2699                        requested_workers: None,
2700                        decision,
2701                    });
2702                }
2703            } else if diagnostics.is_some() {
2704                let shadow = state.shadow_calibration.as_ref().and_then(|shadow| {
2705                    (shadow.entries > 0).then_some((
2706                        shadow.window_start,
2707                        shadow.entries,
2708                        shadow.chunks,
2709                        shadow.work_ns,
2710                    ))
2711                });
2712                if let Some((window_start, entries, chunks, work_ns)) = shadow {
2713                    let sequence = state.allocate_policy_sequence();
2714                    trailing_window = Some(PolicyWindowSnapshot {
2715                        sequence,
2716                        start_entry_ordinal: window_start,
2717                        end_entry_ordinal: window_start.saturating_add(entries),
2718                        observed_entries: entries,
2719                        observed_chunks: chunks,
2720                        observed_work_ns: work_ns,
2721                        ready_directories: state.ready_directories,
2722                        in_flight_directories: state.in_flight_directories,
2723                        active_workers: diagnostics.as_ref().map_or(0, |diagnostics| {
2724                            diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
2725                        }),
2726                        handoff_backlog: diagnostics.as_ref().map_or(0, |diagnostics| {
2727                            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
2728                        }),
2729                        requested_workers: None,
2730                        decision: WorkerPolicyDecision::ObserveIncomplete,
2731                    });
2732                }
2733            }
2734            (state.ready_directories, state.in_flight_directories, trailing_window)
2735        };
2736        if let Some(diagnostics) = &diagnostics {
2737            diagnostics.record_queue_finish(queue_finish.0, queue_finish.1);
2738            if let Some(window) = queue_finish.2 {
2739                diagnostics.record_policy_window(window);
2740            }
2741        }
2742
2743        let mut report = ScanReport::default();
2744        for handle in handles {
2745            match handle.join() {
2746                Ok(worker) => report.absorb(worker),
2747                Err(_) => {
2748                    // A worker panic leaves directories unaccounted for. Preserve that
2749                    // as a partial scan instead of reporting a short tree as complete.
2750                    report.errors.push(Error::io(
2751                        root,
2752                        std::io::Error::other("a scan worker thread panicked"),
2753                    ));
2754                }
2755            }
2756        }
2757        report
2758    });
2759
2760    // Workers finish in filesystem order, so normalize errors before they escape.
2761    report.errors.sort_by_cached_key(ToString::to_string);
2762    report
2763}
2764
2765/// Compile-time adapter for the one directory walker.
2766///
2767/// The filesystem, queue, admission, and diagnostics logic stays singular. Generic
2768/// emission keeps the public streaming path and private detached path branch-free in
2769/// their per-entry loops after monomorphization.
2770trait WalkEmission {
2771    type Directory;
2772
2773    fn begin_directory(&mut self, path: &Path) -> Self::Directory;
2774
2775    #[allow(clippy::too_many_arguments)]
2776    fn record_entry(
2777        &mut self,
2778        root: &Path,
2779        rel_dir: &Path,
2780        depth: usize,
2781        region: RegionId,
2782        name: &OsStr,
2783        kind: EntryKind,
2784        attrs: Attrs,
2785        root_dev: u64,
2786        config: &ScanConfig,
2787        directory: &mut Self::Directory,
2788        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
2789        report: &mut ScanReport,
2790        sender: &std::sync::mpsc::Sender<WalkMessage>,
2791        chunk_send_ns: &mut u64,
2792        diagnostics: Option<&ScanDiagnosticsRecorder>,
2793    ) -> bool;
2794
2795    fn finish_directory(&mut self, directory: Self::Directory);
2796
2797    fn publish_before_discovery(
2798        &mut self,
2799        has_discovered: bool,
2800        sender: &std::sync::mpsc::Sender<WalkMessage>,
2801        chunk_send_ns: &mut u64,
2802        diagnostics: Option<&ScanDiagnosticsRecorder>,
2803    ) -> bool;
2804
2805    fn finish(
2806        &mut self,
2807        sender: &std::sync::mpsc::Sender<WalkMessage>,
2808        report: &mut ScanReport,
2809        diagnostics: Option<&ScanDiagnosticsRecorder>,
2810    );
2811
2812    /// Transient summary can take directory and symlink kind from the listing.
2813    fn skip_dir_symlink_stat(&self) -> bool {
2814        false
2815    }
2816}
2817
2818struct StreamingEmission {
2819    batch: Vec<ObservationOp>,
2820    batch_size: usize,
2821    recycle_tx: Option<std::sync::mpsc::Sender<Vec<ObservationOp>>>,
2822    recycle_rx: Option<std::sync::mpsc::Receiver<Vec<ObservationOp>>>,
2823    skip_dir_symlink_stat: bool,
2824}
2825
2826impl StreamingEmission {
2827    /// An emission with the properties [`SinkMode`] names for `mode`.
2828    ///
2829    /// The recycle channel returns drained `PathBuf` arenas to this worker so glibc
2830    /// frees them on the thread that allocated them.
2831    fn for_sink(batch_size: usize, mode: SinkMode) -> Self {
2832        let (recycle_tx, recycle_rx) = if mode.recycles_batches() {
2833            let (tx, rx) = std::sync::mpsc::channel();
2834            (Some(tx), Some(rx))
2835        } else {
2836            (None, None)
2837        };
2838        Self {
2839            batch: Vec::with_capacity(batch_size),
2840            batch_size,
2841            recycle_tx,
2842            recycle_rx,
2843            skip_dir_symlink_stat: mode.skips_dir_symlink_stat(),
2844        }
2845    }
2846
2847    fn wrap(&self, ops: Vec<ObservationOp>) -> ScannerBatch {
2848        match &self.recycle_tx {
2849            Some(recycle) => ScannerBatch::new(ops).with_recycle(recycle.clone()),
2850            None => ScannerBatch::new(ops),
2851        }
2852    }
2853
2854    fn next_vec(&self) -> Vec<ObservationOp> {
2855        // The retained path keeps its pre-H147 shape: an empty vec that grows by
2856        // doubling. Pre-sizing every batch there was never measured, and the public
2857        // `scan` is what a library caller pays for.
2858        let Some(recycle_rx) = &self.recycle_rx else {
2859            return Vec::new();
2860        };
2861        let mut kept = None;
2862        while let Ok(mut recycled) = recycle_rx.try_recv() {
2863            recycled.clear();
2864            kept = Some(recycled);
2865        }
2866        kept.unwrap_or_else(|| Vec::with_capacity(self.batch_size))
2867    }
2868
2869    fn send_full(
2870        &mut self,
2871        sender: &std::sync::mpsc::Sender<WalkMessage>,
2872        diagnostics: Option<&ScanDiagnosticsRecorder>,
2873    ) -> bool {
2874        let ops = std::mem::take(&mut self.batch);
2875        let sent = send_scanner_batch(sender, self.wrap(ops), diagnostics);
2876        self.batch = self.next_vec();
2877        sent
2878    }
2879}
2880
2881impl WalkEmission for StreamingEmission {
2882    type Directory = ();
2883
2884    fn begin_directory(&mut self, _path: &Path) {}
2885
2886    #[allow(clippy::too_many_arguments)]
2887    fn record_entry(
2888        &mut self,
2889        root: &Path,
2890        rel_dir: &Path,
2891        depth: usize,
2892        region: RegionId,
2893        name: &OsStr,
2894        kind: EntryKind,
2895        attrs: Attrs,
2896        root_dev: u64,
2897        config: &ScanConfig,
2898        _directory: &mut Self::Directory,
2899        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
2900        report: &mut ScanReport,
2901        sender: &std::sync::mpsc::Sender<WalkMessage>,
2902        chunk_send_ns: &mut u64,
2903        diagnostics: Option<&ScanDiagnosticsRecorder>,
2904    ) -> bool {
2905        record_walk_entry(
2906            root,
2907            rel_dir,
2908            depth,
2909            region,
2910            name,
2911            kind,
2912            attrs,
2913            root_dev,
2914            config,
2915            self,
2916            discovered,
2917            report,
2918            sender,
2919            chunk_send_ns,
2920            diagnostics,
2921        )
2922    }
2923
2924    fn finish_directory(&mut self, _directory: Self::Directory) {}
2925
2926    fn publish_before_discovery(
2927        &mut self,
2928        has_discovered: bool,
2929        sender: &std::sync::mpsc::Sender<WalkMessage>,
2930        chunk_send_ns: &mut u64,
2931        diagnostics: Option<&ScanDiagnosticsRecorder>,
2932    ) -> bool {
2933        if self.batch.is_empty() || !has_discovered {
2934            return true;
2935        }
2936        let send_started = std::time::Instant::now();
2937        let sent = self.send_full(sender, diagnostics);
2938        *chunk_send_ns += elapsed_ns(send_started);
2939        sent
2940    }
2941
2942    fn finish(
2943        &mut self,
2944        sender: &std::sync::mpsc::Sender<WalkMessage>,
2945        report: &mut ScanReport,
2946        diagnostics: Option<&ScanDiagnosticsRecorder>,
2947    ) {
2948        if self.batch.is_empty() {
2949            return;
2950        }
2951        let send_started = std::time::Instant::now();
2952        // The walk is over: do not ask `send_full` for a replacement vec that no
2953        // later `record_entry` would use.
2954        let ops = std::mem::take(&mut self.batch);
2955        let _ = send_scanner_batch(sender, self.wrap(ops), diagnostics);
2956        self.batch = Vec::new();
2957        report.attribution.send_ns += elapsed_ns(send_started);
2958    }
2959
2960    fn skip_dir_symlink_stat(&self) -> bool {
2961        self.skip_dir_symlink_stat
2962    }
2963}
2964
2965#[derive(Default)]
2966struct DetachedEmission {
2967    directories: Vec<DetachedDirectory>,
2968}
2969
2970impl WalkEmission for DetachedEmission {
2971    type Directory = DetachedDirectory;
2972
2973    fn begin_directory(&mut self, path: &Path) -> Self::Directory {
2974        DetachedDirectory { path: path.to_path_buf(), children: Vec::new(), control: None }
2975    }
2976
2977    #[allow(clippy::too_many_arguments)]
2978    fn record_entry(
2979        &mut self,
2980        root: &Path,
2981        rel_dir: &Path,
2982        depth: usize,
2983        region: RegionId,
2984        name: &OsStr,
2985        kind: EntryKind,
2986        attrs: Attrs,
2987        root_dev: u64,
2988        config: &ScanConfig,
2989        directory: &mut Self::Directory,
2990        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
2991        report: &mut ScanReport,
2992        _sender: &std::sync::mpsc::Sender<WalkMessage>,
2993        _chunk_send_ns: &mut u64,
2994        _diagnostics: Option<&ScanDiagnosticsRecorder>,
2995    ) -> bool {
2996        record_detached_entry(
2997            root,
2998            rel_dir,
2999            depth,
3000            region,
3001            name,
3002            kind,
3003            attrs,
3004            root_dev,
3005            config,
3006            &mut directory.children,
3007            &mut directory.control,
3008            discovered,
3009            report,
3010        );
3011        true
3012    }
3013
3014    fn finish_directory(&mut self, directory: Self::Directory) {
3015        self.directories.push(directory);
3016    }
3017
3018    fn publish_before_discovery(
3019        &mut self,
3020        _has_discovered: bool,
3021        sender: &std::sync::mpsc::Sender<WalkMessage>,
3022        chunk_send_ns: &mut u64,
3023        diagnostics: Option<&ScanDiagnosticsRecorder>,
3024    ) -> bool {
3025        if self.directories.is_empty() {
3026            return true;
3027        }
3028        let send_started = std::time::Instant::now();
3029        let sent =
3030            send_detached_directories(sender, std::mem::take(&mut self.directories), diagnostics);
3031        *chunk_send_ns += elapsed_ns(send_started);
3032        sent
3033    }
3034
3035    fn finish(
3036        &mut self,
3037        _sender: &std::sync::mpsc::Sender<WalkMessage>,
3038        _report: &mut ScanReport,
3039        _diagnostics: Option<&ScanDiagnosticsRecorder>,
3040    ) {
3041    }
3042}
3043
3044fn walk_detached_worker(
3045    root: &Path,
3046    config: &ScanConfig,
3047    root_dev: u64,
3048    queue: &DirectoryQueue,
3049    sender: &std::sync::mpsc::Sender<WalkMessage>,
3050    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3051) -> ScanReport {
3052    let report = walk_worker_with(
3053        root,
3054        config,
3055        root_dev,
3056        queue,
3057        sender,
3058        diagnostics,
3059        DetachedEmission::default(),
3060    );
3061    // A walker leaves only when the queue is empty with nothing in flight, or when its
3062    // consumer is gone, so the walk is over. The index may still be assembling the
3063    // listings already sent; the counters have stopped, and the phase says why.
3064    if let Some(progress) = &config.progress {
3065        progress.enter(crate::ProgressPhase::Indexing);
3066    }
3067    report
3068}
3069
3070/// One worker's share of the public observation walk.
3071fn walk_worker(
3072    root: &Path,
3073    config: &ScanConfig,
3074    root_dev: u64,
3075    queue: &DirectoryQueue,
3076    sender: &std::sync::mpsc::Sender<WalkMessage>,
3077    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3078) -> ScanReport {
3079    walk_worker_with(
3080        root,
3081        config,
3082        root_dev,
3083        queue,
3084        sender,
3085        diagnostics,
3086        StreamingEmission::for_sink(config.batch_size, SinkMode::Retained),
3087    )
3088}
3089
3090/// One worker's share of the transient summary walk.
3091fn walk_worker_transient_fold(
3092    root: &Path,
3093    config: &ScanConfig,
3094    root_dev: u64,
3095    queue: &DirectoryQueue,
3096    sender: &std::sync::mpsc::Sender<WalkMessage>,
3097    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3098) -> ScanReport {
3099    walk_worker_with(
3100        root,
3101        config,
3102        root_dev,
3103        queue,
3104        sender,
3105        diagnostics,
3106        StreamingEmission::for_sink(config.batch_size, SinkMode::TransientFold),
3107    )
3108}
3109
3110fn walk_worker_with<E: WalkEmission>(
3111    root: &Path,
3112    config: &ScanConfig,
3113    root_dev: u64,
3114    queue: &DirectoryQueue,
3115    sender: &std::sync::mpsc::Sender<WalkMessage>,
3116    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3117    mut emission: E,
3118) -> ScanReport {
3119    let _counter_guard = crate::counters::thread_flush_guard();
3120    let _worker_guard = diagnostics.map(ScanDiagnosticsRecorder::worker_guard);
3121    let worker_started = std::time::Instant::now();
3122    let mut report = ScanReport::default();
3123    let mut tally = ProgressTally::new(config.progress.as_ref());
3124    let mut claimed: Vec<(PathBuf, usize, RegionId)> = Vec::with_capacity(DIR_CLAIM);
3125    let mut discovered: Vec<(PathBuf, usize, RegionId)> = Vec::new();
3126    let mut consumer_gone = false;
3127    #[cfg(target_os = "macos")]
3128    let mut bulk_reader = macos_bulk::Reader::new();
3129
3130    'walk: while let Some(claim) = queue.claim(&mut claimed, &mut report.attribution) {
3131        // One timing pair per claimed chunk, never per entry: the chunk is the unit
3132        // the amortization argument is made in, so it is the unit the evidence is
3133        // collected in.
3134        let chunk_started = std::time::Instant::now();
3135        let mut chunk_send_ns: u64 = 0;
3136        let entries_before = report.entries;
3137        for (rel_dir, depth, region) in claimed.drain(..) {
3138            let abs_dir = root.join(&rel_dir);
3139            let mut directory = emission.begin_directory(&rel_dir);
3140            #[cfg(target_os = "macos")]
3141            {
3142                if let Some(diagnostics) = diagnostics {
3143                    diagnostics.macos_bulk_attempted();
3144                }
3145                if let Some(entries) =
3146                    (!walk_hook_covers(&abs_dir)).then(|| bulk_reader.read(&abs_dir)).flatten()
3147                {
3148                    if let Some(diagnostics) = diagnostics {
3149                        diagnostics.macos_bulk_succeeded();
3150                    }
3151                    report.dirs_read += 1;
3152                    for entry in entries {
3153                        if !emission.record_entry(
3154                            root,
3155                            &rel_dir,
3156                            depth,
3157                            region,
3158                            &entry.name,
3159                            entry.kind,
3160                            entry.attrs,
3161                            root_dev,
3162                            config,
3163                            &mut directory,
3164                            &mut discovered,
3165                            &mut report,
3166                            sender,
3167                            &mut chunk_send_ns,
3168                            diagnostics.map(AsRef::as_ref),
3169                        ) {
3170                            consumer_gone = true;
3171                            break 'walk;
3172                        }
3173                    }
3174                    emission.finish_directory(directory);
3175                    continue;
3176                }
3177                if let Some(diagnostics) = diagnostics {
3178                    diagnostics.macos_bulk_fell_back();
3179                }
3180            }
3181
3182            crate::counters::bump(|c| c.dir_opens += 1);
3183            if let Some(diagnostics) = diagnostics {
3184                diagnostics.portable_attempted();
3185            }
3186
3187            let listing = match fs::read_dir(&abs_dir) {
3188                Ok(listing) => {
3189                    if let Some(diagnostics) = diagnostics {
3190                        diagnostics.portable_succeeded();
3191                    }
3192                    listing
3193                }
3194                Err(e) => {
3195                    report.errors.push(Error::io(abs_dir, e));
3196                    continue;
3197                }
3198            };
3199            report.dirs_read += 1;
3200
3201            for item in listing {
3202                let item = match item {
3203                    Ok(item) => item,
3204                    Err(e) => {
3205                        report.errors.push(Error::io(&abs_dir, e));
3206                        continue;
3207                    }
3208                };
3209                crate::counters::bump(|c| c.dir_entries += 1);
3210                let name = item.file_name();
3211                let (kind, attrs) = match listed_child_kind_and_attrs(
3212                    &item,
3213                    emission.skip_dir_symlink_stat(),
3214                    config.one_filesystem,
3215                ) {
3216                    Ok(Some(observed)) => observed,
3217                    Ok(None) => continue,
3218                    Err(error) => {
3219                        report.errors.push(Error::io(item.path(), error));
3220                        continue;
3221                    }
3222                };
3223                if !emission.record_entry(
3224                    root,
3225                    &rel_dir,
3226                    depth,
3227                    region,
3228                    &name,
3229                    kind,
3230                    attrs,
3231                    root_dev,
3232                    config,
3233                    &mut directory,
3234                    &mut discovered,
3235                    &mut report,
3236                    sender,
3237                    &mut chunk_send_ns,
3238                    diagnostics.map(AsRef::as_ref),
3239                ) {
3240                    // The consumer is gone; nothing further will be read.
3241                    consumer_gone = true;
3242                    break 'walk;
3243                }
3244            }
3245            emission.finish_directory(directory);
3246        }
3247        // Publish facts that authorize newly discovered directories before making
3248        // those directories claimable. Both emission modes preserve this boundary.
3249        if !emission.publish_before_discovery(
3250            !discovered.is_empty(),
3251            sender,
3252            &mut chunk_send_ns,
3253            diagnostics.map(AsRef::as_ref),
3254        ) {
3255            report.attribution.send_ns += chunk_send_ns;
3256            report.attribution.work_ns += elapsed_ns(chunk_started).saturating_sub(chunk_send_ns);
3257            consumer_gone = true;
3258            break 'walk;
3259        }
3260        report.attribution.send_ns += chunk_send_ns;
3261        let chunk_work_ns = elapsed_ns(chunk_started).saturating_sub(chunk_send_ns);
3262        report.attribution.work_ns += chunk_work_ns;
3263        // Progress is reported per chunk for the same reason timing is: the chunk is
3264        // the unit of handoff, so it is the unit the shared counters are touched in.
3265        tally.flush(&report);
3266
3267        // Publish new work before releasing the claim so a worker that finds nothing
3268        // new does not hold work that others could be doing.
3269        if !discovered.is_empty() {
3270            queue.extend(discovered.drain(..), &mut report.attribution);
3271        }
3272        if let Some(target_workers) = claim.release(
3273            report.entries.saturating_sub(entries_before),
3274            chunk_work_ns,
3275            &mut report.attribution,
3276        ) {
3277            // Carry a sender in-band so the consumer can create the reserve workers
3278            // without retaining a channel endpoint that would keep a small scan alive.
3279            // Only the release that completes a slow calibration returns true, so one
3280            // message expands the pool exactly once.
3281            let _ = sender.send(WalkMessage::ScaleUp { sender: sender.clone(), target_workers });
3282        }
3283    }
3284
3285    if !consumer_gone {
3286        emission.finish(sender, &mut report, diagnostics.map(AsRef::as_ref));
3287    }
3288    // A worker that left mid-chunk because its consumer was gone still read what it
3289    // read, and the report it returns says so.
3290    tally.flush(&report);
3291    report.attribution.wall_ns = elapsed_ns(worker_started);
3292    report
3293}
3294
3295#[allow(clippy::too_many_arguments)]
3296fn record_detached_entry(
3297    root: &Path,
3298    rel_dir: &Path,
3299    depth: usize,
3300    region: RegionId,
3301    name: &OsStr,
3302    kind: EntryKind,
3303    attrs: Attrs,
3304    root_dev: u64,
3305    config: &ScanConfig,
3306    children: &mut Vec<DetachedChild>,
3307    control: &mut Option<Op>,
3308    discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3309    report: &mut ScanReport,
3310) {
3311    let disposition = crate::admission::decide(name, kind, config.hidden(), config.exclude_special);
3312    if disposition == crate::admission::Disposition::Reject {
3313        return;
3314    }
3315    // Construct a full path only for the one fixed control name. The scanner's public
3316    // preparation builds one for every retained entry because that path escapes in an
3317    // observation; this private builder keeps ordinary children component-only.
3318    if config.read_controls && name == OsStr::new(crate::control::CONTROL_FILE_NAME) {
3319        let path = rel_dir.join(name);
3320        match read_control_op(config, root, &path, kind) {
3321            // A listing can repeat the control name while the directory changes. The
3322            // later read wins, as the builder keeps the later observation of the entry.
3323            Ok(observed) => *control = observed,
3324            Err(error) => report.errors.push(error),
3325        }
3326    }
3327    if disposition != crate::admission::Disposition::Retain {
3328        return;
3329    }
3330    report.observe(kind, attrs);
3331    // Positions only order repeated names, and no real listing reaches `u32::MAX` entries.
3332    let position = u32::try_from(children.len()).unwrap_or(u32::MAX);
3333    children.push(DetachedChild { name: name.to_os_string(), kind, attrs, position });
3334    if should_descend(kind, attrs, depth, root_dev, config) {
3335        let child_region = if depth == 0 { RegionId::UNASSIGNED } else { region };
3336        discovered.push((rel_dir.join(name), depth + 1, child_region));
3337    }
3338}
3339
3340/// One filesystem entry after the scan's shared admission, control, and descent rules.
3341pub(crate) struct PreparedWalkEntry {
3342    pub(crate) path: PathBuf,
3343    pub(crate) kind: EntryKind,
3344    pub(crate) attrs: Attrs,
3345    pub(crate) retained: bool,
3346    pub(crate) control: Option<Op>,
3347    pub(crate) descend: bool,
3348    pub(crate) control_error: Option<Error>,
3349}
3350
3351/// Apply the producer-independent part of a directory walk to one verified entry.
3352///
3353/// Both blocking and opened-root scans call this after obtaining non-following metadata,
3354/// which keeps admission, fixed controls, and traversal boundaries from drifting.
3355#[allow(clippy::too_many_arguments)]
3356pub(crate) fn prepare_walk_entry(
3357    root: &Path,
3358    rel_dir: &Path,
3359    depth: usize,
3360    name: &OsStr,
3361    kind: EntryKind,
3362    attrs: Attrs,
3363    root_dev: u64,
3364    config: &ScanConfig,
3365) -> Option<PreparedWalkEntry> {
3366    let disposition = crate::admission::decide(name, kind, config.hidden(), config.exclude_special);
3367    if disposition == crate::admission::Disposition::Reject {
3368        return None;
3369    }
3370    let path = rel_dir.join(name);
3371    let (control, control_error) = match read_control_op(config, root, &path, kind) {
3372        Ok(control) => (control, None),
3373        Err(error) => (None, Some(error)),
3374    };
3375    Some(PreparedWalkEntry {
3376        path,
3377        kind,
3378        attrs,
3379        retained: disposition == crate::admission::Disposition::Retain,
3380        control,
3381        descend: should_descend(kind, attrs, depth, root_dev, config),
3382        control_error,
3383    })
3384}
3385
3386#[allow(clippy::too_many_arguments)]
3387fn record_walk_entry(
3388    root: &Path,
3389    rel_dir: &Path,
3390    depth: usize,
3391    region: RegionId,
3392    name: &OsStr,
3393    kind: EntryKind,
3394    attrs: Attrs,
3395    root_dev: u64,
3396    config: &ScanConfig,
3397    emission: &mut StreamingEmission,
3398    discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3399    report: &mut ScanReport,
3400    sender: &std::sync::mpsc::Sender<WalkMessage>,
3401    chunk_send_ns: &mut u64,
3402    diagnostics: Option<&ScanDiagnosticsRecorder>,
3403) -> bool {
3404    let Some(prepared) =
3405        prepare_walk_entry(root, rel_dir, depth, name, kind, attrs, root_dev, config)
3406    else {
3407        return true;
3408    };
3409    if let Some(error) = prepared.control_error {
3410        report.errors.push(error);
3411    }
3412    if !prepared.retained {
3413        if let Some(control) = prepared.control {
3414            emission.batch.push(ObservationOp::unconditional(control));
3415            if emission.batch.len() >= config.batch_size {
3416                let send_started = std::time::Instant::now();
3417                let sent = emission.send_full(sender, diagnostics);
3418                *chunk_send_ns += elapsed_ns(send_started);
3419                return sent;
3420            }
3421        }
3422        return true;
3423    }
3424    report.observe(kind, attrs);
3425    emission.batch.push(ObservationOp::unconditional(Op::Upsert {
3426        path: prepared.path.clone(),
3427        kind,
3428        attrs,
3429    }));
3430    if emission.batch.len() >= config.batch_size {
3431        let send_started = std::time::Instant::now();
3432        let sent = emission.send_full(sender, diagnostics);
3433        *chunk_send_ns += elapsed_ns(send_started);
3434        if !sent {
3435            return false;
3436        }
3437    }
3438    if let Some(control) = prepared.control {
3439        emission.batch.push(ObservationOp::unconditional(control));
3440        if emission.batch.len() >= config.batch_size {
3441            let send_started = std::time::Instant::now();
3442            let sent = emission.send_full(sender, diagnostics);
3443            *chunk_send_ns += elapsed_ns(send_started);
3444            if !sent {
3445                return false;
3446            }
3447        }
3448    }
3449    if prepared.descend {
3450        // A child of the root seeds a new region; everything deeper inherits its
3451        // parent's. Region membership therefore costs one integer copy and never
3452        // inspects a path.
3453        let child_region = if depth == 0 { RegionId::UNASSIGNED } else { region };
3454        discovered.push((prepared.path, depth + 1, child_region));
3455    }
3456    true
3457}
3458
3459/// Observe one control file if the scan's policy asks for control state at all.
3460///
3461/// Every control observation goes through here -- each walk and reconcile site, and the
3462/// watch layer's verification -- so the policy cannot be forgotten at one of them. A
3463/// watch must honor it like a scan does: its scope has to equal the index's, the scope
3464/// carries this bit, and a verifier that read control files regardless would grow a
3465/// partial rule set, from whichever sources events touched, under a scope that says
3466/// there is none.
3467pub(crate) fn read_control_op(
3468    config: &ScanConfig,
3469    root: &Path,
3470    path: &Path,
3471    kind: EntryKind,
3472) -> Result<Option<Op>> {
3473    if !config.read_controls {
3474        return Ok(None);
3475    }
3476    read_control_op_unconditional(root, path, kind, config.control_limits.budget)
3477}
3478
3479/// Read one fixed control source without allowing a raced or hostile file to allocate
3480/// beyond the index-wide control budget.
3481///
3482/// A file longer than the budget is read only to one byte past it. No table under that
3483/// budget can admit a source that long, since every retained byte is charged at least
3484/// once, so the truncated source it sends is refused for the budget rather than parsed.
3485///
3486/// Private to this module, so no caller elsewhere can step around the policy gate in
3487/// `read_control_op`.
3488fn read_control_op_unconditional(
3489    root: &Path,
3490    path: &Path,
3491    kind: EntryKind,
3492    budget: Option<usize>,
3493) -> Result<Option<Op>> {
3494    if !crate::control::is_control_file(path) {
3495        return Ok(None);
3496    }
3497    if kind != EntryKind::File {
3498        return Ok(Some(Op::ControlRemove { path: path.to_path_buf() }));
3499    }
3500    let absolute = root.join(path);
3501    let file = open_control_file(&absolute).map_err(|error| Error::io(&absolute, error))?;
3502    if !file.metadata().map_err(|error| Error::io(&absolute, error))?.file_type().is_file() {
3503        return Ok(Some(Op::ControlRemove { path: path.to_path_buf() }));
3504    }
3505    let read_limit = budget
3506        .map_or(u64::MAX, |budget| u64::try_from(budget).unwrap_or(u64::MAX).saturating_add(1));
3507    let mut source = Vec::new();
3508    file.take(read_limit).read_to_end(&mut source).map_err(|error| Error::io(&absolute, error))?;
3509    crate::counters::bump(|counts| counts.control_reads = counts.control_reads.saturating_add(1));
3510    Ok(Some(Op::ControlUpsert { path: path.to_path_buf(), source }))
3511}
3512
3513#[cfg(unix)]
3514fn open_control_file(path: &Path) -> std::io::Result<fs::File> {
3515    use std::os::unix::fs::OpenOptionsExt as _;
3516
3517    fs::OpenOptions::new().read(true).custom_flags(libc::O_NONBLOCK | libc::O_NOFOLLOW).open(path)
3518}
3519
3520#[cfg(not(unix))]
3521fn open_control_file(path: &Path) -> std::io::Result<fs::File> {
3522    fs::File::open(path)
3523}
3524
3525fn send_scanner_batch(
3526    sender: &std::sync::mpsc::Sender<WalkMessage>,
3527    batch: ScannerBatch,
3528    diagnostics: Option<&ScanDiagnosticsRecorder>,
3529) -> bool {
3530    if let Some(diagnostics) = diagnostics {
3531        diagnostics.handoff_sent();
3532    }
3533    let sent = sender.send(WalkMessage::Batch(batch)).is_ok();
3534    if !sent {
3535        // Balance the reservation when the receiver disappeared before accepting it.
3536        if let Some(diagnostics) = diagnostics {
3537            diagnostics.handoff_received();
3538        }
3539    }
3540    sent
3541}
3542
3543fn send_detached_directories(
3544    sender: &std::sync::mpsc::Sender<WalkMessage>,
3545    directories: Vec<DetachedDirectory>,
3546    diagnostics: Option<&ScanDiagnosticsRecorder>,
3547) -> bool {
3548    if let Some(diagnostics) = diagnostics {
3549        diagnostics.handoff_sent();
3550    }
3551    let sent = sender.send(WalkMessage::DetachedDirectories(directories)).is_ok();
3552    if !sent {
3553        if let Some(diagnostics) = diagnostics {
3554            diagnostics.handoff_received();
3555        }
3556    }
3557    sent
3558}
3559
3560/// Nanoseconds since `started`, saturating rather than panicking on the absurd.
3561fn elapsed_ns(started: std::time::Instant) -> u64 {
3562    u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)
3563}
3564
3565/// A top-level subtree, used to spread workers across the breadth of the tree.
3566///
3567/// Every directory below the root belongs to the region seeded by its depth-1
3568/// ancestor, inherited from its parent rather than recomputed from its path. The root
3569/// itself is [`RegionId::ROOT`], which exists only to bootstrap.
3570#[derive(Clone, Copy, PartialEq, Eq, Debug)]
3571struct RegionId(usize);
3572
3573impl RegionId {
3574    /// The root's own region, which exists only to bootstrap the walk.
3575    const ROOT: Self = Self(0);
3576    /// "Allocate a fresh region for this directory." Resolved by
3577    /// [`DirectoryQueueState::push`], which is the only place holding the lock that
3578    /// owns the region table.
3579    const UNASSIGNED: Self = Self(usize::MAX);
3580}
3581
3582/// Directories still to read, plus enough state to know when the walk is finished.
3583///
3584/// The termination condition is the only subtle part: the queue being empty does not
3585/// mean the walk is done, because a worker that is mid-directory may be about to push
3586/// its children. So a worker holds a claim from the moment it takes work until the
3587/// moment it has published everything that work produced, and the walk ends only when
3588/// the queue is empty *and* no claim is outstanding.
3589///
3590/// # Why breadth-first is region-scheduled rather than a global FIFO
3591///
3592/// A global FIFO orders the *queue*, but claims are unordered: workers take whatever
3593/// is at the front, which on a real tree means several workers grinding through the
3594/// same top-level subtree while others sit untouched. Measured on the branching
3595/// fixture, that left a global-FIFO walk starting the same 7–8 of 12 subtrees at the
3596/// halfway mark as depth-first did — the ordering bought nothing a consumer could see.
3597/// It also made the pending set hold a whole level of the tree, which is where the
3598/// +1.5–3.7% peak RSS in exp-012 came from.
3599///
3600/// So breadth-first keeps work in per-region buckets and hands each free worker a
3601/// *different* region, round-robin. Within a region the bucket is LIFO, which restores
3602/// depth-first's locality and spine-bounded memory. Nothing waits on a level boundary:
3603/// if only one region has work, every worker takes it. The result is a scheduler whose
3604/// shallow preference is expressed in *which subtree a worker picks up*, not in the
3605/// order a single queue drains — which is the property progressive consumers actually
3606/// need.
3607struct DirectoryQueue {
3608    state: std::sync::Mutex<DirectoryQueueState>,
3609    ready: std::sync::Condvar,
3610    order: ScanOrder,
3611    diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
3612}
3613
3614/// One outstanding claim, held for exactly as long as the worker owes the queue the
3615/// work it took.
3616///
3617/// Giving the claim back is the queue's liveness condition, not a courtesy: [`claim`]
3618/// parks every other worker on the condvar while `outstanding` is nonzero, so a single
3619/// claim that is never returned stops the whole walk and the scoped join that waits on
3620/// it. That makes `Drop` the only safe place to put the release, because the paths that
3621/// skip a hand-written call are exactly the ones that matter — the `break` taken when
3622/// the consumer disconnects, and an unwinding panic inside a directory read.
3623///
3624/// [`claim`]: DirectoryQueue::claim
3625struct DirectoryClaim<'a> {
3626    queue: &'a DirectoryQueue,
3627    /// Directories represented by the claim, for exact in-flight accounting.
3628    directories: usize,
3629    /// Whether the worker already returned this claim through [`Self::release`].
3630    released: bool,
3631}
3632
3633impl DirectoryClaim<'_> {
3634    /// Return the claim at the end of a completed chunk, feeding the chunk's own
3635    /// measurements to the shared calibration.
3636    ///
3637    /// Returns the queue's scale-up decision, which is why the normal path cannot be
3638    /// `Drop`: a destructor has neither the chunk's timing nor anywhere to put an
3639    /// answer.
3640    fn release(
3641        mut self,
3642        entries: u64,
3643        work_ns: u64,
3644        timing: &mut WalkAttribution,
3645    ) -> Option<usize> {
3646        self.released = true;
3647        self.queue.release(self.directories, entries, work_ns, timing)
3648    }
3649}
3650
3651impl Drop for DirectoryClaim<'_> {
3652    fn drop(&mut self) {
3653        if self.released {
3654            return;
3655        }
3656        // An abandoned chunk: the consumer went away mid-directory, or a read panicked.
3657        // Either way the partial timing describes an aborted chunk rather than the cost
3658        // of reading directories, so it must not reach the calibration that sizes the
3659        // worker pool. Returning the claim is the whole job.
3660        self.queue.abandon(self.directories);
3661    }
3662}
3663
3664struct DirectoryQueueState {
3665    /// Depth-first's single stack. Unused under breadth-first.
3666    pending: VecDeque<(PathBuf, usize, RegionId)>,
3667    /// Breadth-first's per-region work, indexed by [`RegionId`]. Each is a LIFO stack.
3668    regions: Vec<Vec<(PathBuf, usize, RegionId)>>,
3669    /// Regions with work, in round-robin order. A region appears at most once; the
3670    /// flag array is what keeps that true without scanning the ring.
3671    ready_ring: VecDeque<RegionId>,
3672    /// Whether each region is currently in `ready_ring`.
3673    enqueued: Vec<bool>,
3674    /// Directories currently available for a future claim.
3675    ready_directories: usize,
3676    /// Directories held by outstanding claims.
3677    in_flight_directories: usize,
3678    outstanding: usize,
3679    finished: bool,
3680    controller: Option<WorkerController>,
3681    /// Observation-only windows retained after the shipped one-shot decision.
3682    shadow_calibration: Option<RepeatedCalibration>,
3683    /// Completion-order sequence assigned under the queue lock.
3684    next_policy_sequence: u64,
3685    worker_target: usize,
3686    maximum_workers: usize,
3687}
3688
3689impl DirectoryQueueState {
3690    fn seeded(
3691        root: (PathBuf, usize),
3692        order: ScanOrder,
3693        calibration: Option<WorkerCalibration>,
3694        initial_workers: usize,
3695        maximum_workers: usize,
3696        policy: WorkerPolicyExperiment,
3697    ) -> Self {
3698        let mut state = Self {
3699            pending: VecDeque::new(),
3700            regions: Vec::new(),
3701            ready_ring: VecDeque::new(),
3702            enqueued: Vec::new(),
3703            ready_directories: 0,
3704            in_flight_directories: 0,
3705            outstanding: 0,
3706            finished: false,
3707            controller: calibration.map(|value| WorkerController::new(value, policy)),
3708            shadow_calibration: None,
3709            next_policy_sequence: 0,
3710            worker_target: initial_workers,
3711            maximum_workers,
3712        };
3713        state.push((root.0, root.1, RegionId::ROOT), order);
3714        state
3715    }
3716
3717    /// Push one directory into the structure the order uses.
3718    fn push(&mut self, item: (PathBuf, usize, RegionId), order: ScanOrder) {
3719        self.ready_directories = self.ready_directories.saturating_add(1);
3720        match order {
3721            ScanOrder::DepthFirst => self.pending.push_back(item),
3722            ScanOrder::BreadthFirst => {
3723                let region = if item.2 == RegionId::UNASSIGNED {
3724                    // One region per top-level subtree, numbered as they are found.
3725                    self.regions.len().max(1)
3726                } else {
3727                    item.2.0
3728                };
3729                if region >= self.regions.len() {
3730                    self.regions.resize_with(region + 1, Vec::new);
3731                    self.enqueued.resize(region + 1, false);
3732                }
3733                // Resolve the id *into* the item, so every directory discovered beneath
3734                // this one inherits a concrete region instead of the sentinel. Without
3735                // this the sentinel propagates and each directory allocates a region of
3736                // its own, degenerating the scheduler into round-robin over the whole
3737                // frontier.
3738                let mut item = item;
3739                item.2 = RegionId(region);
3740                self.regions[region].push(item);
3741                if !self.enqueued[region] {
3742                    self.enqueued[region] = true;
3743                    self.ready_ring.push_back(RegionId(region));
3744                }
3745            }
3746        }
3747    }
3748
3749    /// Whether any work is available.
3750    fn is_empty(&self, order: ScanOrder) -> bool {
3751        match order {
3752            ScanOrder::DepthFirst => self.pending.is_empty(),
3753            ScanOrder::BreadthFirst => self.ready_ring.is_empty(),
3754        }
3755    }
3756
3757    /// Take up to `limit` directories from the next region in the round-robin ring.
3758    ///
3759    /// Every region holding work is in the ring exactly once, so popping it always
3760    /// finds work and always moves to a *different* subtree than the previous claim.
3761    /// An earlier version preferred the caller's previous region for locality, which
3762    /// pinned each worker to one subtree: with twelve deep chains and six workers only
3763    /// six subtrees ever advanced, and depth-first — whose four-directory claims
3764    /// happen to fan across the root's children — spread wider than breadth-first did.
3765    /// Locality still comes from the claim being a run of directories out of one
3766    /// region; it must not come from a worker refusing to leave.
3767    fn take(
3768        &mut self,
3769        limit: usize,
3770        order: ScanOrder,
3771        into: &mut Vec<(PathBuf, usize, RegionId)>,
3772    ) -> usize {
3773        let before = into.len();
3774        match order {
3775            ScanOrder::DepthFirst => {
3776                let take = self.pending.len().min(limit);
3777                let start = self.pending.len() - take;
3778                into.extend(self.pending.drain(start..));
3779            }
3780            ScanOrder::BreadthFirst => {
3781                let Some(region) = self.ready_ring.pop_front() else { return 0 };
3782                self.enqueued[region.0] = false;
3783                let bucket = &mut self.regions[region.0];
3784                let take = bucket.len().min(limit);
3785                let start = bucket.len() - take;
3786                into.extend(bucket.drain(start..));
3787                // Re-arm the region only if work remains and it is not already queued,
3788                // so a busy region cannot appear twice and starve the others.
3789                if !bucket.is_empty() && !self.enqueued[region.0] {
3790                    self.enqueued[region.0] = true;
3791                    self.ready_ring.push_back(region);
3792                }
3793            }
3794        }
3795        into.len().saturating_sub(before)
3796    }
3797
3798    fn allocate_policy_sequence(&mut self) -> u64 {
3799        let sequence = self.next_policy_sequence;
3800        self.next_policy_sequence = self.next_policy_sequence.saturating_add(1);
3801        sequence
3802    }
3803}
3804
3805impl DirectoryQueue {
3806    /// Seed the queue with the root, which is region zero until its children fan out.
3807    #[cfg(test)]
3808    fn new(
3809        root: (PathBuf, usize),
3810        order: ScanOrder,
3811        calibration: Option<WorkerCalibration>,
3812        diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
3813    ) -> Self {
3814        Self::new_with_policy(
3815            root,
3816            order,
3817            calibration,
3818            diagnostics,
3819            1,
3820            2,
3821            WorkerPolicyExperiment::ShippedOneShot,
3822        )
3823    }
3824
3825    #[allow(clippy::too_many_arguments)]
3826    fn new_with_policy(
3827        root: (PathBuf, usize),
3828        order: ScanOrder,
3829        calibration: Option<WorkerCalibration>,
3830        diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
3831        initial_workers: usize,
3832        maximum_workers: usize,
3833        policy: WorkerPolicyExperiment,
3834    ) -> Self {
3835        let state = DirectoryQueueState::seeded(
3836            root,
3837            order,
3838            calibration,
3839            initial_workers,
3840            maximum_workers,
3841            policy,
3842        );
3843        Self {
3844            state: std::sync::Mutex::new(state),
3845            ready: std::sync::Condvar::new(),
3846            order,
3847            diagnostics,
3848        }
3849    }
3850
3851    /// Take up to [`DIR_CLAIM`] directories, blocking until there is work or the walk
3852    /// is over. Returns `None` once no more work will ever arrive.
3853    ///
3854    /// Time spent waiting is charged to `timing`: lock acquisition to `lock_wait_ns`
3855    /// when contended, condvar waits to `starved_ns`. The condvar span includes the
3856    /// lock re-acquisition on wake, which slightly overstates starvation rather than
3857    /// understating contention — the fail-honest direction for the number that is
3858    /// supposed to stay near zero.
3859    fn claim<'a>(
3860        &'a self,
3861        into: &mut Vec<(PathBuf, usize, RegionId)>,
3862        timing: &mut WalkAttribution,
3863    ) -> Option<DirectoryClaim<'a>> {
3864        let mut state = self.lock_timed(timing);
3865        loop {
3866            if !state.is_empty(self.order) {
3867                let directories = state.take(DIR_CLAIM, self.order, into);
3868                state.ready_directories = state.ready_directories.saturating_sub(directories);
3869                state.in_flight_directories =
3870                    state.in_flight_directories.saturating_add(directories);
3871                state.outstanding += 1;
3872                timing.claims += 1;
3873                return Some(DirectoryClaim { queue: self, directories, released: false });
3874            }
3875            if state.finished {
3876                return None;
3877            }
3878            if state.outstanding == 0 {
3879                state.finished = true;
3880                self.ready.notify_all();
3881                return None;
3882            }
3883            let started = std::time::Instant::now();
3884            state = self.ready.wait(state).unwrap_or_else(std::sync::PoisonError::into_inner);
3885            timing.starved_ns += elapsed_ns(started);
3886        }
3887    }
3888
3889    fn extend(
3890        &self,
3891        directories: impl Iterator<Item = (PathBuf, usize, RegionId)>,
3892        timing: &mut WalkAttribution,
3893    ) {
3894        let mut state = self.lock_timed(timing);
3895        for item in directories {
3896            state.push(item, self.order);
3897        }
3898        drop(state);
3899        self.ready.notify_all();
3900    }
3901
3902    /// Give up a claim whose chunk never finished. Wakes everyone if it was the last.
3903    ///
3904    /// Reached only from [`DirectoryClaim::drop`], where there is no `WalkAttribution`
3905    /// to charge and nothing worth charging: an abandoned chunk read some unknown
3906    /// fraction of its directories, so its lock wait says nothing about contention
3907    /// during the walk.
3908    fn abandon(&self, directories: usize) {
3909        let mut state = self.lock();
3910        state.outstanding -= 1;
3911        state.in_flight_directories = state.in_flight_directories.saturating_sub(directories);
3912        if state.outstanding == 0 && state.is_empty(self.order) {
3913            state.finished = true;
3914            drop(state);
3915            self.ready.notify_all();
3916        }
3917    }
3918
3919    /// Give up a claim taken by [`claim`]. Wakes everyone if this was the last one.
3920    ///
3921    /// The chunk's own entry count and work time feed the shared service-time
3922    /// calibration, so the decision uses the timing the walk already collects for
3923    /// attribution rather than a second clock. Returns the new worker target when a
3924    /// controller requests expansion. A release that ends the walk returns no target,
3925    /// because there is no longer useful work for a reserve worker to take.
3926    fn release(
3927        &self,
3928        directories: usize,
3929        observed_entries: u64,
3930        observed_work_ns: u64,
3931        timing: &mut WalkAttribution,
3932    ) -> Option<usize> {
3933        let mut state = self.lock_timed(timing);
3934        // Whether this chunk reached the calibration at all. Only chunks released while
3935        // it is still live are policy history; later ones are ordinary walk work.
3936        let calibrating = state.controller.is_some();
3937        let one_shot = state.controller.as_ref().is_some_and(WorkerController::is_one_shot);
3938        let controller_spec = state.controller.as_ref().map(WorkerController::calibration_spec);
3939        let staged_gated = state.controller.as_ref().is_some_and(WorkerController::is_staged_gated);
3940        let completed_window = state
3941            .controller
3942            .as_mut()
3943            .and_then(|value| value.observe(observed_entries, observed_work_ns));
3944        let observed_window = state
3945            .shadow_calibration
3946            .as_mut()
3947            .and_then(|value| value.observe(observed_entries, observed_work_ns));
3948        let window_completed = completed_window.is_some();
3949        state.outstanding -= 1;
3950        state.in_flight_directories = state.in_flight_directories.saturating_sub(directories);
3951        let finished = state.outstanding == 0 && state.is_empty(self.order);
3952        if finished {
3953            state.finished = true;
3954        }
3955        let handoff_backlog = self.diagnostics.as_ref().map_or(0, |diagnostics| {
3956            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
3957        });
3958        let mut requested_workers = None;
3959        let policy_window = completed_window.map(|window| {
3960            let useful_frontier =
3961                state.ready_directories.saturating_add(state.in_flight_directories);
3962            let decision = if !window.slow {
3963                WorkerPolicyDecision::Hold
3964            } else if finished {
3965                WorkerPolicyDecision::HoldNoUsefulWork
3966            } else if staged_gated && useful_frontier <= state.worker_target {
3967                WorkerPolicyDecision::HoldInsufficientFrontier
3968            } else if staged_gated && handoff_backlog >= state.worker_target {
3969                WorkerPolicyDecision::HoldHandoffBacklog
3970            } else {
3971                let target = if staged_gated {
3972                    state.worker_target.saturating_mul(2).min(state.maximum_workers)
3973                } else {
3974                    state.maximum_workers
3975                };
3976                if target > state.worker_target {
3977                    state.worker_target = target;
3978                    requested_workers = Some(target);
3979                    WorkerPolicyDecision::ScaleUp
3980                } else {
3981                    WorkerPolicyDecision::HoldInsufficientFrontier
3982                }
3983            };
3984            let sequence = state.allocate_policy_sequence();
3985            PolicyWindowSnapshot {
3986                sequence,
3987                start_entry_ordinal: window.start_entry_ordinal,
3988                end_entry_ordinal: window.end_entry_ordinal,
3989                observed_entries: window.entries,
3990                observed_chunks: window.chunks,
3991                observed_work_ns: window.work_ns,
3992                ready_directories: state.ready_directories,
3993                in_flight_directories: state.in_flight_directories,
3994                active_workers: self.diagnostics.as_ref().map_or(0, |diagnostics| {
3995                    diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
3996                }),
3997                handoff_backlog,
3998                requested_workers,
3999                decision,
4000            }
4001        });
4002        let shadow_window = observed_window.map(|window| {
4003            let sequence = state.allocate_policy_sequence();
4004            PolicyWindowSnapshot {
4005                sequence,
4006                start_entry_ordinal: window.start_entry_ordinal,
4007                end_entry_ordinal: window.end_entry_ordinal,
4008                observed_entries: window.entries,
4009                observed_chunks: window.chunks,
4010                observed_work_ns: window.work_ns,
4011                ready_directories: state.ready_directories,
4012                in_flight_directories: state.in_flight_directories,
4013                active_workers: self.diagnostics.as_ref().map_or(0, |diagnostics| {
4014                    diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
4015                }),
4016                handoff_backlog,
4017                requested_workers: None,
4018                decision: if window.slow {
4019                    WorkerPolicyDecision::ObserveSlow
4020                } else {
4021                    WorkerPolicyDecision::ObserveFast
4022                },
4023            }
4024        });
4025        let controller_terminates = (one_shot || finished) && window_completed
4026            || requested_workers.is_some_and(|target| target == state.maximum_workers);
4027        if controller_terminates {
4028            // Continue observing after every terminal decision, including a candidate
4029            // that reached the maximum pool. Without this shadow history a slow-prefix
4030            // expansion makes a later fast phase unobservable, precisely the
4031            // irreversible over-expansion case the evidence matrix must detect.
4032            if let (Some(spec), Some(window)) = (controller_spec, completed_window) {
4033                if self.diagnostics.is_some() && !finished {
4034                    state.shadow_calibration =
4035                        Some(RepeatedCalibration::starting_at(spec, window.end_entry_ordinal));
4036                }
4037            }
4038            state.controller = None;
4039        }
4040        drop(state);
4041
4042        // The sequence was assigned while the queue was locked, but the trace lock and
4043        // bounded-vector update stay outside that critical section. Recorder arrival
4044        // may differ from completion order; `finish` sorts the retained prefix.
4045        if let (Some(diagnostics), Some(window)) = (&self.diagnostics, policy_window) {
4046            diagnostics.record_policy_window(window);
4047        }
4048        if let (Some(diagnostics), Some(window)) = (&self.diagnostics, shadow_window) {
4049            diagnostics.record_policy_window(window);
4050        }
4051
4052        // Recorded outside the lock: the sampling is off by default, and a disabled
4053        // counter must not lengthen the critical section it observes.
4054        if calibrating {
4055            record_adaptive_calibration_chunk(
4056                self.diagnostics.as_ref(),
4057                observed_entries,
4058                observed_work_ns,
4059            );
4060        }
4061
4062        if finished {
4063            self.ready.notify_all();
4064        }
4065        requested_workers
4066    }
4067
4068    /// Acquire the state lock, charging any contention to `timing`.
4069    ///
4070    /// The fast path is a `try_lock` that succeeds and costs one counter increment;
4071    /// only the contended path pays for reading the clock. Poisoning is tolerated for
4072    /// the same reason as [`Self::lock`].
4073    fn lock_timed(
4074        &self,
4075        timing: &mut WalkAttribution,
4076    ) -> std::sync::MutexGuard<'_, DirectoryQueueState> {
4077        timing.lock_ops += 1;
4078        match self.state.try_lock() {
4079            Ok(guard) => guard,
4080            Err(std::sync::TryLockError::Poisoned(poisoned)) => poisoned.into_inner(),
4081            Err(std::sync::TryLockError::WouldBlock) => {
4082                timing.lock_contended += 1;
4083                let started = std::time::Instant::now();
4084                let guard = self.lock();
4085                timing.lock_wait_ns += elapsed_ns(started);
4086                guard
4087            }
4088        }
4089    }
4090
4091    /// A poisoned queue means a worker panicked mid-walk. The data behind the lock is
4092    /// a plain work list with no invariant that a panic could have broken, and the
4093    /// caller already reports the panic as a scan error, so recovering the list is
4094    /// strictly better than propagating a second panic into every other worker.
4095    fn lock(&self) -> std::sync::MutexGuard<'_, DirectoryQueueState> {
4096        self.state.lock().unwrap_or_else(std::sync::PoisonError::into_inner)
4097    }
4098}
4099
4100fn record_adaptive_calibration_chunk(
4101    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
4102    entries: u64,
4103    work_ns: u64,
4104) {
4105    if let Some(diagnostics) = diagnostics {
4106        diagnostics.calibration_chunk(entries, work_ns);
4107    }
4108    crate::counters::bump(|counts| {
4109        counts.adaptive_calibration_chunks = counts.adaptive_calibration_chunks.saturating_add(1);
4110        counts.adaptive_calibration_entries =
4111            counts.adaptive_calibration_entries.saturating_add(entries);
4112        counts.adaptive_calibration_work_us =
4113            counts.adaptive_calibration_work_us.saturating_add(work_ns / 1_000);
4114    });
4115}
4116
4117fn record_adaptive_worker_expansion(diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>) {
4118    if let Some(diagnostics) = diagnostics {
4119        diagnostics.worker_expanded();
4120    }
4121    crate::counters::bump(|counts| {
4122        counts.adaptive_scale_ups = counts.adaptive_scale_ups.saturating_add(1);
4123    });
4124}
4125
4126fn scan_detached_directories(
4127    root: &Path,
4128    config: &ScanConfig,
4129    collect_diagnostics: bool,
4130    policy: WorkerPolicyExperiment,
4131) -> Result<(ScanReport, DetachedIndexBuilder, Option<ScanDiagnostics>)> {
4132    if let Some(progress) = &config.progress {
4133        progress.enter(crate::ProgressPhase::Scanning);
4134    }
4135    let root_metadata = {
4136        crate::counters::bump(|counts| counts.stats += 1);
4137        fs::symlink_metadata(root)
4138    }
4139    .map_err(|error| Error::io(root, error))?;
4140    if !root_metadata.is_dir() {
4141        return Err(Error::io(
4142            root,
4143            std::io::Error::new(std::io::ErrorKind::NotADirectory, "scan root is not a directory"),
4144        ));
4145    }
4146    let root_dev = root_device(root, &root_metadata).map_err(|error| Error::io(root, error))?;
4147    let available_parallelism =
4148        std::thread::available_parallelism().map_or(1, std::num::NonZero::get);
4149    let pool = config.worker_pool_for(available_parallelism);
4150    let diagnostics = collect_diagnostics
4151        .then(|| ScanDiagnosticsRecorder::new(pool, available_parallelism, policy));
4152
4153    if config.max_depth == Some(0) {
4154        if let Some(diagnostics) = &diagnostics {
4155            diagnostics.mark_not_run();
4156            diagnostics.record_queue_finish(0, 0);
4157        }
4158        return Ok((
4159            ScanReport::default(),
4160            DetachedIndexBuilder::new(root, config.scope(), config.types_shared())
4161                .with_control_limits(config.control_limits),
4162            diagnostics.as_ref().map(|value| value.finish()),
4163        ));
4164    }
4165
4166    let walk_started = crate::counters::enabled().then(std::time::Instant::now);
4167    let (output, builder) =
4168        scan_concurrent_detached(root, config, root_dev, pool, diagnostics.as_ref(), policy)?;
4169    // Also reached by a walk no worker left, such as a single-threaded one, so every
4170    // cold index ends its walk in the same phase.
4171    if let Some(progress) = &config.progress {
4172        progress.enter(crate::ProgressPhase::Indexing);
4173    }
4174    if let Some(started) = walk_started {
4175        let elapsed = elapsed_ns(started) / 1_000;
4176        crate::counters::bump(|counts| {
4177            counts.detached_walk_us = counts.detached_walk_us.saturating_add(elapsed);
4178        });
4179    }
4180    Ok((output, builder, diagnostics.as_ref().map(|value| value.finish())))
4181}
4182
4183fn consolidate_detached_index(
4184    mut output: ScanReport,
4185    builder: DetachedIndexBuilder,
4186) -> (Index, ScanReport) {
4187    let entries = output.entries;
4188    let consolidate_started = crate::counters::enabled().then(std::time::Instant::now);
4189    let mut index = builder.finish();
4190    if let Some(started) = consolidate_started {
4191        let elapsed = elapsed_ns(started) / 1_000;
4192        crate::counters::bump(|counts| {
4193            counts.detached_builds = counts.detached_builds.saturating_add(1);
4194            counts.detached_entries = counts.detached_entries.saturating_add(entries);
4195            counts.detached_finish_us = counts.detached_finish_us.saturating_add(elapsed);
4196        });
4197    }
4198    index.record_walk_errors(&mut output.errors);
4199    index.set_initial_scan_freshness(&output.errors);
4200    (index, output)
4201}
4202
4203/// Walk `root` and return a fully populated index.
4204pub fn scan_into_index(root: &Path, config: &ScanConfig) -> Result<(Index, ScanReport)> {
4205    config.validate()?;
4206    let root = root.canonicalize().map_err(|error| Error::io(root, error))?;
4207    let (output, builder, _diagnostics) =
4208        scan_detached_directories(&root, config, false, WorkerPolicyExperiment::ShippedOneShot)?;
4209    Ok(consolidate_detached_index(output, builder))
4210}
4211
4212#[cfg(test)]
4213fn scan_into_index_via_scanner(root: &Path, config: &ScanConfig) -> Result<(Index, ScanReport)> {
4214    let mut index = Index::new_with_scope_and_types(root, config.scope(), config.types_shared());
4215    index.set_control_limits(config.control_limits);
4216    let mut apply_error: Option<Error> = None;
4217    let (mut report, _diagnostics) = scan_internal(
4218        root,
4219        config,
4220        &mut |batch| {
4221            if apply_error.is_none() {
4222                if let Err(error) = index.apply_scanner_baseline(batch) {
4223                    apply_error = Some(error);
4224                }
4225            }
4226        },
4227        false,
4228        WorkerPolicyExperiment::ShippedOneShot,
4229        SinkMode::Retained,
4230    )?;
4231    if let Some(error) = apply_error {
4232        return Err(error);
4233    }
4234    index.record_walk_errors(&mut report.errors);
4235    index.set_initial_scan_freshness(&report.errors);
4236    Ok((index, report))
4237}
4238
4239/// Walk `root` into an index and retain the opt-in diagnostic trace for that run.
4240///
4241/// The index and [`ScanReport`] have the same semantics as [`scan_into_index`]. The
4242/// additional trace is versioned independently so evidence tooling can fail closed on
4243/// changes without coupling cache or engine behavior to measurement details.
4244pub fn scan_into_index_with_diagnostics(
4245    root: &Path,
4246    config: &ScanConfig,
4247) -> Result<(Index, ScanReport, ScanDiagnostics)> {
4248    scan_into_index_with_policy_diagnostics(root, config, WorkerPolicyExperiment::ShippedOneShot)
4249}
4250
4251/// Exercise a repository-only worker-controller candidate while building an index.
4252#[doc(hidden)]
4253pub fn scan_into_index_with_policy_diagnostics(
4254    root: &Path,
4255    config: &ScanConfig,
4256    policy: WorkerPolicyExperiment,
4257) -> Result<(Index, ScanReport, ScanDiagnostics)> {
4258    config.validate()?;
4259    let root = root.canonicalize().map_err(|error| Error::io(root, error))?;
4260    let (output, builder, diagnostics) = scan_detached_directories(&root, config, true, policy)?;
4261    let (index, report) = consolidate_detached_index(output, builder);
4262    Ok((index, report, diagnostics.expect("diagnostic detached scan creates a recorder")))
4263}
4264
4265/// Diff the filesystem against an existing index and emit conditional observations.
4266///
4267/// This is cache tier 2: after a snapshot is loaded, a sweep like this is what makes the
4268/// answer trustworthy rather than merely fast. Unchanged entries produce upserts whose
4269/// fingerprints already match, which the index discards as no-ops, so the caller can
4270/// apply the whole stream without filtering it first.
4271///
4272/// Entries the index holds but the filesystem no longer has become [`Op::Remove`],
4273/// detected per directory rather than by accumulating every visited path in memory.
4274///
4275/// This observation-only reference API assumes its emitted stream is applied to the same
4276/// unchanged baseline after the borrow ends. Use [`reconcile`] or [`reconcile_handle`]
4277/// when other producers can write concurrently; those paths capture the stronger
4278/// generation/revision/absence expectations returned by [`Index::expectation`].
4279pub fn revalidate(
4280    index: &Index,
4281    config: &ScanConfig,
4282    sink: &mut dyn FnMut(Observation),
4283) -> Result<ScanReport> {
4284    config.validate_for_scope(index.scope())?;
4285    let root = index.root_path().to_path_buf();
4286    let root_meta = {
4287        crate::counters::bump(|c| c.stats += 1);
4288        fs::symlink_metadata(&root)
4289    }
4290    .map_err(|error| Error::io(&root, error))?;
4291    if !root_meta.is_dir() {
4292        return Err(Error::io(
4293            &root,
4294            std::io::Error::new(
4295                std::io::ErrorKind::NotADirectory,
4296                "revalidation root is not a directory",
4297            ),
4298        ));
4299    }
4300    let root_dev = root_device(&root, &root_meta).map_err(|error| Error::io(&root, error))?;
4301    if let Some(progress) = &config.progress {
4302        progress.enter(crate::ProgressPhase::Revalidating);
4303    }
4304    let mut report = ScanReport::default();
4305    let mut tally = ProgressTally::new(config.progress.as_ref());
4306    let batch_limit = config.batch_size.max(1);
4307    let mut batch: Vec<ObservationOp> = Vec::with_capacity(batch_limit);
4308    if config.max_depth == Some(0) {
4309        if let Some(children) = index.children(Path::new("")) {
4310            for (name, _) in children {
4311                let path = PathBuf::from(name);
4312                batch.push(ObservationOp::if_state(
4313                    Op::Remove { path: path.clone() },
4314                    index.relaxed_expectation(&path),
4315                ));
4316                if batch.len() >= batch_limit {
4317                    sink(Observation::from_ops(std::mem::take(&mut batch)));
4318                    batch.reserve(batch_limit);
4319                }
4320            }
4321        }
4322        if !batch.is_empty() {
4323            sink(Observation::from_ops(batch));
4324        }
4325        return Ok(report);
4326    }
4327    let mut queue: VecDeque<(PathBuf, usize)> = VecDeque::from(vec![(PathBuf::new(), 0)]);
4328
4329    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
4330        let abs_dir = root.join(&rel_dir);
4331        crate::counters::bump(|c| c.dir_opens += 1);
4332        let listing = match fs::read_dir(&abs_dir) {
4333            Ok(listing) => listing,
4334            Err(e) => {
4335                report.errors.push(Error::io(abs_dir, e));
4336                continue;
4337            }
4338        };
4339        report.dirs_read += 1;
4340
4341        let mut seen: BTreeSet<OsString> = BTreeSet::new();
4342        let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
4343        let mut had_control = index.control_table().contains(&control_path);
4344        let mut control_seen = false;
4345        let mut listing_complete = true;
4346        let listing = reconcile_listing(listing, &abs_dir);
4347        for item in listing {
4348            let item = match item {
4349                Ok(item) => item,
4350                Err(e) => {
4351                    listing_complete = false;
4352                    report.errors.push(Error::io(&abs_dir, e));
4353                    continue;
4354                }
4355            };
4356            let name = item.file_name();
4357            // Seeing the name proves it is not absent even when a following metadata
4358            // lookup fails. Record it before any fallible per-entry work so an
4359            // operational error cannot become a false removal in the missing sweep.
4360            seen.insert(name.clone());
4361            let rel_path = rel_dir.join(&name);
4362            let baseline = index.relaxed_expectation(&rel_path);
4363            let (kind, attrs) = match observe_dir_entry(&item) {
4364                Ok(Some(observed)) => observed,
4365                Ok(None) => {
4366                    let entry_held = baseline.state != PathState::Absent;
4367                    for removal in
4368                        vanished_child_removals(&rel_dir, &name, entry_held, &mut had_control)
4369                            .into_iter()
4370                            .flatten()
4371                    {
4372                        batch.push(ObservationOp::if_state(removal, baseline));
4373                    }
4374                    if batch.len() >= batch_limit {
4375                        sink(Observation::from_ops(std::mem::take(&mut batch)));
4376                        batch.reserve(batch_limit);
4377                    }
4378                    continue;
4379                }
4380                Err(e) => {
4381                    control_seen |= name == crate::control::CONTROL_FILE_NAME;
4382                    report.errors.push(Error::io(item.path(), e));
4383                    continue;
4384                }
4385            };
4386            control_seen |= name == crate::control::CONTROL_FILE_NAME;
4387            let disposition =
4388                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
4389            let control = match read_control_op(config, &root, &rel_path, kind) {
4390                Ok(control) => control,
4391                Err(error) => {
4392                    report.errors.push(error);
4393                    None
4394                }
4395            };
4396            if disposition != crate::admission::Disposition::Retain {
4397                if baseline.state != PathState::Absent {
4398                    batch.push(ObservationOp::if_state(
4399                        Op::Remove { path: rel_path.clone() },
4400                        baseline,
4401                    ));
4402                }
4403                if disposition == crate::admission::Disposition::ControlOnly {
4404                    if let Some(control) = control {
4405                        batch.push(ObservationOp::if_state(control, baseline));
4406                    }
4407                }
4408                if batch.len() >= batch_limit {
4409                    sink(Observation::from_ops(std::mem::take(&mut batch)));
4410                    batch.reserve(batch_limit);
4411                }
4412                continue;
4413            }
4414            report.observe(kind, attrs);
4415            batch.push(ObservationOp::if_state(
4416                Op::Upsert { path: rel_path.clone(), kind, attrs },
4417                baseline,
4418            ));
4419            if batch.len() >= batch_limit {
4420                sink(Observation::from_ops(std::mem::take(&mut batch)));
4421                batch.reserve(batch_limit);
4422            }
4423            if let Some(control) = control {
4424                batch.push(ObservationOp::if_state(control, baseline));
4425                if batch.len() >= batch_limit {
4426                    sink(Observation::from_ops(std::mem::take(&mut batch)));
4427                    batch.reserve(batch_limit);
4428                }
4429            }
4430
4431            if should_descend(kind, attrs, depth, root_dev, config) {
4432                queue.push_back((rel_path, depth + 1));
4433            } else if kind.is_dir() {
4434                if let Some(children) = index.children(&rel_path) {
4435                    for (child_name, _) in children {
4436                        let child_path = rel_path.join(child_name);
4437                        batch.push(ObservationOp::if_state(
4438                            Op::Remove { path: child_path.clone() },
4439                            index.relaxed_expectation(&child_path),
4440                        ));
4441                        if batch.len() >= batch_limit {
4442                            sink(Observation::from_ops(std::mem::take(&mut batch)));
4443                            batch.reserve(batch_limit);
4444                        }
4445                    }
4446                }
4447            }
4448        }
4449
4450        // Anything the index still lists here but the filesystem did not return is gone.
4451        if listing_complete {
4452            if let Some(known) = index.children(&rel_dir) {
4453                for (name, _) in known {
4454                    if !seen.contains(name) {
4455                        let path = rel_dir.join(name);
4456                        batch.push(ObservationOp::if_state(
4457                            Op::Remove { path: path.clone() },
4458                            index.relaxed_expectation(&path),
4459                        ));
4460                    }
4461                }
4462            }
4463            if had_control && !control_seen {
4464                batch.push(ObservationOp::if_state(
4465                    Op::ControlRemove { path: control_path.clone() },
4466                    index.relaxed_expectation(&control_path),
4467                ));
4468            }
4469        }
4470        // Per directory: an unchanged tree fills no batch, so the batch cannot be the
4471        // unit here without the counters standing still for the whole walk.
4472        tally.flush(&report);
4473    }
4474
4475    if !batch.is_empty() {
4476        sink(Observation::from_ops(batch));
4477    }
4478    tally.flush(&report);
4479    Ok(report)
4480}
4481
4482/// Reconcile the full index and publish each exact commit as it lands.
4483pub fn reconcile(
4484    index: &mut Index,
4485    config: &ScanConfig,
4486    sink: &mut dyn FnMut(&Commit),
4487) -> Result<ReconcileReport> {
4488    reconcile_subtree(index, Path::new(""), config, sink)
4489}
4490
4491/// Reconcile one relative subtree, applying effective changes during the walk.
4492///
4493/// If an ancestor vanished or became a non-directory, reconciliation widens to that
4494/// ancestor so a child invalidation can converge instead of retrying `ENOTDIR` forever.
4495pub fn reconcile_subtree(
4496    index: &mut Index,
4497    subtree: &Path,
4498    config: &ScanConfig,
4499    sink: &mut dyn FnMut(&Commit),
4500) -> Result<ReconcileReport> {
4501    reconcile_target(&mut ReconcileTarget::Direct(index), subtree, config, sink)
4502}
4503
4504/// Reconcile a shared index while allowing readers between applied batches.
4505pub fn reconcile_handle(
4506    handle: &IndexHandle,
4507    config: &ScanConfig,
4508    sink: &mut dyn FnMut(&Commit),
4509) -> Result<ReconcileReport> {
4510    reconcile_subtree_handle(handle, Path::new(""), config, sink)
4511}
4512
4513/// Reconcile one subtree of a shared index, widening to a missing/non-directory ancestor
4514/// when necessary.
4515pub fn reconcile_subtree_handle(
4516    handle: &IndexHandle,
4517    subtree: &Path,
4518    config: &ScanConfig,
4519    sink: &mut dyn FnMut(&Commit),
4520) -> Result<ReconcileReport> {
4521    reconcile_target(&mut ReconcileTarget::Shared(handle), subtree, config, sink)
4522}
4523
4524/// Internal effects of one opened-root multi-path reconciliation.
4525#[derive(Debug, Default)]
4526pub(crate) struct ReconcilePathsReport {
4527    pub(crate) reconciliation: ReconcileReport,
4528    pub(crate) accepted: Vec<PathBuf>,
4529    pub(crate) rejected: Vec<crate::RejectedRefreshPath>,
4530}
4531
4532/// Reconcile one bounded path set under an opened-root lifecycle controller.
4533///
4534/// Classification precedes I/O, overlapping descendants fold into one walk, and all
4535/// surviving scopes enter `Reconciling` before the first is read. `forbid_expansion`
4536/// is the conservative resource-stop rule: removals and same-file verification remain
4537/// legal, while work that could retain another file or discover children is refused.
4538pub(crate) fn reconcile_paths_handle_controlled(
4539    handle: &IndexHandle,
4540    paths: &[PathBuf],
4541    config: &ScanConfig,
4542    forbid_expansion: bool,
4543    control: &dyn ReconcileControl,
4544    sink: &mut dyn FnMut(&Commit),
4545) -> Result<ReconcilePathsReport> {
4546    let mut target = ReconcileTarget::Controlled { handle, control };
4547    reconcile_paths_target(&mut target, paths, config, forbid_expansion, sink)
4548}
4549
4550fn reconcile_paths_target(
4551    target: &mut ReconcileTarget<'_>,
4552    paths: &[PathBuf],
4553    config: &ScanConfig,
4554    forbid_expansion: bool,
4555    sink: &mut dyn FnMut(&Commit),
4556) -> Result<ReconcilePathsReport> {
4557    config.validate_for_scope(target.scope()?)?;
4558    let mut report = ReconcilePathsReport::default();
4559    let mut accepted = BTreeSet::new();
4560
4561    for requested in paths {
4562        let reject = |reason| crate::RejectedRefreshPath { path: requested.clone(), reason };
4563        let Ok(path) = normalize_subtree(requested) else {
4564            report.rejected.push(reject(crate::RefreshRejection::OutsideRoot));
4565            continue;
4566        };
4567        if config.max_depth.is_some_and(|maximum| path.components().count() > maximum) {
4568            report.rejected.push(reject(crate::RefreshRejection::BeyondDepth));
4569            continue;
4570        }
4571        // This is lexical admission before the final kind is observed. Treating the
4572        // boundary as a file preserves the fixed hidden `.gitignore` control exception;
4573        // the verified walk still applies the real kind and special-object policy.
4574        if crate::admission::decide_path(&path, EntryKind::File, config.hidden(), false)
4575            == crate::admission::Disposition::Reject
4576        {
4577            report.rejected.push(reject(crate::RefreshRejection::NotAdmitted));
4578            continue;
4579        }
4580        if forbid_expansion && refresh_may_expand(target, &path, &mut report.reconciliation.scan)? {
4581            report.rejected.push(reject(crate::RefreshRejection::ResourceBudget));
4582            continue;
4583        }
4584        accepted.insert(path);
4585    }
4586
4587    report.accepted = accepted.into_iter().collect();
4588    let mut resolved = Vec::new();
4589    let mut unsafe_roots = Vec::new();
4590    for requested_root in covering_roots(report.accepted.clone()) {
4591        match resolve_subtree_root(target, &requested_root, config) {
4592            Ok(root) => resolved.push(root),
4593            Err(Error::SubtreeOutsideScanScope { .. }) => unsafe_roots.push(requested_root),
4594            Err(error) => return Err(error),
4595        }
4596    }
4597    if !unsafe_roots.is_empty() {
4598        let mut retained = Vec::with_capacity(report.accepted.len());
4599        for path in std::mem::take(&mut report.accepted) {
4600            if unsafe_roots.iter().any(|root| path.starts_with(root)) {
4601                report.rejected.push(crate::RejectedRefreshPath {
4602                    path,
4603                    reason: crate::RefreshRejection::UnsafeAncestry,
4604                });
4605            } else {
4606                retained.push(path);
4607            }
4608        }
4609        report.accepted = retained;
4610    }
4611    let walked = covering_roots(resolved);
4612    if walked.is_empty() {
4613        return Ok(report);
4614    }
4615
4616    let mut opened = Vec::with_capacity(walked.len());
4617    for subtree in walked {
4618        let (started_at, commit) = target.begin_reconcile(&subtree)?;
4619        if let Some(commit) = commit.as_ref() {
4620            sink(commit);
4621        }
4622        opened.push((subtree, started_at));
4623    }
4624
4625    // Each subtree closes on its own walk's outcome, so a subtree that could not be read
4626    // neither marks a verified sibling partial nor withholds the completeness its listing
4627    // earned.
4628    let mut failure = None;
4629    let mut outcomes = Vec::with_capacity(opened.len());
4630    for (subtree, started_at) in &opened {
4631        if failure.is_some() {
4632            outcomes.push((false, false));
4633            continue;
4634        }
4635        match reconcile_target_inner(
4636            target,
4637            subtree,
4638            *started_at,
4639            config,
4640            MAX_DEFERRED_RECONCILE_OPS,
4641            sink,
4642        ) {
4643            Ok(mut reconciliation) => {
4644                outcomes.push((
4645                    reconciliation.is_complete(),
4646                    reconciliation.apply.stale == 0 && reconciliation.apply.resource_refused == 0,
4647                ));
4648                reconciliation.listed_incomplete = reconciliation.take_recordable_completeness();
4649                merge_reconcile_report(&mut report.reconciliation, reconciliation);
4650            }
4651            Err(error) => {
4652                outcomes.push((false, false));
4653                failure = Some(error);
4654            }
4655        }
4656    }
4657
4658    let listed_incomplete = std::mem::take(&mut report.reconciliation.listed_incomplete);
4659    let root = target.root_path()?;
4660    normalize_walk_errors(&root, &mut report.reconciliation.scan.errors);
4661    let failed_paths = failure_paths(target, &report.reconciliation.scan.errors)?;
4662    for ((subtree, started_at), (complete, disproves_old)) in opened.into_iter().zip(outcomes) {
4663        let commit = target.finish_reconcile(
4664            &subtree,
4665            started_at,
4666            complete,
4667            &listed_incomplete,
4668            &failed_paths,
4669            ReconcileErrors {
4670                errors: &report.reconciliation.scan.errors,
4671                terminal: failure.as_ref(),
4672                disproves_old,
4673            },
4674        )?;
4675        if let Some(commit) = commit.commit.as_ref() {
4676            sink(commit);
4677        }
4678        report.reconciliation.retry_required |= commit.retry;
4679    }
4680
4681    match failure {
4682        Some(error) => Err(error),
4683        None => Ok(report),
4684    }
4685}
4686
4687/// Root-relative paths whose filesystem facts a failed reconciliation could not verify.
4688///
4689/// An unscoped error returns an empty set, which makes the closer conservatively mark the
4690/// whole requested subtree partial. Precise I/O paths let verified siblings remain fresh.
4691fn failure_paths(target: &ReconcileTarget<'_>, errors: &[Error]) -> Result<Vec<PathBuf>> {
4692    if errors.is_empty() {
4693        return Ok(Vec::new());
4694    }
4695    let root = target.root_path()?;
4696    let mut paths = Vec::with_capacity(errors.len());
4697    for error in errors {
4698        let Some(path) = crate::Issue::from_error_under(&root, error).path else {
4699            return Ok(Vec::new());
4700        };
4701        if path.is_absolute() {
4702            return Ok(Vec::new());
4703        }
4704        paths.push(path);
4705    }
4706    paths.sort();
4707    paths.dedup();
4708    Ok(paths)
4709}
4710
4711/// Whether verification could increase the retained-file set.
4712///
4713/// This deliberately recognizes only cases that prove non-expansion. At a resource
4714/// boundary, uncertainty is a refusal rather than permission to exceed the bound.
4715fn refresh_may_expand(
4716    target: &ReconcileTarget<'_>,
4717    path: &Path,
4718    work: &mut ScanReport,
4719) -> Result<bool> {
4720    let current = target.expectation(path)?.state;
4721    let absolute = target.root_path()?.join(path);
4722    let observed = match fs::symlink_metadata(&absolute) {
4723        Ok(metadata) => {
4724            let Ok((kind, attrs)) = observe(&absolute, &metadata) else {
4725                return Ok(true);
4726            };
4727            work.observe(kind, attrs);
4728            Some(kind)
4729        }
4730        Err(error)
4731            if matches!(
4732                error.kind(),
4733                std::io::ErrorKind::NotFound | std::io::ErrorKind::NotADirectory
4734            ) =>
4735        {
4736            None
4737        }
4738        Err(_) => return Ok(true),
4739    };
4740    Ok(!matches!(
4741        (current, observed),
4742        (PathState::Present { kind: EntryKind::File, .. }, Some(EntryKind::File)) | (_, None)
4743    ))
4744}
4745
4746/// Drop every path covered by a shallower member of the same sorted set.
4747fn covering_roots(mut paths: Vec<PathBuf>) -> Vec<PathBuf> {
4748    paths.sort();
4749    paths.dedup();
4750    if paths.first().is_some_and(|first| first.as_os_str().is_empty()) {
4751        return vec![PathBuf::new()];
4752    }
4753    let mut roots: Vec<PathBuf> = Vec::with_capacity(paths.len());
4754    for path in paths {
4755        if roots.last().is_some_and(|kept| path.starts_with(kept)) {
4756            continue;
4757        }
4758        roots.push(path);
4759    }
4760    roots
4761}
4762
4763fn reconcile_target(
4764    target: &mut ReconcileTarget<'_>,
4765    subtree: &Path,
4766    config: &ScanConfig,
4767    sink: &mut dyn FnMut(&Commit),
4768) -> Result<ReconcileReport> {
4769    config.validate_for_scope(target.scope()?)?;
4770    if let Some(progress) = &config.progress {
4771        progress.enter(crate::ProgressPhase::Revalidating);
4772    }
4773    let subtree = normalize_subtree(subtree)?;
4774    if config.max_depth.is_some_and(|maximum| subtree.components().count() > maximum) {
4775        return Err(Error::SubtreeOutsideScanScope { path: subtree, scope: config.scope() });
4776    }
4777    let subtree = resolve_subtree_root(target, &subtree, config)?;
4778    let (started_at, started) = target.begin_reconcile(&subtree)?;
4779    if let Some(commit) = started.as_ref() {
4780        sink(commit);
4781    }
4782    match reconcile_target_inner(
4783        target,
4784        &subtree,
4785        started_at,
4786        config,
4787        MAX_DEFERRED_RECONCILE_OPS,
4788        sink,
4789    ) {
4790        Ok(mut report) => {
4791            let root = target.root_path()?;
4792            normalize_walk_errors(&root, &mut report.scan.errors);
4793            let listed_incomplete = report.take_recordable_completeness();
4794            let failed_paths = failure_paths(target, &report.scan.errors)?;
4795            let finished = target.finish_reconcile(
4796                &subtree,
4797                started_at,
4798                report.is_complete(),
4799                &listed_incomplete,
4800                &failed_paths,
4801                ReconcileErrors {
4802                    errors: &report.scan.errors,
4803                    terminal: None,
4804                    disproves_old: report.apply.stale == 0 && report.apply.resource_refused == 0,
4805                },
4806            )?;
4807            if let Some(commit) = finished.commit.as_ref() {
4808                sink(commit);
4809            }
4810            report.retry_required |= finished.retry;
4811            Ok(report)
4812        }
4813        Err(error) => {
4814            let finished = target.finish_reconcile(
4815                &subtree,
4816                started_at,
4817                false,
4818                &[],
4819                &[],
4820                ReconcileErrors { errors: &[], terminal: Some(&error), disproves_old: false },
4821            )?;
4822            if let Some(commit) = finished.commit.as_ref() {
4823                sink(commit);
4824            }
4825            Err(error)
4826        }
4827    }
4828}
4829
4830fn reconcile_target_inner(
4831    target: &mut ReconcileTarget<'_>,
4832    subtree: &Path,
4833    started_at: u64,
4834    config: &ScanConfig,
4835    max_deferred_ops: usize,
4836    sink: &mut dyn FnMut(&Commit),
4837) -> Result<ReconcileReport> {
4838    let root = target.root_path()?;
4839    let root_meta = {
4840        crate::counters::bump(|c| c.stats += 1);
4841        fs::symlink_metadata(&root)
4842    }
4843    .map_err(|error| Error::io(&root, error))?;
4844    if !root_meta.is_dir() {
4845        return Err(Error::io(
4846            &root,
4847            std::io::Error::new(
4848                std::io::ErrorKind::NotADirectory,
4849                "reconciliation root is not a directory",
4850            ),
4851        ));
4852    }
4853    let root_dev = root_device(&root, &root_meta).map_err(|error| Error::io(&root, error))?;
4854    let start_depth = subtree.components().count();
4855    let mut report =
4856        ReconcileReport { reconcile_epoch: Some(started_at), ..ReconcileReport::default() };
4857    let mut tally = ProgressTally::new(config.progress.as_ref());
4858    let mut retry_frontier = None;
4859    let mut batch: Vec<ObservationOp> = Vec::with_capacity(config.batch_size.max(1));
4860
4861    if config.max_depth == Some(0) {
4862        remove_known_children(target, Path::new(""), config, &mut batch, sink, &mut report)?;
4863        return Ok(report);
4864    }
4865
4866    if !subtree.as_os_str().is_empty() {
4867        let baseline = target.expectation(subtree)?;
4868        let absolute = root.join(subtree);
4869        let meta = match fs::symlink_metadata(&absolute) {
4870            Ok(meta) => meta,
4871            Err(error) if error.kind() == std::io::ErrorKind::NotFound => {
4872                batch.push(ObservationOp::if_state(
4873                    Op::Remove { path: subtree.to_path_buf() },
4874                    baseline,
4875                ));
4876                if target.has_control(subtree)? {
4877                    batch.push(ObservationOp::if_state(
4878                        Op::ControlRemove { path: subtree.to_path_buf() },
4879                        baseline,
4880                    ));
4881                }
4882                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
4883                return Ok(report);
4884            }
4885            Err(error) => {
4886                report.scan.errors.push(Error::io(&absolute, error));
4887                if baseline.state != PathState::Absent {
4888                    batch.push(ObservationOp::if_state(
4889                        Op::Remove { path: subtree.to_path_buf() },
4890                        baseline,
4891                    ));
4892                }
4893                if target.has_control(subtree)? {
4894                    batch.push(ObservationOp::if_state(
4895                        Op::ControlRemove { path: subtree.to_path_buf() },
4896                        baseline,
4897                    ));
4898                }
4899                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
4900                return Ok(report);
4901            }
4902        };
4903        let (kind, attrs) = match observe(&absolute, &meta) {
4904            Ok(observed) => observed,
4905            Err(error) => {
4906                report.scan.errors.push(Error::io(absolute, error));
4907                return Ok(report);
4908            }
4909        };
4910        let disposition =
4911            crate::admission::decide_path(subtree, kind, config.hidden(), config.exclude_special);
4912        if disposition != crate::admission::Disposition::Retain {
4913            if baseline.state != PathState::Absent {
4914                batch.push(ObservationOp::if_state(
4915                    Op::Remove { path: subtree.to_path_buf() },
4916                    baseline,
4917                ));
4918            }
4919            if disposition == crate::admission::Disposition::ControlOnly {
4920                match read_control_op(config, &root, subtree, kind) {
4921                    Ok(Some(control)) => {
4922                        batch.push(ObservationOp::if_state(control, baseline));
4923                    }
4924                    Ok(None) => {}
4925                    Err(error) => {
4926                        if target.has_control(subtree)? {
4927                            batch.push(ObservationOp::if_state(
4928                                Op::ControlRemove { path: subtree.to_path_buf() },
4929                                baseline,
4930                            ));
4931                        }
4932                        report.scan.errors.push(error);
4933                    }
4934                }
4935            }
4936            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
4937            return Ok(report);
4938        }
4939        report.scan.observe(kind, attrs);
4940        push_reconcile_upsert(target, subtree, kind, attrs, baseline, &mut batch, &mut report);
4941        // A retained control file at the root of the walk reads its rules here, as the
4942        // listing walk does for every retained entry it lists: a file does not descend,
4943        // so nothing below would read them, and the table kept the old source while the
4944        // pass reported complete and marked the path fresh. In the same batch as the
4945        // upsert, so both are arbitrated against one baseline.
4946        match read_control_op(config, &root, subtree, kind) {
4947            Ok(Some(control)) => batch.push(ObservationOp::if_state(control, baseline)),
4948            Ok(None) => {}
4949            Err(error) => {
4950                if target.has_control(subtree)? {
4951                    batch.push(ObservationOp::if_state(
4952                        Op::ControlRemove { path: subtree.to_path_buf() },
4953                        baseline,
4954                    ));
4955                }
4956                report.scan.errors.push(error);
4957            }
4958        }
4959        flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
4960        if !should_descend(kind, attrs, start_depth.saturating_sub(1), root_dev, config) {
4961            if kind.is_dir() {
4962                remove_known_children(target, subtree, config, &mut batch, sink, &mut report)?;
4963            }
4964            tally.flush(&report.scan);
4965            return Ok(report);
4966        }
4967    }
4968
4969    if subtree.as_os_str().is_empty() && config.reconciliation_worker_threads() > 1 {
4970        if let ReconcileTarget::Direct(index) = target {
4971            match reconcile_direct_parallel(index, &root, root_dev, config, max_deferred_ops, sink)?
4972            {
4973                DirectParallelOutcome::Complete(parallel) => return Ok(parallel),
4974                DirectParallelOutcome::RetrySerial { prefix, remaining } => {
4975                    report = prefix;
4976                    report.reconcile_epoch = Some(started_at);
4977                    retry_frontier = Some(remaining);
4978                    // The wave workers reported the prefix themselves, and the wave
4979                    // that overflowed as well: progress counts that wave's reads twice,
4980                    // once there and once as the serial retry rereads it, while this
4981                    // report counts each directory once. Work done, not the answer.
4982                    tally.skip_to(&report.scan);
4983                }
4984            }
4985        }
4986    }
4987
4988    let mut queue: VecDeque<(PathBuf, usize)> = retry_frontier
4989        .unwrap_or_else(|| VecDeque::from(vec![(subtree.to_path_buf(), start_depth)]));
4990    #[cfg(target_os = "macos")]
4991    let mut bulk_reader = (config.worker_threads() > 1).then(macos_bulk::Reader::new);
4992    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
4993        let (mut known, records_completeness) = target.listing_baseline(&rel_dir)?;
4994        let errors_before = report.scan.errors.len();
4995        let abs_dir = root.join(&rel_dir);
4996        let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
4997        let mut had_control = target.has_control(&control_path)?;
4998        let mut control_seen = false;
4999        let mut listing_complete = true;
5000        let process_entry = |name: OsString,
5001                             kind: EntryKind,
5002                             attrs: Attrs,
5003                             baseline: PathExpectation,
5004                             control_seen: &mut bool,
5005                             target: &mut ReconcileTarget<'_>,
5006                             queue: &mut VecDeque<(PathBuf, usize)>,
5007                             batch: &mut Vec<ObservationOp>,
5008                             sink: &mut dyn FnMut(&Commit),
5009                             report: &mut ReconcileReport|
5010         -> Result<()> {
5011            let rel_path = rel_dir.join(&name);
5012            *control_seen |= name == crate::control::CONTROL_FILE_NAME;
5013            let disposition =
5014                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
5015            if disposition != crate::admission::Disposition::Retain {
5016                if baseline.state != PathState::Absent {
5017                    batch.push(ObservationOp::if_state(
5018                        Op::Remove { path: rel_path.clone() },
5019                        baseline,
5020                    ));
5021                }
5022                if disposition == crate::admission::Disposition::ControlOnly {
5023                    match read_control_op(config, &root, &rel_path, kind) {
5024                        Ok(Some(control)) => {
5025                            batch.push(ObservationOp::if_state(control, baseline));
5026                        }
5027                        Ok(None) => {}
5028                        Err(error) => {
5029                            if target.has_control(&rel_path)? {
5030                                batch.push(ObservationOp::if_state(
5031                                    Op::ControlRemove { path: rel_path.clone() },
5032                                    baseline,
5033                                ));
5034                            }
5035                            report.scan.errors.push(error);
5036                        }
5037                    }
5038                }
5039                if batch.len() >= config.batch_size.max(1) {
5040                    flush_reconcile_batch(target, batch, sink, report)?;
5041                }
5042                return Ok(());
5043            }
5044            report.scan.observe(kind, attrs);
5045            push_reconcile_upsert(target, &rel_path, kind, attrs, baseline, batch, report);
5046            if batch.len() >= config.batch_size.max(1) {
5047                flush_reconcile_batch(target, batch, sink, report)?;
5048            }
5049            match read_control_op(config, &root, &rel_path, kind) {
5050                Ok(Some(control)) => {
5051                    batch.push(ObservationOp::if_state(control, baseline));
5052                    if batch.len() >= config.batch_size.max(1) {
5053                        flush_reconcile_batch(target, batch, sink, report)?;
5054                    }
5055                }
5056                Ok(None) => {}
5057                Err(error) => {
5058                    if target.has_control(&rel_path)? {
5059                        batch.push(ObservationOp::if_state(
5060                            Op::ControlRemove { path: rel_path.clone() },
5061                            baseline,
5062                        ));
5063                    }
5064                    report.scan.errors.push(error);
5065                }
5066            }
5067
5068            if should_descend(kind, attrs, depth, root_dev, config) {
5069                queue.push_back((rel_path, depth + 1));
5070            } else if kind.is_dir() {
5071                remove_known_children(target, &rel_path, config, batch, sink, report)?;
5072            }
5073            Ok(())
5074        };
5075
5076        #[cfg(target_os = "macos")]
5077        let used_bulk = if walk_hook_covers(&abs_dir) {
5078            false
5079        } else if let Some(entries) = bulk_reader.as_mut().and_then(|reader| reader.read(&abs_dir))
5080        {
5081            report.scan.dirs_read += 1;
5082            for entry in entries {
5083                let baseline = match known.remove(&entry.name) {
5084                    Some(baseline) => baseline,
5085                    None => target.expectation(&rel_dir.join(&entry.name))?,
5086                };
5087                process_entry(
5088                    entry.name,
5089                    entry.kind,
5090                    entry.attrs,
5091                    baseline,
5092                    &mut control_seen,
5093                    target,
5094                    &mut queue,
5095                    &mut batch,
5096                    sink,
5097                    &mut report,
5098                )?;
5099            }
5100            true
5101        } else {
5102            false
5103        };
5104        #[cfg(not(target_os = "macos"))]
5105        let used_bulk = false;
5106
5107        if !used_bulk {
5108            crate::counters::bump(|c| c.dir_opens += 1);
5109            let listing = match fs::read_dir(&abs_dir) {
5110                Ok(listing) => listing,
5111                Err(error) => {
5112                    report.scan.errors.push(Error::io(&abs_dir, error));
5113                    remove_known_children(target, &rel_dir, config, &mut batch, sink, &mut report)?;
5114                    if had_control {
5115                        let baseline = target.expectation(&control_path)?;
5116                        batch.push(ObservationOp::if_state(
5117                            Op::ControlRemove { path: control_path },
5118                            baseline,
5119                        ));
5120                        flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5121                    }
5122                    continue;
5123                }
5124            };
5125            report.scan.dirs_read += 1;
5126            let listing = reconcile_listing(listing, &abs_dir);
5127            for item in listing {
5128                let item = match item {
5129                    Ok(item) => item,
5130                    Err(error) => {
5131                        listing_complete = false;
5132                        report.scan.errors.push(Error::io(&abs_dir, error));
5133                        continue;
5134                    }
5135                };
5136                let name = item.file_name();
5137                // Seeing the name proves it is not absent even if the following
5138                // metadata lookup fails. Remove it from the missing set before that
5139                // fallible lookup so an operational error cannot turn an existing
5140                // entry into a deletion.
5141                let baseline = match known.remove(&name) {
5142                    Some(baseline) => baseline,
5143                    None => target.expectation(&rel_dir.join(&name))?,
5144                };
5145                let (kind, attrs) = match observe_dir_entry(&item) {
5146                    Ok(Some(observed)) => observed,
5147                    Ok(None) => {
5148                        let entry_held = baseline.state != PathState::Absent;
5149                        for removal in
5150                            vanished_child_removals(&rel_dir, &name, entry_held, &mut had_control)
5151                                .into_iter()
5152                                .flatten()
5153                        {
5154                            batch.push(ObservationOp::if_state(removal, baseline));
5155                        }
5156                        if batch.len() >= config.batch_size.max(1) {
5157                            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5158                        }
5159                        continue;
5160                    }
5161                    Err(error) => {
5162                        control_seen |= name == crate::control::CONTROL_FILE_NAME;
5163                        report.scan.errors.push(Error::io(item.path(), error));
5164                        if baseline.state != PathState::Absent {
5165                            batch.push(ObservationOp::if_state(
5166                                Op::Remove { path: rel_dir.join(&name) },
5167                                baseline,
5168                            ));
5169                        }
5170                        if name == crate::control::CONTROL_FILE_NAME && had_control {
5171                            batch.push(ObservationOp::if_state(
5172                                Op::ControlRemove { path: rel_dir.join(&name) },
5173                                baseline,
5174                            ));
5175                            had_control = false;
5176                        }
5177                        if batch.len() >= config.batch_size.max(1) {
5178                            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5179                        }
5180                        continue;
5181                    }
5182                };
5183                process_entry(
5184                    name,
5185                    kind,
5186                    attrs,
5187                    baseline,
5188                    &mut control_seen,
5189                    target,
5190                    &mut queue,
5191                    &mut batch,
5192                    sink,
5193                    &mut report,
5194                )?;
5195            }
5196        }
5197
5198        for (name, baseline) in known {
5199            batch.push(ObservationOp::if_state(Op::Remove { path: rel_dir.join(name) }, baseline));
5200            if batch.len() >= config.batch_size.max(1) {
5201                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5202            }
5203        }
5204        if had_control && !control_seen {
5205            let baseline = target.expectation(&control_path)?;
5206            batch.push(ObservationOp::if_state(Op::ControlRemove { path: control_path }, baseline));
5207        }
5208        if listing_complete {
5209            // Only a directory with no error inside its own processing vouches for its
5210            // child set; an error under a sibling or a descendant is that directory's
5211            // to answer for, as discovery decides completeness per directory.
5212            if records_completeness && report.scan.errors.len() == errors_before {
5213                report.listed_incomplete.push(rel_dir);
5214            }
5215        }
5216        // Per directory, as in `revalidate`: an unchanged tree hands the sink nothing.
5217        tally.flush(&report.scan);
5218    }
5219
5220    flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5221    report.scan.errors.sort_by_cached_key(ToString::to_string);
5222    tally.flush(&report.scan);
5223    Ok(report)
5224}
5225
5226#[derive(Debug, Default)]
5227struct DeferredReconcile {
5228    scan: ScanReport,
5229    unchanged: u64,
5230    operations: Vec<Op>,
5231    discovered: Vec<(PathBuf, usize, RegionId)>,
5232    listed_incomplete: Vec<PathBuf>,
5233}
5234
5235enum DirectParallelOutcome {
5236    Complete(ReconcileReport),
5237    RetrySerial { prefix: ReconcileReport, remaining: VecDeque<(PathBuf, usize)> },
5238}
5239
5240/// Reconcile an exclusive full tree in bounded immutable-baseline waves.
5241///
5242/// No index write occurs while a wave's workers hold shared baseline references. That
5243/// lets each worker discard exact no-ops where they are observed instead of funnelling
5244/// every entry through one consumer. Effective changes still enter through ordinary
5245/// observations between waves, preserving both the index's sole mutation contract and
5246/// progressive delta delivery.
5247///
5248/// Unlike every other reconciliation path, the operations a wave defers are
5249/// unconditional: they carry no [`ObservationOp::if_state`] guard. Three properties
5250/// have to hold together for that to be safe, and a change to any one of them puts the
5251/// guards back. The target is an exclusive `&mut Index`, so no other producer can
5252/// commit between a worker's read and the wave's write. Nothing is applied while
5253/// workers run, so no baseline a worker compared against can go stale beneath it. And
5254/// a directory is only reconciled in a wave after the wave that discovered it has
5255/// committed, so a parent is never absent when its children arrive.
5256fn reconcile_direct_parallel(
5257    index: &mut Index,
5258    root: &Path,
5259    root_dev: u64,
5260    config: &ScanConfig,
5261    max_deferred_ops: usize,
5262    sink: &mut dyn FnMut(&Commit),
5263) -> Result<DirectParallelOutcome> {
5264    let mut frontier = DirectoryQueueState::seeded(
5265        (PathBuf::new(), 0),
5266        config.order,
5267        None,
5268        1,
5269        1,
5270        WorkerPolicyExperiment::ShippedOneShot,
5271    );
5272    let mut report = ReconcileReport::default();
5273    while !frontier.is_empty(config.order) {
5274        let mut wave = Vec::with_capacity(RECONCILE_WAVE_DIRECTORIES);
5275        while wave.len() < RECONCILE_WAVE_DIRECTORIES && !frontier.is_empty(config.order) {
5276            let remaining = RECONCILE_WAVE_DIRECTORIES - wave.len();
5277            frontier.take(remaining.min(DIR_CLAIM), config.order, &mut wave);
5278        }
5279
5280        let next = std::sync::atomic::AtomicUsize::new(0);
5281        let deferred_count = std::sync::atomic::AtomicUsize::new(0);
5282        let overflowed = std::sync::atomic::AtomicBool::new(false);
5283        let workers = config.reconciliation_worker_threads().min(wave.len());
5284        let baseline: &Index = index;
5285        let results: Vec<DeferredReconcile> = std::thread::scope(|scope| {
5286            let handles: Vec<_> = (0..workers)
5287                .map(|_| {
5288                    scope.spawn(|| {
5289                        reconcile_wave_worker(
5290                            baseline,
5291                            root,
5292                            root_dev,
5293                            config,
5294                            &wave,
5295                            &next,
5296                            &deferred_count,
5297                            &overflowed,
5298                            max_deferred_ops,
5299                        )
5300                    })
5301                })
5302                .collect();
5303
5304            handles
5305                .into_iter()
5306                .map(|handle| {
5307                    if let Ok(worker) = handle.join() {
5308                        return worker;
5309                    }
5310                    let mut worker = DeferredReconcile::default();
5311                    worker.scan.errors.push(Error::io(
5312                        root,
5313                        std::io::Error::other("a reconciliation worker thread panicked"),
5314                    ));
5315                    worker
5316                })
5317                .collect()
5318        });
5319
5320        // Nothing from an overflowing wave was applied, so the ordinary incremental
5321        // reconciler can resume at that wave. Completed waves and their statistics are
5322        // retained exactly once; restarting from the root would count their unchanged
5323        // entries again and misreport the logical reconciliation pass.
5324        if overflowed.load(std::sync::atomic::Ordering::Relaxed) {
5325            let mut remaining: VecDeque<_> =
5326                wave.into_iter().map(|(path, depth, _region)| (path, depth)).collect();
5327            let mut deferred = Vec::with_capacity(DIR_CLAIM);
5328            while !frontier.is_empty(config.order) {
5329                frontier.take(DIR_CLAIM, config.order, &mut deferred);
5330                remaining.extend(deferred.drain(..).map(|(path, depth, _region)| (path, depth)));
5331            }
5332            return Ok(DirectParallelOutcome::RetrySerial { prefix: report, remaining });
5333        }
5334
5335        let operation_count = deferred_count.load(std::sync::atomic::Ordering::Relaxed);
5336        let mut operations = Vec::with_capacity(operation_count);
5337        for worker in results {
5338            report.listed_incomplete.extend(worker.listed_incomplete);
5339            report.scan.absorb(worker.scan);
5340            report.apply.unchanged += worker.unchanged;
5341            report.observations = report.observations.saturating_add(worker.unchanged);
5342            operations.extend(worker.operations);
5343            for directory in worker.discovered {
5344                frontier.push(directory, config.order);
5345            }
5346        }
5347        apply_deferred_reconcile(
5348            index,
5349            &mut operations,
5350            config,
5351            sink,
5352            &mut report.apply,
5353            &mut report.observations,
5354        )?;
5355    }
5356    report.scan.errors.sort_by_cached_key(ToString::to_string);
5357    Ok(DirectParallelOutcome::Complete(report))
5358}
5359
5360fn apply_deferred_reconcile(
5361    index: &mut Index,
5362    operations: &mut Vec<Op>,
5363    config: &ScanConfig,
5364    sink: &mut dyn FnMut(&Commit),
5365    stats: &mut ApplyStats,
5366    observations: &mut u64,
5367) -> Result<()> {
5368    // Parent upserts establish real directory attributes before children arrive.
5369    // Removals run deepest first so a parent removal never precedes an independently
5370    // observed descendant operation. Deterministic causal order also makes emitted
5371    // commits stable for callers.
5372    operations.sort_by(|left, right| {
5373        let left_remove = matches!(left, Op::Remove { .. });
5374        let right_remove = matches!(right, Op::Remove { .. });
5375        left_remove.cmp(&right_remove).then_with(|| {
5376            let left_depth = left.path().components().count();
5377            let right_depth = right.path().components().count();
5378            if left_remove {
5379                right_depth.cmp(&left_depth).then_with(|| left.path().cmp(right.path()))
5380            } else {
5381                left_depth.cmp(&right_depth).then_with(|| left.path().cmp(right.path()))
5382            }
5383        })
5384    });
5385
5386    let batch_limit = config.batch_size.max(1);
5387    let mut batch = Vec::with_capacity(batch_limit.min(operations.len()));
5388    for operation in operations.drain(..) {
5389        batch.push(operation);
5390        if batch.len() >= batch_limit {
5391            flush_direct_reconcile_batch(index, &mut batch, sink, stats, observations)?;
5392        }
5393    }
5394    flush_direct_reconcile_batch(index, &mut batch, sink, stats, observations)
5395}
5396
5397#[allow(clippy::too_many_arguments)]
5398fn reconcile_wave_worker(
5399    index: &Index,
5400    root: &Path,
5401    root_dev: u64,
5402    config: &ScanConfig,
5403    wave: &[(PathBuf, usize, RegionId)],
5404    next: &std::sync::atomic::AtomicUsize,
5405    deferred_count: &std::sync::atomic::AtomicUsize,
5406    overflowed: &std::sync::atomic::AtomicBool,
5407    max_deferred_ops: usize,
5408) -> DeferredReconcile {
5409    let _counter_guard = crate::counters::thread_flush_guard();
5410    let mut result = DeferredReconcile::default();
5411    let mut tally = ProgressTally::new(config.progress.as_ref());
5412    #[cfg(target_os = "macos")]
5413    let mut bulk_reader = macos_bulk::Reader::new();
5414
5415    loop {
5416        let start = next.fetch_add(DIR_CLAIM, std::sync::atomic::Ordering::Relaxed);
5417        if start >= wave.len() {
5418            break;
5419        }
5420        let end = start.saturating_add(DIR_CLAIM).min(wave.len());
5421        for (rel_dir, depth, region) in &wave[start..end] {
5422            let errors_before = result.scan.errors.len();
5423            let mut known = collect_child_expectations(index, rel_dir);
5424            let abs_dir = root.join(rel_dir);
5425            let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
5426            let mut had_control = index.control_table().contains(&control_path);
5427            let mut control_seen = false;
5428            let mut control_errors = Vec::new();
5429            let mut vanished = Vec::new();
5430            let mut unverified = Vec::new();
5431            let mut control_read_failed = false;
5432            let mut listing_open_failed = false;
5433
5434            {
5435                let mut process_entry =
5436                    |name: OsString,
5437                     kind: EntryKind,
5438                     attrs: Attrs,
5439                     baseline: PathExpectation,
5440                     control_seen: &mut bool| {
5441                        let rel_path = rel_dir.join(&name);
5442                        *control_seen |= name == crate::control::CONTROL_FILE_NAME;
5443                        let disposition = crate::admission::decide(
5444                            &name,
5445                            kind,
5446                            config.hidden(),
5447                            config.exclude_special,
5448                        );
5449                        if disposition == crate::admission::Disposition::Reject {
5450                            if baseline.state != PathState::Absent {
5451                                defer_reconcile_op(
5452                                    Op::Remove { path: rel_path },
5453                                    &mut result.operations,
5454                                    deferred_count,
5455                                    overflowed,
5456                                    max_deferred_ops,
5457                                );
5458                            }
5459                            return;
5460                        }
5461                        if disposition == crate::admission::Disposition::ControlOnly {
5462                            match read_control_op(config, root, &rel_path, kind) {
5463                                Ok(Some(Op::ControlUpsert { path, source })) => {
5464                                    if !index.control_table().source_is(&path, &source) {
5465                                        defer_reconcile_op(
5466                                            Op::ControlUpsert { path, source },
5467                                            &mut result.operations,
5468                                            deferred_count,
5469                                            overflowed,
5470                                            max_deferred_ops,
5471                                        );
5472                                    }
5473                                }
5474                                Ok(Some(Op::ControlRemove { path })) => {
5475                                    if index.control_table().contains(&path) {
5476                                        defer_reconcile_op(
5477                                            Op::ControlRemove { path },
5478                                            &mut result.operations,
5479                                            deferred_count,
5480                                            overflowed,
5481                                            max_deferred_ops,
5482                                        );
5483                                    }
5484                                }
5485                                Ok(Some(_) | None) => {}
5486                                Err(error) => {
5487                                    control_errors.push(error);
5488                                    control_read_failed = true;
5489                                }
5490                            }
5491                            return;
5492                        }
5493                        result.scan.entries += 1;
5494                        if kind == EntryKind::File {
5495                            result.scan.files_walked += 1;
5496                            result.scan.bytes_walked += attrs.size;
5497                            result.scan.allocated_walked += attrs.allocated;
5498                        }
5499                        if baseline.state == (PathState::Present { kind, attrs }) {
5500                            result.unchanged += 1;
5501                        } else {
5502                            defer_reconcile_op(
5503                                Op::Upsert { path: rel_path.clone(), kind, attrs },
5504                                &mut result.operations,
5505                                deferred_count,
5506                                overflowed,
5507                                max_deferred_ops,
5508                            );
5509                        }
5510                        match read_control_op(config, root, &rel_path, kind) {
5511                            Ok(Some(Op::ControlUpsert { path, source })) => {
5512                                if !index.control_table().source_is(&path, &source) {
5513                                    defer_reconcile_op(
5514                                        Op::ControlUpsert { path, source },
5515                                        &mut result.operations,
5516                                        deferred_count,
5517                                        overflowed,
5518                                        max_deferred_ops,
5519                                    );
5520                                }
5521                            }
5522                            Ok(Some(_) | None) => {}
5523                            Err(error) => {
5524                                control_errors.push(error);
5525                                control_read_failed |= name == crate::control::CONTROL_FILE_NAME;
5526                            }
5527                        }
5528
5529                        if should_descend(kind, attrs, *depth, root_dev, config) {
5530                            let child_region =
5531                                if *depth == 0 { RegionId::UNASSIGNED } else { *region };
5532                            result.discovered.push((rel_path, depth + 1, child_region));
5533                        } else if kind.is_dir() {
5534                            for name in collect_child_expectations(index, &rel_path).into_keys() {
5535                                defer_reconcile_op(
5536                                    Op::Remove { path: rel_path.join(name) },
5537                                    &mut result.operations,
5538                                    deferred_count,
5539                                    overflowed,
5540                                    max_deferred_ops,
5541                                );
5542                            }
5543                        }
5544                    };
5545
5546                #[cfg(target_os = "macos")]
5547                let used_bulk = if let Some(entries) =
5548                    (!walk_hook_covers(&abs_dir)).then(|| bulk_reader.read(&abs_dir)).flatten()
5549                {
5550                    result.scan.dirs_read += 1;
5551                    for entry in entries {
5552                        let baseline = known
5553                            .remove(&entry.name)
5554                            .unwrap_or_else(|| index.expectation(&rel_dir.join(&entry.name)));
5555                        process_entry(
5556                            entry.name,
5557                            entry.kind,
5558                            entry.attrs,
5559                            baseline,
5560                            &mut control_seen,
5561                        );
5562                    }
5563                    true
5564                } else {
5565                    false
5566                };
5567                #[cfg(not(target_os = "macos"))]
5568                let used_bulk = false;
5569
5570                if !used_bulk {
5571                    crate::counters::bump(|c| c.dir_opens += 1);
5572                    let listing = match fs::read_dir(&abs_dir) {
5573                        Ok(listing) => Some(listing),
5574                        Err(error) => {
5575                            result.scan.errors.push(Error::io(&abs_dir, error));
5576                            listing_open_failed = true;
5577                            None
5578                        }
5579                    };
5580                    if let Some(listing) = listing {
5581                        result.scan.dirs_read += 1;
5582                        let listing = reconcile_listing(listing, &abs_dir);
5583                        for item in listing {
5584                            let item = match item {
5585                                Ok(item) => item,
5586                                Err(error) => {
5587                                    result.scan.errors.push(Error::io(&abs_dir, error));
5588                                    continue;
5589                                }
5590                            };
5591                            let name = item.file_name();
5592                            // Match the serial path: an entry whose name was enumerated is
5593                            // not missing merely because its metadata could not be read.
5594                            let baseline = known
5595                                .remove(&name)
5596                                .unwrap_or_else(|| index.expectation(&rel_dir.join(&name)));
5597                            let (kind, attrs) = match observe_dir_entry(&item) {
5598                                Ok(Some(observed)) => observed,
5599                                Ok(None) => {
5600                                    // Removed once this directory's listing is done.
5601                                    vanished.push((name, baseline.state != PathState::Absent));
5602                                    continue;
5603                                }
5604                                Err(error) => {
5605                                    control_seen |= name == crate::control::CONTROL_FILE_NAME;
5606                                    result.scan.errors.push(Error::io(item.path(), error));
5607                                    unverified.push((name, baseline.state != PathState::Absent));
5608                                    continue;
5609                                }
5610                            };
5611                            process_entry(name, kind, attrs, baseline, &mut control_seen);
5612                        }
5613                    }
5614                }
5615            }
5616            control_read_failed |= listing_open_failed && had_control;
5617            result.scan.errors.append(&mut control_errors);
5618            if index.directory_complete(rel_dir) != Some(true)
5619                && result.scan.errors.len() == errors_before
5620            {
5621                result.listed_incomplete.push(rel_dir.clone());
5622            }
5623            for (name, entry_held) in vanished {
5624                for removal in vanished_child_removals(rel_dir, &name, entry_held, &mut had_control)
5625                    .into_iter()
5626                    .flatten()
5627                {
5628                    defer_reconcile_op(
5629                        removal,
5630                        &mut result.operations,
5631                        deferred_count,
5632                        overflowed,
5633                        max_deferred_ops,
5634                    );
5635                }
5636            }
5637            for (name, entry_held) in unverified {
5638                if entry_held {
5639                    defer_reconcile_op(
5640                        Op::Remove { path: rel_dir.join(&name) },
5641                        &mut result.operations,
5642                        deferred_count,
5643                        overflowed,
5644                        max_deferred_ops,
5645                    );
5646                }
5647                if name == crate::control::CONTROL_FILE_NAME {
5648                    control_read_failed = true;
5649                }
5650            }
5651            for (name, _) in known {
5652                defer_reconcile_op(
5653                    Op::Remove { path: rel_dir.join(name) },
5654                    &mut result.operations,
5655                    deferred_count,
5656                    overflowed,
5657                    max_deferred_ops,
5658                );
5659            }
5660            if had_control && (!control_seen || control_read_failed) {
5661                defer_reconcile_op(
5662                    Op::ControlRemove { path: control_path },
5663                    &mut result.operations,
5664                    deferred_count,
5665                    overflowed,
5666                    max_deferred_ops,
5667                );
5668            }
5669        }
5670        // Once per claimed chunk, as the cold walker reports, so a long wave on a slow
5671        // filesystem moves the counters while it runs rather than when it lands.
5672        tally.flush(&result.scan);
5673    }
5674    result
5675}
5676
5677/// What a listed child gone at its stat removes: its entry, if the baseline holds one, and
5678/// its rules, if it is the directory's control file and the table holds them.
5679///
5680/// The stat's `NotFound` is positive evidence that both are gone, so neither waits for a
5681/// complete listing; only a name the listing never returned has to, because it may merely
5682/// be unread. Removing a retained control file's entry drops its rules too, but a
5683/// hidden-pruned one has no entry, and without its own removal one unreadable sibling would
5684/// leave its rules applied with no file behind them. Clears `had_control` once the rules
5685/// are removed, so the listing's closing removals do not repeat it.
5686fn vanished_child_removals(
5687    dir: &Path,
5688    name: &OsStr,
5689    entry_held: bool,
5690    had_control: &mut bool,
5691) -> [Option<Op>; 2] {
5692    let path = dir.join(name);
5693    let rules = (*had_control && name == crate::control::CONTROL_FILE_NAME).then(|| {
5694        *had_control = false;
5695        Op::ControlRemove { path: path.clone() }
5696    });
5697    [entry_held.then_some(Op::Remove { path }), rules]
5698}
5699
5700fn defer_reconcile_op(
5701    operation: Op,
5702    operations: &mut Vec<Op>,
5703    deferred_count: &std::sync::atomic::AtomicUsize,
5704    overflowed: &std::sync::atomic::AtomicBool,
5705    max_deferred_ops: usize,
5706) {
5707    let position = deferred_count.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
5708    if position < max_deferred_ops {
5709        operations.push(operation);
5710    } else {
5711        overflowed.store(true, std::sync::atomic::Ordering::Relaxed);
5712    }
5713}
5714
5715fn flush_direct_reconcile_batch(
5716    index: &mut Index,
5717    batch: &mut Vec<Op>,
5718    sink: &mut dyn FnMut(&Commit),
5719    stats: &mut ApplyStats,
5720    observations: &mut u64,
5721) -> Result<()> {
5722    if batch.is_empty() {
5723        return Ok(());
5724    }
5725    *observations = observations.saturating_add(u64::try_from(batch.len()).unwrap_or(u64::MAX));
5726    let outcome = index.apply(&Observation::new(std::mem::take(batch)))?;
5727    merge_apply_stats(stats, outcome.stats);
5728    if let Some(commit) = outcome.commit.as_ref() {
5729        sink(commit);
5730    }
5731    Ok(())
5732}
5733
5734/// Drain and reconcile every pending invalidation, collapsing nested requests.
5735///
5736/// An invalidation whose reconciliation comes back incomplete -- a subtree that could not
5737/// be read, or a conditional commit that lost a race -- is queued again, so the next call
5738/// retries it. That suits a caller that drains when it chooses. A caller that drains after
5739/// every event would re-walk an unreadable subtree each time, and belongs on
5740/// [`reconcile_pending_handle`], which settles it instead.
5741pub fn reconcile_pending(
5742    index: &mut Index,
5743    config: &ScanConfig,
5744    sink: &mut dyn FnMut(&Commit),
5745) -> Result<ReconcileReport> {
5746    let mut target = ReconcileTarget::Direct(index);
5747    reconcile_pending_target(&mut target, config, sink)
5748}
5749
5750/// Drain and reconcile invalidations on a shared index.
5751///
5752/// Unlike [`reconcile_pending`], a subtree that could not be read is not queued again.
5753/// `Watcher::apply_next` drains after every event, and a retry there re-walks the same
5754/// unreadable subtree on each unrelated one, for the life of the watch. The error is a
5755/// settled boundary instead: the subtree stays [`crate::Freshness::Partial`], the returned
5756/// report names the error once, and the watcher retains its cause as an issue. Only a lost
5757/// race -- a stale conditional commit -- is queued for the next call. Invalidate the subtree
5758/// again to retry it deliberately.
5759pub fn reconcile_pending_handle(
5760    handle: &IndexHandle,
5761    config: &ScanConfig,
5762    sink: &mut dyn FnMut(&Commit),
5763) -> Result<ReconcileReport> {
5764    let mut target = ReconcileTarget::Shared(handle);
5765    reconcile_pending_target(&mut target, config, sink)
5766}
5767
5768/// Drain and reconcile invalidations under an opened-root lifecycle and resource bound.
5769#[cfg(feature = "watch")]
5770pub(crate) fn reconcile_pending_handle_controlled(
5771    handle: &IndexHandle,
5772    config: &ScanConfig,
5773    control: &dyn ReconcileControl,
5774    sink: &mut dyn FnMut(&Commit),
5775) -> Result<ReconcileReport> {
5776    let mut target = ReconcileTarget::Controlled { handle, control };
5777    reconcile_pending_target(&mut target, config, sink)
5778}
5779
5780fn reconcile_pending_target(
5781    target: &mut ReconcileTarget<'_>,
5782    config: &ScanConfig,
5783    sink: &mut dyn FnMut(&Commit),
5784) -> Result<ReconcileReport> {
5785    config.validate_for_scope(target.scope()?)?;
5786    let roots = take_invalidation_roots(target)?;
5787    let mut combined = ReconcileReport::default();
5788    for (position, (root, reason)) in roots.iter().enumerate() {
5789        match reconcile_target(target, root, config, sink) {
5790            Ok(report) => {
5791                if target.retries_incomplete(&report) {
5792                    target.restore_pending_invalidations(vec![(root.clone(), *reason)])?;
5793                }
5794                merge_reconcile_report(&mut combined, report);
5795            }
5796            Err(error) => {
5797                target.restore_pending_invalidations(roots[position..].to_vec())?;
5798                return Err(error);
5799            }
5800        }
5801    }
5802    Ok(combined)
5803}
5804
5805fn take_invalidation_roots(
5806    target: &mut ReconcileTarget<'_>,
5807) -> Result<Vec<(PathBuf, crate::InvalidateReason)>> {
5808    let mut pending = target.take_pending_invalidations()?;
5809    pending.sort_by(|(left, _), (right, _)| {
5810        left.components().count().cmp(&right.components().count()).then_with(|| left.cmp(right))
5811    });
5812    let mut roots: Vec<(PathBuf, crate::InvalidateReason)> = Vec::new();
5813    for (path, reason) in pending {
5814        if roots.iter().any(|(root, _)| path.starts_with(root)) {
5815            continue;
5816        }
5817        roots.push((path, reason));
5818    }
5819
5820    Ok(roots)
5821}
5822
5823fn remove_known_children(
5824    target: &mut ReconcileTarget<'_>,
5825    path: &Path,
5826    config: &ScanConfig,
5827    batch: &mut Vec<ObservationOp>,
5828    sink: &mut dyn FnMut(&Commit),
5829    report: &mut ReconcileReport,
5830) -> Result<()> {
5831    for (name, baseline) in target.child_states(path)? {
5832        batch.push(ObservationOp::if_state(Op::Remove { path: path.join(name) }, baseline));
5833        if batch.len() >= config.batch_size.max(1) {
5834            flush_reconcile_batch(target, batch, sink, report)?;
5835        }
5836    }
5837    flush_reconcile_batch(target, batch, sink, report)
5838}
5839
5840fn push_reconcile_upsert(
5841    target: &ReconcileTarget<'_>,
5842    path: &Path,
5843    kind: EntryKind,
5844    attrs: Attrs,
5845    baseline: PathExpectation,
5846    batch: &mut Vec<ObservationOp>,
5847    report: &mut ReconcileReport,
5848) {
5849    // An exclusive Index borrow cannot race another index producer. If filesystem
5850    // metadata exactly matches the captured state, applying this upsert can only be a
5851    // no-op, so avoid allocating an owned op and walking the index again. Shared
5852    // reconciliation keeps the conditional observation so ABA arbitration remains
5853    // authoritative between its read and write lock boundaries.
5854    if target.direct_upsert_is_unchanged(baseline, kind, attrs) {
5855        report.observations = report.observations.saturating_add(1);
5856        report.apply.unchanged = report.apply.unchanged.saturating_add(1);
5857        return;
5858    }
5859    batch.push(ObservationOp::if_state(
5860        Op::Upsert { path: path.to_path_buf(), kind, attrs },
5861        baseline,
5862    ));
5863}
5864
5865fn flush_reconcile_batch(
5866    target: &mut ReconcileTarget<'_>,
5867    batch: &mut Vec<ObservationOp>,
5868    sink: &mut dyn FnMut(&Commit),
5869    report: &mut ReconcileReport,
5870) -> Result<()> {
5871    if batch.is_empty() {
5872        return Ok(());
5873    }
5874    report.observations =
5875        report.observations.saturating_add(u64::try_from(batch.len()).unwrap_or(u64::MAX));
5876    let started_at = report.reconcile_epoch.expect("reconciliation report has an owner");
5877    let outcome = target.apply(started_at, &Observation::from_ops(std::mem::take(batch)))?;
5878    merge_apply_stats(&mut report.apply, outcome.stats);
5879    if let Some(commit) = outcome.commit.as_ref() {
5880        sink(commit);
5881    }
5882    Ok(())
5883}
5884
5885fn merge_apply_stats(total: &mut ApplyStats, addition: ApplyStats) {
5886    total.inserted += addition.inserted;
5887    total.updated += addition.updated;
5888    total.removed += addition.removed;
5889    total.unchanged += addition.unchanged;
5890    total.invalidated += addition.invalidated;
5891    total.controls += addition.controls;
5892    total.reclassified += addition.reclassified;
5893    total.stale += addition.stale;
5894    total.resource_refused += addition.resource_refused;
5895}
5896
5897fn merge_reconcile_report(total: &mut ReconcileReport, addition: ReconcileReport) {
5898    total.retry_required |= addition.retry_required;
5899    total.scan.dirs_read += addition.scan.dirs_read;
5900    total.scan.entries += addition.scan.entries;
5901    total.scan.files_walked += addition.scan.files_walked;
5902    total.scan.bytes_walked += addition.scan.bytes_walked;
5903    total.scan.allocated_walked += addition.scan.allocated_walked;
5904    total.scan.errors.extend(addition.scan.errors);
5905    total.observations = total.observations.saturating_add(addition.observations);
5906    merge_apply_stats(&mut total.apply, addition.apply);
5907    total.listed_incomplete.extend(addition.listed_incomplete);
5908}
5909
5910fn should_descend(
5911    kind: EntryKind,
5912    attrs: Attrs,
5913    parent_depth: usize,
5914    root_dev: u64,
5915    config: &ScanConfig,
5916) -> bool {
5917    crate::admission::should_descend(
5918        kind,
5919        attrs,
5920        parent_depth,
5921        root_dev,
5922        config.max_depth,
5923        config.one_filesystem,
5924    )
5925}
5926
5927pub(crate) fn normalize_subtree(path: &Path) -> Result<PathBuf> {
5928    let mut normalized = PathBuf::new();
5929    for component in path.components() {
5930        match component {
5931            Component::Normal(part) => normalized.push(part),
5932            Component::CurDir => {}
5933            Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
5934                return Err(Error::PathEscapesRoot(path.to_path_buf()));
5935            }
5936        }
5937    }
5938    Ok(normalized)
5939}
5940
5941fn resolve_subtree_root(
5942    target: &ReconcileTarget<'_>,
5943    subtree: &Path,
5944    config: &ScanConfig,
5945) -> Result<PathBuf> {
5946    if subtree.as_os_str().is_empty() {
5947        return Ok(PathBuf::new());
5948    }
5949    let root = target.root_path()?;
5950    crate::counters::bump(|c| c.stats += 1);
5951    let Ok(root_metadata) = fs::symlink_metadata(&root) else {
5952        // The applying pass reports operational root failures as partial.
5953        return Ok(subtree.to_path_buf());
5954    };
5955    if !root_metadata.is_dir() {
5956        return Ok(subtree.to_path_buf());
5957    }
5958    let Ok(root_dev) = root_device(&root, &root_metadata) else {
5959        return Ok(subtree.to_path_buf());
5960    };
5961    let mut prefix = PathBuf::new();
5962    let mut components = subtree.components().peekable();
5963    while let Some(component) = components.next() {
5964        if components.peek().is_none() {
5965            break; // The boundary entry itself remains visible even when descent stops.
5966        }
5967        prefix.push(component.as_os_str());
5968        let metadata = match fs::symlink_metadata(root.join(&prefix)) {
5969            Ok(metadata) => metadata,
5970            Err(error)
5971                if matches!(
5972                    error.kind(),
5973                    std::io::ErrorKind::NotFound | std::io::ErrorKind::NotADirectory
5974                ) =>
5975            {
5976                return Ok(prefix);
5977            }
5978            Err(_) => break, // The applying pass records operational failures as partial.
5979        };
5980        if metadata.file_type().is_symlink() {
5981            return Err(Error::SubtreeOutsideScanScope {
5982                path: subtree.to_path_buf(),
5983                scope: config.scope(),
5984            });
5985        }
5986        if !metadata.is_dir() {
5987            return Ok(prefix);
5988        }
5989        let Ok(attrs) = attrs_from(&root.join(&prefix), &metadata) else {
5990            break;
5991        };
5992        if config.one_filesystem && root_dev != 0 && attrs.dev != 0 && attrs.dev != root_dev {
5993            return Err(Error::SubtreeOutsideScanScope {
5994                path: subtree.to_path_buf(),
5995                scope: config.scope(),
5996            });
5997        }
5998    }
5999    Ok(subtree.to_path_buf())
6000}
6001
6002/// Read an entry's kind and roll-up attributes out of its metadata.
6003///
6004/// Exposed so the watch layer verifies entries exactly the way the walker records them —
6005/// two stat interpretations that could drift would show up as an index that disagrees
6006/// with itself depending on which producer last touched a path.
6007///
6008/// On Windows the observation comes from a fresh non-following handle, and `meta` is
6009/// what answers for an entry whose handle cannot be opened because it is locked or
6010/// access is denied — the same fallback std's `metadata` makes, with identity and change
6011/// time unavailable for that entry.
6012pub fn observe(path: &Path, meta: &fs::Metadata) -> std::io::Result<(EntryKind, Attrs)> {
6013    #[cfg(windows)]
6014    {
6015        windows_metadata::observe(path, || Ok(meta.clone()))
6016    }
6017    #[cfg(not(windows))]
6018    {
6019        Ok((kind_from(meta), attrs_from(path, meta)?))
6020    }
6021}
6022
6023pub(crate) fn observe_dir_entry(
6024    entry: &fs::DirEntry,
6025) -> std::io::Result<Option<(EntryKind, Attrs)>> {
6026    #[cfg(windows)]
6027    {
6028        crate::counters::bump(|c| c.stats += 1);
6029        #[cfg(test)]
6030        {
6031            let path = entry.path();
6032            if let Some(error) =
6033                walk_hook(&path).and_then(|hook| hook(WalkHookPoint::ChildMetadata(&path)))
6034            {
6035                return missing_as_none(Err(error));
6036            }
6037        }
6038        // The listing already holds the entry's enumeration data; it is read only when
6039        // the handle cannot be opened, so the ordinary path allocates nothing more.
6040        missing_as_none(windows_metadata::observe(&entry.path(), || entry.metadata()))
6041    }
6042    #[cfg(not(windows))]
6043    {
6044        let Some(meta) = listed_child_metadata(entry)? else {
6045            return Ok(None);
6046        };
6047        Ok(Some((kind_from(&meta), attrs_from(Path::new(""), &meta)?)))
6048    }
6049}
6050
6051#[cfg(not(windows))]
6052fn kind_from(meta: &fs::Metadata) -> EntryKind {
6053    let file_type = meta.file_type();
6054    if file_type.is_symlink() {
6055        EntryKind::Symlink
6056    } else if file_type.is_dir() {
6057        EntryKind::Dir
6058    } else if file_type.is_file() {
6059        EntryKind::File
6060    } else {
6061        EntryKind::Other
6062    }
6063}
6064
6065#[cfg(unix)]
6066#[allow(clippy::unnecessary_wraps)] // Windows observation is fallible; keep one call contract.
6067pub(crate) fn attrs_from(_path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
6068    use std::os::unix::fs::MetadataExt;
6069    Ok(Attrs {
6070        size: meta.size(),
6071        // st_blocks is in 512-byte units by POSIX convention regardless of the
6072        // filesystem's own block size.
6073        allocated: meta.blocks().saturating_mul(512),
6074        mtime_ns: compose_ns(meta.mtime(), meta.mtime_nsec()),
6075        ctime_ns: compose_ns(meta.ctime(), meta.ctime_nsec()),
6076        inode: meta.ino(),
6077        dev: meta.dev(),
6078    })
6079}
6080
6081#[cfg(unix)]
6082fn compose_ns(secs: i64, nanos: i64) -> i64 {
6083    secs.saturating_mul(1_000_000_000).saturating_add(nanos)
6084}
6085
6086#[cfg(windows)]
6087pub(crate) fn attrs_from(path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
6088    windows_metadata::observe(path, || Ok(meta.clone())).map(|(_, attrs)| attrs)
6089}
6090
6091#[cfg(not(any(unix, windows)))]
6092#[allow(clippy::unnecessary_wraps)] // Windows observation is fallible; keep one call contract.
6093pub(crate) fn attrs_from(_path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
6094    let mtime_ns = meta.modified().map_or(0, system_time_ns);
6095    Ok(Attrs {
6096        size: meta.len(),
6097        // No allocated size without platform-specific calls; apparent size is the
6098        // honest fallback rather than a guess at block rounding.
6099        allocated: meta.len(),
6100        mtime_ns,
6101        // Windows has no ctime in the Unix sense. Leaving it zero means the fingerprint
6102        // degrades to size + mtime there, which is what every portable tool does.
6103        ctime_ns: 0,
6104        inode: 0,
6105        dev: 0,
6106    })
6107}
6108
6109/// The device a walk's root is on, which bounds a one-filesystem walk.
6110///
6111/// Only the device is needed, and on Windows it is read without demanding a consistent
6112/// observation of the root's times, which change whenever a child is created or removed.
6113pub(crate) fn root_device(root: &Path, meta: &fs::Metadata) -> std::io::Result<u64> {
6114    #[cfg(windows)]
6115    {
6116        let _ = meta;
6117        windows_metadata::volume_serial(root)
6118    }
6119    #[cfg(not(windows))]
6120    {
6121        attrs_from(root, meta).map(|attrs| attrs.dev)
6122    }
6123}
6124
6125pub(crate) fn attrs_from_file(file: &fs::File, meta: &fs::Metadata) -> std::io::Result<Attrs> {
6126    #[cfg(windows)]
6127    {
6128        let _ = meta;
6129        windows_metadata::attrs_from_file(file)
6130    }
6131    #[cfg(not(windows))]
6132    {
6133        let _ = file;
6134        attrs_from(Path::new(""), meta)
6135    }
6136}
6137
6138#[cfg(any(not(any(unix, windows)), test))]
6139fn system_time_ns(time: std::time::SystemTime) -> i64 {
6140    match time.duration_since(std::time::UNIX_EPOCH) {
6141        Ok(duration) => i64::try_from(duration.as_nanos()).unwrap_or(i64::MAX),
6142        Err(error) => {
6143            i64::try_from(error.duration().as_nanos()).map_or(i64::MIN, i64::saturating_neg)
6144        }
6145    }
6146}
6147
6148#[cfg(test)]
6149mod tests {
6150    use super::*;
6151    use std::fs::File;
6152    use std::io::Write;
6153
6154    fn write_file(path: &Path, contents: &[u8]) {
6155        if let Some(parent) = path.parent() {
6156            fs::create_dir_all(parent).expect("create parent");
6157        }
6158        let mut f = File::create(path).expect("create file");
6159        f.write_all(contents).expect("write");
6160    }
6161
6162    fn sample_tree() -> tempfile::TempDir {
6163        let dir = tempfile::tempdir().expect("tempdir");
6164        write_file(&dir.path().join("a.txt"), b"hello");
6165        write_file(&dir.path().join("src/main.rs"), b"fn main() {}");
6166        write_file(&dir.path().join("src/deep/nested.rs"), b"// nested");
6167        dir
6168    }
6169
6170    /// A counter that silently reads zero is worse than a missing one, because a report
6171    /// full of zeroes invites the conclusion that the work did not happen.
6172    ///
6173    /// This has already gone wrong twice: once when the per-entry counter was added to
6174    /// the serial walk while the parallel walk went uninstrumented, and once when a
6175    /// clippy fix hoisted a `read_dir` out of a match scrutinee and took the counter
6176    /// with it. Both builds compiled, passed every other test, and reported zero. This
6177    /// asserts the relationships a real walk must satisfy, so the next such edit fails
6178    /// here instead of in a report someone believes.
6179    #[test]
6180    fn a_walk_moves_every_counter_it_should() {
6181        // Both walkers, because they are separate loops with separate call sites. The
6182        // first version of this test only exercised the parallel one, and deleting the
6183        // serial walker's counter still passed — a guard covering one path gives false
6184        // confidence about the other.
6185        let _serial = crate::counters::test_serial();
6186        crate::counters::enable(true);
6187        for threads in [Some(1), Some(4)] {
6188            let dir = sample_tree();
6189            let config = ScanConfig { threads, ..ScanConfig::default() };
6190
6191            // Deltas around the scan, not absolute totals. The counters are
6192            // process-global, so a test running beside this one can add to them — and
6193            // `test_serial` cannot prevent that, since it only serializes tests that
6194            // take it, not every test that happens to walk a tree.
6195            let before = crate::counters::snapshot();
6196            let report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
6197            crate::counters::flush_thread();
6198            let after = crate::counters::snapshot();
6199            let observed_entries = after.dir_entries - before.dir_entries;
6200            let observed_opens = after.dir_opens - before.dir_opens;
6201            let observed_stats = after.stats - before.stats;
6202
6203            // `>=` rather than `==`, and the direction is the whole point: concurrent
6204            // work can only inflate these, never deflate them. So a counter that is too
6205            // low means a path ran uninstrumented, which is the failure worth catching
6206            // and the one that has actually happened — the macOS bulk reader reported
6207            // zero opens against three real ones. Equality would catch double-counting
6208            // too, and would be flaky for it.
6209            assert!(
6210                observed_entries >= report.entries,
6211                "every enumerated entry is counted at {threads:?}: {observed_entries} < {}",
6212                report.entries
6213            );
6214            assert!(
6215                observed_opens >= report.dirs_read,
6216                "every directory open is counted at {threads:?}: {observed_opens} < {}",
6217                report.dirs_read
6218            );
6219            assert!(
6220                observed_stats >= report.entries,
6221                "every entry is stated at {threads:?}: {observed_stats} < {}",
6222                report.entries
6223            );
6224            #[cfg(target_os = "macos")]
6225            {
6226                let observed_enum = after.dir_enumeration_calls - before.dir_enumeration_calls;
6227                // The serial walker is the portable `read_dir` path, which cannot see
6228                // getdents multiplicity. Enumeration calls are a bulk-backend fact.
6229                if threads != Some(1) {
6230                    assert!(
6231                        observed_enum >= report.dirs_read,
6232                        "every successful bulk directory issues at least one enumeration \
6233                         call at {threads:?}: {observed_enum} < {}",
6234                        report.dirs_read
6235                    );
6236                }
6237            }
6238
6239            // Deliberately not asserted: `allocs` stays zero in a library test, because
6240            // allocation counting needs a binary to install `CountingAlloc` as its
6241            // global allocator and a test harness installs its own. The probe covers
6242            // that half; this covers the counters the library itself drives.
6243        }
6244        crate::counters::enable(false);
6245    }
6246
6247    #[test]
6248    fn summary_fold_skips_stat_on_directories_and_symlinks() {
6249        let _serial = crate::counters::test_serial();
6250        crate::counters::enable(true);
6251        let dir = tempfile::tempdir().expect("tempdir");
6252        fs::create_dir(dir.path().join("src")).expect("directory");
6253        write_file(&dir.path().join("a.txt"), b"hi");
6254        #[cfg(unix)]
6255        std::os::unix::fs::symlink("a.txt", dir.path().join("link")).expect("symlink");
6256        let config = ScanConfig { threads: Some(1), read_controls: false, ..ScanConfig::default() };
6257
6258        crate::counters::test_thread_reset();
6259        let scan_report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
6260        let scan_stats = crate::counters::test_thread_snapshot().stats;
6261
6262        crate::counters::test_thread_reset();
6263        let fold_report = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
6264        let fold_stats = crate::counters::test_thread_snapshot().stats;
6265        crate::counters::enable(false);
6266
6267        assert_eq!(fold_report.entries, scan_report.entries);
6268        assert_eq!(fold_report.files_walked, scan_report.files_walked);
6269        assert_eq!(fold_report.bytes_walked, scan_report.bytes_walked);
6270        // Windows observes every listed entry through a fresh handle on both paths, so the
6271        // fold performs exactly the retained walk's observations there; the skip is a
6272        // non-Windows saving.
6273        #[cfg(not(windows))]
6274        assert!(
6275            fold_stats < scan_stats,
6276            "fold {fold_stats} should skip directory/symlink stats versus scan {scan_stats}"
6277        );
6278        #[cfg(unix)]
6279        assert_eq!(scan_stats.saturating_sub(fold_stats), 2);
6280        #[cfg(not(any(unix, windows)))]
6281        assert_eq!(scan_stats.saturating_sub(fold_stats), 1);
6282        #[cfg(windows)]
6283        assert_eq!(fold_stats, scan_stats);
6284    }
6285
6286    #[cfg(unix)]
6287    #[test]
6288    fn summary_fold_still_stats_directories_when_bound_to_one_filesystem() {
6289        let _serial = crate::counters::test_serial();
6290        crate::counters::enable(true);
6291        let dir = tempfile::tempdir().expect("tempdir");
6292        fs::create_dir(dir.path().join("src")).expect("directory");
6293        write_file(&dir.path().join("a.txt"), b"hi");
6294        std::os::unix::fs::symlink("a.txt", dir.path().join("link")).expect("symlink");
6295        let config = ScanConfig {
6296            threads: Some(1),
6297            read_controls: false,
6298            one_filesystem: true,
6299            ..ScanConfig::default()
6300        };
6301
6302        crate::counters::test_thread_reset();
6303        let scan_report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
6304        let scan_stats = crate::counters::test_thread_snapshot().stats;
6305
6306        crate::counters::test_thread_reset();
6307        let fold_report = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
6308        let fold_stats = crate::counters::test_thread_snapshot().stats;
6309        crate::counters::enable(false);
6310
6311        assert_eq!(fold_report.entries, scan_report.entries);
6312        assert_eq!(scan_stats.saturating_sub(fold_stats), 1);
6313    }
6314
6315    #[test]
6316    fn summary_fold_reuses_cleared_recycled_batches() {
6317        // Four workers and a batch of three force StreamingEmission to send more than
6318        // once per worker on this tree. Without `recycled.clear()`, the next send
6319        // re-folds the previous ops and files/bytes/dirs double-count.
6320        const DIRS: usize = 16;
6321        const FILES_PER_DIR: usize = 40;
6322        let dir = tempfile::tempdir().expect("tempdir");
6323        let mut expected_bytes = 0u64;
6324        for directory in 0..DIRS {
6325            let child = dir.path().join(format!("d{directory:02}"));
6326            fs::create_dir(&child).expect("directory");
6327            for file in 0..FILES_PER_DIR {
6328                let size = directory * FILES_PER_DIR + file + 1;
6329                expected_bytes += size as u64;
6330                write_file(&child.join(format!("f{file:02}.dat")), &vec![b'x'; size]);
6331            }
6332        }
6333        let expected_files = (DIRS * FILES_PER_DIR) as u64;
6334        let expected_dirs = DIRS as u64;
6335        let expected_entries = expected_files + expected_dirs;
6336        let config = ScanConfig {
6337            threads: Some(4),
6338            batch_size: 3,
6339            read_controls: false,
6340            ..ScanConfig::default()
6341        };
6342        let mut files = 0u64;
6343        let mut bytes = 0u64;
6344        let mut dirs = 0u64;
6345        let mut ops = 0u64;
6346        let report = scan_summary_fold(dir.path(), &config, &mut |observed| {
6347            ops += 1;
6348            let Op::Upsert { kind, attrs, .. } = &observed.op else {
6349                return;
6350            };
6351            match kind {
6352                EntryKind::File => {
6353                    files += 1;
6354                    bytes += attrs.size;
6355                }
6356                EntryKind::Dir => dirs += 1,
6357                EntryKind::Symlink | EntryKind::Other => {}
6358            }
6359        })
6360        .expect("fold");
6361        assert_eq!(files, expected_files);
6362        assert_eq!(bytes, expected_bytes);
6363        assert_eq!(dirs, expected_dirs);
6364        assert_eq!(ops, report.entries);
6365        assert_eq!(report.entries, expected_entries);
6366        assert_eq!(report.files_walked, expected_files);
6367        assert_eq!(report.bytes_walked, expected_bytes);
6368    }
6369
6370    /// An automatic walk too short to fill its calibration window must say so.
6371    ///
6372    /// The failure this guards is quiet: such a walk runs on its initial pool, which is
6373    /// indistinguishable in the artifacts from a walk that measured the filesystem and
6374    /// chose to hold — unless the undecided case is recorded separately. Reading the
6375    /// first as the second is how a policy with no evidence behind it comes to look
6376    /// like a policy with evidence behind it.
6377    #[test]
6378    fn a_short_automatic_walk_records_an_undecided_policy() {
6379        let _serial = crate::counters::test_serial();
6380        let available = std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get);
6381        if automatic_worker_pool(available).calibration.is_none() {
6382            // A host reporting one processor has no reserve to unlock, so there is no
6383            // policy here to leave undecided.
6384            return;
6385        }
6386
6387        crate::counters::enable(true);
6388        let dir = sample_tree();
6389        let config = ScanConfig { threads: None, ..ScanConfig::default() };
6390        let before = crate::counters::snapshot();
6391        scan(dir.path(), &config, &mut |_| {}).expect("scan");
6392        crate::counters::flush_thread();
6393        let after = crate::counters::snapshot();
6394        crate::counters::enable(false);
6395
6396        // A strict increase, so a counter inflated by a test running beside this one
6397        // cannot turn the assertion into a false pass.
6398        assert!(
6399            after.adaptive_policy_undecided > before.adaptive_policy_undecided,
6400            "a three-file tree cannot fill a {ADAPTIVE_SCAN_CALIBRATION_ENTRIES}-entry window"
6401        );
6402    }
6403
6404    #[test]
6405    fn diagnostics_make_a_fixed_pool_and_backend_choice_explicit() {
6406        let dir = sample_tree();
6407        let config = ScanConfig { threads: Some(1), ..ScanConfig::default() };
6408
6409        let (report, diagnostics) =
6410            scan_with_diagnostics(dir.path(), &config, &mut |_| {}).expect("diagnostic scan");
6411
6412        assert_eq!(diagnostics.schema, SCAN_DIAGNOSTICS_SCHEMA);
6413        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Fixed);
6414        assert_eq!(diagnostics.worker_policy.initial_workers, 1);
6415        assert_eq!(diagnostics.worker_policy.maximum_workers, 1);
6416        assert_eq!(diagnostics.worker_policy.peak_active_workers, 1);
6417        assert!(diagnostics.worker_policy.windows.is_empty());
6418        assert!(!diagnostics.worker_policy.events_truncated);
6419        assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
6420        assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
6421        assert_eq!(diagnostics.backend.portable_directory_reads, report.dirs_read);
6422
6423        #[cfg(target_os = "macos")]
6424        {
6425            assert_eq!(diagnostics.backend.macos_bulk_attempts, Some(0));
6426            assert_eq!(diagnostics.backend.macos_bulk_successes, Some(0));
6427            assert_eq!(diagnostics.backend.macos_bulk_fallbacks, Some(0));
6428            assert!(diagnostics.backend.unavailable_reason.is_none());
6429        }
6430        #[cfg(not(target_os = "macos"))]
6431        {
6432            assert_eq!(diagnostics.backend.macos_bulk_attempts, None);
6433            assert_eq!(diagnostics.backend.macos_bulk_successes, None);
6434            assert_eq!(diagnostics.backend.macos_bulk_fallbacks, None);
6435            assert_eq!(
6436                diagnostics.backend.unavailable_reason,
6437                Some("macOS bulk directory enumeration is unavailable on this platform")
6438            );
6439        }
6440    }
6441
6442    #[test]
6443    fn diagnostics_fail_closed_when_an_automatic_window_is_incomplete() {
6444        let available = std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get);
6445        let pool = automatic_worker_pool(available);
6446        if pool.calibration.is_none() {
6447            return;
6448        }
6449        let dir = sample_tree();
6450        let config = ScanConfig { threads: None, ..ScanConfig::default() };
6451
6452        let (report, diagnostics) =
6453            scan_with_diagnostics(dir.path(), &config, &mut |_| {}).expect("diagnostic scan");
6454
6455        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Undecided);
6456        assert_eq!(diagnostics.worker_policy.available_parallelism, available);
6457        assert_eq!(diagnostics.worker_policy.initial_workers, pool.initial);
6458        assert_eq!(diagnostics.worker_policy.maximum_workers, pool.maximum);
6459        assert_eq!(
6460            diagnostics.worker_policy.calibration_window_entries,
6461            Some(ADAPTIVE_SCAN_CALIBRATION_ENTRIES)
6462        );
6463        assert_eq!(
6464            diagnostics.worker_policy.slow_threshold_ns_per_entry,
6465            Some(ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY)
6466        );
6467        assert_eq!(diagnostics.worker_policy.windows.len(), 1);
6468        let window = &diagnostics.worker_policy.windows[0];
6469        assert_eq!(window.sequence, 0);
6470        assert_eq!(window.start_entry_ordinal, 0);
6471        assert_eq!(window.end_entry_ordinal, report.entries);
6472        assert_eq!(window.observed_entries, report.entries);
6473        assert_eq!(window.decision, WorkerPolicyDecision::Undecided);
6474        assert!(window.end_entry_ordinal < ADAPTIVE_SCAN_CALIBRATION_ENTRIES);
6475        assert!(window.active_workers <= diagnostics.worker_policy.peak_active_workers);
6476        assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
6477        assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
6478        assert!(diagnostics.worker_policy.handoff_backlog_high_water >= 1);
6479    }
6480
6481    #[test]
6482    fn diagnostic_trace_is_bounded_and_marks_truncation() {
6483        let recorder = ScanDiagnosticsRecorder::new(
6484            WorkerPool::fixed(2),
6485            2,
6486            WorkerPolicyExperiment::ShippedOneShot,
6487        );
6488        for sequence in 0..=MAX_POLICY_TRACE_EVENTS {
6489            recorder.record_policy_window(PolicyWindowSnapshot {
6490                sequence: sequence as u64,
6491                start_entry_ordinal: sequence as u64,
6492                end_entry_ordinal: sequence as u64 + 1,
6493                observed_entries: 1,
6494                observed_chunks: 1,
6495                observed_work_ns: 10,
6496                ready_directories: 1,
6497                in_flight_directories: 1,
6498                active_workers: 1,
6499                handoff_backlog: 0,
6500                requested_workers: None,
6501                decision: WorkerPolicyDecision::Hold,
6502            });
6503        }
6504
6505        let diagnostics = recorder.finish();
6506        assert_eq!(diagnostics.worker_policy.windows.len(), MAX_POLICY_TRACE_EVENTS);
6507        assert!(diagnostics.worker_policy.events_truncated);
6508    }
6509
6510    #[test]
6511    fn diagnostic_trace_preserves_queue_order_when_recorders_arrive_out_of_order() {
6512        let pool =
6513            WorkerPool { initial: 2, maximum: 4, calibration: Some(WorkerCalibration::new(1, 1)) };
6514        let recorder =
6515            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::RepeatedWindows);
6516        let snapshot = |sequence, decision| PolicyWindowSnapshot {
6517            sequence,
6518            start_entry_ordinal: sequence,
6519            end_entry_ordinal: sequence + 1,
6520            observed_entries: 1,
6521            observed_chunks: 1,
6522            observed_work_ns: 1,
6523            ready_directories: 0,
6524            in_flight_directories: 0,
6525            active_workers: 1,
6526            handoff_backlog: 0,
6527            requested_workers: None,
6528            decision,
6529        };
6530
6531        recorder.record_policy_window(snapshot(1, WorkerPolicyDecision::HoldNoUsefulWork));
6532        recorder.record_policy_window(snapshot(0, WorkerPolicyDecision::Hold));
6533
6534        let diagnostics = recorder.finish();
6535        assert_eq!(
6536            diagnostics
6537                .worker_policy
6538                .windows
6539                .iter()
6540                .map(|window| window.sequence)
6541                .collect::<Vec<_>>(),
6542            vec![0, 1]
6543        );
6544        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::HeldNoUsefulWork);
6545    }
6546
6547    #[test]
6548    fn diagnostic_policy_aggregates_cross_check_runtime_counters() {
6549        let _serial = crate::counters::test_serial();
6550        let pool = WorkerPool {
6551            initial: 2,
6552            maximum: 4,
6553            calibration: Some(WorkerCalibration::new(17, 100)),
6554        };
6555        let recorder =
6556            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::ShippedOneShot);
6557
6558        crate::counters::enable(true);
6559        let before = crate::counters::snapshot();
6560        record_adaptive_calibration_chunk(Some(&recorder), 17, 2_100);
6561        record_adaptive_worker_expansion(Some(&recorder));
6562        crate::counters::flush_thread();
6563        let after = crate::counters::snapshot();
6564        crate::counters::enable(false);
6565
6566        recorder.record_policy_window(PolicyWindowSnapshot {
6567            sequence: 0,
6568            start_entry_ordinal: 0,
6569            end_entry_ordinal: 17,
6570            observed_entries: 17,
6571            observed_chunks: 1,
6572            observed_work_ns: 2_100,
6573            ready_directories: 2,
6574            in_flight_directories: 2,
6575            active_workers: 2,
6576            handoff_backlog: 0,
6577            requested_workers: Some(4),
6578            decision: WorkerPolicyDecision::ScaleUp,
6579        });
6580        let diagnostics = recorder.finish();
6581        let policy = diagnostics.worker_policy;
6582        assert_eq!(policy.calibration_chunks, 1);
6583        assert_eq!(policy.calibration_entries, 17);
6584        assert_eq!(policy.calibration_work_ns, 2_100);
6585        assert_eq!(policy.worker_expansions, 1);
6586        assert_eq!(policy.windows[0].observed_chunks, policy.calibration_chunks);
6587        assert_eq!(policy.windows[0].observed_entries, policy.calibration_entries);
6588        assert_eq!(policy.windows[0].observed_work_ns, policy.calibration_work_ns);
6589
6590        // Other tests can record while this process-global interval is enabled, so the
6591        // counter delta may be larger but must never be smaller than this run-scoped
6592        // trace. The shared helpers above make the two observations one event.
6593        assert!(
6594            after.adaptive_calibration_chunks - before.adaptive_calibration_chunks
6595                >= policy.calibration_chunks
6596        );
6597        assert!(
6598            after.adaptive_calibration_entries - before.adaptive_calibration_entries
6599                >= policy.calibration_entries
6600        );
6601        assert!(
6602            after.adaptive_calibration_work_us - before.adaptive_calibration_work_us
6603                >= policy.calibration_work_ns / 1_000
6604        );
6605        assert!(after.adaptive_scale_ups - before.adaptive_scale_ups >= policy.worker_expansions);
6606    }
6607
6608    #[test]
6609    fn diagnostic_index_scan_preserves_the_regular_result() {
6610        let dir = branching_tree();
6611        let config = ScanConfig { threads: Some(4), ..ScanConfig::default() };
6612        let (plain, plain_report) = scan_into_index(dir.path(), &config).expect("plain scan");
6613        let (diagnostic, diagnostic_report, diagnostics) =
6614            scan_into_index_with_diagnostics(dir.path(), &config).expect("diagnostic scan");
6615
6616        assert_eq!(index_fingerprint(&plain), index_fingerprint(&diagnostic));
6617        assert_eq!(plain_report.entries, diagnostic_report.entries);
6618        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Fixed);
6619    }
6620
6621    #[test]
6622    fn detached_bootstrap_matches_the_streaming_reducer_for_each_worker_count() {
6623        let dir = branching_tree();
6624        for threads in 1..=4 {
6625            let config = ScanConfig {
6626                read_controls: false,
6627                threads: Some(threads),
6628                ..ScanConfig::default()
6629            };
6630            let _ = detached_and_streaming_indexes(dir.path(), &config);
6631        }
6632    }
6633
6634    #[test]
6635    fn detached_control_bootstrap_matches_the_streaming_reducer_for_each_worker_count() {
6636        let dir = controlled_branching_tree();
6637        for threads in 1..=4 {
6638            let config =
6639                ScanConfig { read_controls: true, threads: Some(threads), ..ScanConfig::default() };
6640            let _ = detached_and_streaming_indexes(dir.path(), &config);
6641        }
6642    }
6643
6644    #[test]
6645    fn detached_bootstrap_preserves_the_exact_first_mutation() {
6646        let dir = branching_tree();
6647        let config = ScanConfig { read_controls: false, threads: Some(4), ..ScanConfig::default() };
6648        let (mut detached, mut streaming) = detached_and_streaming_indexes(dir.path(), &config);
6649        let created = dir.path().join("t3/m2/after-bootstrap.rs");
6650        write_file(&created, b"new fact");
6651        let attrs =
6652            attrs_from(&created, &fs::symlink_metadata(&created).expect("new file metadata"))
6653                .expect("observe new file");
6654        let observation = Observation::new(vec![Op::Upsert {
6655            path: PathBuf::from("t3/m2/after-bootstrap.rs"),
6656            kind: EntryKind::File,
6657            attrs,
6658        }]);
6659
6660        let detached_outcome = detached.apply(&observation).expect("detached mutation");
6661        let streaming_outcome = streaming.apply(&observation).expect("streaming mutation");
6662        assert_eq!(detached_outcome, streaming_outcome);
6663        assert_indexes_equal(&detached, &streaming);
6664    }
6665
6666    #[test]
6667    fn detached_control_bootstrap_preserves_the_exact_first_mutation() {
6668        let dir = controlled_branching_tree();
6669        let config = ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
6670        let (mut detached, mut streaming) = detached_and_streaming_indexes(dir.path(), &config);
6671        let observation = Observation::new(vec![Op::ControlUpsert {
6672            path: PathBuf::from(".gitignore"),
6673            source: b"leaf-2.dat\n".to_vec(),
6674        }]);
6675
6676        let detached_outcome = detached.apply(&observation).expect("detached control mutation");
6677        let streaming_outcome = streaming.apply(&observation).expect("streaming control mutation");
6678        assert_eq!(detached_outcome, streaming_outcome);
6679        assert_indexes_equal(&detached, &streaming);
6680    }
6681
6682    fn observed_coverage(index: &Index) -> crate::control::ControlObservation {
6683        match index.control_coverage() {
6684            crate::control::ControlCoverage::Observed(observation) => observation,
6685            crate::control::ControlCoverage::NotObserved => panic!("controls were observed"),
6686        }
6687    }
6688
6689    /// Both bootstrap lanes refuse a line over the limit and a file over the budget, and
6690    /// neither ends the scan or makes it partial. Both refusals are order-independent, so
6691    /// the lanes agree on exactly which files they refused.
6692    #[test]
6693    fn both_bootstrap_lanes_refuse_over_bound_controls_without_ending_the_scan() {
6694        let dir = tempfile::tempdir().expect("tempdir");
6695        let mut long_line = b"*.log\n".to_vec();
6696        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
6697        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
6698        write_file(&dir.path().join("guarded/kept.log"), b"guarded");
6699        write_file(
6700            &dir.path().join("huge/.gitignore"),
6701            &b"x\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
6702        );
6703        write_file(&dir.path().join("applied/.gitignore"), b"*.log\n");
6704        write_file(&dir.path().join("applied/dropped.log"), b"applied");
6705        let config = ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
6706
6707        let (detached, _) = detached_and_streaming_indexes(dir.path(), &config);
6708        let (_, report) = scan_into_index(dir.path(), &config).expect("scan");
6709
6710        assert!(report.is_complete(), "{:?}", report.errors);
6711        let coverage = observed_coverage(&detached);
6712        assert_eq!((coverage.applied, coverage.refused), (1, 2));
6713        assert_eq!(
6714            coverage.refusals,
6715            vec![
6716                crate::control::RefusedControl {
6717                    path: PathBuf::from("guarded/.gitignore"),
6718                    reason: crate::control::ControlRefusalReason::LineLimit,
6719                },
6720                crate::control::RefusedControl {
6721                    path: PathBuf::from("huge/.gitignore"),
6722                    reason: crate::control::ControlRefusalReason::Budget,
6723                },
6724            ]
6725        );
6726        assert_eq!(
6727            detached.is_ignored(Path::new("guarded/kept.log")).expect("observed"),
6728            Some(false)
6729        );
6730        assert_eq!(
6731            detached.is_ignored(Path::new("applied/dropped.log")).expect("observed"),
6732            Some(true)
6733        );
6734    }
6735
6736    /// The control counters attribute what a scan's control state cost: files read, sources
6737    /// refused, and sources that shared a retained content instead of parsing their own.
6738    ///
6739    /// Off by default and compiled in, like every counter, so the numbers a speed check
6740    /// reads come from the shipped path rather than an instrumented build.
6741    #[test]
6742    fn control_counters_attribute_reads_refusals_and_sharing() {
6743        let _serial = crate::counters::test_serial();
6744        let dir = tempfile::tempdir().expect("tempdir");
6745        let shared = b"*.log\n".to_vec();
6746        write_file(&dir.path().join(".gitignore"), &shared);
6747        write_file(&dir.path().join("twin/.gitignore"), &shared);
6748        let mut long_line = b"*.tmp\n".to_vec();
6749        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
6750        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
6751        let config = ScanConfig { read_controls: true, threads: Some(1), ..ScanConfig::default() };
6752
6753        crate::counters::enable(true);
6754        // Deltas around the scan rather than absolute totals, for the reason
6755        // `a_walk_moves_every_counter_it_should` gives: the counters are process-global,
6756        // `test_serial` only serializes the tests that take it, and every report in this
6757        // binary now reads `.gitignore` by default, so a test running beside this one can
6758        // add control reads of its own.
6759        let before = crate::counters::snapshot();
6760        let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
6761        crate::counters::flush_thread();
6762        let after = crate::counters::snapshot();
6763        crate::counters::enable(false);
6764
6765        assert!(report.is_complete(), "{:?}", report.errors);
6766        assert_eq!(observed_coverage(&index).refused, 1);
6767        // `>=` in the one direction concurrency can move them. A count that is too low
6768        // means a path ran uninstrumented, which is the defect worth catching; too high
6769        // is another test's tree, which is not.
6770        for (label, observed, expected) in [
6771            ("one read per .gitignore", after.control_reads - before.control_reads, 3),
6772            ("the line over the limit", after.control_refused - before.control_refused, 1),
6773            (
6774                "the twin shares one parsed content",
6775                after.control_sources_shared - before.control_sources_shared,
6776                1,
6777            ),
6778        ] {
6779            assert!(observed >= expected, "{label}: counted {observed}, expected {expected}");
6780        }
6781    }
6782
6783    /// Both limits are part of the scope, and each lifts only its own refusals: no budget
6784    /// still refuses a long line, and no line limit still refuses a file past the budget.
6785    #[test]
6786    fn each_control_limit_is_scope_and_lifts_only_its_own_refusals() {
6787        use crate::control::{ControlLimits, ControlRefusalReason, RefusedControl};
6788
6789        let with = |limits| ScanConfig { control_limits: limits, ..ScanConfig::default() };
6790        let defaults = ControlLimits::default();
6791        let default = ScanConfig::default();
6792        let no_budget = with(ControlLimits { budget: None, ..defaults });
6793        let no_line_limit = with(ControlLimits { line_limit: None, ..defaults });
6794        let configs = [
6795            default.clone(),
6796            with(ControlLimits { budget: Some(16 * 1024 * 1024), ..defaults }),
6797            no_budget.clone(),
6798            with(ControlLimits { line_limit: Some(64 * 1024), ..defaults }),
6799            no_line_limit.clone(),
6800            with(ControlLimits { budget: None, line_limit: None }),
6801            // The same values in each other's places are a different scope.
6802            with(ControlLimits { budget: defaults.line_limit, line_limit: defaults.budget }),
6803        ];
6804        let scopes: Vec<ScanScope> = configs.iter().map(ScanConfig::scope).collect();
6805        for (index, scope) in scopes.iter().enumerate() {
6806            assert!(scope.observes_controls());
6807            assert!(scopes[index + 1..].iter().all(|other| other != scope), "{scopes:?}");
6808        }
6809        for config in &configs {
6810            let blind = ScanConfig { read_controls: false, ..config.clone() };
6811            assert_eq!(blind.scope().ignore_rules_fingerprint, 0, "unobserved has one scope");
6812        }
6813
6814        let dir = tempfile::tempdir().expect("tempdir");
6815        let mut long_line = b"*.log\n".to_vec();
6816        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
6817        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
6818        write_file(&dir.path().join("guarded/dropped.log"), b"log");
6819        write_file(
6820            &dir.path().join("huge/.gitignore"),
6821            &b"x\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
6822        );
6823        let refused = |path: &str, reason| RefusedControl { path: PathBuf::from(path), reason };
6824        let (bounded, _) = scan_into_index(dir.path(), &default).expect("default scan");
6825        assert_eq!(observed_coverage(&bounded).refused, 2);
6826
6827        let (budget_lifted, _) = detached_and_streaming_indexes(dir.path(), &no_budget);
6828        let coverage = observed_coverage(&budget_lifted);
6829        assert_eq!(coverage.limits, no_budget.control_limits);
6830        assert_eq!(
6831            coverage.refusals,
6832            [refused("guarded/.gitignore", ControlRefusalReason::LineLimit)]
6833        );
6834        assert_eq!(
6835            budget_lifted.is_ignored(Path::new("guarded/dropped.log")).expect("observed"),
6836            Some(false)
6837        );
6838
6839        let (line_limit_lifted, _) = detached_and_streaming_indexes(dir.path(), &no_line_limit);
6840        let coverage = observed_coverage(&line_limit_lifted);
6841        assert_eq!(coverage.refusals, [refused("huge/.gitignore", ControlRefusalReason::Budget)]);
6842        assert_eq!(
6843            line_limit_lifted.is_ignored(Path::new("guarded/dropped.log")).expect("observed"),
6844            Some(true)
6845        );
6846        assert_eq!(line_limit_lifted.scope(), no_line_limit.scope());
6847    }
6848
6849    /// The synthetic tree that ended a cold scan (fdu-1onj): 1,105 directories, each with
6850    /// a distinct 510-byte `.gitignore` of short rules, plus one line over the limit. The
6851    /// scan completes with every size exact and names what it refused, on both lanes.
6852    #[test]
6853    fn a_tree_past_both_control_bounds_completes_with_exact_sizes() {
6854        const DIRECTORIES: usize = 1_105;
6855        let dir = tempfile::tempdir().expect("tempdir");
6856        for directory in 0..DIRECTORIES {
6857            let mut source = Vec::new();
6858            for line in 0..63 {
6859                source.extend(format!("p{directory:04}{line:02}\n").bytes());
6860            }
6861            source.extend(format!("q{directory:04}\n").bytes());
6862            assert_eq!(source.len(), 510);
6863            let root = dir.path().join(format!("d{directory:04}"));
6864            write_file(&root.join(".gitignore"), &source);
6865            write_file(&root.join("file.txt"), b"contents");
6866        }
6867        write_file(
6868            &dir.path().join("a-guard/.gitignore"),
6869            &vec![b'x'; crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1],
6870        );
6871        let observing =
6872            ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
6873        let blind = ScanConfig { read_controls: false, ..observing.clone() };
6874
6875        let (unobserved, _) = scan_into_index(dir.path(), &blind).expect("controls-off scan");
6876        let canonical = dir.path().canonicalize().expect("canonical root");
6877        let lanes = [
6878            scan_into_index(dir.path(), &observing).expect("detached scan"),
6879            scan_into_index_via_scanner(&canonical, &observing).expect("streaming scan"),
6880        ];
6881        for (index, report) in &lanes {
6882            assert!(report.is_complete(), "{:?}", report.errors);
6883            assert_eq!(index.total(), unobserved.total(), "sizes do not depend on controls");
6884            let coverage = observed_coverage(index);
6885            assert!(coverage.refused > 1, "the budget refused sources: {coverage:?}");
6886            assert_eq!(
6887                coverage.applied + coverage.refused,
6888                u64::try_from(DIRECTORIES + 1).expect("small")
6889            );
6890            assert_eq!(coverage.refusals.len(), crate::MAX_RETAINED_ISSUES);
6891            assert!(!coverage.lists_every_refusal());
6892            assert_eq!(
6893                coverage.refusals[0],
6894                crate::control::RefusedControl {
6895                    path: PathBuf::from("a-guard/.gitignore"),
6896                    reason: crate::control::ControlRefusalReason::LineLimit,
6897                }
6898            );
6899            assert!(
6900                index.control_table().retained_cost() <= crate::control::DEFAULT_CONTROL_BUDGET
6901            );
6902        }
6903    }
6904
6905    #[test]
6906    fn fingerprint_metadata_observes_mutation_after_directory_enumeration() {
6907        let dir = tempfile::tempdir().expect("tempdir");
6908        let path = dir.path().join("changing.bin");
6909        write_file(&path, b"before");
6910        let entry = fs::read_dir(dir.path())
6911            .expect("read directory")
6912            .next()
6913            .expect("one entry")
6914            .expect("read entry");
6915
6916        write_file(&path, b"after mutation");
6917
6918        let metadata = metadata_for_fingerprint(&entry).expect("fresh metadata");
6919        assert_eq!(metadata.len(), b"after mutation".len() as u64);
6920    }
6921
6922    /// A tree wide and deep enough that workers genuinely interleave.
6923    ///
6924    /// A three-file fixture would pass every one of these tests with a broken queue,
6925    /// because one worker would finish before another started.
6926    fn branching_tree() -> tempfile::TempDir {
6927        let dir = tempfile::tempdir().expect("tempdir");
6928        for top in 0..12 {
6929            for middle in 0..6 {
6930                for leaf in 0..7 {
6931                    write_file(
6932                        &dir.path().join(format!("t{top}/m{middle}/leaf-{leaf}.dat")),
6933                        &vec![b'x'; leaf * 13],
6934                    );
6935                }
6936            }
6937            // A deep chain alongside the wide fan-out, so depth and width are both
6938            // exercised by the same walk.
6939            write_file(&dir.path().join(format!("t{top}/a/b/c/d/e/deep.txt")), b"deep");
6940        }
6941        dir
6942    }
6943
6944    fn controlled_branching_tree() -> tempfile::TempDir {
6945        let dir = branching_tree();
6946        write_file(&dir.path().join(".gitignore"), b"leaf-1.dat\nt7/\n");
6947        write_file(&dir.path().join("t3/.gitignore"), b"!m2/leaf-1.dat\n*.tmp\n");
6948        write_file(&dir.path().join("t3/m2/generated.tmp"), b"ignored by nested control");
6949        write_file(&dir.path().join("t7/.gitignore"), b"!m0/leaf-1.dat\n");
6950        fs::create_dir_all(dir.path().join("t5/.gitignore")).expect("non-file control directory");
6951        write_file(&dir.path().join("t5/.gitignore/ordinary.txt"), b"ordinary child");
6952        dir
6953    }
6954
6955    fn index_fingerprint(index: &Index) -> Vec<(PathBuf, EntryKind, Attrs)> {
6956        let mut entries: Vec<(PathBuf, EntryKind, Attrs)> = Vec::new();
6957        let mut queue = vec![PathBuf::new()];
6958        while let Some(path) = queue.pop() {
6959            let Some(children) = index.children(&path) else {
6960                continue;
6961            };
6962            let names: Vec<PathBuf> = children.map(|(name, _id)| path.join(name)).collect();
6963            for child_path in names {
6964                let kind = index.kind(&child_path).expect("child has a kind");
6965                let attrs = *index.attrs(&child_path).expect("child has attrs");
6966                entries.push((child_path.clone(), kind, attrs));
6967                if kind.is_dir() {
6968                    queue.push(child_path);
6969                }
6970            }
6971        }
6972        entries.sort_by(|left, right| left.0.cmp(&right.0));
6973        entries
6974    }
6975
6976    fn detached_and_streaming_indexes(root: &Path, config: &ScanConfig) -> (Index, Index) {
6977        let canonical = root.canonicalize().expect("canonical test root");
6978        let (streaming, streaming_report) =
6979            scan_into_index_via_scanner(&canonical, config).expect("streaming oracle");
6980        let (detached, detached_report) = scan_into_index(root, config).expect("detached scan");
6981
6982        assert_eq!(detached_report.dirs_read, streaming_report.dirs_read);
6983        assert_eq!(detached_report.entries, streaming_report.entries);
6984        assert_eq!(detached_report.files_walked, streaming_report.files_walked);
6985        assert_eq!(detached_report.bytes_walked, streaming_report.bytes_walked);
6986        assert_eq!(
6987            detached_report.errors.iter().map(ToString::to_string).collect::<Vec<_>>(),
6988            streaming_report.errors.iter().map(ToString::to_string).collect::<Vec<_>>()
6989        );
6990        assert_indexes_equal(&detached, &streaming);
6991        (detached, streaming)
6992    }
6993
6994    fn assert_indexes_equal(left: &Index, right: &Index) {
6995        assert_eq!(index_fingerprint(left), index_fingerprint(right));
6996        assert_eq!(left.total(), right.total());
6997        assert_eq!(left.partition_total().ok(), right.partition_total().ok());
6998        assert_eq!(left.scope(), right.scope());
6999        assert_eq!(left.freshness(), right.freshness());
7000        assert_eq!(left.state(), right.state());
7001        assert_eq!(left.clock(), right.clock());
7002        assert_eq!(left.len(), right.len());
7003        assert_eq!(left.issues(), right.issues());
7004        assert_eq!(left.observes_controls(), right.observes_controls());
7005        assert_eq!(left.control_coverage(), right.control_coverage());
7006        assert_eq!(
7007            left.control_table()
7008                .sources()
7009                .map(|(path, source)| (path, source.to_vec()))
7010                .collect::<Vec<_>>(),
7011            right
7012                .control_table()
7013                .sources()
7014                .map(|(path, source)| (path, source.to_vec()))
7015                .collect::<Vec<_>>()
7016        );
7017        for (path, _, _) in index_fingerprint(left) {
7018            assert_eq!(left.is_ignored(&path).ok(), right.is_ignored(&path).ok(), "{path:?}");
7019        }
7020    }
7021
7022    /// A small tree whose mutation crosses every structural reconciliation boundary.
7023    fn reconciliation_transition_tree() -> tempfile::TempDir {
7024        let dir = tempfile::tempdir().expect("tempdir");
7025        write_file(&dir.path().join("changed.txt"), b"before");
7026        write_file(&dir.path().join("removed.txt"), b"remove me");
7027        write_file(&dir.path().join("directory-to-file/old.rs"), b"old child");
7028        write_file(&dir.path().join("file-to-directory"), b"old file");
7029        write_file(&dir.path().join("removed-tree/nested/gone.md"), b"gone");
7030        write_file(&dir.path().join("stable/deep/kept.rs"), b"kept");
7031        dir
7032    }
7033
7034    fn mutate_reconciliation_transition_tree(root: &Path) {
7035        write_file(&root.join("changed.txt"), b"after, with a distinct size");
7036        fs::remove_file(root.join("removed.txt")).expect("remove root file");
7037
7038        fs::remove_dir_all(root.join("directory-to-file")).expect("remove old directory");
7039        write_file(&root.join("directory-to-file"), b"replacement file");
7040
7041        fs::remove_file(root.join("file-to-directory")).expect("remove old file");
7042        write_file(&root.join("file-to-directory/new.txt"), b"replacement child");
7043
7044        fs::remove_dir_all(root.join("removed-tree")).expect("remove nested tree");
7045        write_file(&root.join("added-tree/nested/new.md"), b"new nested file");
7046    }
7047
7048    fn effective_ops(commits: &[Commit]) -> Vec<Op> {
7049        let mut operations: Vec<_> = commits
7050            .iter()
7051            .flat_map(|commit| commit.changes.iter())
7052            .filter_map(|change| match change {
7053                crate::EffectiveChange::Inserted { path, kind, attrs } => {
7054                    Some(Op::Upsert { path: path.clone(), kind: *kind, attrs: *attrs })
7055                }
7056                crate::EffectiveChange::Updated { path, kind, current, .. } => {
7057                    Some(Op::Upsert { path: path.clone(), kind: *kind, attrs: *current })
7058                }
7059                crate::EffectiveChange::Removed { path, .. } => {
7060                    Some(Op::Remove { path: path.clone() })
7061                }
7062                crate::EffectiveChange::Invalidated { path, reason } => {
7063                    Some(Op::InvalidateSubtree { path: path.clone(), reason: *reason })
7064                }
7065                crate::EffectiveChange::ControlUpdated { .. }
7066                | crate::EffectiveChange::ControlRefusalUpdated { .. }
7067                | crate::EffectiveChange::Reclassified { .. } => None,
7068            })
7069            .collect();
7070        operations.sort_by(|left, right| left.path().cmp(right.path()));
7071        operations
7072    }
7073
7074    fn commit_touches(commit: &Commit, path: &Path) -> bool {
7075        commit.changes.iter().any(|change| change.path() == path)
7076    }
7077
7078    #[test]
7079    fn parallel_and_serial_walks_produce_the_same_index() {
7080        let dir = branching_tree();
7081        let serial_config = ScanConfig { threads: Some(1), ..ScanConfig::default() };
7082        let (serial, serial_report) =
7083            scan_into_index(dir.path(), &serial_config).expect("serial scan");
7084        assert!(serial_report.is_complete());
7085
7086        for threads in [2_usize, 3, 8] {
7087            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
7088            let (parallel, report) = scan_into_index(dir.path(), &config).expect("parallel scan");
7089            assert!(report.is_complete(), "{threads} threads reported errors");
7090            assert_eq!(report.entries, serial_report.entries, "{threads} threads");
7091            assert_eq!(report.dirs_read, serial_report.dirs_read, "{threads} threads");
7092            assert_eq!(report.files_walked, serial_report.files_walked, "{threads} threads");
7093            assert_eq!(report.bytes_walked, serial_report.bytes_walked, "{threads} threads");
7094            // Public roll-ups carry extension names even though the internal merge path
7095            // uses ids whose assignment order differs between serial and parallel walks.
7096            let (serial_total, parallel_total) = (serial.total(), parallel.total());
7097            assert_eq!(
7098                (
7099                    parallel_total.files,
7100                    parallel_total.dirs,
7101                    parallel_total.bytes,
7102                    parallel_total.allocated,
7103                    parallel_total.newest_mtime_ns,
7104                ),
7105                (
7106                    serial_total.files,
7107                    serial_total.dirs,
7108                    serial_total.bytes,
7109                    serial_total.allocated,
7110                    serial_total.newest_mtime_ns,
7111                ),
7112                "{threads} threads roll-up"
7113            );
7114            assert_eq!(
7115                parallel_total.by_ext, serial_total.by_ext,
7116                "{threads} threads per-extension roll-up"
7117            );
7118            assert_eq!(
7119                index_fingerprint(&parallel),
7120                index_fingerprint(&serial),
7121                "{threads} threads produced a different index"
7122            );
7123        }
7124    }
7125
7126    #[test]
7127    fn parallel_walk_emits_every_entry_exactly_once() {
7128        let dir = branching_tree();
7129        let config = ScanConfig { threads: Some(4), batch_size: 16, ..ScanConfig::default() };
7130        let mut seen: BTreeMap<PathBuf, usize> = BTreeMap::new();
7131        let report = scan(dir.path(), &config, &mut |observation| {
7132            for op in &observation.ops {
7133                if let Op::Upsert { path, .. } = &op.op {
7134                    *seen.entry(path.clone()).or_default() += 1;
7135                }
7136            }
7137        })
7138        .expect("parallel scan");
7139
7140        assert!(report.is_complete());
7141        assert_eq!(seen.len() as u64, report.entries, "entry count disagrees with the report");
7142        let duplicated: Vec<_> =
7143            seen.iter().filter(|(_path, count)| **count != 1).map(|(path, _)| path).collect();
7144        assert!(duplicated.is_empty(), "paths emitted more than once: {duplicated:?}");
7145    }
7146
7147    #[test]
7148    fn parallel_walk_honours_max_depth() {
7149        let dir = branching_tree();
7150        for threads in [1_usize, 4] {
7151            let config =
7152                ScanConfig { threads: Some(threads), max_depth: Some(2), ..ScanConfig::default() };
7153            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
7154            assert!(report.is_complete());
7155            for (path, _kind, _attrs) in index_fingerprint(&index) {
7156                assert!(
7157                    path.components().count() <= 2,
7158                    "{threads} threads kept {path:?} past the depth limit"
7159                );
7160            }
7161        }
7162    }
7163
7164    #[test]
7165    fn scan_order_never_changes_the_resulting_index() {
7166        let dir = branching_tree();
7167        let depth_first =
7168            ScanConfig { order: ScanOrder::DepthFirst, threads: Some(1), ..ScanConfig::default() };
7169        let (expected, expected_report) =
7170            scan_into_index(dir.path(), &depth_first).expect("depth-first scan");
7171
7172        for (order, threads) in
7173            [(ScanOrder::BreadthFirst, 1), (ScanOrder::BreadthFirst, 4), (ScanOrder::DepthFirst, 4)]
7174        {
7175            let config = ScanConfig { order, threads: Some(threads), ..ScanConfig::default() };
7176            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
7177            assert_eq!(report.entries, expected_report.entries, "{order:?}/{threads}");
7178            assert_eq!(report.dirs_read, expected_report.dirs_read, "{order:?}/{threads}");
7179            // Public roll-ups resolve internal ids, so their named maps are stable even
7180            // when traversal order changes id assignment.
7181            let (totals, expected_totals) = (index.total(), expected.total());
7182            assert_eq!(
7183                (totals.files, totals.dirs, totals.bytes, totals.allocated),
7184                (
7185                    expected_totals.files,
7186                    expected_totals.dirs,
7187                    expected_totals.bytes,
7188                    expected_totals.allocated
7189                ),
7190                "{order:?}/{threads} roll-up"
7191            );
7192            assert_eq!(
7193                totals.newest_mtime_ns, expected_totals.newest_mtime_ns,
7194                "{order:?}/{threads} newest mtime"
7195            );
7196            assert_eq!(
7197                totals.by_ext, expected_totals.by_ext,
7198                "{order:?}/{threads} extension tallies"
7199            );
7200            assert_eq!(
7201                index_fingerprint(&index),
7202                index_fingerprint(&expected),
7203                "{order:?}/{threads} produced a different index"
7204            );
7205        }
7206    }
7207
7208    #[test]
7209    fn a_single_worker_breadth_first_walk_is_strictly_level_ordered() {
7210        // The strict guarantee, which holds only with one worker. With several, the
7211        // queue is ordered but the claims are not: a fast worker can enqueue and claim
7212        // depth d+2 while a slow worker still holds depth d+1. See
7213        // `breadth_first_starts_every_top_level_subtree_early` for the property the
7214        // default configuration actually provides, which is the one consumers rely on.
7215        let dir = branching_tree();
7216        let config = ScanConfig {
7217            order: ScanOrder::BreadthFirst,
7218            threads: Some(1),
7219            batch_size: 1,
7220            ..ScanConfig::default()
7221        };
7222        let mut depths_in_order: Vec<usize> = Vec::new();
7223        scan(dir.path(), &config, &mut |observation| {
7224            for op in &observation.ops {
7225                if let Op::Upsert { path, kind, .. } = &op.op {
7226                    if kind.is_dir() {
7227                        depths_in_order.push(path.components().count());
7228                    }
7229                }
7230            }
7231        })
7232        .expect("scan");
7233
7234        assert!(depths_in_order.len() > 10, "fixture should have many directories");
7235        assert!(
7236            depths_in_order.windows(2).all(|pair| pair[0] <= pair[1]),
7237            "directory depths were not non-decreasing: {depths_in_order:?}"
7238        );
7239    }
7240
7241    /// How many of the fixture's twelve top-level subtrees have received any file by
7242    /// the time half the files have been emitted.
7243    ///
7244    /// This is the product metric — "is a mid-scan ranking meaningful?" — rather than
7245    /// first-touch, which cannot distinguish the orders at all: reading the root
7246    /// enumerates all twelve children at once either way. What a ranking needs is that
7247    /// the subtrees grow *together*.
7248    fn subtrees_started_at_halfway(order: ScanOrder, threads: usize, dir: &Path) -> usize {
7249        let config =
7250            ScanConfig { order, batch_size: 1, threads: Some(threads), ..ScanConfig::default() };
7251
7252        let mut files: Vec<PathBuf> = Vec::new();
7253        scan(dir, &config, &mut |observation| {
7254            for op in &observation.ops {
7255                if let Op::Upsert { path, kind, .. } = &op.op {
7256                    if !kind.is_dir() {
7257                        files.push(path.clone());
7258                    }
7259                }
7260            }
7261        })
7262        .expect("scan");
7263
7264        let halfway = files.len() / 2;
7265        let mut started: BTreeSet<PathBuf> = BTreeSet::new();
7266        for path in files.iter().take(halfway) {
7267            if let Some(top) = path.components().next() {
7268                started.insert(PathBuf::from(top.as_os_str()));
7269            }
7270        }
7271        started.len()
7272    }
7273
7274    #[test]
7275    fn a_parallel_walk_accounts_for_where_its_time_went() {
7276        // The attribution identity: every named cause is a disjoint slice of worker
7277        // wall time, so the parts can never exceed the whole, and the counters that
7278        // amortization depends on are actually incremented. This is the instrument
7279        // the scheduler experiments will read; if it drifts, they measure noise.
7280        let dir = branching_tree();
7281        let config = ScanConfig { threads: Some(4), batch_size: 64, ..ScanConfig::default() };
7282        let report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
7283        let a = report.attribution;
7284
7285        assert!(a.claims > 0, "a parallel walk claims chunks: {a:?}");
7286        assert!(a.work_ns > 0, "reading directories takes time: {a:?}");
7287        assert!(a.wall_ns > 0);
7288        // claim() locks at least once per successful claim, and release() locks once
7289        // per claim cycle too.
7290        assert!(a.lock_ops >= a.claims * 2, "lock ops out of step with claims: {a:?}");
7291        assert!(
7292            a.accounted_ns() <= a.wall_ns,
7293            "attributed slices are disjoint intervals inside worker wall: {a:?}"
7294        );
7295    }
7296
7297    #[test]
7298    fn a_serial_walk_has_no_coordination_to_attribute() {
7299        // Serial semantics: wall is the loop, "send" is the inline sink (the consumer
7300        // actually running), work is the rest — and the coordination counters stay
7301        // zero because there is no queue lock and no channel.
7302        let dir = branching_tree();
7303        let config = ScanConfig { threads: Some(1), batch_size: 64, ..ScanConfig::default() };
7304        let mut observations = 0usize;
7305        let report = scan(dir.path(), &config, &mut |_| observations += 1).expect("scan");
7306        let a = report.attribution;
7307
7308        assert!(observations > 0, "the sink ran, so send_ns measured something real");
7309        assert!(a.work_ns > 0 && a.wall_ns >= a.work_ns);
7310        assert_eq!(
7311            (a.claims, a.lock_ops, a.lock_contended, a.starved_ns, a.lock_wait_ns),
7312            (0, 0, 0, 0, 0),
7313            "no queue, no lock, nothing to wait on: {a:?}"
7314        );
7315    }
7316
7317    /// Twelve top-level subtrees, each a branching tree several levels deep.
7318    ///
7319    /// Branching matters: an earlier fixture gave every level exactly one child, which
7320    /// pinned the frontier at twelve directories and made both orders behave
7321    /// identically — a LIFO cannot dive when there is nothing to dive into. With two
7322    /// children per level, depth-first pushes siblings and immediately descends into
7323    /// the last one, which is the behaviour that leaves other subtrees behind.
7324    ///
7325    /// It is also deliberately uniform. A version using one deep spur beside shallow
7326    /// siblings made the result depend on whether `readdir` returned the spur early:
7327    /// it passed on APFS and failed on ext4.
7328    fn deep_forest() -> tempfile::TempDir {
7329        let dir = tempfile::tempdir().expect("tempdir");
7330        for top in 0..12 {
7331            let mut level: Vec<PathBuf> = vec![dir.path().join(format!("t{top}"))];
7332            for _ in 0..5 {
7333                let mut next = Vec::new();
7334                for parent in &level {
7335                    for child in 0..2 {
7336                        let path = parent.join(format!("c{child}"));
7337                        for file in 0..3 {
7338                            write_file(&path.join(format!("f{file}.dat")), b"xxxxxxxxxx");
7339                        }
7340                        next.push(path);
7341                    }
7342                }
7343                level = next;
7344            }
7345        }
7346        dir
7347    }
7348
7349    /// Files accumulated by the *least advanced* top-level subtree in the first
7350    /// quarter of the walk.
7351    ///
7352    /// Counting subtrees merely *started* cannot discriminate on a tree whose root
7353    /// fans out twelve ways: every scheduler touches all twelve immediately, because
7354    /// reading the root enumerates them. What differs is whether they then advance
7355    /// together, so the question is how far behind the laggard is.
7356    fn leanest_subtree_early(order: ScanOrder, threads: usize, dir: &Path) -> usize {
7357        let config =
7358            ScanConfig { order, batch_size: 1, threads: Some(threads), ..ScanConfig::default() };
7359        let mut files: Vec<PathBuf> = Vec::new();
7360        scan(dir, &config, &mut |observation| {
7361            for op in &observation.ops {
7362                if let Op::Upsert { path, kind, .. } = &op.op {
7363                    if !kind.is_dir() {
7364                        files.push(path.clone());
7365                    }
7366                }
7367            }
7368        })
7369        .expect("scan");
7370
7371        let quarter = files.len() / 4;
7372        let mut per_top: BTreeMap<PathBuf, usize> = BTreeMap::new();
7373        for path in files.iter().take(quarter) {
7374            if let Some(top) = path.components().next() {
7375                *per_top.entry(PathBuf::from(top.as_os_str())).or_default() += 1;
7376            }
7377        }
7378        (0..12)
7379            .map(|top| per_top.get(&PathBuf::from(format!("t{top}"))).copied().unwrap_or(0))
7380            .min()
7381            .unwrap_or(0)
7382    }
7383
7384    #[test]
7385    fn deep_subtrees_do_not_delay_their_siblings() {
7386        // The orientation property, and the reason breadth-first is the default: when
7387        // every top-level subtree is deep, depth-first pours its early effort down
7388        // whichever ones it picked up and leaves the rest at zero, while the region
7389        // scheduler advances all twelve together. A user watching the top level fill
7390        // in sees a meaningful ranking in the first case and a misleading one in the
7391        // second.
7392        //
7393        // Asserted at one worker only, and that bound is deliberate. This metric reads
7394        // *emission* order, and under several workers emission reflects which worker
7395        // finished first as much as which region was claimed — so it varies with core
7396        // count. Measured on a six-core machine the margin is wide (33-37 files against
7397        // 6); on a CI runner with fewer cores both orders can report zero. That makes it
7398        // a benchmark-grade observation, recorded in exp-013, not a unit-test assertion.
7399        //
7400        // The scheduling property itself *is* asserted deterministically, against the
7401        // queue rather than through a walk, by
7402        // `the_region_scheduler_spreads_workers_over_distinct_subtrees`.
7403        let dir = deep_forest();
7404        let breadth = leanest_subtree_early(ScanOrder::BreadthFirst, 1, dir.path());
7405        let depth = leanest_subtree_early(ScanOrder::DepthFirst, 1, dir.path());
7406        assert!(
7407            breadth > depth,
7408            "breadth-first should leave its least advanced top-level subtree further \
7409             along: {breadth} files against {depth}"
7410        );
7411    }
7412
7413    #[test]
7414    fn the_region_scheduler_spreads_workers_over_distinct_subtrees() {
7415        // The scheduler invariant, checked directly on the queue rather than through a
7416        // walk: consecutive claims by *different* workers must land in different
7417        // regions while several regions have work. This is what the round-robin ready
7418        // ring buys, and it is the thing a global FIFO could not promise.
7419        let queue = DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, None, None);
7420        let mut timing = WalkAttribution::default();
7421
7422        // Bootstrap: drain the root, then seed four top-level regions.
7423        let mut claimed = Vec::new();
7424        let root = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
7425        claimed.clear();
7426        queue.extend(
7427            (0..4).map(|top| (PathBuf::from(format!("t{top}")), 1, RegionId::UNASSIGNED)),
7428            &mut timing,
7429        );
7430        assert!(root.release(0, 0, &mut timing).is_none());
7431
7432        // Four workers with no affinity must each be handed a different region. The
7433        // claims are held for the whole loop, as four concurrent workers would hold
7434        // them, because releasing between them would let one worker take every region.
7435        let mut regions = BTreeSet::new();
7436        let mut held = Vec::new();
7437        for _ in 0..4 {
7438            let mut claimed = Vec::new();
7439            held.push(queue.claim(&mut claimed, &mut timing).expect("a region has work"));
7440            regions.insert(claimed[0].2.0);
7441            assert_eq!(claimed.len(), 1, "one directory per region so far");
7442        }
7443        assert_eq!(regions.len(), 4, "each claim took a distinct region: {regions:?}");
7444    }
7445
7446    #[test]
7447    fn breadth_first_spreads_early_work_across_top_level_subtrees() {
7448        // The justification for making breadth-first the default: at the halfway point
7449        // more of the tree's top-level subtrees have started filling, so a consumer
7450        // ranking by size mid-scan is comparing partial values rather than a mix of
7451        // final values and zeros.
7452        //
7453        // Pinned with one worker, where the ordering guarantee is strict and the result
7454        // is deterministic. The multi-worker case is deliberately NOT asserted here:
7455        // measured on this fixture the advantage disappears under the default worker
7456        // count (both orders start 7-8 subtrees, run to run), because emission order is
7457        // then dominated by worker scheduling rather than by queue order. That is a
7458        // real limitation of the current design, recorded in the plan and tracked
7459        // rather than papered over with a test tuned until it passed.
7460        let dir = branching_tree();
7461        let breadth = subtrees_started_at_halfway(ScanOrder::BreadthFirst, 1, dir.path());
7462        let depth = subtrees_started_at_halfway(ScanOrder::DepthFirst, 1, dir.path());
7463
7464        assert!(
7465            breadth > depth,
7466            "breadth-first should have more top-level subtrees underway at the halfway \
7467             point, but started {breadth} against depth-first's {depth}"
7468        );
7469    }
7470
7471    #[test]
7472    fn scan_order_does_not_change_the_cache_scope() {
7473        // Order is operational, like the worker count: it changes when observations
7474        // appear, never which ones, so it must not be able to invalidate a snapshot.
7475        let breadth = ScanConfig { order: ScanOrder::BreadthFirst, ..ScanConfig::default() };
7476        let depth = ScanConfig { order: ScanOrder::DepthFirst, ..ScanConfig::default() };
7477        assert_eq!(breadth.scope(), depth.scope());
7478    }
7479
7480    #[test]
7481    fn worker_threads_are_bounded_and_never_zero() {
7482        let zero = ScanConfig { threads: Some(0), ..ScanConfig::default() };
7483        assert_eq!(zero.worker_threads(), 1, "zero threads must fall back to the serial walk");
7484        let absurd = ScanConfig { threads: Some(usize::MAX), ..ScanConfig::default() };
7485        assert_eq!(absurd.worker_threads(), MAX_SCAN_THREADS);
7486        // The automatic choice is capped well below what a caller may request, because
7487        // the measured knee is far below the core count on a large machine.
7488        let automatic = ScanConfig { threads: None, ..ScanConfig::default() };
7489        assert!((1..=DEFAULT_SCAN_THREADS_CAP).contains(&automatic.worker_threads()));
7490    }
7491
7492    #[test]
7493    fn automatic_worker_pool_keeps_a_conservative_start_and_bounded_reserve() {
7494        assert_eq!(automatic_worker_pool(1), WorkerPool::fixed(1));
7495        assert_eq!(
7496            automatic_worker_pool(4),
7497            WorkerPool {
7498                initial: 4,
7499                maximum: 8,
7500                calibration: Some(WorkerCalibration::new(
7501                    ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
7502                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
7503                )),
7504            }
7505        );
7506        assert_eq!(
7507            automatic_worker_pool(10),
7508            WorkerPool {
7509                initial: DEFAULT_SCAN_THREADS_CAP,
7510                maximum: ADAPTIVE_SCAN_THREADS_CAP,
7511                calibration: Some(WorkerCalibration::new(
7512                    ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
7513                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
7514                )),
7515            }
7516        );
7517    }
7518
7519    #[test]
7520    fn an_abandoned_claim_does_not_strand_the_other_workers() {
7521        // The liveness property behind `DirectoryClaim`. A worker that stops mid-chunk
7522        // — consumer gone, or a panic unwinding through the directory read — still owes
7523        // the queue its claim, and `claim` parks everyone else until `outstanding`
7524        // reaches zero. Before the claim was an RAII guard both of those exits skipped
7525        // the release, and every remaining worker waited on the condvar forever while
7526        // the scoped join waited on them.
7527        let queue = std::sync::Arc::new(DirectoryQueue::new(
7528            (PathBuf::new(), 0),
7529            ScanOrder::BreadthFirst,
7530            None,
7531            None,
7532        ));
7533        let mut timing = WalkAttribution::default();
7534        let mut claimed = Vec::new();
7535
7536        // One worker takes the root and abandons it without publishing anything.
7537        drop(queue.claim(&mut claimed, &mut timing).expect("root is claimable"));
7538
7539        // A second worker must now be told the walk is over rather than parking.
7540        let waiter = queue.clone();
7541        let (done, finished) = std::sync::mpsc::sync_channel(1);
7542        std::thread::spawn(move || {
7543            let mut timing = WalkAttribution::default();
7544            let mut claimed = Vec::new();
7545            let outcome = waiter.claim(&mut claimed, &mut timing).is_some();
7546            done.send(outcome).expect("publish the claim outcome");
7547        });
7548
7549        assert_eq!(
7550            finished.recv_timeout(std::time::Duration::from_secs(5)),
7551            Ok(false),
7552            "the queue must report the walk finished instead of parking the worker"
7553        );
7554    }
7555
7556    #[test]
7557    fn automatic_queue_activates_its_reserve_only_for_slow_initial_work() {
7558        let slow = WorkerCalibration::new(3, 10);
7559        let queue =
7560            DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, Some(slow), None);
7561        let mut timing = WalkAttribution::default();
7562        let mut claimed = Vec::new();
7563
7564        let claim = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
7565        queue.extend([(PathBuf::from("child"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
7566        assert!(claim.release(2, 20, &mut timing).is_none());
7567
7568        claimed.clear();
7569        let claim = queue.claim(&mut claimed, &mut timing).expect("child is claimable");
7570        queue.extend([(PathBuf::from("grandchild"), 2, claimed[0].2)].into_iter(), &mut timing);
7571        assert_eq!(claim.release(1, 10, &mut timing), Some(2));
7572
7573        let fast = WorkerCalibration::new(3, 11);
7574        let queue =
7575            DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, Some(fast), None);
7576        let mut timing = WalkAttribution::default();
7577        let mut claimed = Vec::new();
7578        let claim = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
7579        assert!(claim.release(3, 30, &mut timing).is_none());
7580        assert!(queue.lock().controller.is_none(), "calibration decides only once");
7581    }
7582
7583    #[test]
7584    fn repeated_windows_reconsider_a_late_slow_phase_in_entry_order() {
7585        let calibration = WorkerCalibration::new(4, 10);
7586        let mut controller =
7587            WorkerController::new(calibration, WorkerPolicyExperiment::RepeatedWindows);
7588
7589        let fast = controller.observe(4, 20).expect("first complete window");
7590        let slow = controller.observe(4, 80).expect("second complete window");
7591
7592        assert!(!fast.slow);
7593        assert!(slow.slow);
7594        assert_eq!((fast.start_entry_ordinal, fast.end_entry_ordinal), (0, 4));
7595        assert_eq!((slow.start_entry_ordinal, slow.end_entry_ordinal), (4, 8));
7596    }
7597
7598    #[test]
7599    fn shipped_trace_retains_post_decision_windows_without_changing_policy() {
7600        let calibration = WorkerCalibration::new(2, 10);
7601        let pool = WorkerPool { initial: 2, maximum: 4, calibration: Some(calibration) };
7602        let recorder =
7603            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::ShippedOneShot);
7604        let queue = DirectoryQueue::new_with_policy(
7605            (PathBuf::new(), 0),
7606            ScanOrder::BreadthFirst,
7607            Some(calibration),
7608            Some(recorder.clone()),
7609            pool.initial,
7610            pool.maximum,
7611            WorkerPolicyExperiment::ShippedOneShot,
7612        );
7613        let mut timing = WalkAttribution::default();
7614        let mut claimed = Vec::new();
7615
7616        let first = queue.claim(&mut claimed, &mut timing).expect("first window");
7617        queue.extend([(PathBuf::from("late"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
7618        assert_eq!(first.release(2, 10, &mut timing), None, "fast prefix holds");
7619
7620        claimed.clear();
7621        let late = queue.claim(&mut claimed, &mut timing).expect("late phase");
7622        queue.extend([(PathBuf::from("tail"), 2, RegionId::UNASSIGNED)].into_iter(), &mut timing);
7623        assert_eq!(late.release(2, 40, &mut timing), None, "shadow cannot scale");
7624
7625        let diagnostics = recorder.finish();
7626        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Held);
7627        assert_eq!(
7628            diagnostics
7629                .worker_policy
7630                .windows
7631                .iter()
7632                .map(|window| window.decision)
7633                .collect::<Vec<_>>(),
7634            vec![WorkerPolicyDecision::Hold, WorkerPolicyDecision::ObserveSlow]
7635        );
7636    }
7637
7638    #[test]
7639    fn staged_controller_requires_a_useful_frontier_then_stays_bounded() {
7640        let calibration = WorkerCalibration::new(1, 10);
7641        let pool = WorkerPool { initial: 2, maximum: 8, calibration: Some(calibration) };
7642        let recorder =
7643            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::StagedGatedWindows);
7644        let queue = DirectoryQueue::new_with_policy(
7645            (PathBuf::new(), 0),
7646            ScanOrder::BreadthFirst,
7647            Some(calibration),
7648            Some(recorder.clone()),
7649            pool.initial,
7650            pool.maximum,
7651            WorkerPolicyExperiment::StagedGatedWindows,
7652        );
7653        let mut timing = WalkAttribution::default();
7654        let mut claimed = Vec::new();
7655
7656        let root = queue.claim(&mut claimed, &mut timing).expect("root");
7657        queue.extend([(PathBuf::from("narrow"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
7658        assert_eq!(root.release(1, 20, &mut timing), None);
7659
7660        claimed.clear();
7661        let narrow = queue.claim(&mut claimed, &mut timing).expect("narrow child");
7662        queue.extend(
7663            (0..9).map(|index| (PathBuf::from(format!("wide-{index}")), 2, RegionId::UNASSIGNED)),
7664            &mut timing,
7665        );
7666        assert_eq!(narrow.release(1, 20, &mut timing), Some(4));
7667
7668        claimed.clear();
7669        let wide = queue.claim(&mut claimed, &mut timing).expect("wide claim");
7670        queue.extend(
7671            (0..9).map(|index| (PathBuf::from(format!("wider-{index}")), 3, RegionId::UNASSIGNED)),
7672            &mut timing,
7673        );
7674        assert_eq!(wide.release(1, 20, &mut timing), Some(8));
7675
7676        let diagnostics = recorder.finish();
7677        let decisions: Vec<_> = diagnostics
7678            .worker_policy
7679            .windows
7680            .iter()
7681            .map(|window| (window.decision, window.requested_workers))
7682            .collect();
7683        assert_eq!(
7684            decisions,
7685            vec![
7686                (WorkerPolicyDecision::HoldInsufficientFrontier, None),
7687                (WorkerPolicyDecision::ScaleUp, Some(4)),
7688                (WorkerPolicyDecision::ScaleUp, Some(8)),
7689            ]
7690        );
7691        assert!(
7692            diagnostics
7693                .worker_policy
7694                .windows
7695                .windows(2)
7696                .all(|pair| pair[0].end_entry_ordinal <= pair[1].start_entry_ordinal)
7697        );
7698        assert!(diagnostics.worker_policy.windows.iter().all(|window| {
7699            window.requested_workers.is_none_or(|workers| workers <= pool.maximum)
7700        }));
7701    }
7702
7703    #[test]
7704    fn staged_controller_does_not_add_producers_to_a_delayed_handoff() {
7705        let calibration = WorkerCalibration::new(1, 10);
7706        let pool = WorkerPool { initial: 2, maximum: 8, calibration: Some(calibration) };
7707        let recorder =
7708            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::StagedGatedWindows);
7709        recorder.handoff_sent();
7710        recorder.handoff_sent();
7711        let queue = DirectoryQueue::new_with_policy(
7712            (PathBuf::new(), 0),
7713            ScanOrder::BreadthFirst,
7714            Some(calibration),
7715            Some(recorder.clone()),
7716            pool.initial,
7717            pool.maximum,
7718            WorkerPolicyExperiment::StagedGatedWindows,
7719        );
7720        let mut timing = WalkAttribution::default();
7721        let mut claimed = Vec::new();
7722        let claim = queue.claim(&mut claimed, &mut timing).expect("root");
7723        queue.extend(
7724            (0..9).map(|index| (PathBuf::from(format!("ready-{index}")), 1, RegionId::UNASSIGNED)),
7725            &mut timing,
7726        );
7727
7728        assert_eq!(claim.release(1, 20, &mut timing), None);
7729        recorder.handoff_received();
7730        recorder.handoff_received();
7731        let diagnostics = recorder.finish();
7732        assert_eq!(
7733            diagnostics.worker_policy.windows[0].decision,
7734            WorkerPolicyDecision::HoldHandoffBacklog
7735        );
7736    }
7737
7738    #[test]
7739    fn candidate_retains_post_expansion_shadow_history() {
7740        let calibration = WorkerCalibration::new(1, 10);
7741        let pool = WorkerPool { initial: 2, maximum: 4, calibration: Some(calibration) };
7742        let recorder =
7743            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::RepeatedWindows);
7744        let queue = DirectoryQueue::new_with_policy(
7745            (PathBuf::new(), 0),
7746            ScanOrder::BreadthFirst,
7747            Some(calibration),
7748            Some(recorder.clone()),
7749            pool.initial,
7750            pool.maximum,
7751            WorkerPolicyExperiment::RepeatedWindows,
7752        );
7753        let mut timing = WalkAttribution::default();
7754        let mut claimed = Vec::new();
7755
7756        let slow = queue.claim(&mut claimed, &mut timing).expect("slow prefix");
7757        queue.extend(
7758            (0..4).map(|index| (PathBuf::from(format!("fast-{index}")), 1, RegionId::UNASSIGNED)),
7759            &mut timing,
7760        );
7761        assert_eq!(slow.release(1, 20, &mut timing), Some(4));
7762
7763        claimed.clear();
7764        let fast = queue.claim(&mut claimed, &mut timing).expect("fast suffix");
7765        queue.extend([(PathBuf::from("tail"), 2, RegionId::UNASSIGNED)].into_iter(), &mut timing);
7766        assert_eq!(fast.release(1, 1, &mut timing), None);
7767
7768        let diagnostics = recorder.finish();
7769        assert_eq!(
7770            diagnostics
7771                .worker_policy
7772                .windows
7773                .iter()
7774                .map(|window| window.decision)
7775                .collect::<Vec<_>>(),
7776            vec![WorkerPolicyDecision::ScaleUp, WorkerPolicyDecision::ObserveFast]
7777        );
7778    }
7779
7780    #[test]
7781    fn every_experimental_controller_preserves_exactness_and_shutdown() {
7782        let dir = branching_tree();
7783        let serial = ScanConfig { threads: Some(1), ..ScanConfig::default() };
7784        let (reference, _) = scan_into_index(dir.path(), &serial).expect("serial reference");
7785        let automatic = ScanConfig { threads: None, ..ScanConfig::default() };
7786
7787        for policy in [
7788            WorkerPolicyExperiment::ShippedOneShot,
7789            WorkerPolicyExperiment::RepeatedWindows,
7790            WorkerPolicyExperiment::StagedGatedWindows,
7791        ] {
7792            let (index, report, diagnostics) =
7793                scan_into_index_with_policy_diagnostics(dir.path(), &automatic, policy)
7794                    .expect("candidate scan finishes");
7795            assert!(report.is_complete(), "{policy:?}: {:?}", report.errors);
7796            assert_eq!(index_fingerprint(&reference), index_fingerprint(&index), "{policy:?}");
7797            assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
7798            assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
7799            assert_eq!(diagnostics.worker_policy.handoff_backlog_at_finish, 0);
7800            assert!(
7801                diagnostics.worker_policy.workers_spawned
7802                    <= diagnostics.worker_policy.maximum_workers
7803            );
7804        }
7805    }
7806
7807    /// A deterministic model of the automatic worker policy under *completion* order.
7808    ///
7809    /// The scaling decision is driven by chunk releases, and chunks complete in whatever
7810    /// order the filesystem and the workers produce them — not in traversal order. On a
7811    /// homogeneous tree that distinction is invisible, because every prefix looks like
7812    /// every other. On a heterogeneous one it decides the answer.
7813    ///
7814    /// These tests exist because the alternative is a stopwatch on a real tree, which
7815    /// measures one host on one day and cannot separate a policy defect from ambient
7816    /// noise. Replaying an explicit completion order through the shipped calibration
7817    /// isolates the policy exactly, and does so identically on every platform.
7818    ///
7819    /// They characterize behavior; they do not endorse a replacement. Which controller
7820    /// is *faster* is a question only the held-out Apple Silicon/APFS matrix can answer.
7821    mod completion_order {
7822        use super::{ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, WorkerCalibration};
7823
7824        /// One chunk release: entries observed and worker time spent observing them.
7825        #[derive(Clone, Copy, Debug)]
7826        struct Chunk {
7827            entries: u64,
7828            work_ns: u64,
7829        }
7830
7831        impl Chunk {
7832            /// A run of `entries` entries costing `per_entry_ns` each.
7833            const fn at(entries: u64, per_entry_ns: u64) -> Self {
7834                Self { entries, work_ns: entries.saturating_mul(per_entry_ns) }
7835            }
7836        }
7837
7838        /// What a policy concluded over one completion order.
7839        #[derive(Debug, PartialEq, Eq)]
7840        enum Outcome {
7841            /// The policy found the filesystem slow and expanded the reserve.
7842            ScaledUp { after_chunks: usize },
7843            /// The policy found the filesystem fast and held the initial pool.
7844            Held { after_chunks: usize },
7845            /// The walk ended before the policy observed enough to conclude anything.
7846            ///
7847            /// Distinct from [`Outcome::Held`] on purpose: nothing was measured, so a
7848            /// held pool here is an absence of evidence rather than a decision.
7849            Undecided,
7850        }
7851
7852        /// Mean cost per entry over a whole trace, which is what the threshold *means*.
7853        fn whole_trace_ns_per_entry(trace: &[Chunk]) -> u64 {
7854            let entries: u64 = trace.iter().map(|chunk| chunk.entries).sum();
7855            let work_ns: u64 = trace.iter().map(|chunk| chunk.work_ns).sum();
7856            assert!(entries > 0, "a trace must observe entries");
7857            work_ns / entries
7858        }
7859
7860        /// Replay a completion order through the *shipped* calibration.
7861        ///
7862        /// This drives [`WorkerCalibration::observe`] itself rather than restating its
7863        /// arithmetic, so the model cannot quietly drift from the policy it is evidence
7864        /// about. The loop mirrors `DirectoryQueue::release`: fold each chunk in, and
7865        /// stop at the first one that produces a verdict.
7866        fn shipped(window: u64, threshold_ns: u64, trace: &[Chunk]) -> Outcome {
7867            let mut calibration = WorkerCalibration::new(window, threshold_ns);
7868            for (index, chunk) in trace.iter().enumerate() {
7869                if let Some(slow) = calibration.observe(chunk.entries, chunk.work_ns) {
7870                    let after_chunks = index + 1;
7871                    return if slow {
7872                        Outcome::ScaledUp { after_chunks }
7873                    } else {
7874                        Outcome::Held { after_chunks }
7875                    };
7876                }
7877            }
7878            Outcome::Undecided
7879        }
7880
7881        /// Entries per chunk in the traces below. Four fill the 16,384-entry window.
7882        const CHUNK: u64 = 4_096;
7883        /// A shallow, cache-warm phase: metadata already resident.
7884        const FAST: Chunk = Chunk::at(CHUNK, 2_000);
7885        /// A deep, cold phase: the latency-bound regime the reserve exists to hide.
7886        const SLOW: Chunk = Chunk::at(CHUNK, 90_000);
7887
7888        #[test]
7889        fn completion_order_alone_flips_the_shipped_decision() {
7890            // The defect, stated as an experiment: hold the *tree* constant and vary
7891            // only the order its chunks complete in. Both traces contain the same four
7892            // fast and four slow chunks, so they describe the same filesystem work.
7893            let fast_phase_first = [FAST, FAST, FAST, FAST, SLOW, SLOW, SLOW, SLOW];
7894            let interleaved = [SLOW, FAST, SLOW, FAST, SLOW, FAST, SLOW, FAST];
7895
7896            let window = 4 * CHUNK;
7897            let threshold = ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY;
7898
7899            // Whole-walk truth is identical, and by the policy's own threshold both
7900            // walks are latency-bound: 46 µs per entry against a 30 µs trigger.
7901            let truth = whole_trace_ns_per_entry(&fast_phase_first);
7902            assert_eq!(truth, whole_trace_ns_per_entry(&interleaved));
7903            assert!(
7904                truth >= threshold,
7905                "both traces are slow walks by the shipped threshold: {truth} < {threshold}"
7906            );
7907
7908            // Yet the decision depends entirely on which chunks happened to finish
7909            // first. One walk hides latency; the other runs the whole slow phase on the
7910            // starting pool, having concluded from an unrepresentative prefix.
7911            assert_eq!(
7912                shipped(window, threshold, &fast_phase_first),
7913                Outcome::Held { after_chunks: 4 },
7914                "a fast prefix holds the pool for a walk that is slow overall"
7915            );
7916            assert_eq!(
7917                shipped(window, threshold, &interleaved),
7918                Outcome::ScaledUp { after_chunks: 4 },
7919                "the same tree scales up when its slow chunks land in the window"
7920            );
7921        }
7922
7923        #[test]
7924        fn a_slow_phase_after_the_window_is_never_reconsidered() {
7925            // The heterogeneous-tree case from the field report. A small fast region
7926            // fills the window, and everything after it is slow — but the calibration
7927            // is already gone, so no amount of later evidence can reopen the decision.
7928            let mut trace = vec![FAST; 4];
7929            trace.extend(std::iter::repeat_n(SLOW, 400));
7930
7931            let observed = whole_trace_ns_per_entry(&trace);
7932            assert!(
7933                observed >= 2 * ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
7934                "the walk is overwhelmingly latency-bound: {observed} ns per entry"
7935            );
7936
7937            assert_eq!(
7938                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
7939                Outcome::Held { after_chunks: 4 },
7940                "1% of the walk decided the worker policy for the other 99%"
7941            );
7942        }
7943
7944        #[test]
7945        fn slow_in_flight_work_is_censored_by_fast_completions() {
7946            // Four slow chunks have already been claimed, but their filesystem calls
7947            // remain in flight while four cache-warm chunks complete. Completion-order
7948            // calibration cannot see owed work: the fast completions close the window
7949            // and permanently hold before any slow claim returns.
7950            let completed_before_slow_returns = [FAST, FAST, FAST, FAST];
7951            assert_eq!(
7952                shipped(
7953                    4 * CHUNK,
7954                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
7955                    &completed_before_slow_returns,
7956                ),
7957                Outcome::Held { after_chunks: 4 }
7958            );
7959
7960            let mut eventual_completions = completed_before_slow_returns.to_vec();
7961            eventual_completions.extend([SLOW, SLOW, SLOW, SLOW]);
7962            assert!(
7963                whole_trace_ns_per_entry(&eventual_completions)
7964                    >= ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY
7965            );
7966            assert_eq!(
7967                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &eventual_completions,),
7968                Outcome::Held { after_chunks: 4 }
7969            );
7970        }
7971
7972        #[test]
7973        fn a_slow_prefix_can_scale_a_walk_that_is_fast_overall() {
7974            // The mirror-image error. A one-way expansion reacts correctly to the
7975            // prefix by its local threshold, but the prefix is under 1% of this walk
7976            // and the whole trace is firmly in the fast regime. A repeated trigger
7977            // alone cannot undo an expansion; staged growth limits exposure but does
7978            // not make reversible parking unnecessary.
7979            let mut trace = vec![SLOW; 4];
7980            trace.extend(std::iter::repeat_n(FAST, 400));
7981            let observed = whole_trace_ns_per_entry(&trace);
7982            assert!(observed < ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY);
7983            assert_eq!(
7984                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
7985                Outcome::ScaledUp { after_chunks: 4 }
7986            );
7987            assert_eq!(
7988                sliding(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
7989                Outcome::ScaledUp { after_chunks: 4 }
7990            );
7991        }
7992
7993        #[test]
7994        fn a_walk_shorter_than_the_window_decides_nothing() {
7995            // Fails closed rather than reporting a held pool: a walk this short never
7996            // observed enough to have an opinion, and an artifact that recorded `Held`
7997            // would claim a measurement that was never taken.
7998            let trace = [FAST, SLOW];
7999            assert_eq!(
8000                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
8001                Outcome::Undecided
8002            );
8003        }
8004
8005        /// A screening-only candidate: a window that slides instead of closing once.
8006        ///
8007        /// Present as *evidence about a design*, not as a proposed change. It keeps the
8008        /// shipped trigger and pool bounds and alters only when the question is asked,
8009        /// which is the narrowest edit that could address the order sensitivity above.
8010        /// Whether it is faster on a real tree is unmeasured here and unmeasurable in a
8011        /// virtualized non-APFS environment; selecting it would need the held-out Apple
8012        /// Silicon matrix that this workstream has not yet been able to run.
8013        struct SlidingWindow {
8014            window_entries: u64,
8015            threshold_ns: u64,
8016            recent: std::collections::VecDeque<Chunk>,
8017            entries: u64,
8018            work_ns: u64,
8019        }
8020
8021        impl SlidingWindow {
8022            fn new(window_entries: u64, threshold_ns: u64) -> Self {
8023                Self {
8024                    window_entries,
8025                    threshold_ns,
8026                    recent: std::collections::VecDeque::new(),
8027                    entries: 0,
8028                    work_ns: 0,
8029                }
8030            }
8031
8032            /// Fold in a chunk and re-ask the question over the trailing window.
8033            fn observe(&mut self, chunk: Chunk) -> Option<bool> {
8034                self.recent.push_back(chunk);
8035                self.entries = self.entries.saturating_add(chunk.entries);
8036                self.work_ns = self.work_ns.saturating_add(chunk.work_ns);
8037
8038                // Drop from the front while the window stays full without the oldest
8039                // chunk, so the answer describes recent work rather than the whole walk.
8040                while let Some(oldest) = self.recent.front().copied() {
8041                    if self.entries - oldest.entries < self.window_entries {
8042                        break;
8043                    }
8044                    self.recent.pop_front();
8045                    self.entries -= oldest.entries;
8046                    self.work_ns -= oldest.work_ns;
8047                }
8048
8049                (self.entries >= self.window_entries)
8050                    .then(|| self.work_ns / self.entries >= self.threshold_ns)
8051            }
8052        }
8053
8054        /// Replay a completion order through the candidate, stopping at its first
8055        /// scale-up. The shipped pool only grows, so a later verdict cannot undo one.
8056        fn sliding(window: u64, threshold_ns: u64, trace: &[Chunk]) -> Outcome {
8057            let mut policy = SlidingWindow::new(window, threshold_ns);
8058            let mut decided = None;
8059            for (index, chunk) in trace.iter().enumerate() {
8060                if let Some(slow) = policy.observe(*chunk) {
8061                    let after_chunks = index + 1;
8062                    if slow {
8063                        return Outcome::ScaledUp { after_chunks };
8064                    }
8065                    decided.get_or_insert(Outcome::Held { after_chunks });
8066                }
8067            }
8068            decided.unwrap_or(Outcome::Undecided)
8069        }
8070
8071        #[test]
8072        fn screening_a_sliding_window_against_the_order_sensitivity() {
8073            let window = 4 * CHUNK;
8074            let threshold = ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY;
8075
8076            // The pair that splits the shipped policy reaches one answer here, and it
8077            // is the answer the whole-trace mean supports in both orders.
8078            let fast_phase_first = [FAST, FAST, FAST, FAST, SLOW, SLOW, SLOW, SLOW];
8079            let interleaved = [SLOW, FAST, SLOW, FAST, SLOW, FAST, SLOW, FAST];
8080            // Both reach the same verdict; they differ only in how long the fast prefix
8081            // delays it, which is the behavior a trailing window is supposed to have.
8082            assert_eq!(
8083                sliding(window, threshold, &fast_phase_first),
8084                Outcome::ScaledUp { after_chunks: 6 }
8085            );
8086            assert_eq!(
8087                sliding(window, threshold, &interleaved),
8088                Outcome::ScaledUp { after_chunks: 4 }
8089            );
8090
8091            // And the late slow phase is reached rather than missed: two slow chunks
8092            // after the window closes are enough to pull the trailing mean over.
8093            let mut late = vec![FAST; 4];
8094            late.extend(std::iter::repeat_n(SLOW, 400));
8095            assert_eq!(sliding(window, threshold, &late), Outcome::ScaledUp { after_chunks: 6 });
8096
8097            // A genuinely fast tree must still hold the pool: the candidate has to keep
8098            // the property the shipped policy gets right, or it is not a candidate.
8099            let uniformly_fast = vec![FAST; 40];
8100            assert_eq!(
8101                sliding(window, threshold, &uniformly_fast),
8102                Outcome::Held { after_chunks: 4 }
8103            );
8104
8105            // A short walk still decides nothing, for the same reason as above.
8106            assert_eq!(sliding(window, threshold, &[FAST, SLOW]), Outcome::Undecided);
8107        }
8108    }
8109
8110    #[test]
8111    fn thread_count_does_not_change_the_cache_scope() {
8112        // Threads are an operational choice. If they leaked into the scope, changing
8113        // the pool size would invalidate every snapshot on disk.
8114        let serial = ScanConfig { threads: Some(1), ..ScanConfig::default() };
8115        let parallel = ScanConfig { threads: Some(8), ..ScanConfig::default() };
8116        assert_eq!(serial.scope(), parallel.scope());
8117    }
8118
8119    #[test]
8120    fn scan_populates_an_index_end_to_end() {
8121        let dir = sample_tree();
8122        let (index, report) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
8123
8124        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8125        let total = index.total();
8126        assert_eq!(total.files, 3);
8127        assert_eq!(total.dirs, 2);
8128        assert_eq!(total.bytes, 5 + 12 + 9);
8129        assert_eq!(total.by_ext[".rs"].files, 2);
8130        assert_eq!(total.by_ext[".txt"].files, 1);
8131
8132        let src = index.rollup(Path::new("src")).expect("src");
8133        assert_eq!(src.files, 2);
8134        assert_eq!(src.dirs, 1);
8135    }
8136
8137    #[test]
8138    fn cold_scan_routes_control_sources_through_both_walkers() {
8139        let dir = tempfile::tempdir().expect("tempdir");
8140        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8141        write_file(&dir.path().join("debug.log"), b"ignored");
8142        write_file(&dir.path().join("keep.rs"), b"visible");
8143
8144        for threads in [1, 4] {
8145            let config =
8146                ScanConfig { read_controls: true, threads: Some(threads), ..ScanConfig::default() };
8147            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
8148
8149            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8150            assert!(
8151                index
8152                    .controls()
8153                    .expect("control state observed")
8154                    .source_is(Path::new(".gitignore"), b"*.log\n")
8155            );
8156            assert_eq!(
8157                index.is_ignored(Path::new("debug.log")).expect("control state observed"),
8158                Some(true)
8159            );
8160            assert_eq!(
8161                index.is_ignored(Path::new("keep.rs")).expect("control state observed"),
8162                Some(false)
8163            );
8164            let partitions = index.partition_total().expect("control state observed");
8165            assert_eq!(partitions.all.files, 3);
8166            assert_eq!(partitions.unignored.files, 2);
8167        }
8168    }
8169
8170    #[cfg(unix)]
8171    #[test]
8172    fn raced_fifo_control_source_is_rejected_without_blocking() {
8173        let dir = tempfile::tempdir().expect("tempdir");
8174        let control = dir.path().join(".gitignore");
8175        let status = match std::process::Command::new("mkfifo").arg(&control).status() {
8176            Ok(status) => status,
8177            Err(error) if error.kind() == std::io::ErrorKind::NotFound => return,
8178            Err(error) => panic!("create fifo: {error}"),
8179        };
8180        assert!(status.success(), "mkfifo exited with {status}");
8181
8182        let root = dir.path().to_path_buf();
8183        let (sender, receiver) = std::sync::mpsc::channel();
8184        std::thread::spawn(move || {
8185            let result = read_control_op_unconditional(
8186                &root,
8187                Path::new(".gitignore"),
8188                EntryKind::File,
8189                Some(crate::control::DEFAULT_CONTROL_BUDGET),
8190            );
8191            sender.send(result).ok();
8192        });
8193        let result = receiver
8194            .recv_timeout(std::time::Duration::from_secs(1))
8195            .expect("a raced FIFO must not block the scan worker")
8196            .expect("the non-regular replacement is a normal control removal");
8197
8198        assert!(matches!(result, Some(Op::ControlRemove { .. })));
8199    }
8200
8201    #[test]
8202    fn hidden_admission_keeps_exact_allowlist_and_control_signals_only() {
8203        let dir = tempfile::tempdir().expect("tempdir");
8204        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8205        write_file(&dir.path().join("debug.log"), b"ignored");
8206        write_file(&dir.path().join(".secret/token"), b"hidden");
8207        write_file(&dir.path().join(".github/workflows/check.yml"), b"visible");
8208        let hidden = std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([".github"]));
8209
8210        for threads in [1, 4] {
8211            let config = ScanConfig {
8212                hidden: Some(std::sync::Arc::clone(&hidden)),
8213                threads: Some(threads),
8214                read_controls: true,
8215                ..ScanConfig::default()
8216            };
8217            let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
8218
8219            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8220            assert!(index.lookup(Path::new(".gitignore")).is_none());
8221            assert!(index.lookup(Path::new(".secret")).is_none());
8222            assert!(index.lookup(Path::new(".secret/token")).is_none());
8223            assert!(index.lookup(Path::new(".github/workflows/check.yml")).is_some());
8224            assert!(
8225                index
8226                    .controls()
8227                    .expect("control state observed")
8228                    .source_is(Path::new(".gitignore"), b"*.log\n")
8229            );
8230            assert_eq!(
8231                index.is_ignored(Path::new("debug.log")).expect("control state observed"),
8232                Some(true)
8233            );
8234
8235            fs::remove_file(dir.path().join(".gitignore")).expect("remove control");
8236            if threads > 1 {
8237                fs::create_dir(dir.path().join(".gitignore")).expect("replace with directory");
8238            }
8239            let reconciled = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
8240            assert!(reconciled.is_complete());
8241            assert!(index.controls().expect("control state observed").is_empty());
8242            if threads > 1 {
8243                fs::remove_dir(dir.path().join(".gitignore")).expect("remove directory");
8244            }
8245            write_file(&dir.path().join(".gitignore"), b"*.log\n");
8246        }
8247    }
8248
8249    #[cfg(unix)]
8250    #[test]
8251    fn excluded_special_objects_never_enter_cold_or_reconciled_facts() {
8252        use std::os::unix::net::UnixListener;
8253
8254        let dir = tempfile::tempdir().expect("tempdir");
8255        let socket_path = dir.path().join("service.sock");
8256        let _listener = UnixListener::bind(&socket_path).expect("bind socket");
8257        write_file(&dir.path().join("replacement"), b"ordinary");
8258        let (kept, kept_report) =
8259            scan_into_index(dir.path(), &ScanConfig::default()).expect("default scan");
8260        assert!(kept_report.is_complete());
8261        assert_eq!(kept.kind(Path::new("service.sock")), Some(EntryKind::Other));
8262
8263        let serial_config =
8264            ScanConfig { exclude_special: true, threads: Some(1), ..ScanConfig::default() };
8265        let parallel_config =
8266            ScanConfig { exclude_special: true, threads: Some(4), ..ScanConfig::default() };
8267        let (mut serial, serial_report) =
8268            scan_into_index(dir.path(), &serial_config).expect("serial scan");
8269        let (mut parallel, parallel_report) =
8270            scan_into_index(dir.path(), &parallel_config).expect("parallel scan");
8271
8272        for (index, report) in [(&serial, &serial_report), (&parallel, &parallel_report)] {
8273            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8274            assert!(index.lookup(Path::new("service.sock")).is_none());
8275            assert!(index.lookup(Path::new("replacement")).is_some());
8276        }
8277
8278        fs::remove_file(dir.path().join("replacement")).expect("remove file");
8279        let _replacement =
8280            UnixListener::bind(dir.path().join("replacement")).expect("bind replacement socket");
8281        let serial_reconciled =
8282            reconcile(&mut serial, &serial_config, &mut |_| {}).expect("serial reconcile");
8283        let parallel_reconciled =
8284            reconcile(&mut parallel, &parallel_config, &mut |_| {}).expect("parallel reconcile");
8285
8286        assert!(serial_reconciled.is_complete());
8287        assert!(parallel_reconciled.is_complete());
8288        assert!(serial.lookup(Path::new("replacement")).is_none());
8289        assert!(parallel.lookup(Path::new("replacement")).is_none());
8290        assert_eq!(index_fingerprint(&serial), index_fingerprint(&parallel));
8291    }
8292
8293    #[test]
8294    fn control_sources_respect_a_single_operation_batch_bound() {
8295        let dir = tempfile::tempdir().expect("tempdir");
8296        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8297        write_file(&dir.path().join("debug.log"), b"ignored");
8298
8299        for threads in [1, 4] {
8300            let config = ScanConfig {
8301                read_controls: true,
8302                threads: Some(threads),
8303                batch_size: 1,
8304                ..ScanConfig::default()
8305            };
8306            let mut largest = 0;
8307            let report = scan(dir.path(), &config, &mut |observation| {
8308                largest = largest.max(observation.len());
8309            })
8310            .expect("scan");
8311
8312            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8313            assert_eq!(largest, 1);
8314        }
8315    }
8316
8317    #[test]
8318    fn cold_scan_matches_the_metabrowser_nested_control_fixture() {
8319        let dir = tempfile::tempdir().expect("tempdir");
8320        write_file(&dir.path().join(".gitignore"), b"node_modules/\n*.pyc\n");
8321        write_file(&dir.path().join("src/app.py"), b"x");
8322        write_file(&dir.path().join("src/thing.pyc"), b"x");
8323        write_file(&dir.path().join("src/generated/.gitignore"), b"*.gen\n");
8324        write_file(&dir.path().join("src/generated/out.gen"), b"x");
8325        write_file(&dir.path().join("node_modules/.gitignore"), b"!keep-me.py\n");
8326        write_file(&dir.path().join("node_modules/keep-me.py"), b"x");
8327
8328        let (index, report) = scan_into_index(
8329            dir.path(),
8330            &ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() },
8331        )
8332        .expect("scan fixture");
8333
8334        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8335        assert_eq!(
8336            index.is_ignored(Path::new("src/app.py")).expect("control state observed"),
8337            Some(false)
8338        );
8339        assert_eq!(
8340            index.is_ignored(Path::new("src/thing.pyc")).expect("control state observed"),
8341            Some(true)
8342        );
8343        assert_eq!(
8344            index.is_ignored(Path::new("src/generated")).expect("control state observed"),
8345            Some(false)
8346        );
8347        assert_eq!(
8348            index.is_ignored(Path::new("src/generated/out.gen")).expect("control state observed"),
8349            Some(true)
8350        );
8351        assert_eq!(
8352            index.is_ignored(Path::new("node_modules")).expect("control state observed"),
8353            Some(true)
8354        );
8355        assert_eq!(
8356            index.is_ignored(Path::new("node_modules/keep-me.py")).expect("control state observed"),
8357            Some(true)
8358        );
8359    }
8360
8361    #[test]
8362    fn reconciliation_observes_same_metadata_control_edits_and_last_deletion() {
8363        let dir = tempfile::tempdir().expect("tempdir");
8364        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8365        write_file(&dir.path().join("debug.log"), b"ignored");
8366        let config = ScanConfig { read_controls: true, threads: Some(1), ..ScanConfig::default() };
8367        let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
8368        assert!(report.is_complete());
8369        assert_eq!(
8370            index.is_ignored(Path::new("debug.log")).expect("control state observed"),
8371            Some(true)
8372        );
8373
8374        // Same-length content proves control identity is not inferred from stat-tier
8375        // metadata, which can remain unchanged on coarse filesystems.
8376        write_file(&dir.path().join(".gitignore"), b"*.tmp\n");
8377        let edited = reconcile(&mut index, &config, &mut |_| {}).expect("edit reconcile");
8378        assert!(edited.is_complete());
8379        assert_eq!(edited.apply.controls, 1);
8380        assert_eq!(edited.apply.reclassified, 1);
8381        assert!(
8382            index
8383                .controls()
8384                .expect("control state observed")
8385                .source_is(Path::new(".gitignore"), b"*.tmp\n")
8386        );
8387        assert_eq!(
8388            index.is_ignored(Path::new("debug.log")).expect("control state observed"),
8389            Some(false)
8390        );
8391
8392        fs::remove_file(dir.path().join(".gitignore")).expect("remove control");
8393        let removed = reconcile(&mut index, &config, &mut |_| {}).expect("remove reconcile");
8394        assert!(removed.is_complete());
8395        assert_eq!(removed.apply.controls, 1);
8396        assert!(index.controls().expect("control state observed").is_empty());
8397        let partitions = index.partition_total().expect("control state observed");
8398        assert_eq!(partitions.all, partitions.unignored);
8399    }
8400
8401    #[cfg(unix)]
8402    #[test]
8403    fn directory_entry_metadata_does_not_follow_symlinks() {
8404        use std::os::unix::fs::symlink;
8405
8406        let root = tempfile::tempdir().expect("root");
8407        let outside = tempfile::tempdir().expect("outside");
8408        write_file(&outside.path().join("must-not-be-scanned.txt"), b"outside");
8409        symlink(outside.path(), root.path().join("link")).expect("symlink");
8410
8411        let (index, report) = scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
8412
8413        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8414        assert_eq!(index.kind(Path::new("link")), Some(EntryKind::Symlink));
8415        assert!(index.lookup(Path::new("link/must-not-be-scanned.txt")).is_none());
8416        assert_eq!(index.total().files, 0);
8417    }
8418
8419    #[test]
8420    fn cold_scan_establishes_a_baseline_without_change_history() {
8421        let dir = sample_tree();
8422        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
8423
8424        assert_eq!(index.clock(), crate::Clock::ZERO);
8425        assert!(index.since(crate::Clock::ZERO).commits.is_empty());
8426    }
8427
8428    #[test]
8429    fn max_depth_stops_descent() {
8430        let dir = sample_tree();
8431        let config = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
8432        let (index, _) = scan_into_index(dir.path(), &config).expect("scan");
8433
8434        assert!(index.lookup(Path::new("src")).is_some());
8435        assert!(index.lookup(Path::new("src/main.rs")).is_none());
8436    }
8437
8438    #[test]
8439    fn zero_max_depth_keeps_only_the_index_root() {
8440        let dir = sample_tree();
8441        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
8442        let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
8443
8444        assert!(index.is_empty());
8445        assert_eq!(report.entries, 0);
8446        assert_eq!(report.dirs_read, 0);
8447    }
8448
8449    #[test]
8450    fn direct_scan_records_the_canonical_root() {
8451        let dir = sample_tree();
8452        let aliased = dir.path().join(".");
8453        let (index, _) = scan_into_index(&aliased, &ScanConfig::default()).expect("scan");
8454
8455        assert_eq!(index.root_path(), dir.path().canonicalize().expect("canonical root"));
8456    }
8457
8458    #[test]
8459    fn unsupported_symlink_following_is_rejected_on_cold_and_warm_paths() {
8460        let dir = sample_tree();
8461        let unsupported = ScanConfig { follow_symlinks: true, ..ScanConfig::default() };
8462        assert!(matches!(
8463            scan_into_index(dir.path(), &unsupported),
8464            Err(Error::UnsupportedScanConfig(_))
8465        ));
8466
8467        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
8468        assert!(matches!(
8469            revalidate(&index, &unsupported, &mut |_| {}),
8470            Err(Error::UnsupportedScanConfig(_))
8471        ));
8472    }
8473
8474    #[test]
8475    fn revalidation_uses_the_same_depth_boundary_as_cold_scan() {
8476        let dir = sample_tree();
8477        let config = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
8478        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
8479        write_file(&dir.path().join("src/added-after-scan.txt"), b"new");
8480
8481        let mut observations = Vec::new();
8482        revalidate(&index, &config, &mut |observation| observations.push(observation))
8483            .expect("revalidate");
8484        for observation in &observations {
8485            index.apply_ok(observation);
8486        }
8487
8488        assert!(index.lookup(Path::new("src/added-after-scan.txt")).is_none());
8489    }
8490
8491    #[test]
8492    fn zero_depth_revalidation_prunes_cached_root_children() {
8493        let dir = tempfile::tempdir().expect("tempdir");
8494        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
8495        let mut index = Index::new_with_scope(dir.path(), config.scope());
8496        index.apply_baseline_ok(&Observation::new(vec![Op::Upsert {
8497            path: PathBuf::from("stale.txt"),
8498            kind: EntryKind::File,
8499            attrs: Attrs::default(),
8500        }]));
8501
8502        let mut observations = Vec::new();
8503        let report = revalidate(&index, &config, &mut |observation| {
8504            observations.push(observation);
8505        })
8506        .expect("revalidate");
8507        for observation in &observations {
8508            index.apply_ok(observation);
8509        }
8510
8511        assert!(index.is_empty());
8512        assert_eq!(report.dirs_read, 0);
8513    }
8514
8515    #[test]
8516    fn zero_depth_applying_reconciliation_prunes_cached_root_children() {
8517        let dir = tempfile::tempdir().expect("tempdir");
8518        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
8519        let mut index = Index::new_with_scope(dir.path(), config.scope());
8520        index.apply_baseline_ok(&Observation::new(vec![Op::Upsert {
8521            path: PathBuf::from("stale.txt"),
8522            kind: EntryKind::File,
8523            attrs: Attrs::default(),
8524        }]));
8525
8526        let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
8527
8528        assert!(index.is_empty());
8529        assert_eq!(report.scan.dirs_read, 0);
8530    }
8531
8532    #[test]
8533    fn filesystem_boundary_is_part_of_the_shared_descent_policy() {
8534        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
8535        let attrs = Attrs { dev: 22, ..Attrs::default() };
8536        assert!(!should_descend(EntryKind::Dir, attrs, 0, 11, &config));
8537        assert!(should_descend(EntryKind::Dir, Attrs { dev: 11, ..attrs }, 0, 11, &config,));
8538    }
8539
8540    /// A cold scan's index records its own pass start, the stamp a snapshot of it writes:
8541    /// never earlier than an instant taken before the scan, so it is not a stale or zero
8542    /// stamp, and never later than one taken after it. The builder constructs the index,
8543    /// and so takes the stamp, before the walk begins.
8544    #[test]
8545    fn a_cold_scan_stamps_its_own_pass_start() {
8546        let nanos = || {
8547            i64::try_from(
8548                std::time::SystemTime::now()
8549                    .duration_since(std::time::UNIX_EPOCH)
8550                    .expect("after the epoch")
8551                    .as_nanos(),
8552            )
8553            .expect("nanoseconds")
8554        };
8555        let dir = sample_tree();
8556        let before = nanos();
8557        let (index, report) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
8558        let after = nanos();
8559        assert!(report.is_complete() && report.entries > 0, "{report:?}");
8560        let stamp = index.writing_pass_started_at_ns();
8561        assert!(before <= stamp && stamp <= after, "{before} <= {stamp} <= {after}");
8562    }
8563
8564    #[test]
8565    fn scanning_a_file_is_an_error_not_a_panic() {
8566        let dir = sample_tree();
8567        let err = scan_into_index(&dir.path().join("a.txt"), &ScanConfig::default());
8568        assert!(err.is_err());
8569    }
8570
8571    #[test]
8572    fn deltas_arrive_in_batches_of_the_configured_size() {
8573        let dir = tempfile::tempdir().expect("tempdir");
8574        for i in 0..25 {
8575            write_file(&dir.path().join(format!("f{i}.txt")), b"x");
8576        }
8577        let config = ScanConfig { batch_size: 10, ..ScanConfig::default() };
8578        let mut sizes = Vec::new();
8579        scan(dir.path(), &config, &mut |d| sizes.push(d.len())).expect("scan");
8580
8581        assert!(sizes.len() >= 3, "expected several batches, got {sizes:?}");
8582        assert!(sizes.iter().all(|&n| n <= 10));
8583        assert_eq!(sizes.iter().sum::<usize>(), 25);
8584    }
8585
8586    #[test]
8587    fn invalid_batch_sizes_are_rejected_before_allocation() {
8588        let zero = ScanConfig { batch_size: 0, ..ScanConfig::default() };
8589        let unbounded = ScanConfig { batch_size: usize::MAX, ..ScanConfig::default() };
8590
8591        assert!(matches!(zero.validate(), Err(Error::UnsupportedScanConfig(_))));
8592        assert!(matches!(unbounded.validate(), Err(Error::UnsupportedScanConfig(_))));
8593    }
8594
8595    #[test]
8596    fn reconciliation_scope_budget_publishes_before_returning_retry() {
8597        let directory = tempfile::tempdir().expect("root");
8598        let config = ScanConfig::default();
8599        let (index, _) = scan_into_index(directory.path(), &config).expect("scan root-only tree");
8600        let handle = IndexHandle::new(index);
8601        let mut started = false;
8602        let mut published_partial = false;
8603        let report = reconcile_handle(&handle, &config, &mut |commit| {
8604            if !started {
8605                started = true;
8606                // No entries are added: distinct absent children must not grow history
8607                // for the paused root pass without bound.
8608                for child in ["missing-a", "missing-b", "missing-c"] {
8609                    let nested = reconcile_subtree_handle(&handle, Path::new(child), &config, &mut |_| {}).expect("newer absent scope");
8610                    assert!(nested.is_complete());
8611                }
8612            }
8613            if commit.state.iter().any(|state| matches!(state,
8614                crate::StateTransition::IndexState { current, .. }
8615                    if current.coverage == crate::Coverage::Partial(crate::CoverageReason::Inaccessible))) {
8616                assert_eq!(handle.read_with(Index::state).expect("coherent state").coverage,
8617                    crate::Coverage::Partial(crate::CoverageReason::Inaccessible));
8618                published_partial = true;
8619            }
8620        }).expect("interrupted pass returns retryable report");
8621        assert!(!report.is_complete());
8622        assert!(report.retry_required);
8623        assert!(published_partial, "the transition precedes the caller's retry result");
8624        let recovered = reconcile_handle(&handle, &config, &mut |_| {}).expect("retry");
8625        assert!(recovered.is_complete());
8626        assert_eq!(
8627            handle.read_with(Index::state).expect("recovered state").coverage,
8628            crate::Coverage::Complete
8629        );
8630    }
8631
8632    #[test]
8633    fn stale_arbitration_keeps_a_reconciliation_incomplete() {
8634        let report = ReconcileReport {
8635            scan: ScanReport::default(),
8636            apply: ApplyStats { stale: 1, ..ApplyStats::default() },
8637            observations: 1,
8638            ..ReconcileReport::default()
8639        };
8640
8641        assert!(!report.is_complete());
8642    }
8643
8644    #[test]
8645    fn portable_system_time_conversion_preserves_pre_epoch_values() {
8646        let before_epoch = std::time::UNIX_EPOCH
8647            // Windows timestamps have 100 ns granularity, so use a duration that every
8648            // supported platform can represent without rounding back to the epoch.
8649            .checked_sub(std::time::Duration::from_secs(1))
8650            .expect("represent pre-epoch fixture");
8651
8652        assert_eq!(system_time_ns(before_epoch), -1_000_000_000);
8653        assert_eq!(system_time_ns(std::time::UNIX_EPOCH), 0);
8654    }
8655
8656    #[cfg(not(unix))]
8657    #[test]
8658    fn one_filesystem_fails_when_device_identity_is_unavailable() {
8659        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
8660
8661        assert!(matches!(config.validate(), Err(Error::UnsupportedScanConfig(_))));
8662    }
8663
8664    #[test]
8665    fn revalidate_is_a_no_op_against_an_unchanged_tree() {
8666        let dir = sample_tree();
8667        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
8668        let before = index.total();
8669
8670        let mut deltas = Vec::new();
8671        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
8672        let mut unchanged = 0;
8673        for delta in &deltas {
8674            unchanged += index.apply_ok(delta).unchanged;
8675        }
8676
8677        assert_eq!(unchanged, 5, "3 files + 2 dirs all already known");
8678        assert_eq!(index.total(), before);
8679    }
8680
8681    #[cfg(windows)]
8682    #[test]
8683    fn windows_reconcile_detects_same_size_rewrite_with_preserved_mtime() {
8684        let root = tempfile::tempdir().expect("tempdir");
8685        let path = root.path().join("same.txt");
8686        write_file(&path, b"first");
8687        let modified = fs::metadata(&path).expect("metadata").modified().expect("mtime");
8688        let (mut index, _) =
8689            scan_into_index(root.path(), &ScanConfig::default()).expect("initial scan");
8690        let before = *index.attrs(Path::new("same.txt")).expect("initial attrs");
8691
8692        // NTFS stamps change time from the system clock, which advances in ticks of up to
8693        // 15.625 ms, so a rewrite stamped in the same tick as the first write is
8694        // indistinguishable from it. Wait for the clock to leave that tick rather than for
8695        // a fixed interval; the precise clock `SystemTime` reads runs at most one tick ahead.
8696        let stamped = std::time::UNIX_EPOCH
8697            + std::time::Duration::from_nanos(
8698                u64::try_from(before.ctime_ns).expect("change time after the epoch"),
8699            );
8700        while std::time::SystemTime::now() <= stamped + std::time::Duration::from_millis(20) {
8701            std::thread::sleep(std::time::Duration::from_millis(5));
8702        }
8703        write_file(&path, b"other");
8704        File::options()
8705            .write(true)
8706            .open(&path)
8707            .expect("open rewritten file")
8708            .set_times(std::fs::FileTimes::new().set_modified(modified))
8709            .expect("restore mtime");
8710        reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
8711
8712        let after = *index.attrs(Path::new("same.txt")).expect("rewritten attrs");
8713        assert_eq!((after.size, after.mtime_ns), (before.size, before.mtime_ns));
8714        assert_ne!(after.ctime_ns, before.ctime_ns, "change time detects the rewrite");
8715        assert_ne!(after.fingerprint(), before.fingerprint());
8716    }
8717
8718    #[cfg(windows)]
8719    #[test]
8720    fn windows_reconcile_detects_path_identity_replacement() {
8721        let root = tempfile::tempdir().expect("tempdir");
8722        let path = root.path().join("replace.txt");
8723        let displaced = root.path().join("displaced.txt");
8724        write_file(&path, b"first");
8725        let modified = fs::metadata(&path).expect("metadata").modified().expect("mtime");
8726        let (mut index, _) =
8727            scan_into_index(root.path(), &ScanConfig::default()).expect("initial scan");
8728        let before = *index.attrs(Path::new("replace.txt")).expect("initial attrs");
8729
8730        fs::rename(&path, &displaced).expect("retain old file identity");
8731        write_file(&path, b"other");
8732        File::options()
8733            .write(true)
8734            .open(&path)
8735            .expect("open replacement")
8736            .set_times(std::fs::FileTimes::new().set_modified(modified))
8737            .expect("restore mtime");
8738        reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
8739
8740        let after = *index.attrs(Path::new("replace.txt")).expect("replacement attrs");
8741        assert_eq!((after.size, after.mtime_ns), (before.size, before.mtime_ns));
8742        assert_ne!(
8743            (after.dev, after.inode),
8744            (before.dev, before.inode),
8745            "volume serial and file index identify the replacement"
8746        );
8747        assert_ne!(after.fingerprint(), before.fingerprint());
8748    }
8749
8750    #[test]
8751    fn direct_reconciliation_counts_unchanged_entries_and_publishes_state_commits() {
8752        let dir = sample_tree();
8753        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
8754        let before_total = index.total();
8755        let before_clock = index.clock();
8756        let mut commits = Vec::new();
8757
8758        let report = reconcile(&mut index, &ScanConfig::default(), &mut |commit| {
8759            commits.push(commit.clone());
8760        })
8761        .expect("reconcile");
8762
8763        assert!(report.is_complete());
8764        assert_eq!(report.apply.unchanged, 5, "3 files + 2 dirs all already known");
8765        assert_eq!(commits.len(), 2);
8766        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
8767        assert_eq!(
8768            index.clock(),
8769            crate::Clock(before_clock.0 + 2),
8770            "start and finish are state commits"
8771        );
8772        let commits = index.since(before_clock).commits;
8773        assert_eq!(commits.len(), 2);
8774        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
8775        assert_eq!(index.total(), before_total);
8776    }
8777
8778    #[test]
8779    fn parallel_and_serial_reconciliation_produce_the_same_index() {
8780        let dir = sample_tree();
8781        let portable_config =
8782            ScanConfig { threads: Some(1), batch_size: 2, ..ScanConfig::default() };
8783        let (mut portable, _) =
8784            scan_into_index(dir.path(), &portable_config).expect("portable baseline");
8785        let mut bulk = portable.clone();
8786
8787        fs::remove_file(dir.path().join("a.txt")).expect("remove file");
8788        write_file(&dir.path().join("src/main.rs"), b"fn main() { much longer }");
8789        write_file(&dir.path().join("src/added.md"), b"new file");
8790
8791        let portable_report = reconcile(&mut portable, &portable_config, &mut |_| {})
8792            .expect("portable reconciliation");
8793        let bulk_config = ScanConfig { threads: Some(2), ..portable_config };
8794        let bulk_report =
8795            reconcile(&mut bulk, &bulk_config, &mut |_| {}).expect("bulk reconciliation");
8796
8797        assert!(portable_report.is_complete());
8798        assert!(bulk_report.is_complete());
8799        assert_eq!(bulk_report.scan.attribution, WalkAttribution::default());
8800        assert_eq!(bulk_report.scan.entries, portable_report.scan.entries);
8801        assert_eq!(bulk_report.scan.dirs_read, portable_report.scan.dirs_read);
8802        assert_eq!(bulk_report.apply, portable_report.apply);
8803        assert_eq!(index_fingerprint(&bulk), index_fingerprint(&portable));
8804        assert_eq!(bulk.total(), portable.total());
8805    }
8806
8807    #[test]
8808    fn parallel_reconciliation_workers_publish_directory_counters() {
8809        let _serial = crate::counters::test_serial();
8810        let dir = sample_tree();
8811        let baseline = ScanConfig { threads: Some(1), ..ScanConfig::default() };
8812        let (mut index, scan) = scan_into_index(dir.path(), &baseline).expect("baseline scan");
8813        assert!(scan.is_complete());
8814
8815        crate::counters::enable(true);
8816        crate::counters::reset();
8817        let config = ScanConfig { threads: Some(4), ..baseline };
8818        let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconciliation");
8819        crate::counters::flush_thread();
8820        let counts = crate::counters::snapshot();
8821        crate::counters::reset();
8822        crate::counters::enable(false);
8823
8824        assert!(report.is_complete());
8825        assert!(
8826            counts.dir_opens >= report.scan.dirs_read,
8827            "parallel worker directory opens were folded: {counts:?}, report={report:?}"
8828        );
8829    }
8830
8831    #[test]
8832    fn parallel_reconciliation_matches_serial_across_structural_transitions() {
8833        for max_depth in [None, Some(1), Some(2)] {
8834            for order in [ScanOrder::BreadthFirst, ScanOrder::DepthFirst] {
8835                let dir = reconciliation_transition_tree();
8836                let reference_config = ScanConfig {
8837                    order,
8838                    max_depth,
8839                    threads: Some(1),
8840                    batch_size: 2,
8841                    ..ScanConfig::default()
8842                };
8843                let (baseline, baseline_report) =
8844                    scan_into_index(dir.path(), &reference_config).expect("baseline scan");
8845                assert!(baseline_report.is_complete());
8846                mutate_reconciliation_transition_tree(dir.path());
8847
8848                let mut serial = baseline.clone();
8849                let mut serial_commits = Vec::new();
8850                let serial_report = reconcile(&mut serial, &reference_config, &mut |commit| {
8851                    serial_commits.push(commit.clone());
8852                })
8853                .expect("serial reconciliation");
8854                let (fresh, fresh_report) =
8855                    scan_into_index(dir.path(), &reference_config).expect("fresh oracle");
8856                assert!(serial_report.is_complete(), "serial {order:?}/{max_depth:?}");
8857                assert!(fresh_report.is_complete(), "fresh {order:?}/{max_depth:?}");
8858                assert_eq!(
8859                    index_fingerprint(&serial),
8860                    index_fingerprint(&fresh),
8861                    "serial did not converge to a fresh scan for {order:?}/{max_depth:?}"
8862                );
8863
8864                for workers in [2, 4] {
8865                    let mut parallel = baseline.clone();
8866                    let config = ScanConfig { threads: Some(workers), ..reference_config.clone() };
8867                    let mut parallel_commits = Vec::new();
8868                    let report = reconcile(&mut parallel, &config, &mut |commit| {
8869                        parallel_commits.push(commit.clone());
8870                    })
8871                    .expect("parallel reconciliation");
8872                    let context = format!("{order:?}/{max_depth:?}/{workers} workers");
8873
8874                    assert!(report.is_complete(), "{context}: unexpected partial report");
8875                    assert_eq!(report.scan.entries, serial_report.scan.entries, "{context}");
8876                    assert_eq!(report.scan.dirs_read, serial_report.scan.dirs_read, "{context}");
8877                    assert_eq!(report.apply, serial_report.apply, "{context}");
8878                    assert_eq!(
8879                        effective_ops(&parallel_commits),
8880                        effective_ops(&serial_commits),
8881                        "{context}: effective delta differs"
8882                    );
8883                    assert_eq!(
8884                        index_fingerprint(&parallel),
8885                        index_fingerprint(&serial),
8886                        "{context}: final index differs"
8887                    );
8888                    let (parallel_total, serial_total) = (parallel.total(), serial.total());
8889                    assert_eq!(
8890                        (
8891                            parallel_total.files,
8892                            parallel_total.dirs,
8893                            parallel_total.bytes,
8894                            parallel_total.allocated,
8895                            parallel_total.newest_mtime_ns,
8896                        ),
8897                        (
8898                            serial_total.files,
8899                            serial_total.dirs,
8900                            serial_total.bytes,
8901                            serial_total.allocated,
8902                            serial_total.newest_mtime_ns,
8903                        ),
8904                        "{context}: roll-up differs"
8905                    );
8906                    assert_eq!(
8907                        parallel_total.by_ext, serial_total.by_ext,
8908                        "{context}: extension roll-up differs"
8909                    );
8910                }
8911            }
8912        }
8913    }
8914
8915    #[cfg(unix)]
8916    #[test]
8917    fn revalidation_metadata_errors_do_not_delete_enumerated_entries() {
8918        use std::os::unix::fs::PermissionsExt;
8919
8920        if !crate::test_support::require_permission_bits() {
8921            return;
8922        }
8923
8924        let dir = sample_tree();
8925        let config = ScanConfig::default();
8926        let (mut index, baseline_report) =
8927            scan_into_index(dir.path(), &config).expect("baseline scan");
8928        assert!(baseline_report.is_complete());
8929        let before = index_fingerprint(&index);
8930        let original_permissions = fs::metadata(dir.path()).expect("root metadata").permissions();
8931
8932        fs::set_permissions(dir.path(), fs::Permissions::from_mode(0o400))
8933            .expect("remove search permission");
8934        let mut observations = Vec::new();
8935        let outcome = revalidate(&index, &config, &mut |observation| {
8936            observations.push(observation);
8937        });
8938        fs::set_permissions(dir.path(), original_permissions).expect("restore permissions");
8939
8940        let report = outcome.expect("operational metadata errors are a partial report");
8941        assert!(!report.errors.is_empty(), "the fixture did not induce metadata errors");
8942        for observation in &observations {
8943            index.apply_ok(observation);
8944        }
8945        assert_eq!(index_fingerprint(&index), before);
8946        assert!(index.attrs(Path::new("a.txt")).is_some(), "existing entry was removed");
8947    }
8948
8949    #[cfg(unix)]
8950    #[test]
8951    fn reconciliation_metadata_errors_drop_unverified_entries_like_a_cold_scan() {
8952        use std::os::unix::fs::PermissionsExt;
8953
8954        if !crate::test_support::require_permission_bits() {
8955            return;
8956        }
8957
8958        // macOS's parallel path may satisfy the whole directory through
8959        // getattrlistbulk even without search permission. One worker pins the portable
8960        // fallback there; Linux also exercises the parallel portable worker.
8961        let worker_counts = if cfg!(target_os = "macos") { vec![1] } else { vec![1, 2] };
8962        for workers in worker_counts {
8963            let dir = sample_tree();
8964            let config =
8965                ScanConfig { threads: Some(workers), batch_size: 2, ..ScanConfig::default() };
8966            let (mut index, baseline_report) =
8967                scan_into_index(dir.path(), &config).expect("baseline scan");
8968            assert!(baseline_report.is_complete());
8969            let before = index_fingerprint(&index);
8970            let original_permissions =
8971                fs::metadata(dir.path()).expect("root metadata").permissions();
8972
8973            // Reading names requires read permission; looking up their metadata also
8974            // requires search permission. This makes enumeration succeed and each
8975            // metadata lookup fail, the boundary where an encountered name used to be
8976            // misclassified as a deletion.
8977            fs::set_permissions(dir.path(), fs::Permissions::from_mode(0o400))
8978                .expect("remove search permission");
8979            let outcome = reconcile(&mut index, &config, &mut |_| {});
8980            let cold = scan_into_index(dir.path(), &config);
8981            fs::set_permissions(dir.path(), original_permissions).expect("restore permissions");
8982
8983            let report = outcome.expect("operational metadata errors are a partial report");
8984            let (cold, cold_report) = cold.expect("cold partial scan");
8985            assert!(!report.scan.errors.is_empty(), "the fixture did not induce metadata errors");
8986            assert!(!report.is_complete());
8987            assert!(!cold_report.is_complete());
8988            assert_eq!(index_fingerprint(&index), index_fingerprint(&cold));
8989            assert!(
8990                index_fingerprint(&index).is_empty(),
8991                "neither warm nor cold may retain attributes it could not verify"
8992            );
8993            assert_eq!(index.directory_complete(Path::new("")), Some(false));
8994            assert_eq!(cold.directory_complete(Path::new("")), Some(false));
8995            assert!(!before.is_empty(), "the fixture began with retained facts");
8996        }
8997    }
8998
8999    #[cfg(unix)]
9000    #[test]
9001    fn failed_listing_withdraws_retained_completeness_and_recovers() {
9002        use std::os::unix::fs::PermissionsExt;
9003
9004        if !crate::test_support::require_permission_bits() {
9005            return;
9006        }
9007        for workers in [1, 2] {
9008            let dir = tempfile::tempdir().expect("root");
9009            write_file(&dir.path().join("ancestor/blocked/unknown.txt"), b"unknown");
9010            write_file(&dir.path().join("healthy/known.txt"), b"known");
9011            let config = ScanConfig { threads: Some(workers), ..ScanConfig::default() };
9012            let (mut warm, baseline) = scan_into_index(dir.path(), &config).expect("baseline");
9013            assert!(baseline.is_complete());
9014            let blocked = dir.path().join("ancestor/blocked");
9015            fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny listing");
9016            let probe = fs::read_dir(&blocked);
9017            let mut commits = Vec::new();
9018            let refreshed =
9019                reconcile(&mut warm, &config, &mut |commit| commits.push(commit.clone()));
9020            let cold = scan_into_index(dir.path(), &config);
9021            fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore");
9022            assert_eq!(
9023                probe.expect_err("real denied listing").kind(),
9024                std::io::ErrorKind::PermissionDenied
9025            );
9026            assert!(!refreshed.expect("partial refresh").is_complete());
9027            let (cold, report) = cold.expect("partial cold scan");
9028            assert!(!report.is_complete());
9029            for index in [&warm, &cold] {
9030                for path in ["", "ancestor", "healthy"] {
9031                    assert_eq!(
9032                        index.directory_complete(Path::new(path)),
9033                        Some(true),
9034                        "workers={workers}: {path}"
9035                    );
9036                }
9037                assert_eq!(index.directory_complete(Path::new("ancestor/blocked")), Some(false));
9038                assert_eq!(index.freshness_at(Path::new("ancestor")), crate::Freshness::Partial);
9039            }
9040            assert!(commits.iter().flat_map(|commit| &commit.state).any(|state| matches!(state,
9041                crate::StateTransition::IndexState { current, .. } if current.coverage != crate::Coverage::Complete
9042            )), "failure is published");
9043            assert!(reconcile(&mut warm, &config, &mut |_| {}).expect("recovery").is_complete());
9044            assert_eq!(warm.directory_complete(Path::new("ancestor/blocked")), Some(true));
9045            assert_eq!(warm.state().coverage, crate::Coverage::Complete);
9046        }
9047    }
9048
9049    #[test]
9050    fn unreadable_directory_warm_answer_matches_cold_verified_tree() {
9051        for workers in [1, 2] {
9052            let dir = tempfile::tempdir().expect("tempdir");
9053            write_file(&dir.path().join("blocked/old.txt"), b"old");
9054            write_file(&dir.path().join("verified.txt"), b"verified");
9055            let config = ScanConfig { threads: Some(workers), ..ScanConfig::default() };
9056            let (mut warm, baseline) = scan_into_index(dir.path(), &config).expect("baseline");
9057            assert!(baseline.is_complete());
9058
9059            let blocked = dir.path().join("blocked").canonicalize().expect("blocked path");
9060            let hook = install_child_metadata_hook(dir.path(), move |path| {
9061                (path == blocked)
9062                    .then(|| std::io::Error::from(std::io::ErrorKind::PermissionDenied))
9063            });
9064            let warm_report = reconcile(&mut warm, &config, &mut |_| {}).expect("warm partial");
9065            let (cold, cold_report) = scan_into_index(dir.path(), &config).expect("cold partial");
9066            drop(hook);
9067
9068            assert!(!warm_report.is_complete(), "workers={workers}");
9069            assert!(!cold_report.is_complete(), "workers={workers}");
9070            assert!(warm.lookup(Path::new("blocked")).is_none(), "workers={workers}");
9071            assert!(warm.lookup(Path::new("blocked/old.txt")).is_none(), "workers={workers}");
9072            assert_eq!(index_fingerprint(&warm), index_fingerprint(&cold), "workers={workers}");
9073            assert!(warm.lookup(Path::new("verified.txt")).is_some(), "workers={workers}");
9074        }
9075    }
9076
9077    #[test]
9078    fn deferred_change_overflow_retries_without_applying_a_partial_wave() {
9079        let dir = sample_tree();
9080        let config = ScanConfig { threads: Some(2), batch_size: 2, ..ScanConfig::default() };
9081        let (mut index, _) = scan_into_index(dir.path(), &config).expect("baseline");
9082        let before = index_fingerprint(&index);
9083
9084        fs::remove_file(dir.path().join("a.txt")).expect("remove file");
9085        write_file(&dir.path().join("added.md"), b"new file");
9086        write_file(&dir.path().join("src/main.rs"), b"fn main() { much longer }");
9087
9088        let root = index.root_path().to_path_buf();
9089        let root_meta = {
9090            crate::counters::bump(|c| c.stats += 1);
9091            fs::symlink_metadata(&root)
9092        }
9093        .expect("root metadata");
9094        let mut commits = Vec::new();
9095        let outcome = reconcile_direct_parallel(
9096            &mut index,
9097            &root,
9098            root_device(&root, &root_meta).expect("root device"),
9099            &config,
9100            1,
9101            &mut |commit| commits.push(commit.clone()),
9102        )
9103        .expect("parallel attempt");
9104        let DirectParallelOutcome::RetrySerial { prefix, remaining } = outcome else {
9105            panic!("the deliberately tiny deferred budget must trigger the retry");
9106        };
9107
9108        assert_eq!(prefix.apply, ApplyStats::default());
9109        assert_eq!(remaining, VecDeque::from([(PathBuf::new(), 0)]));
9110        assert!(commits.is_empty());
9111        assert_eq!(index_fingerprint(&index), before);
9112
9113        let serial = ScanConfig { threads: Some(1), ..config };
9114        let report = reconcile(&mut index, &serial, &mut |_| {}).expect("serial retry");
9115        let (expected, expected_report) = scan_into_index(dir.path(), &serial).expect("oracle");
9116        assert!(report.is_complete());
9117        assert!(expected_report.is_complete());
9118        assert_eq!(index_fingerprint(&index), index_fingerprint(&expected));
9119        assert_eq!(index.total().files, expected.total().files);
9120        assert_eq!(index.total().dirs, expected.total().dirs);
9121        assert_eq!(index.total().bytes, expected.total().bytes);
9122        assert_eq!(index.total().allocated, expected.total().allocated);
9123        assert_eq!(index.total().newest_mtime_ns, expected.total().newest_mtime_ns);
9124        assert_eq!(index.total().by_ext, expected.total().by_ext);
9125    }
9126
9127    #[test]
9128    fn late_overflow_resumes_without_double_counting_completed_waves() {
9129        let dir = tempfile::tempdir().expect("tempdir");
9130        // The root wave discovers more than one full wave of directories. A change in
9131        // the second wave then forces the serial fallback only after the first wave's
9132        // unchanged entries have already been counted.
9133        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
9134            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), b"unchanged");
9135        }
9136        let parallel = ScanConfig { threads: Some(2), ..ScanConfig::default() };
9137        let (baseline, _) = scan_into_index(dir.path(), &parallel).expect("baseline");
9138        let mut candidate = baseline.clone();
9139        let mut serial_oracle = baseline;
9140
9141        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
9142            write_file(
9143                &dir.path().join(format!("d{directory:04}/file.txt")),
9144                b"changed after the first wave",
9145            );
9146        }
9147
9148        let candidate_report = reconcile_target_inner(
9149            &mut ReconcileTarget::Direct(&mut candidate),
9150            Path::new(""),
9151            0,
9152            &parallel,
9153            0,
9154            &mut |_| {},
9155        )
9156        .expect("late-overflow reconciliation");
9157        let serial = ScanConfig { threads: Some(1), ..parallel };
9158        let oracle_report =
9159            reconcile(&mut serial_oracle, &serial, &mut |_| {}).expect("serial oracle");
9160
9161        assert_eq!(candidate_report.apply, oracle_report.apply);
9162        assert_eq!(candidate_report.scan.entries, oracle_report.scan.entries);
9163        assert_eq!(candidate_report.scan.dirs_read, oracle_report.scan.dirs_read);
9164        assert_eq!(index_fingerprint(&candidate), index_fingerprint(&serial_oracle));
9165    }
9166
9167    /// The four counts a walk reports, in the order [`crate::ProgressSnapshot`] shows them.
9168    fn walked(report: &ScanReport) -> (u64, u64, u64, u64) {
9169        (report.dirs_read, report.files_walked, report.bytes_walked, report.allocated_walked)
9170    }
9171
9172    fn reported(progress: &crate::Progress) -> (u64, u64, u64, u64) {
9173        let snapshot = progress.snapshot();
9174        (snapshot.directories, snapshot.files, snapshot.bytes, snapshot.allocated)
9175    }
9176
9177    /// Each walker is a separate loop with its own reporting sites, so each is checked:
9178    /// the detached cold walk, the streaming walk, the transient summary fold, the
9179    /// reference revalidation, exclusive reconciliation serial and in parallel waves,
9180    /// and shared-handle reconciliation. The tree is wide enough that every parallel
9181    /// walker claims several chunks and the small batch size fills several batches, so a
9182    /// walker that reported only its final state would still fail on the counts a
9183    /// mid-walk chunk added twice or not at all.
9184    #[test]
9185    fn every_walker_reports_exactly_what_its_report_counts() {
9186        let dir = tempfile::tempdir().expect("tempdir");
9187        for directory in 0..12 {
9188            for file in 0..5 {
9189                write_file(
9190                    &dir.path().join(format!("d{directory}/f{file}.txt")),
9191                    &vec![b'x'; directory * 5 + file + 1],
9192                );
9193            }
9194        }
9195        for threads in [1, 4] {
9196            let context = format!("threads={threads}");
9197            let progress = crate::Progress::new();
9198            let cold_config = ScanConfig {
9199                threads: Some(threads),
9200                batch_size: 4,
9201                progress: Some(progress.clone()),
9202                ..ScanConfig::default()
9203            };
9204            let (mut index, cold) = scan_into_index(dir.path(), &cold_config).expect("cold scan");
9205            assert_eq!(cold.dirs_read, 13, "{context}: the root and twelve children");
9206            assert_eq!(
9207                progress.snapshot().phase,
9208                crate::ProgressPhase::Indexing,
9209                "{context}: the detached walk ends by assembling the index"
9210            );
9211            assert_eq!(reported(&progress), walked(&cold), "{context}: detached cold walk");
9212
9213            let progress = crate::Progress::new();
9214            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
9215            let streamed = scan(dir.path(), &config, &mut |_| {}).expect("streaming scan");
9216            assert_eq!(walked(&streamed), walked(&cold), "{context}");
9217            assert_eq!(reported(&progress), walked(&streamed), "{context}: streaming walk");
9218
9219            let progress = crate::Progress::new();
9220            let config = ScanConfig {
9221                read_controls: false,
9222                progress: Some(progress.clone()),
9223                ..cold_config.clone()
9224            };
9225            let folded = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
9226            assert_eq!(walked(&folded), walked(&cold), "{context}");
9227            assert_eq!(reported(&progress), walked(&folded), "{context}: summary fold");
9228
9229            let progress = crate::Progress::new();
9230            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
9231            let revalidated = revalidate(&index, &config, &mut |_| {}).expect("revalidate");
9232            assert_eq!(progress.snapshot().phase, crate::ProgressPhase::Revalidating, "{context}");
9233            assert_eq!(reported(&progress), walked(&revalidated), "{context}: revalidate");
9234
9235            // Changes, so reconciliation defers and applies operations rather than
9236            // discarding every entry as unchanged.
9237            write_file(&dir.path().join(format!("d0/new{threads}.txt")), b"added");
9238            fs::remove_file(dir.path().join(format!("d1/f{}.txt", threads - 1))).expect("remove");
9239            write_file(&dir.path().join("d2/f0.txt"), &vec![b'y'; 40 + threads]);
9240            // An independent walk of the changed tree. The handle and the reconcile's own
9241            // report both come from the walker's counts, so agreeing with each other
9242            // would not show that the walker counted anything; agreeing with this does.
9243            let (_, fresh) = scan_into_index(dir.path(), &ScanConfig::default()).expect("fresh");
9244            let progress = crate::Progress::new();
9245            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
9246            let reconciled = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
9247            assert!(reconciled.apply.mutated(), "{context}: the changes were applied");
9248            assert_eq!(progress.snapshot().phase, crate::ProgressPhase::Revalidating, "{context}");
9249            assert_eq!(reported(&progress), walked(&reconciled.scan), "{context}: reconcile");
9250            assert_eq!(walked(&reconciled.scan), walked(&fresh), "{context}: the whole tree");
9251
9252            let handle = crate::IndexHandle::new(index);
9253            let progress = crate::Progress::new();
9254            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config };
9255            let shared = reconcile_handle(&handle, &config, &mut |_| {}).expect("shared");
9256            assert_eq!(reported(&progress), walked(&shared.scan), "{context}: shared handle");
9257        }
9258    }
9259
9260    /// A wave that overflows its deferred-operation budget is thrown away and rewalked
9261    /// serially. The report counts each directory once, as the logical pass does, and
9262    /// progress counts the wave's reads both times, as the filesystem did them: work
9263    /// done, not the answer. This is the one walker relation that is not equality, and
9264    /// the difference is exactly the rewalked wave.
9265    #[test]
9266    fn progress_counts_a_rewalked_wave_twice_where_the_report_counts_it_once() {
9267        let dir = tempfile::tempdir().expect("tempdir");
9268        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
9269            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), b"unchanged");
9270        }
9271        // A file in the wave that completes, so the serial rewalk has counts of that wave
9272        // to carry forward as already added rather than add again.
9273        write_file(&dir.path().join("root.txt"), b"counted by the wave that completes");
9274        let parallel = ScanConfig { threads: Some(2), ..ScanConfig::default() };
9275        let (mut index, _) = scan_into_index(dir.path(), &parallel).expect("baseline");
9276        // Larger than any filesystem stores inline in the inode, so each copy occupies
9277        // blocks of its own wherever the test runs.
9278        let changed = &vec![b'c'; 8_193];
9279        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
9280            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), changed);
9281        }
9282
9283        let progress = crate::Progress::new();
9284        let observed = ScanConfig { progress: Some(progress.clone()), ..parallel };
9285        let report = reconcile_target_inner(
9286            &mut ReconcileTarget::Direct(&mut index),
9287            Path::new(""),
9288            0,
9289            &observed,
9290            0,
9291            &mut |_| {},
9292        )
9293        .expect("late-overflow reconciliation");
9294
9295        // The root wave changes nothing and completes; the second wave holds exactly
9296        // one full wave of changed directories, overflows, and is rewalked with the one
9297        // directory the wave left behind.
9298        let rewalked = u64::try_from(RECONCILE_WAVE_DIRECTORIES).expect("fits");
9299        let snapshot = progress.snapshot();
9300        assert_eq!(snapshot.directories, report.scan.dirs_read + rewalked);
9301        assert_eq!(snapshot.files, report.scan.files_walked + rewalked);
9302        assert_eq!(
9303            snapshot.bytes,
9304            report.scan.bytes_walked + rewalked * u64::try_from(changed.len()).expect("fits")
9305        );
9306        let allocated = index.attrs(Path::new("d0000/file.txt")).expect("indexed").allocated;
9307        assert!(allocated > 0, "a file with content occupies blocks");
9308        assert_eq!(snapshot.allocated, report.scan.allocated_walked + rewalked * allocated);
9309    }
9310
9311    /// Several invalidated roots are reconciled one at a time and their reports summed,
9312    /// so every walked count has to survive the sum, allocated bytes included.
9313    #[test]
9314    fn a_multi_root_reconcile_sums_every_walked_count() {
9315        let dir = tempfile::tempdir().expect("tempdir");
9316        for directory in ["a", "b", "c"] {
9317            for file in 0..3 {
9318                write_file(&dir.path().join(format!("{directory}/f{file}.txt")), b"before");
9319            }
9320        }
9321        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9322        for directory in ["a", "b"] {
9323            write_file(&dir.path().join(format!("{directory}/f0.txt")), &vec![b'x'; 5_000]);
9324        }
9325        index.apply_ok(&Observation::new(
9326            ["a", "b"]
9327                .into_iter()
9328                .map(|directory| Op::InvalidateSubtree {
9329                    path: PathBuf::from(directory),
9330                    reason: crate::InvalidateReason::Requested,
9331                })
9332                .collect(),
9333        ));
9334        let report =
9335            reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
9336
9337        let attrs: Vec<Attrs> = ["a", "b"]
9338            .into_iter()
9339            .flat_map(|directory| (0..3).map(move |file| format!("{directory}/f{file}.txt")))
9340            .map(|path| *index.attrs(Path::new(&path)).expect("indexed"))
9341            .collect();
9342        assert_eq!(report.scan.files_walked, 6, "the two roots' files, and not c's");
9343        assert_eq!(report.scan.bytes_walked, attrs.iter().map(|attrs| attrs.size).sum::<u64>());
9344        assert_eq!(
9345            report.scan.allocated_walked,
9346            attrs.iter().map(|attrs| attrs.allocated).sum::<u64>()
9347        );
9348    }
9349
9350    #[test]
9351    fn shared_reconciliation_retains_conditional_no_op_arbitration() {
9352        let dir = sample_tree();
9353        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9354        let handle = crate::IndexHandle::new(index);
9355        let before_clock = handle.clock().expect("clock");
9356        let mut commits = Vec::new();
9357
9358        let report = reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
9359            commits.push(commit.clone());
9360        })
9361        .expect("reconcile");
9362
9363        assert!(report.is_complete());
9364        assert_eq!(report.apply.unchanged, 5, "3 files + 2 dirs all already known");
9365        assert_eq!(commits.len(), 2);
9366        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
9367        assert_eq!(
9368            handle.clock().expect("clock"),
9369            crate::Clock(before_clock.0 + 2),
9370            "start and finish are state commits"
9371        );
9372        let commits = handle.since(before_clock).expect("state commits").commits;
9373        assert_eq!(commits.len(), 2);
9374        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
9375    }
9376
9377    #[test]
9378    fn revalidate_detects_additions_edits_and_deletions() {
9379        let dir = sample_tree();
9380        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9381
9382        fs::remove_file(dir.path().join("a.txt")).expect("remove");
9383        write_file(&dir.path().join("src/main.rs"), b"fn main() { longer }");
9384        write_file(&dir.path().join("added.md"), b"new");
9385
9386        let mut deltas = Vec::new();
9387        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
9388        let mut stats = crate::index::ApplyStats::default();
9389        for delta in &deltas {
9390            let s = index.apply_ok(delta);
9391            stats.inserted += s.inserted;
9392            stats.updated += s.updated;
9393            stats.removed += s.removed;
9394        }
9395
9396        assert_eq!(stats.inserted, 1, "added.md");
9397        assert_eq!(stats.updated, 1, "main.rs grew");
9398        assert_eq!(stats.removed, 1, "a.txt is gone");
9399
9400        let total = index.total();
9401        assert_eq!(total.files, 3);
9402        assert_eq!(total.bytes, 20 + 9 + 3);
9403        assert!(!total.by_ext.contains_key(".txt"));
9404        assert_eq!(total.by_ext[".md"].files, 1);
9405    }
9406
9407    #[test]
9408    fn revalidate_removes_a_whole_vanished_directory() {
9409        let dir = sample_tree();
9410        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9411        fs::remove_dir_all(dir.path().join("src")).expect("remove dir");
9412
9413        let mut deltas = Vec::new();
9414        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
9415        for delta in &deltas {
9416            index.apply_ok(delta);
9417        }
9418
9419        let total = index.total();
9420        assert_eq!(total.files, 1);
9421        assert_eq!(total.dirs, 0);
9422        assert!(index.lookup(Path::new("src")).is_none());
9423    }
9424
9425    #[test]
9426    fn pending_invalidation_reconciles_the_requested_subtree() {
9427        let dir = sample_tree();
9428        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9429        write_file(&dir.path().join("src/added.rs"), b"new");
9430        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
9431            path: PathBuf::from("src"),
9432            reason: crate::InvalidateReason::Requested,
9433        }]));
9434        assert_eq!(index.freshness_at(Path::new("src")), crate::Freshness::Stale);
9435
9436        let mut applied = Vec::new();
9437        let report = reconcile_pending(&mut index, &ScanConfig::default(), &mut |delta| {
9438            applied.push(delta.clone());
9439        })
9440        .expect("reconcile pending");
9441
9442        assert!(report.is_complete());
9443        assert!(index.lookup(Path::new("src/added.rs")).is_some());
9444        assert_eq!(index.freshness_at(Path::new("src")), crate::Freshness::Fresh);
9445        assert!(index.take_pending_invalidations().is_empty());
9446        assert!(applied.iter().any(|commit| commit_touches(commit, Path::new("src/added.rs"))));
9447    }
9448
9449    /// A retained `.gitignore` reconciled as the root of its own walk re-reads its rules.
9450    /// A file does not descend, so the subtree-root branch was the only place that could
9451    /// read them, and it did not: the table kept `*.log` while the pass reported complete.
9452    #[test]
9453    fn reconciling_a_retained_control_file_as_the_subtree_root_rereads_its_rules() {
9454        let dir = tempfile::tempdir().expect("tempdir");
9455        write_file(&dir.path().join(".gitignore"), b"*.log\n");
9456        write_file(&dir.path().join("a.log"), b"log");
9457        let config = ScanConfig { read_controls: true, ..ScanConfig::default() };
9458        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
9459        assert_eq!(
9460            index.is_ignored(Path::new("a.log")).expect("control state observed"),
9461            Some(true)
9462        );
9463
9464        write_file(&dir.path().join(".gitignore"), b"# nothing is ignored now\n");
9465        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
9466            path: PathBuf::from(".gitignore"),
9467            reason: crate::InvalidateReason::Requested,
9468        }]));
9469        let report = reconcile_pending(&mut index, &config, &mut |_| {}).expect("reconcile");
9470
9471        assert!(report.is_complete(), "{:?}", report.scan.errors);
9472        assert_eq!(index.freshness_at(Path::new(".gitignore")), crate::Freshness::Fresh);
9473        assert_eq!(
9474            index.is_ignored(Path::new("a.log")).expect("control state observed"),
9475            Some(false)
9476        );
9477    }
9478
9479    /// A control file the pass cannot verify contributes neither stale rules nor an entry.
9480    #[cfg(unix)]
9481    #[test]
9482    fn reconciling_an_unreadable_control_file_root_drops_its_rules_and_stays_partial() {
9483        use std::os::unix::fs::PermissionsExt;
9484
9485        if !crate::test_support::require_permission_bits() {
9486            return;
9487        }
9488
9489        let dir = tempfile::tempdir().expect("tempdir");
9490        let control = dir.path().join(".gitignore");
9491        write_file(&control, b"*.log\n");
9492        write_file(&dir.path().join("a.log"), b"log");
9493        let config = ScanConfig { read_controls: true, ..ScanConfig::default() };
9494        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
9495
9496        write_file(&control, b"# rewritten, then made unreadable\n");
9497        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("chmod");
9498        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
9499            path: PathBuf::from(".gitignore"),
9500            reason: crate::InvalidateReason::Requested,
9501        }]));
9502        let report = reconcile_pending(&mut index, &config, &mut |_| {});
9503        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore");
9504        let report = report.expect("reconcile");
9505
9506        assert!(!report.is_complete());
9507        assert_eq!(report.scan.errors.len(), 1, "{:?}", report.scan.errors);
9508        assert_eq!(index.freshness_at(Path::new(".gitignore")), crate::Freshness::Partial);
9509        assert!(
9510            !index.controls().expect("control state observed").contains(Path::new(".gitignore"))
9511        );
9512        assert_eq!(
9513            index.is_ignored(Path::new("a.log")).expect("control state observed"),
9514            Some(false)
9515        );
9516    }
9517
9518    #[test]
9519    fn handle_reconciliation_publishes_after_each_delta_is_applied() {
9520        let dir = sample_tree();
9521        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9522        let handle = crate::IndexHandle::new(index);
9523        let reader = handle.clone();
9524        write_file(&dir.path().join("added.md"), b"new");
9525
9526        let mut observed_after_apply = false;
9527        reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
9528            if commit_touches(commit, Path::new("added.md")) {
9529                observed_after_apply =
9530                    reader.kind(Path::new("added.md")).expect("query index").is_some();
9531            }
9532        })
9533        .expect("reconcile handle");
9534
9535        assert!(observed_after_apply);
9536    }
9537
9538    /// Delete `name` under `root` from inside its own metadata lookup, after the listing
9539    /// returned it, on whichever thread performs the lookup.
9540    fn delete_between_listing_and_stat(root: &Path, name: &'static str) -> WalkHookGuard {
9541        install_child_metadata_hook(root, move |path| {
9542            delete_if_named(path, name);
9543            None
9544        })
9545    }
9546
9547    /// As [`delete_between_listing_and_stat`], and every reconciliation listing under `root`
9548    /// also ends in an error, so none of them is complete.
9549    fn delete_between_listing_and_stat_in_a_failing_listing(
9550        root: &Path,
9551        name: &'static str,
9552    ) -> WalkHookGuard {
9553        install_walk_hook(root, move |point| match point {
9554            WalkHookPoint::ChildMetadata(path) => {
9555                delete_if_named(path, name);
9556                None
9557            }
9558            WalkHookPoint::ListingEnd => Some(std::io::Error::other("injected listing error")),
9559        })
9560    }
9561
9562    fn delete_if_named(path: &Path, name: &str) {
9563        if path.file_name() == Some(OsStr::new(name)) {
9564            fs::remove_file(path).expect("delete between listing and stat");
9565        }
9566    }
9567
9568    /// A name the listing returned that is gone by the time it is stat'd was deleted, on
9569    /// the serial path and on the parallel waves a full-root pass takes by default. The
9570    /// walk removes it; recorded as an error, it would settle as a phantom entry with
9571    /// permanent partial freshness.
9572    #[test]
9573    fn a_child_deleted_between_listing_and_stat_is_removed_rather_than_reported() {
9574        for threads in [1, 4] {
9575            let dir = tempfile::tempdir().expect("tempdir");
9576            write_file(&dir.path().join("keep.txt"), b"keep");
9577            write_file(&dir.path().join("gone.txt"), b"gone");
9578            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
9579            let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
9580            assert!(index.lookup(Path::new("gone.txt")).is_some());
9581
9582            index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
9583                path: PathBuf::new(),
9584                reason: crate::InvalidateReason::Requested,
9585            }]));
9586            let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
9587            let report = reconcile_pending(&mut index, &config, &mut |_| {});
9588            drop(hook);
9589            let report = report.expect("reconcile");
9590
9591            assert!(report.is_complete(), "threads {threads}: {:?}", report.scan.errors);
9592            assert_eq!(report.apply.removed, 1, "threads {threads}");
9593            assert!(index.lookup(Path::new("gone.txt")).is_none(), "threads {threads}");
9594            assert!(index.lookup(Path::new("keep.txt")).is_some(), "threads {threads}");
9595            assert_eq!(index.freshness(), crate::Freshness::Fresh, "threads {threads}");
9596        }
9597    }
9598
9599    /// A vanished control file takes its rules with it on both reconcile paths, even when the
9600    /// rest of its listing fails: the stat's `NotFound` is the evidence. A retained file's
9601    /// rules go with its entry's removal; a hidden-pruned one has no entry to remove, so
9602    /// without its own removal its rules would go on ignoring its siblings.
9603    #[test]
9604    fn a_control_file_deleted_between_listing_and_stat_takes_its_rules_with_it() {
9605        for prune_hidden in [false, true] {
9606            for (threads, listing_fails) in [(1, false), (4, false), (1, true), (4, true)] {
9607                let case = format!(
9608                    "prune hidden {prune_hidden}, threads {threads}, listing fails {listing_fails}"
9609                );
9610                let dir = tempfile::tempdir().expect("tempdir");
9611                write_file(&dir.path().join(".gitignore"), b"*.log\n");
9612                write_file(&dir.path().join("a.log"), b"log");
9613                let config = ScanConfig {
9614                    read_controls: true,
9615                    threads: Some(threads),
9616                    hidden: prune_hidden
9617                        .then(|| std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([""; 0]))),
9618                    ..ScanConfig::default()
9619                };
9620                let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
9621                assert_eq!(
9622                    index.is_ignored(Path::new("a.log")).expect("control state observed"),
9623                    Some(true),
9624                    "{case}"
9625                );
9626
9627                index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
9628                    path: PathBuf::new(),
9629                    reason: crate::InvalidateReason::Requested,
9630                }]));
9631                let hook = if listing_fails {
9632                    delete_between_listing_and_stat_in_a_failing_listing(dir.path(), ".gitignore")
9633                } else {
9634                    delete_between_listing_and_stat(dir.path(), ".gitignore")
9635                };
9636                let report = reconcile_pending(&mut index, &config, &mut |_| {});
9637                drop(hook);
9638                let report = report.expect("reconcile");
9639
9640                assert_eq!(
9641                    report.is_complete(),
9642                    !listing_fails,
9643                    "{case}: {:?}",
9644                    report.scan.errors
9645                );
9646                assert!(index.lookup(Path::new(".gitignore")).is_none(), "{case}");
9647                assert!(
9648                    !index
9649                        .controls()
9650                        .expect("control state observed")
9651                        .contains(Path::new(".gitignore")),
9652                    "{case}"
9653                );
9654                assert_eq!(
9655                    index.is_ignored(Path::new("a.log")).expect("control state observed"),
9656                    Some(false),
9657                    "{case}"
9658                );
9659            }
9660        }
9661    }
9662
9663    /// A cold walk records a name gone by its stat as it records a name the listing never
9664    /// returned: not at all, and without an error that would make the walk partial.
9665    #[test]
9666    fn a_cold_walk_omits_a_child_deleted_between_listing_and_stat() {
9667        for threads in [1, 4] {
9668            let dir = tempfile::tempdir().expect("tempdir");
9669            write_file(&dir.path().join("keep.txt"), b"keep");
9670            write_file(&dir.path().join("gone.txt"), b"gone");
9671            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
9672
9673            let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
9674            let scanned = scan_into_index_via_scanner(dir.path(), &config);
9675            drop(hook);
9676            let (index, report) = scanned.expect("scan");
9677
9678            assert!(report.is_complete(), "threads {threads}: {:?}", report.errors);
9679            assert!(index.lookup(Path::new("gone.txt")).is_none(), "threads {threads}");
9680            assert!(index.lookup(Path::new("keep.txt")).is_some(), "threads {threads}");
9681        }
9682    }
9683
9684    /// Revalidation emits the removal a reconciliation would apply.
9685    #[test]
9686    fn revalidation_removes_a_child_deleted_between_listing_and_stat() {
9687        let dir = tempfile::tempdir().expect("tempdir");
9688        write_file(&dir.path().join("keep.txt"), b"keep");
9689        write_file(&dir.path().join("gone.txt"), b"gone");
9690        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9691
9692        let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
9693        let mut observations = Vec::new();
9694        let report = revalidate(&index, &ScanConfig::default(), &mut |observation| {
9695            observations.push(observation);
9696        });
9697        drop(hook);
9698        let report = report.expect("revalidate");
9699        for observation in &observations {
9700            index.apply_ok(observation);
9701        }
9702
9703        assert!(report.is_complete(), "{:?}", report.errors);
9704        assert!(index.lookup(Path::new("gone.txt")).is_none());
9705        assert!(index.lookup(Path::new("keep.txt")).is_some());
9706    }
9707
9708    /// Revalidation emits the same rules removal when the rest of the listing fails.
9709    #[test]
9710    fn revalidation_removes_the_rules_of_a_control_file_deleted_in_a_failing_listing() {
9711        for prune_hidden in [false, true] {
9712            let dir = tempfile::tempdir().expect("tempdir");
9713            write_file(&dir.path().join(".gitignore"), b"*.log\n");
9714            write_file(&dir.path().join("a.log"), b"log");
9715            let config = ScanConfig {
9716                read_controls: true,
9717                hidden: prune_hidden
9718                    .then(|| std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([""; 0]))),
9719                ..ScanConfig::default()
9720            };
9721            let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
9722            assert_eq!(
9723                index.is_ignored(Path::new("a.log")).expect("control state observed"),
9724                Some(true)
9725            );
9726
9727            let hook =
9728                delete_between_listing_and_stat_in_a_failing_listing(dir.path(), ".gitignore");
9729            let mut observations = Vec::new();
9730            let report = revalidate(&index, &config, &mut |observation| {
9731                observations.push(observation);
9732            });
9733            drop(hook);
9734            let report = report.expect("revalidate");
9735            for observation in &observations {
9736                index.apply_ok(observation);
9737            }
9738
9739            assert!(!report.is_complete(), "prune hidden {prune_hidden}");
9740            assert!(index.lookup(Path::new(".gitignore")).is_none(), "prune hidden {prune_hidden}");
9741            assert!(
9742                !index
9743                    .controls()
9744                    .expect("control state observed")
9745                    .contains(Path::new(".gitignore")),
9746                "prune hidden {prune_hidden}"
9747            );
9748            assert_eq!(
9749                index.is_ignored(Path::new("a.log")).expect("control state observed"),
9750                Some(false),
9751                "prune hidden {prune_hidden}"
9752            );
9753        }
9754    }
9755
9756    #[test]
9757    fn reconciliation_does_not_clear_a_newer_invalidation() {
9758        let dir = sample_tree();
9759        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9760        let handle = crate::IndexHandle::new(index);
9761        write_file(&dir.path().join("added.md"), b"new");
9762
9763        let invalidator = handle.clone();
9764        let mut saw_reconciling = false;
9765        reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
9766            if commit_touches(commit, Path::new("added.md")) {
9767                saw_reconciling =
9768                    invalidator.freshness().expect("query") == crate::Freshness::Reconciling;
9769                invalidator
9770                    .apply(&Observation::new(vec![Op::InvalidateSubtree {
9771                        path: PathBuf::new(),
9772                        reason: crate::InvalidateReason::WatchOverflow,
9773                    }]))
9774                    .expect("new invalidation");
9775            }
9776        })
9777        .expect("reconcile handle");
9778
9779        assert!(saw_reconciling);
9780        assert_eq!(handle.freshness().expect("query"), crate::Freshness::Stale);
9781    }
9782
9783    #[test]
9784    fn failed_reconciliation_marks_the_scope_partial() {
9785        let dir = sample_tree();
9786        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9787        fs::remove_dir_all(dir.path()).expect("remove root");
9788
9789        assert!(reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).is_err());
9790        assert_eq!(index.freshness(), crate::Freshness::Partial);
9791    }
9792
9793    #[test]
9794    fn successful_subtree_retry_restores_complete_root_coverage() {
9795        let dir = tempfile::tempdir().expect("tempdir");
9796        write_file(&dir.path().join("blocked/known.txt"), b"known");
9797        let config = ScanConfig::default();
9798        let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
9799        assert!(report.is_complete());
9800        let blocked = dir.path().join("blocked");
9801        let fault = install_walk_hook(&blocked, |_| {
9802            Some(std::io::Error::new(
9803                std::io::ErrorKind::PermissionDenied,
9804                "deterministic subtree refusal",
9805            ))
9806        });
9807
9808        let failed = reconcile_subtree(&mut index, Path::new("blocked"), &config, &mut |_| {})
9809            .expect("partial");
9810        assert!(!failed.scan.is_complete());
9811        assert_eq!(
9812            index.state().coverage,
9813            crate::Coverage::Partial(crate::CoverageReason::Inaccessible)
9814        );
9815        drop(fault);
9816
9817        let recovered = reconcile_subtree(&mut index, Path::new("blocked"), &config, &mut |_| {})
9818            .expect("retry");
9819
9820        assert!(recovered.scan.is_complete());
9821        assert_eq!(index.state().coverage, crate::Coverage::Complete);
9822    }
9823
9824    #[test]
9825    fn failed_pending_reconciliation_remains_queued_for_retry() {
9826        let dir = tempfile::tempdir().expect("tempdir");
9827        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9828        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
9829            path: PathBuf::new(),
9830            reason: crate::InvalidateReason::Requested,
9831        }]));
9832        fs::remove_dir_all(dir.path()).expect("remove root");
9833
9834        assert!(reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {}).is_err());
9835        assert_eq!(
9836            index.take_pending_invalidations(),
9837            vec![(PathBuf::new(), crate::InvalidateReason::Requested)]
9838        );
9839        assert_eq!(index.freshness(), crate::Freshness::Partial);
9840    }
9841
9842    #[cfg(unix)]
9843    #[test]
9844    fn partial_cold_scan_keeps_verified_siblings_complete() {
9845        use std::os::unix::fs::PermissionsExt;
9846        if !crate::test_support::require_permission_bits() {
9847            return;
9848        }
9849        let root = tempfile::tempdir().expect("root");
9850        write_file(&root.path().join("blocked/unknown.txt"), b"unread");
9851        write_file(&root.path().join("healthy/nested/known.txt"), b"known");
9852        let blocked = root.path().join("blocked");
9853        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
9854        let scan = ScanConfig::default();
9855        let detached = scan_into_index(root.path(), &scan);
9856        let streamed = scan_into_index_via_scanner(root.path(), &scan);
9857        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
9858        for result in [detached, streamed] {
9859            let (index, report) = result.expect("partial scan still returns its facts");
9860            assert!(!report.is_complete(), "permission fixture must fail the blocked listing");
9861            assert!(
9862                index
9863                    .issues()
9864                    .iter()
9865                    .any(|issue| issue.path.as_deref() == Some(Path::new("blocked")))
9866            );
9867            assert_eq!(index.freshness_at(Path::new("")), crate::Freshness::Partial);
9868            assert_eq!(index.directory_complete(Path::new("")), Some(true));
9869            assert_eq!(index.directory_complete(Path::new("blocked")), Some(false));
9870            assert_eq!(index.freshness_at(Path::new("blocked")), crate::Freshness::Partial);
9871            assert!(index.lookup(Path::new("blocked/unknown.txt")).is_none());
9872            for sibling in ["healthy", "healthy/nested"] {
9873                assert_eq!(index.directory_complete(Path::new(sibling)), Some(true), "{sibling}");
9874                assert_eq!(
9875                    index.freshness_at(Path::new(sibling)),
9876                    crate::Freshness::Fresh,
9877                    "{sibling}"
9878                );
9879            }
9880            assert!(index.lookup(Path::new("healthy/nested/known.txt")).is_some());
9881            assert!(
9882                !crate::stored_state::entries_writable(&index),
9883                "partial root cannot persist metadata"
9884            );
9885        }
9886    }
9887
9888    #[cfg(unix)]
9889    #[test]
9890    fn partial_pending_reconciliation_remains_queued_for_retry() {
9891        use std::os::unix::fs::PermissionsExt;
9892
9893        if !crate::test_support::require_permission_bits() {
9894            return;
9895        }
9896        let dir = tempfile::tempdir().expect("tempdir");
9897        write_file(&dir.path().join("blocked/known.txt"), b"known");
9898        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9899        let blocked = dir.path().join("blocked");
9900        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
9901        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
9902            path: PathBuf::from("blocked"),
9903            reason: crate::InvalidateReason::VerificationFailed,
9904        }]));
9905
9906        let report = reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {})
9907            .expect("permission failure is a partial report");
9908        let pending = index.take_pending_invalidations();
9909        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
9910        assert!(!report.is_complete(), "permission fixture must make reconciliation partial");
9911        assert_eq!(
9912            pending,
9913            vec![(PathBuf::from("blocked"), crate::InvalidateReason::VerificationFailed)]
9914        );
9915        assert_eq!(index.freshness_at(Path::new("blocked")), crate::Freshness::Partial);
9916    }
9917
9918    /// The shared API settles an unreadable subtree instead of queueing it again.
9919    ///
9920    /// Its per-event driver, `Watcher::apply_next`, drains after every event, so a retry
9921    /// re-walked the same unreadable subtree on each unrelated event, forever. The subtree
9922    /// stays partial and the report still names the error, once.
9923    #[cfg(unix)]
9924    #[test]
9925    fn partial_shared_pending_reconciliation_settles_instead_of_retrying() {
9926        use std::os::unix::fs::PermissionsExt;
9927
9928        if !crate::test_support::require_permission_bits() {
9929            return;
9930        }
9931        let dir = tempfile::tempdir().expect("tempdir");
9932        write_file(&dir.path().join("blocked/known.txt"), b"known");
9933        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9934        let handle = crate::IndexHandle::new(index);
9935        let blocked = dir.path().join("blocked");
9936        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
9937        handle
9938            .apply(&Observation::new(vec![Op::InvalidateSubtree {
9939                path: PathBuf::from("blocked"),
9940                reason: crate::InvalidateReason::VerificationFailed,
9941            }]))
9942            .expect("invalidate");
9943
9944        let report = reconcile_pending_handle(&handle, &ScanConfig::default(), &mut |_| {})
9945            .expect("permission failure is a partial report");
9946        let pending = handle.take_pending_invalidations().expect("pending");
9947        let freshness = handle.freshness_at(Path::new("blocked")).expect("freshness");
9948        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
9949        assert!(!report.is_complete(), "permission fixture must make reconciliation partial");
9950        assert!(pending.is_empty(), "{pending:?}");
9951        assert_eq!(freshness, crate::Freshness::Partial);
9952        assert!(!report.scan.errors.is_empty());
9953    }
9954
9955    #[test]
9956    fn pending_scope_mismatch_does_not_drain_the_retry_queue() {
9957        let dir = tempfile::tempdir().expect("tempdir");
9958        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
9959        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
9960        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
9961            path: PathBuf::new(),
9962            reason: crate::InvalidateReason::Requested,
9963        }]));
9964
9965        let error = reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {})
9966            .expect_err("mismatched scope must fail");
9967
9968        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
9969        assert_eq!(
9970            index.take_pending_invalidations(),
9971            vec![(PathBuf::new(), crate::InvalidateReason::Requested)]
9972        );
9973    }
9974
9975    #[test]
9976    fn reconciliation_rejects_a_scope_mismatch_before_mutating() {
9977        let dir = tempfile::tempdir().expect("tempdir");
9978        write_file(&dir.path().join("deep/nested.txt"), b"nested");
9979        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
9980        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
9981        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
9982
9983        let error = reconcile(&mut index, &ScanConfig::default(), &mut |_| {})
9984            .expect_err("mismatched scope must fail");
9985
9986        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
9987        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
9988        assert_eq!(index.freshness(), crate::Freshness::Fresh);
9989    }
9990
9991    #[test]
9992    fn subtree_reconciliation_rejects_a_path_beyond_the_depth_scope() {
9993        let dir = tempfile::tempdir().expect("tempdir");
9994        write_file(&dir.path().join("deep/nested.txt"), b"nested");
9995        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
9996        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
9997
9998        let result =
9999            reconcile_subtree(&mut index, Path::new("deep/nested.txt"), &shallow, &mut |_| {});
10000
10001        assert!(matches!(result, Err(Error::SubtreeOutsideScanScope { .. })));
10002        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
10003        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10004    }
10005
10006    #[cfg(unix)]
10007    #[test]
10008    fn subtree_reconciliation_does_not_follow_an_ancestor_symlink() {
10009        use std::os::unix::fs::symlink;
10010
10011        let root = tempfile::tempdir().expect("root");
10012        let outside = tempfile::tempdir().expect("outside");
10013        write_file(&outside.path().join("secret.txt"), b"secret");
10014        symlink(outside.path(), root.path().join("link")).expect("symlink");
10015        let config = ScanConfig::default();
10016        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
10017
10018        let result =
10019            reconcile_subtree(&mut index, Path::new("link/secret.txt"), &config, &mut |_| {});
10020
10021        assert!(matches!(result, Err(Error::SubtreeOutsideScanScope { .. })));
10022        assert!(index.lookup(Path::new("link/secret.txt")).is_none());
10023        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10024    }
10025
10026    #[test]
10027    fn subtree_reconciliation_widens_to_a_non_directory_ancestor() {
10028        let root = tempfile::tempdir().expect("root");
10029        write_file(&root.path().join("parent/child.txt"), b"old");
10030        let config = ScanConfig::default();
10031        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
10032        fs::remove_dir_all(root.path().join("parent")).expect("remove directory");
10033        write_file(&root.path().join("parent"), b"replacement");
10034
10035        let report =
10036            reconcile_subtree(&mut index, Path::new("parent/child.txt"), &config, &mut |_| {})
10037                .expect("reconcile widened ancestor");
10038
10039        assert!(report.is_complete());
10040        assert_eq!(index.kind(Path::new("parent")), Some(EntryKind::File));
10041        assert!(index.lookup(Path::new("parent/child.txt")).is_none());
10042        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10043    }
10044
10045    #[test]
10046    fn subtree_reconciliation_widens_to_a_missing_ancestor() {
10047        let root = tempfile::tempdir().expect("root");
10048        write_file(&root.path().join("parent/child.txt"), b"old");
10049        let config = ScanConfig::default();
10050        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
10051        fs::remove_dir_all(root.path().join("parent")).expect("remove directory");
10052
10053        let report =
10054            reconcile_subtree(&mut index, Path::new("parent/child.txt"), &config, &mut |_| {})
10055                .expect("reconcile widened ancestor");
10056
10057        assert!(report.is_complete());
10058        assert!(index.lookup(Path::new("parent")).is_none());
10059        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10060    }
10061
10062    #[test]
10063    fn observation_only_revalidation_rejects_a_scope_mismatch() {
10064        let dir = tempfile::tempdir().expect("tempdir");
10065        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
10066        let (index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
10067        let mut observations = Vec::new();
10068
10069        let error = revalidate(&index, &ScanConfig::default(), &mut |observation| {
10070            observations.push(observation);
10071        })
10072        .expect_err("mismatched scope must fail");
10073
10074        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
10075        assert!(observations.is_empty());
10076    }
10077
10078    #[cfg(unix)]
10079    #[test]
10080    fn a_new_filesystem_boundary_prunes_cached_descendants() {
10081        use std::os::unix::fs::MetadataExt;
10082
10083        let root = Path::new("/");
10084        let root_dev = {
10085            crate::counters::bump(|c| c.stats += 1);
10086            fs::symlink_metadata(root)
10087        }
10088        .expect("stat root")
10089        .dev();
10090        let Some(mount) = [Path::new("/dev"), Path::new("/proc"), Path::new("/sys")]
10091            .into_iter()
10092            .find(|candidate| {
10093                fs::symlink_metadata(candidate)
10094                    .is_ok_and(|metadata| metadata.is_dir() && metadata.dev() != root_dev)
10095            })
10096        else {
10097            return; // This host exposes no convenient cross-device directory.
10098        };
10099        let relative = mount.strip_prefix(root).expect("mount is below root");
10100        let stale_child = relative.join(".fdu-stale-snapshot-entry");
10101        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
10102        let mount_meta = fs::symlink_metadata(mount).expect("stat mount");
10103        let mut index = Index::new_with_scope(root, config.scope());
10104        index.apply_baseline_ok(&Observation::new(vec![
10105            Op::Upsert {
10106                path: relative.to_path_buf(),
10107                kind: EntryKind::Dir,
10108                attrs: attrs_from(mount, &mount_meta).expect("mount attrs"),
10109            },
10110            Op::Upsert {
10111                path: stale_child.clone(),
10112                kind: EntryKind::File,
10113                attrs: Attrs { size: 10, allocated: 10, ..Attrs::default() },
10114            },
10115        ]));
10116
10117        let error = reconcile_subtree(&mut index, &stale_child, &config, &mut |_| {})
10118            .expect_err("a descendant below the mount boundary is outside scope");
10119        assert!(matches!(error, Error::SubtreeOutsideScanScope { .. }));
10120        assert!(index.lookup(&stale_child).is_some());
10121
10122        reconcile_subtree(&mut index, relative, &config, &mut |_| {}).expect("reconcile mount");
10123
10124        assert!(index.lookup(relative).is_some(), "the mount point itself stays visible");
10125        assert!(index.lookup(&stale_child).is_none(), "out-of-scope descendants are pruned");
10126    }
10127
10128    #[test]
10129    fn subtree_reconciliation_rejects_paths_outside_the_root() {
10130        let dir = sample_tree();
10131        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10132
10133        assert!(matches!(
10134            reconcile_subtree(
10135                &mut index,
10136                Path::new("../outside"),
10137                &ScanConfig::default(),
10138                &mut |_| {},
10139            ),
10140            Err(Error::PathEscapesRoot(_))
10141        ));
10142        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10143    }
10144
10145    #[test]
10146    fn normalized_walk_errors_keep_index_and_one_shot_status_in_lockstep() {
10147        let root = Path::new("/root");
10148        let mut order: Vec<_> = (0..66).rev().collect();
10149        order.push(65);
10150        let mut errors = order
10151            .into_iter()
10152            .map(|number| {
10153                Error::io(
10154                    root.join(format!("file-{number:02}")),
10155                    std::io::Error::new(std::io::ErrorKind::PermissionDenied, "denied"),
10156                )
10157            })
10158            .collect::<Vec<_>>();
10159
10160        normalize_walk_errors(root, &mut errors);
10161        assert_eq!(errors.len(), 66, "the repeated cause is removed once");
10162
10163        let mut index = crate::Index::new(root);
10164        index.record_walk_errors(&mut errors);
10165        let status = crate::query::TreeStatus::of_walk(
10166            root,
10167            &mut ScanReport { errors, ..ScanReport::default() },
10168        );
10169
10170        assert_eq!(status.errors, index.issues());
10171        assert_eq!(status.errors_omitted, index.state().issues.omitted);
10172        assert_eq!(status.errors.len(), crate::MAX_RETAINED_ISSUES);
10173        assert_eq!(status.errors_omitted, 2);
10174    }
10175
10176    #[cfg(target_os = "linux")]
10177    #[test]
10178    fn scan_and_revalidate_keep_non_utf8_names_distinct() {
10179        use std::ffi::OsString;
10180        use std::os::unix::ffi::OsStringExt;
10181
10182        let dir = tempfile::tempdir().expect("tempdir");
10183        let first = PathBuf::from(OsString::from_vec(vec![b'n', 0x80]));
10184        let second = PathBuf::from(OsString::from_vec(vec![b'n', 0x81]));
10185        write_file(&dir.path().join(&first), b"a");
10186        write_file(&dir.path().join(&second), b"bb");
10187
10188        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10189        assert_eq!(index.total().files, 2);
10190        assert_eq!(index.total().bytes, 3);
10191
10192        fs::remove_file(dir.path().join(&first)).expect("remove first");
10193        let mut observations = Vec::new();
10194        revalidate(&index, &ScanConfig::default(), &mut |observation| {
10195            observations.push(observation);
10196        })
10197        .expect("revalidate");
10198        for observation in &observations {
10199            index.apply_ok(observation);
10200        }
10201        assert!(index.lookup(&first).is_none());
10202        assert!(index.lookup(&second).is_some());
10203        assert_eq!(index.total().bytes, 2);
10204    }
10205}