Skip to main content

fdu_core/
scan.rs

1//! The scan layer: walking a tree, producing observations, and applying reconciliation.
2//!
3//! Public scans emit upsert observations, and a revalidation sweep is the diff between
4//! what the index believes and what the filesystem says. Both speak the same
5//! [`Observation`] vocabulary as the watch layer. A detached one-shot index may consume
6//! equivalent parent-first directory groups privately because no observer can see its
7//! construction; every later mutation still crosses the shared observation boundary.
8//!
9//! # Status
10//!
11//! The serial walk is the portable `read_dir` plus non-following metadata reference.
12//! Parallel scans use the same path on most platforms; on macOS they first try a
13//! measured `getattrlistbulk` backend that returns directory entries and stat-tier
14//! metadata together. Unsupported filesystems, malformed results, mount points, and
15//! firmlinks fail closed to the portable path for the complete containing directory.
16//! Every backend produces the same [`Observation`] contract.
17
18use std::collections::{BTreeMap, BTreeSet, VecDeque};
19use std::ffi::{OsStr, OsString};
20use std::fmt::Write as _;
21use std::fs;
22use std::io::Read as _;
23use std::path::{Component, Path, PathBuf};
24
25use crate::ApplyStats;
26use crate::engine_contract::{
27    Attrs, Commit, EntryKind, Error, Observation, ObservationOp, Op, PathExpectation, PathState,
28    Result, ScanScope,
29};
30use crate::index::{
31    DetachedIndexBuilder, Index, IndexHandle, ReconcileErrors, ReconcileFinish,
32    collect_child_expectations,
33};
34use crate::query::ScopeAxis;
35use crate::stored_state::{ControlTierIdentity, EntryScope, EntryTierIdentity, SnapshotIdentity};
36
37// Keep the FFI exception at the platform boundary. The rest of the engine, including
38// every consumer of these observations, remains under the workspace's unsafe-code
39// denial.
40#[cfg(target_os = "macos")]
41#[allow(unsafe_code)]
42mod macos_bulk;
43
44#[cfg(windows)]
45#[allow(unsafe_code)]
46mod windows_metadata;
47
48/// How many ops accumulate before an observation is handed to the sink.
49///
50/// Batching matters for more than syscall economy: consumers coalesce per path within a
51/// batch and stat once per batch, and a live UI wants partial results while a large tree
52/// is still being walked rather than one delta at the end.
53const DEFAULT_BATCH_SIZE: usize = crate::platform_tuning::tuning().batch_size.get();
54
55/// Largest producer batch accepted before work must be published incrementally.
56pub const MAX_SCAN_BATCH_SIZE: usize = 64 * 1024;
57
58/// Most changed paths an exclusive parallel reconciliation may defer before applying.
59///
60/// Workers compare against one immutable index image, so mutations wait until the wave
61/// joins. Bounding that change set keeps a churned tree from turning the fast unchanged
62/// path into an unbounded allocation; overflow discards the wave and retries through
63/// the incremental serial reconciler.
64const MAX_DEFERRED_RECONCILE_OPS: usize = MAX_SCAN_BATCH_SIZE;
65
66/// Directories compared against one immutable index baseline before changes are applied.
67///
68/// The wave is large enough to amortize scoped worker creation and small enough that a
69/// changed tree publishes progress throughout a long reconciliation.
70const RECONCILE_WAVE_DIRECTORIES: usize =
71    crate::platform_tuning::tuning().reconcile_wave_directories.get();
72
73/// Identity of the fixed stat-tier reducer set.
74const REDUCERS_FINGERPRINT: u64 = 1;
75
76/// The order directories are visited in.
77///
78/// This changes *when* observations are produced, never *which* ones: both orders
79/// visit every entry exactly once and leave an identical index behind. It therefore
80/// stays out of [`ScanScope`] and cannot invalidate a cache, exactly like the worker
81/// count.
82///
83/// The choice only matters to a consumer that reads the index while the walk is still
84/// running, and there it matters a great deal.
85///
86/// # Strength of the guarantee
87///
88/// **These are scheduling preferences, not strict orders, whenever more than one worker
89/// is running** — which is the default.
90///
91/// The queue is ordered, but the *claims* are not. Workers take directories from the
92/// shared queue in the policy's order; a worker that finishes early can enqueue its
93/// children and another worker can claim them while a slower worker still holds
94/// unfinished work from a shallower level. Nothing releases a level barrier, because
95/// a barrier would idle every fast worker at each level boundary and give back most of
96/// the parallel producer's win.
97///
98/// So:
99///
100/// - With `threads: Some(1)`, [`ScanOrder::BreadthFirst`] is strict: no directory is
101///   read before one closer to the root.
102/// - With several workers it is *shallow-first*: shallow work is always preferred when
103///   a worker chooses, and deeper observations can still interleave.
104///
105/// That weaker property is what the browser use case actually needs — every top-level
106/// subtree starts filling early, so a mid-scan ranking is meaningful — and it is the
107/// property the tests pin. A caller that needs strict level order must ask for one
108/// worker and pay for it.
109#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
110pub enum ScanOrder {
111    /// Shallow directories before deep ones.
112    ///
113    /// The default, because it is the order whose partial results mean something.
114    /// Roll-ups are maintained per directory as the walk proceeds, so a consumer that
115    /// looks mid-scan sees top-level totals grow together — bars fill, rankings
116    /// converge — instead of one subtree finishing while its siblings read zero.
117    /// Interrupting early leaves a usefully complete picture of the top of the tree.
118    ///
119    /// Under several workers this is a preference rather than a guarantee; see the
120    /// type-level note above.
121    ///
122    /// Note that totals only grow *while an additive walk is running*. Monotonicity
123    /// comes from the producer being additive, not from the order — the order decides
124    /// which subtrees get to grow early.
125    #[default]
126    BreadthFirst,
127    /// One subtree toward completion before starting the next.
128    ///
129    /// Lower peak memory, since the frontier is bounded by depth rather than by the
130    /// width of a level, and better locality within a subtree. The cost is that
131    /// partial results are actively misleading: one child of the root approaches its
132    /// final total while its siblings read zero, so anything ranking by size mid-scan
133    /// ranks confidently and wrongly. Correct for a caller that only reads the
134    /// finished index and wants the smallest footprint.
135    ///
136    /// Under several workers this too is a preference: several subtrees will be in
137    /// flight at once, one per worker.
138    DepthFirst,
139}
140
141/// Knobs for a scan.
142#[derive(Clone, Debug)]
143// Four booleans, each an independent admission or observation switch with its own
144// semantic-scope consequence, not an enum in disguise: any combination is legal and
145// means what its fields say. The lint suspects flag-soup states; this is a config
146// surface whose fields are documented one by one.
147#[allow(clippy::struct_excessive_bools)]
148pub struct ScanConfig {
149    /// Maximum relative entry depth to retain. Zero keeps only the index root and `None`
150    /// means unlimited.
151    pub max_depth: Option<usize>,
152    /// Ops per emitted observation. Must be between one and [`MAX_SCAN_BATCH_SIZE`].
153    pub batch_size: usize,
154    /// Follow symlinks to directories. Off by default: following them turns a tree walk
155    /// into a graph walk with cycles, and every surveyed tool defaults to off.
156    pub follow_symlinks: bool,
157    /// Stay on the filesystem the root lives on.
158    pub one_filesystem: bool,
159    /// Hidden-component admission, or `None` to retain every component.
160    pub hidden: Option<std::sync::Arc<crate::admission::HiddenPolicy>>,
161    /// Exclude filesystem objects other than files, directories, and symlinks.
162    pub exclude_special: bool,
163    /// Directory-reading worker threads.
164    ///
165    /// A tree walk is a pile of independent, latency-bound directory reads, so it
166    /// scales with threads far better than most work does. One means the serial
167    /// walker, which stays the reference implementation and the thing every result is
168    /// checked against. [`None`] asks for a bounded default derived from the
169    /// machine's available parallelism. The automatic pool starts conservatively and
170    /// unlocks more latency-hiding workers only when initial chunk timing identifies a
171    /// slow filesystem path.
172    ///
173    /// This is an operational knob, not a semantic one: it changes how fast the same
174    /// observations are produced, never which observations they are. That is why it
175    /// stays out of [`ScanScope`] and cannot invalidate a cache.
176    pub threads: Option<usize>,
177    /// The order directories are visited in. See [`ScanOrder`].
178    pub order: ScanOrder,
179    /// File-type rules to classify against, or `None` for the ones compiled into fdu.
180    ///
181    /// Unlike [`Self::threads`] this *is* semantic: a different taxonomy classifies the
182    /// same tree differently, which is why its fingerprint rides in [`ScanScope`] and a
183    /// change to it invalidates a snapshot. Shared rather than owned because a scan
184    /// clones its config per wave and a registry is read-only once built.
185    pub types: Option<std::sync::Arc<crate::classify::TypeRegistry>>,
186    /// Observe `.gitignore` control files and retain ignore classification.
187    ///
188    /// On by default on every surface: an [`Index`] from [`crate::open`] or a scan keeps
189    /// the exact control state it exposes and a watch maintains -- which entries are
190    /// ignored, and the ignored and unignored partitions of every roll-up -- and a one-shot
191    /// report from [`crate::prepare_report`] shows the ignored share of every row
192    /// (fdu-elnn). It costs a read of every `.gitignore` in the tree. A file past the
193    /// [`Self::control_limits`] is refused and named in [`Index::control_coverage`] rather
194    /// than ending the scan; a file that cannot be read is an error at its path, which
195    /// makes the result partial.
196    ///
197    /// Off, the scan performs no control-file I/O and retains no control table, and that is
198    /// stamped into [`ScanScope`], so an index-returning call never serves a snapshot taken
199    /// one way as the other. An [`Index`] built that way answers [`Index::is_ignored`],
200    /// [`Index::controls`], and the partition accessors with
201    /// [`crate::Error::ControlStateNotObserved`], never with "not ignored", and refuses
202    /// control input; a report's rows carry no ignored share, and a selection by ignored
203    /// state is refused. The command line spells it `--no-gitignore`.
204    ///
205    /// An opened root ([`crate::OpenedIndex`]) always observes control state, because its
206    /// ignored and unignored partitions are part of what it serves.
207    pub read_controls: bool,
208    /// Which ignored population shapes retained entries and content candidates.
209    pub population: crate::query::IgnoredEntries,
210    /// The budget and the line limit `.gitignore` files are applied under, each a size or
211    /// unbounded. See [`crate::control::ControlLimits`].
212    ///
213    /// A source that would take the table past the budget, or that has a line longer than
214    /// the line limit, is refused: its rules do not apply, the scan continues with every
215    /// size exact, and [`Index::control_coverage`] names it and the limit that fired. The
216    /// command line spells these `--gitignore-budget SIZE|all` and
217    /// `--gitignore-line-limit SIZE|all`; the Python API spells them `control_budget` and
218    /// `control_line_limit`.
219    ///
220    /// Semantic, like [`Self::read_controls`]: the limits decide which rules apply, so both
221    /// are part of [`ScanScope`] and a snapshot taken under other limits is not reused.
222    /// Ignored when control state is not observed.
223    pub control_limits: crate::control::ControlLimits,
224    /// Where to report how much of the walk has been done, or `None` to report nothing.
225    ///
226    /// An observer rather than a knob: it changes neither which observations a walk
227    /// produces nor how it produces them, so it is no part of [`ScanScope`] or of any
228    /// snapshot identity, and two configs that differ only here are the same scan.
229    /// Honoured by every walker in this module -- the cold scans, the summary fold,
230    /// [`revalidate`], and each `reconcile` entry point -- which enter
231    /// [`ProgressPhase::Scanning`](crate::ProgressPhase) or
232    /// [`ProgressPhase::Revalidating`](crate::ProgressPhase) and add their counts once
233    /// per chunk of directories, never per entry. See [`crate::Progress`] for what the
234    /// counts mean and what holds when a walk returns.
235    pub progress: Option<crate::Progress>,
236}
237
238impl Default for ScanConfig {
239    fn default() -> Self {
240        Self {
241            max_depth: None,
242            batch_size: DEFAULT_BATCH_SIZE,
243            follow_symlinks: false,
244            one_filesystem: false,
245            hidden: None,
246            exclude_special: false,
247            threads: None,
248            order: ScanOrder::default(),
249            types: None,
250            read_controls: crate::query::Request::DEFAULTS.read_controls,
251            population: crate::query::IgnoredEntries::Include,
252            control_limits: crate::query::Request::DEFAULTS.control_limits,
253            progress: None,
254        }
255    }
256}
257
258/// Why watching cannot narrow its structural scan boundary, said once for every surface.
259/// Ignored population is a supported retained-scope choice because control edits
260/// reconcile the governing directory and rebuild that population.
261///
262/// The CLI used to carry this guidance and the library carried "requires event-scope
263/// filtering", which names the implementation rather than the caller's next move -- so a
264/// library caller hitting the same wall got jargon and the CLI user got help. Two
265/// messages for one rule also drift, and the parity harness could not tell they were the
266/// same rule.
267///
268/// The knobs are named by the calling surface: `--scan-depth` on the command line,
269/// `max_depth` through the API. Everything else is identical, so the harness can verify
270/// mechanically that both surfaces state the same rule.
271pub const WATCH_SCOPE_GUIDANCE: &str = concat!(
272    "watching requires full scope and cannot be combined with max_depth or one_filesystem: ",
273    "a watcher cannot filter backend events against a narrowed boundary. Selection such as ",
274    "depth, include, and modified_since does work while watching, because it filters the ",
275    "retained index rather than narrowing the scan"
276);
277
278impl ScanConfig {
279    /// Classify this scan with `types` and include their derived identity in its scope.
280    #[must_use]
281    pub fn with_types(mut self, types: std::sync::Arc<crate::classify::TypeRegistry>) -> Self {
282        self.types = Some(types);
283        self
284    }
285
286    /// The file-type rules in effect: the supplied registry, or the compiled default.
287    pub fn types(&self) -> &crate::classify::TypeRegistry {
288        match &self.types {
289            Some(types) => types,
290            None => crate::classify::TypeRegistry::compiled(),
291        }
292    }
293
294    /// Share the file-type rules with an index that retains them.
295    pub(crate) fn types_shared(&self) -> std::sync::Arc<crate::classify::TypeRegistry> {
296        self.types
297            .as_ref()
298            .map_or_else(crate::classify::TypeRegistry::compiled_shared, std::sync::Arc::clone)
299    }
300
301    /// Hidden-component policy in effect.
302    pub fn hidden(&self) -> &crate::admission::HiddenPolicy {
303        self.hidden.as_deref().unwrap_or_else(|| crate::admission::HiddenPolicy::keep_all())
304    }
305
306    /// Semantic cache identity, excluding operational batching choices.
307    ///
308    /// Composed from [`Self::snapshot_identity`], so the scope an index records and the
309    /// tier identities a snapshot of it carries are one value in two shapes.
310    ///
311    /// No longer `const`: the type-rule fingerprint is now a property of the registry in
312    /// effect rather than a compiled-in constant, which is the whole point of letting a
313    /// caller supply one. A snapshot taken under different rules must not be reused.
314    pub fn scope(&self) -> ScanScope {
315        self.snapshot_identity().scan_scope()
316    }
317
318    /// Which entries this scan retains, the part of its scope no `.gitignore` setting
319    /// changes.
320    pub fn entry_scope(&self) -> EntryScope {
321        EntryScope {
322            max_depth: self.max_depth,
323            follow_symlinks: self.follow_symlinks,
324            one_filesystem: self.one_filesystem,
325            hidden_fingerprint: self.hidden().fingerprint(),
326            exclude_special: self.exclude_special,
327            population: self.population,
328            control_fingerprint: if self.population == crate::query::IgnoredEntries::Include {
329                0
330            } else {
331                self.control_identity().ignore_rules_fingerprint()
332            },
333        }
334    }
335
336    /// Whether this scan observes `.gitignore` control state, and under which limits.
337    ///
338    /// The limits are part of the identity only when control state is observed: a scan
339    /// that reads no control file applies none, whatever [`Self::control_limits`] says.
340    pub fn control_identity(&self) -> ControlTierIdentity {
341        if self.read_controls {
342            ControlTierIdentity::Observed { limits: self.control_limits }
343        } else {
344            ControlTierIdentity::NotObserved
345        }
346    }
347
348    /// The identity of every tier a snapshot of this scan holds.
349    pub fn snapshot_identity(&self) -> SnapshotIdentity {
350        SnapshotIdentity {
351            entries: EntryTierIdentity {
352                engine: crate::snapshot::engine_fingerprint(),
353                scope: self.entry_scope(),
354                type_rules_fingerprint: self.types().fingerprint(),
355                reducers_fingerprint: REDUCERS_FINGERPRINT,
356            },
357            controls: self.control_identity(),
358        }
359    }
360
361    /// Resolve [`Self::threads`] to the workers active when a scan begins.
362    #[cfg(any(target_os = "macos", test))]
363    fn worker_threads(&self) -> usize {
364        self.worker_pool().initial
365    }
366
367    /// Resolve the worker count for immutable-baseline reconciliation waves.
368    fn reconciliation_worker_threads(&self) -> usize {
369        match self.threads {
370            Some(threads) => threads.clamp(1, MAX_SCAN_THREADS),
371            None => std::thread::available_parallelism()
372                .map_or(1, std::num::NonZero::get)
373                .clamp(1, DEFAULT_RECONCILE_THREADS_CAP),
374        }
375    }
376
377    /// Resolve the initial and maximum worker counts for one scan.
378    #[cfg(any(target_os = "macos", test))]
379    fn worker_pool(&self) -> WorkerPool {
380        self.worker_pool_for(std::thread::available_parallelism().map_or(1, std::num::NonZero::get))
381    }
382
383    /// Resolve the worker pool from one captured operating-system parallelism value.
384    fn worker_pool_for(&self, available_parallelism: usize) -> WorkerPool {
385        match self.threads {
386            Some(threads) => WorkerPool::fixed(threads.clamp(1, MAX_SCAN_THREADS)),
387            None => automatic_worker_pool(available_parallelism),
388        }
389    }
390
391    /// The scope axis this build cannot honour, if any.
392    ///
393    /// The one statement of the capability rule, so it is asked rather than restated.
394    /// [`Request::validate`](crate::query::Request::validate) asks it before any stored
395    /// state is read, which is what makes a scope this build cannot honour refuse the same
396    /// way on every route, every cache policy, and both surfaces; [`Self::validate`] asks
397    /// it for the engine-internal callers -- a bound root, a raw scan, an observation --
398    /// that never carry a request.
399    pub(crate) const fn unsupported_axis(&self) -> Option<ScopeAxis> {
400        if self.follow_symlinks {
401            return Some(ScopeAxis::FollowSymlinks);
402        }
403        #[cfg(not(unix))]
404        if self.one_filesystem {
405            return Some(ScopeAxis::OneFilesystem);
406        }
407        None
408    }
409
410    pub(crate) fn validate(&self) -> Result<()> {
411        if !self.read_controls && self.population != crate::query::IgnoredEntries::Include {
412            return Err(Error::UnsupportedScanConfig(
413                "ignored population requires .gitignore observation",
414            ));
415        }
416        if self.batch_size == 0 || self.batch_size > MAX_SCAN_BATCH_SIZE {
417            return Err(Error::UnsupportedScanConfig(
418                "batch_size must be nonzero and no greater than MAX_SCAN_BATCH_SIZE",
419            ));
420        }
421        if let Some(axis) = self.unsupported_axis() {
422            return Err(Error::UnsupportedScanConfig(axis.reason()));
423        }
424        Ok(())
425    }
426
427    pub(crate) fn validate_for_scope(&self, indexed: ScanScope) -> Result<()> {
428        self.validate()?;
429        let requested = self.scope();
430        if indexed != requested {
431            return Err(Error::ScanScopeMismatch { indexed, requested });
432        }
433        Ok(())
434    }
435
436    /// Scope equality, plus the boundary a watcher cannot filter its backend's events
437    /// against.
438    ///
439    /// The rule belongs to the request model, which refuses a watch of a narrowed scope
440    /// before anything is opened ([`RequestError::WatchScope`](crate::query::RequestError));
441    /// this is the same rule where a watcher is bound without a request -- an opened root
442    /// that observes, and each batch the adapter applies -- and it renders the one
443    /// guidance string the model renders.
444    #[cfg(feature = "watch")]
445    pub(crate) fn validate_for_watch_scope(&self, indexed: ScanScope) -> Result<()> {
446        self.validate_for_scope(indexed)?;
447        if self.max_depth.is_some() || self.one_filesystem {
448            return Err(Error::UnsupportedScanConfig(WATCH_SCOPE_GUIDANCE));
449        }
450        Ok(())
451    }
452}
453
454impl Default for ScanScope {
455    fn default() -> Self {
456        ScanConfig::default().scope()
457    }
458}
459
460/// What a scan did, including the errors it walked past.
461///
462/// Unreadable directories are skipped rather than aborting the scan — a permission-denied
463/// subdirectory should not cost you the other 499,000 files — but they are reported
464/// rather than swallowed, so a caller can tell a complete answer from a partial one.
465#[derive(Debug, Default)]
466pub struct ScanReport {
467    /// Directories successfully listed.
468    pub dirs_read: u64,
469    /// Entries observed, directories included.
470    pub entries: u64,
471    /// Regular files whose metadata was observed.
472    pub files_walked: u64,
473    /// Apparent bytes represented by the regular files whose metadata was observed.
474    pub bytes_walked: u64,
475    /// Allocated bytes of those files: what the default size metric counts, and what a
476    /// sparse disk image or a clone makes far smaller than their apparent bytes.
477    pub allocated_walked: u64,
478    /// Paths that could not be read, with the reason.
479    pub errors: Vec<Error>,
480    /// Where the walk's time went, summed across workers.
481    pub attribution: WalkAttribution,
482}
483
484impl ScanReport {
485    /// True when every directory in scope was read successfully.
486    pub fn is_complete(&self) -> bool {
487        self.errors.is_empty()
488    }
489
490    /// Fold one worker's share of a parallel walk into the whole-walk report.
491    fn absorb(&mut self, other: Self) {
492        self.dirs_read += other.dirs_read;
493        self.entries += other.entries;
494        self.files_walked += other.files_walked;
495        self.bytes_walked += other.bytes_walked;
496        self.allocated_walked += other.allocated_walked;
497        self.errors.extend(other.errors);
498        self.attribution.absorb(other.attribution);
499    }
500
501    /// Record one successfully stated directory entry.
502    fn observe(&mut self, kind: EntryKind, attrs: Attrs) {
503        self.entries += 1;
504        if kind == EntryKind::File {
505            self.files_walked += 1;
506            self.bytes_walked += attrs.size;
507            self.allocated_walked += attrs.allocated;
508        }
509    }
510}
511
512/// One walker's running share of the progress counters.
513///
514/// Each walker keeps its own [`ScanReport`]; this remembers how much of that report it
515/// has already added to the shared [`crate::Progress`] cells, so each addition is the
516/// difference since the last. It lives on the worker's stack beside the report rather
517/// than inside it, so a report absorbed into another never carries a stale baseline.
518struct ProgressTally<'a> {
519    progress: Option<&'a crate::Progress>,
520    directories: u64,
521    files: u64,
522    bytes: u64,
523    allocated: u64,
524}
525
526impl<'a> ProgressTally<'a> {
527    const fn new(progress: Option<&'a crate::Progress>) -> Self {
528        Self { progress, directories: 0, files: 0, bytes: 0, allocated: 0 }
529    }
530
531    /// Add what `report` has counted since the last call.
532    ///
533    /// The one `Option` check is the whole cost when no handle is attached. Called once
534    /// per chunk of directories a walker hands over, never per entry.
535    fn flush(&mut self, report: &ScanReport) {
536        let Some(progress) = self.progress else { return };
537        let directories = report.dirs_read - self.directories;
538        let files = report.files_walked - self.files;
539        let bytes = report.bytes_walked - self.bytes;
540        let allocated = report.allocated_walked - self.allocated;
541        if directories != 0 || files != 0 || bytes != 0 || allocated != 0 {
542            progress.add_walked(directories, files, bytes, allocated);
543            self.directories = report.dirs_read;
544            self.files = report.files_walked;
545            self.bytes = report.bytes_walked;
546            self.allocated = report.allocated_walked;
547        }
548    }
549
550    /// Treat everything `report` holds as already added.
551    ///
552    /// For a walker that continues a report whose counts other workers added themselves.
553    fn skip_to(&mut self, report: &ScanReport) {
554        self.directories = report.dirs_read;
555        self.files = report.files_walked;
556        self.bytes = report.bytes_walked;
557        self.allocated = report.allocated_walked;
558    }
559}
560
561/// Normalize filesystem failures before one of the bounded status collectors retains them.
562///
563/// A walk may encounter the same inaccessible path from several worker paths. The report is
564/// already the full, transient set for this pass, so sorting and deduplicating it here avoids
565/// allocating or formatting a second unbounded set solely to decide which 64 details survive.
566/// I/O causes are keyed by their native root-relative path and the issue category, exactly the
567/// cause identity retained by an index. Other engine failures are left distinct: walker errors
568/// are I/O failures, and treating arbitrary engine errors as equivalent without constructing
569/// their bounded issue representation would lose information.
570pub(crate) fn normalize_walk_errors(root: &Path, errors: &mut Vec<Error>) {
571    errors.sort_by(|left, right| match (left, right) {
572        (
573            Error::Io { path: left_path, source: left_source },
574            Error::Io { path: right_path, source: right_source },
575        ) => left_path
576            .strip_prefix(root)
577            .unwrap_or(left_path)
578            .cmp(right_path.strip_prefix(root).unwrap_or(right_path))
579            .then_with(|| {
580                walk_issue_kind_rank(left_source).cmp(&walk_issue_kind_rank(right_source))
581            }),
582        (Error::Io { .. }, _) => std::cmp::Ordering::Less,
583        (_, Error::Io { .. }) => std::cmp::Ordering::Greater,
584        _ => std::cmp::Ordering::Equal,
585    });
586    errors.dedup_by(|right, left| match (left, right) {
587        (
588            Error::Io { path: left_path, source: left_source },
589            Error::Io { path: right_path, source: right_source },
590        ) => {
591            left_path.strip_prefix(root).unwrap_or(left_path)
592                == right_path.strip_prefix(root).unwrap_or(right_path)
593                && walk_issue_kind_rank(left_source) == walk_issue_kind_rank(right_source)
594        }
595        _ => false,
596    });
597}
598
599fn walk_issue_kind_rank(error: &std::io::Error) -> u8 {
600    match error.kind() {
601        std::io::ErrorKind::PermissionDenied => 0,
602        std::io::ErrorKind::NotFound => 1,
603        std::io::ErrorKind::InvalidData | std::io::ErrorKind::InvalidInput => 2,
604        _ => 5,
605    }
606}
607
608/// Schema carried by [`ScanDiagnostics`].
609///
610/// Diagnostics are an opt-in measurement contract rather than stable human output.
611/// Consumers must reject an unknown schema instead of guessing that fields retained
612/// their meaning.
613pub const SCAN_DIAGNOSTICS_SCHEMA: &str = "fdu-scan-diagnostics-v1";
614
615/// Maximum policy-window records retained by one diagnostic scan.
616///
617/// The bound is on controller evaluations, not filesystem entries. A controller that
618/// needs more history must mark the artifact truncated; claim-grade consumers reject
619/// that artifact rather than silently analyzing an incomplete policy history.
620const MAX_POLICY_TRACE_EVENTS: usize = 256;
621
622/// Opt-in, run-scoped evidence about a filesystem scan.
623///
624/// Obtain this through [`scan_with_diagnostics`] or
625/// [`scan_into_index_with_diagnostics`]. Keeping it out of [`ScanReport`] preserves the
626/// existing scan API and keeps ordinary callers off the measurement path entirely.
627#[derive(Clone, Debug, PartialEq, Eq)]
628pub struct ScanDiagnostics {
629    /// Version of this diagnostic contract.
630    pub schema: &'static str,
631    /// Automatic worker-controller history and queue state.
632    pub worker_policy: WorkerPolicyDiagnostics,
633    /// Directory-enumeration backends used by this run.
634    pub backend: ScanBackendDiagnostics,
635}
636
637/// Repository-only controller variants used by the performance evidence probe.
638///
639/// These variants are not selected by [`scan`] or [`scan_with_diagnostics`]; both keep
640/// the shipped one-shot policy. The explicit experimental APIs make candidate behavior
641/// measurable without hiding a production change behind an environment variable.
642#[doc(hidden)]
643#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
644pub enum WorkerPolicyExperiment {
645    /// The production controller: one prefix window and at most one expansion.
646    #[default]
647    ShippedOneShot,
648    /// Re-evaluate independent windows until a slow phase requests the full reserve.
649    RepeatedWindows,
650    /// Re-evaluate independent windows, gate on useful frontier/handoff backlog, and grow
651    /// the pool in stages.
652    StagedGatedWindows,
653}
654
655/// Final state of the automatic worker controller.
656#[derive(Clone, Copy, Debug, PartialEq, Eq)]
657pub enum WorkerPolicyOutcome {
658    /// The scan did no walking, for example because `max_depth` was zero.
659    NotRun,
660    /// A fixed pool had no adaptive decision to make.
661    Fixed,
662    /// The walk ended before an adaptive window became observable.
663    Undecided,
664    /// The controller measured a window and retained the initial pool.
665    Held,
666    /// The controller requested and activated the reserve workers.
667    ScaledUp,
668    /// Slow work was observed only after no useful queued or in-flight work remained.
669    HeldNoUsefulWork,
670}
671
672/// One controller evaluation over a half-open range of completed entry ordinals.
673#[derive(Clone, Debug, PartialEq, Eq)]
674pub struct WorkerPolicyWindow {
675    /// Monotonic record number within this scan.
676    pub sequence: u64,
677    /// First completed-entry ordinal represented by this window, inclusive.
678    pub start_entry_ordinal: u64,
679    /// Ordinal immediately after the last represented entry.
680    pub end_entry_ordinal: u64,
681    /// Entries contributing to the service-time signal.
682    pub observed_entries: u64,
683    /// Completed directory claims contributing to the service-time signal.
684    pub observed_chunks: u64,
685    /// Worker time contributing to the service-time signal.
686    pub observed_work_ns: u64,
687    /// Derived service time, or null when no entry made the signal observable.
688    pub work_ns_per_entry: Option<u64>,
689    /// Why `work_ns_per_entry` is null.
690    pub work_ns_per_entry_unavailable_reason: Option<&'static str>,
691    /// Directories ready to claim when the controller evaluated the window.
692    pub ready_directories: usize,
693    /// Claimed directories still being processed at that point.
694    pub in_flight_directories: usize,
695    /// Live worker threads at that point, including workers waiting for a claim.
696    pub active_workers: usize,
697    /// Observation batches sent but not yet received by the consumer.
698    pub handoff_backlog: usize,
699    /// Worker target requested by a scale decision.
700    pub requested_workers: Option<usize>,
701    /// What the controller concluded from this window.
702    pub decision: WorkerPolicyDecision,
703}
704
705/// Decision represented by a [`WorkerPolicyWindow`].
706#[derive(Clone, Copy, Debug, PartialEq, Eq)]
707pub enum WorkerPolicyDecision {
708    /// The walk ended before the window could support a decision.
709    Undecided,
710    /// The observed window retained the current pool.
711    Hold,
712    /// The observed window activated reserve workers.
713    ScaleUp,
714    /// The trigger fired after all useful work had drained.
715    HoldNoUsefulWork,
716    /// A complete window held because reserve workers had no useful frontier to claim.
717    HoldInsufficientFrontier,
718    /// A complete window held because the unbounded handoff backlog was already high.
719    HoldHandoffBacklog,
720    /// A post-decision observation window remained below the slow threshold.
721    ObserveFast,
722    /// A post-decision observation window met the slow threshold.
723    ObserveSlow,
724    /// A trailing partial window carried no new terminal decision.
725    Incomplete,
726    /// A trailing partial post-decision observation carried no policy decision.
727    ObserveIncomplete,
728}
729
730/// Worker-controller configuration, trace, and terminal queue state.
731#[derive(Clone, Debug, PartialEq, Eq)]
732pub struct WorkerPolicyDiagnostics {
733    /// Controller variant exercised by this scan.
734    pub controller: &'static str,
735    /// Parallelism reported by the operating system when the scan began.
736    pub available_parallelism: usize,
737    /// Workers in the pool before any adaptive decision.
738    pub initial_workers: usize,
739    /// Hard maximum workers this scan could activate.
740    pub maximum_workers: usize,
741    /// Entry target for an adaptive window, or null for a fixed pool.
742    pub calibration_window_entries: Option<u64>,
743    /// Slow-service trigger, or null for a fixed pool.
744    pub slow_threshold_ns_per_entry: Option<u64>,
745    /// Directory chunks folded into live controller windows.
746    pub calibration_chunks: u64,
747    /// Entries folded into live controller windows.
748    pub calibration_entries: u64,
749    /// Worker time folded into live controller windows.
750    pub calibration_work_ns: u64,
751    /// Expansion messages that caused the consumer to create more workers.
752    pub worker_expansions: u64,
753    /// Terminal policy outcome.
754    pub outcome: WorkerPolicyOutcome,
755    /// Explanation when no adaptive evaluation exists.
756    pub outcome_reason: Option<&'static str>,
757    /// Total worker threads created during the walk.
758    pub workers_spawned: usize,
759    /// Maximum simultaneously live worker threads, including workers waiting for work.
760    pub peak_active_workers: usize,
761    /// Ready directories at scan completion; a complete walk must leave zero.
762    pub ready_directories_at_finish: usize,
763    /// In-flight directories at scan completion; a complete walk must leave zero.
764    pub in_flight_directories_at_finish: usize,
765    /// Observation batches outstanding at scan completion.
766    pub handoff_backlog_at_finish: usize,
767    /// Maximum outstanding observation batches during the scan.
768    pub handoff_backlog_high_water: usize,
769    /// Bounded controller history.
770    pub windows: Vec<WorkerPolicyWindow>,
771    /// True when controller history exceeded the 256-event diagnostic bound.
772    pub events_truncated: bool,
773}
774
775/// Directory enumeration backends used by one scan.
776#[derive(Clone, Debug, PartialEq, Eq)]
777pub struct ScanBackendDiagnostics {
778    /// Portable `read_dir` calls attempted.
779    pub portable_attempts: u64,
780    /// Portable directory listings completed successfully.
781    pub portable_directory_reads: u64,
782    /// macOS bulk enumeration attempts, or null off macOS.
783    pub macos_bulk_attempts: Option<u64>,
784    /// Successful macOS bulk listings, or null off macOS.
785    pub macos_bulk_successes: Option<u64>,
786    /// Bulk attempts that fell back to portable enumeration, or null off macOS.
787    pub macos_bulk_fallbacks: Option<u64>,
788    /// Why macOS fields are null.
789    pub unavailable_reason: Option<&'static str>,
790}
791
792impl ScanDiagnostics {
793    /// Serialize this versioned diagnostic contract as compact JSON.
794    ///
795    /// This deliberately lives beside the contract instead of in a benchmark binary:
796    /// claim-grade installed-command measurements and the repository probe must emit
797    /// byte-for-byte equivalent evidence without adding a serialization dependency to
798    /// the core crate.
799    pub fn to_json(&self) -> String {
800        let policy = &self.worker_policy;
801        let backend = &self.backend;
802        let mut windows = String::from("[");
803        for (index, window) in policy.windows.iter().enumerate() {
804            if index > 0 {
805                windows.push(',');
806            }
807            let _ = write!(
808                windows,
809                concat!(
810                    "{{\"active_workers\":{},\"decision\":\"{}\",",
811                    "\"end_entry_ordinal\":{},\"handoff_backlog\":{},",
812                    "\"in_flight_directories\":{},\"observed_chunks\":{},",
813                    "\"observed_entries\":{},",
814                    "\"observed_work_ns\":{},\"ready_directories\":{},",
815                    "\"requested_workers\":{},\"sequence\":{},\"start_entry_ordinal\":{},",
816                    "\"work_ns_per_entry\":{},",
817                    "\"work_ns_per_entry_unavailable_reason\":{}}}"
818                ),
819                window.active_workers,
820                worker_policy_decision_name(window.decision),
821                window.end_entry_ordinal,
822                window.handoff_backlog,
823                window.in_flight_directories,
824                window.observed_chunks,
825                window.observed_entries,
826                window.observed_work_ns,
827                window.ready_directories,
828                json_optional_usize(window.requested_workers),
829                window.sequence,
830                window.start_entry_ordinal,
831                json_optional_u64(window.work_ns_per_entry),
832                json_optional_string(window.work_ns_per_entry_unavailable_reason),
833            );
834        }
835        windows.push(']');
836        format!(
837            concat!(
838                "{{\"backend\":{{\"macos_bulk_attempts\":{},",
839                "\"macos_bulk_fallbacks\":{},\"macos_bulk_successes\":{},",
840                "\"portable_attempts\":{},\"portable_directory_reads\":{},",
841                "\"unavailable_reason\":{}}},",
842                "\"schema\":\"{}\",\"worker_policy\":{{",
843                "\"available_parallelism\":{},\"calibration_chunks\":{},",
844                "\"calibration_entries\":{},\"calibration_window_entries\":{},",
845                "\"calibration_work_ns\":{},",
846                "\"controller\":\"{}\",",
847                "\"events_truncated\":{},\"handoff_backlog_at_finish\":{},",
848                "\"handoff_backlog_high_water\":{},\"in_flight_directories_at_finish\":{},",
849                "\"initial_workers\":{},\"maximum_workers\":{},\"outcome\":\"{}\",",
850                "\"outcome_reason\":{},\"peak_active_workers\":{},",
851                "\"ready_directories_at_finish\":{},\"slow_threshold_ns_per_entry\":{},",
852                "\"windows\":{},\"worker_expansions\":{},\"workers_spawned\":{}}}}}"
853            ),
854            json_optional_u64(backend.macos_bulk_attempts),
855            json_optional_u64(backend.macos_bulk_fallbacks),
856            json_optional_u64(backend.macos_bulk_successes),
857            backend.portable_attempts,
858            backend.portable_directory_reads,
859            json_optional_string(backend.unavailable_reason),
860            self.schema,
861            policy.available_parallelism,
862            policy.calibration_chunks,
863            policy.calibration_entries,
864            json_optional_u64(policy.calibration_window_entries),
865            policy.calibration_work_ns,
866            policy.controller,
867            policy.events_truncated,
868            policy.handoff_backlog_at_finish,
869            policy.handoff_backlog_high_water,
870            policy.in_flight_directories_at_finish,
871            policy.initial_workers,
872            policy.maximum_workers,
873            worker_policy_outcome_name(policy.outcome),
874            json_optional_string(policy.outcome_reason),
875            policy.peak_active_workers,
876            policy.ready_directories_at_finish,
877            json_optional_u64(policy.slow_threshold_ns_per_entry),
878            windows,
879            policy.worker_expansions,
880            policy.workers_spawned,
881        )
882    }
883}
884
885const fn worker_policy_outcome_name(value: WorkerPolicyOutcome) -> &'static str {
886    match value {
887        WorkerPolicyOutcome::NotRun => "not_run",
888        WorkerPolicyOutcome::Fixed => "fixed",
889        WorkerPolicyOutcome::Undecided => "undecided",
890        WorkerPolicyOutcome::Held => "held",
891        WorkerPolicyOutcome::ScaledUp => "scaled_up",
892        WorkerPolicyOutcome::HeldNoUsefulWork => "held_no_useful_work",
893    }
894}
895
896const fn worker_policy_decision_name(value: WorkerPolicyDecision) -> &'static str {
897    match value {
898        WorkerPolicyDecision::Undecided => "undecided",
899        WorkerPolicyDecision::Hold => "hold",
900        WorkerPolicyDecision::ScaleUp => "scale_up",
901        WorkerPolicyDecision::HoldNoUsefulWork => "hold_no_useful_work",
902        WorkerPolicyDecision::HoldInsufficientFrontier => "hold_insufficient_frontier",
903        WorkerPolicyDecision::HoldHandoffBacklog => "hold_handoff_backlog",
904        WorkerPolicyDecision::ObserveFast => "observe_fast",
905        WorkerPolicyDecision::ObserveSlow => "observe_slow",
906        WorkerPolicyDecision::Incomplete => "incomplete",
907        WorkerPolicyDecision::ObserveIncomplete => "observe_incomplete",
908    }
909}
910
911fn json_optional_string(value: Option<&str>) -> String {
912    value.map_or_else(|| "null".into(), |value| format!("\"{}\"", json_escape(value)))
913}
914
915fn json_optional_u64(value: Option<u64>) -> String {
916    value.map_or_else(|| "null".into(), |value| value.to_string())
917}
918
919fn json_optional_usize(value: Option<usize>) -> String {
920    value.map_or_else(|| "null".into(), |value| value.to_string())
921}
922
923fn json_escape(value: &str) -> String {
924    let mut escaped = String::new();
925    for character in value.chars() {
926        match character {
927            '"' => escaped.push_str("\\\""),
928            '\\' => escaped.push_str("\\\\"),
929            '\u{08}' => escaped.push_str("\\b"),
930            '\u{0c}' => escaped.push_str("\\f"),
931            '\n' => escaped.push_str("\\n"),
932            '\r' => escaped.push_str("\\r"),
933            '\t' => escaped.push_str("\\t"),
934            character if character <= '\u{1f}' => {
935                let _ = write!(escaped, "\\u{:04x}", u32::from(character));
936            }
937            character => escaped.push(character),
938        }
939    }
940    escaped
941}
942
943/// Where a walk's time went, so "blocked" is never one undifferentiated number.
944///
945/// The performance loop's standing question is whether a walk is bound by disk I/O,
946/// by CPU, or by coordination, and process-level counters cannot answer it: user and
947/// system time say how much CPU was burned, but a fused "blocked" number cannot say
948/// whether workers were waiting on the filesystem, on the queue lock, or on nothing
949/// at all because the queue was empty. These counters split that out at the source.
950///
951/// Everything is measured in *chunks*, never per file: one timing pair per claimed
952/// run of directories, per contended lock, per batch handoff. On the 60k-entry
953/// reference tree that is a few thousand `Instant` reads against hundreds of
954/// milliseconds of walking — the instrumentation follows the same amortization rule
955/// it exists to verify.
956///
957/// In a parallel walk the fields sum over workers, so `wall_ns` is worker-seconds
958/// (it can exceed the scan's wall clock) and every other duration is a disjoint
959/// slice of it: `work_ns + starved_ns + lock_wait_ns + send_ns <= wall_ns`, with the
960/// remainder being uninstrumented odds and ends (uncontended lock ops, loop
961/// bookkeeping). A serial walk fills only `wall_ns`, `work_ns`, and `send_ns` —
962/// there is no coordination to attribute.
963#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
964pub struct WalkAttribution {
965    /// Total time workers spent in the walk loop, summed across workers.
966    pub wall_ns: u64,
967    /// Reading directories and stating entries — the real work, syscalls plus the
968    /// compute between them. Separating disk from CPU *within* this span needs the
969    /// process-level user/system counters alongside; per-syscall timing would break
970    /// the chunk-amortization rule.
971    pub work_ns: u64,
972    /// Waiting on the queue's condvar because no work was available. Starvation:
973    /// either the frontier is momentarily narrower than the worker pool, or the walk
974    /// is ending.
975    pub starved_ns: u64,
976    /// Waiting to acquire the queue lock when another worker held it. This is the
977    /// contention the shared-queue design bets stays negligible; now it is measured
978    /// instead of argued.
979    pub lock_wait_ns: u64,
980    /// Handing observation batches to the consumer: the channel send in a parallel
981    /// walk, the inline sink call — which is the consumer actually running — in a
982    /// serial one.
983    pub send_ns: u64,
984    /// Chunks of directories claimed from the queue.
985    pub claims: u64,
986    /// Queue lock acquisitions, contended or not.
987    pub lock_ops: u64,
988    /// Lock acquisitions that found the lock already held.
989    pub lock_contended: u64,
990}
991
992impl WalkAttribution {
993    /// Fold one worker's counters into the whole-walk totals.
994    fn absorb(&mut self, other: Self) {
995        self.wall_ns += other.wall_ns;
996        self.work_ns += other.work_ns;
997        self.starved_ns += other.starved_ns;
998        self.lock_wait_ns += other.lock_wait_ns;
999        self.send_ns += other.send_ns;
1000        self.claims += other.claims;
1001        self.lock_ops += other.lock_ops;
1002        self.lock_contended += other.lock_contended;
1003    }
1004
1005    /// Time attributed to a named cause, as opposed to `wall_ns`'s total.
1006    pub fn accounted_ns(&self) -> u64 {
1007        self.work_ns + self.starved_ns + self.lock_wait_ns + self.send_ns
1008    }
1009}
1010
1011/// Filesystem and index effects from an applying reconciliation pass.
1012#[derive(Debug, Default)]
1013pub struct ReconcileReport {
1014    /// Filesystem walk effects and partial errors.
1015    ///
1016    /// [`ScanReport::attribution`] remains zero for reconciliation because neither the
1017    /// serial nor parallel path has complete, comparable instrumentation yet. Zero
1018    /// means "not measured" here, not "no work".
1019    pub scan: ScanReport,
1020    /// Index arbitration and mutation effects.
1021    pub apply: ApplyStats,
1022    /// Exact producer operations considered, including no-op controls that do not
1023    /// increment an effect counter or create a commit.
1024    pub(crate) observations: u64,
1025    /// Directories this pass listed in full, with no error inside them, that the index did
1026    /// not yet hold as complete.
1027    ///
1028    /// The closing commit records each one's child set as authoritative, as discovery's
1029    /// own listing commit does, whether or not the rest of the pass completed: one transient
1030    /// child error
1031    /// elsewhere used to keep every directory the pass listed incomplete, and a directory
1032    /// first listed by such a pass stayed `Unknown { Building }` under a complete root.
1033    pub(crate) listed_incomplete: Vec<PathBuf>,
1034    /// Ownership epoch for conditional reconciliation batches.
1035    reconcile_epoch: Option<u64>,
1036    /// Retry after bounded verification evidence was superseded.
1037    retry_required: bool,
1038}
1039
1040impl ReconcileReport {
1041    /// True when the filesystem walk was complete and no conditional observation lost
1042    /// a race with another producer.
1043    pub fn is_complete(&self) -> bool {
1044        self.scan.is_complete()
1045            && self.apply.stale == 0
1046            && self.apply.resource_refused == 0
1047            && !self.retry_required
1048    }
1049
1050    /// True when a newer verification retired this pass's bounded evidence before it
1051    /// closed, so its scope is published partial and must be walked again.
1052    pub(crate) const fn retry_required(&self) -> bool {
1053        self.retry_required
1054    }
1055
1056    /// The directories whose listings this pass can vouch for, taken out of the report.
1057    ///
1058    /// None when a conditional commit lost a race or was refused: a child of any listed
1059    /// directory may then be missing from the index until the retry that race earns, and
1060    /// the retry records completeness for what it lists.
1061    pub(crate) fn take_recordable_completeness(&mut self) -> Vec<PathBuf> {
1062        let listed = std::mem::take(&mut self.listed_incomplete);
1063        if self.apply.stale > 0 || self.apply.resource_refused > 0 { Vec::new() } else { listed }
1064    }
1065}
1066
1067enum ReconcileTarget<'a> {
1068    Direct(&'a mut Index),
1069    Shared(&'a IndexHandle),
1070    Controlled { handle: &'a IndexHandle, control: &'a dyn ReconcileControl },
1071}
1072
1073/// Lifecycle checkpoints used by an owned long-running reconciliation.
1074///
1075/// The ordinary one-shot APIs use no controller. An [`crate::OpenedIndex`] supplies one
1076/// so close can stop a refresh before another write, and deterministic tests can pause
1077/// after filesystem verification but before conditional arbitration.
1078pub(crate) trait ReconcileControl {
1079    /// Fail when the owning operation may no longer publish state.
1080    fn check_active(&self) -> Result<()>;
1081
1082    /// Boundary after filesystem verification and before a conditional fact commit.
1083    fn before_conditional_commit(&self) -> Result<()>;
1084
1085    /// Atomic file-retention limit shared with every producer for this opened root.
1086    fn max_files(&self) -> Option<u64>;
1087}
1088
1089impl ReconcileTarget<'_> {
1090    fn scope(&self) -> Result<ScanScope> {
1091        match self {
1092            Self::Direct(index) => Ok(index.scope()),
1093            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.scope(),
1094        }
1095    }
1096
1097    fn root_path(&self) -> Result<PathBuf> {
1098        match self {
1099            Self::Direct(index) => Ok(index.root_path().to_path_buf()),
1100            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.root_path(),
1101        }
1102    }
1103
1104    fn expectation(&self, path: &Path) -> Result<PathExpectation> {
1105        match self {
1106            Self::Direct(index) => Ok(index.expectation(path)),
1107            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.expectation(path),
1108        }
1109    }
1110
1111    fn child_states(&self, path: &Path) -> Result<BTreeMap<OsString, PathExpectation>> {
1112        match self {
1113            Self::Direct(index) => Ok(collect_child_expectations(index, path)),
1114            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.child_states(path),
1115        }
1116    }
1117
1118    /// Child baselines for one directory listing, and whether a complete listing of it
1119    /// would be news to the index's directory completeness.
1120    ///
1121    /// A directory whose upsert has not been flushed yet is not held at all and counts as
1122    /// incomplete.
1123    fn listing_baseline(&self, path: &Path) -> Result<(BTreeMap<OsString, PathExpectation>, bool)> {
1124        match self {
1125            Self::Direct(index) => Ok((
1126                collect_child_expectations(index, path),
1127                index.directory_complete(path) != Some(true),
1128            )),
1129            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.listing_baseline(path),
1130        }
1131    }
1132
1133    fn has_control(&self, path: &Path) -> Result<bool> {
1134        match self {
1135            Self::Direct(index) => Ok(index.control_table().contains(path)),
1136            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.has_control(path),
1137        }
1138    }
1139
1140    fn control_table(&self) -> Result<crate::control::ControlTable> {
1141        match self {
1142            Self::Direct(index) => Ok(index.control_table().clone()),
1143            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1144                handle.read_with(|index| index.control_table().clone())
1145            }
1146        }
1147    }
1148
1149    fn control_classification_known(&self, path: &Path) -> Result<bool> {
1150        match self {
1151            Self::Direct(index) => Ok(index.control_classification_known(path)),
1152            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1153                handle.read_with(|index| index.control_classification_known(path))
1154            }
1155        }
1156    }
1157
1158    fn apply(&mut self, started_at: u64, observation: &Observation) -> Result<crate::ApplyOutcome> {
1159        match self {
1160            Self::Direct(index) => index.apply(observation),
1161            Self::Shared(handle) => handle.apply_reconcile(started_at, observation),
1162            Self::Controlled { handle, control } => {
1163                control.before_conditional_commit()?;
1164                handle.apply_opened_reconcile(started_at, observation, control.max_files())
1165            }
1166        }
1167    }
1168
1169    fn direct_upsert_is_unchanged(
1170        &self,
1171        baseline: PathExpectation,
1172        kind: EntryKind,
1173        attrs: Attrs,
1174    ) -> bool {
1175        matches!(self, Self::Direct(_)) && baseline.state == (PathState::Present { kind, attrs })
1176    }
1177
1178    fn take_pending_invalidations(&mut self) -> Result<Vec<(PathBuf, crate::InvalidateReason)>> {
1179        match self {
1180            Self::Direct(index) => Ok(index.take_pending_invalidations()),
1181            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1182                handle.take_pending_invalidations()
1183            }
1184        }
1185    }
1186
1187    fn restore_pending_invalidations(
1188        &mut self,
1189        invalidations: Vec<(PathBuf, crate::InvalidateReason)>,
1190    ) -> Result<()> {
1191        match self {
1192            Self::Direct(index) => index.restore_pending_invalidations(invalidations),
1193            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1194                handle.restore_pending_invalidations(invalidations)?;
1195            }
1196        }
1197        Ok(())
1198    }
1199
1200    /// Whether an invalidation whose reconciliation came back incomplete is queued again.
1201    ///
1202    /// A caller of the one-shot API owns its index exclusively and drains the queue when it
1203    /// chooses, so an unreadable subtree stays queued for it to retry. The shared API and an
1204    /// opened root are drained after every observed event -- by `Watcher::apply_next` and by
1205    /// the opened root's observer -- where that retry is a full walk of the same unreadable
1206    /// subtree per unrelated event, for the life of the session. There only a lost race is
1207    /// worth retrying: a stale conditional commit, or one the budget refused. A scan error
1208    /// is a settled boundary: the subtree stays partial, as it does at the observation
1209    /// handoff, and the report names the error once.
1210    fn retries_incomplete(&self, report: &ReconcileReport) -> bool {
1211        match self {
1212            Self::Direct(_) => !report.is_complete(),
1213            Self::Shared(_) | Self::Controlled { .. } => {
1214                report.apply.stale > 0 || report.apply.resource_refused > 0 || report.retry_required
1215            }
1216        }
1217    }
1218
1219    fn begin_reconcile(&mut self, path: &Path) -> Result<(u64, Option<Commit>)> {
1220        match self {
1221            Self::Direct(index) => index.begin_reconcile(path),
1222            Self::Shared(handle) => handle.begin_reconcile(path),
1223            Self::Controlled { handle, control } => {
1224                control.check_active()?;
1225                handle.begin_reconcile(path)
1226            }
1227        }
1228    }
1229
1230    fn finish_reconcile(
1231        &mut self,
1232        path: &Path,
1233        started_at: u64,
1234        complete: bool,
1235        listed_incomplete: &[PathBuf],
1236        failed_paths: &[PathBuf],
1237        errors: ReconcileErrors<'_>,
1238    ) -> Result<ReconcileFinish> {
1239        match self {
1240            Self::Direct(index) => index.finish_reconcile(
1241                path,
1242                started_at,
1243                complete,
1244                listed_incomplete,
1245                failed_paths,
1246                errors,
1247            ),
1248            Self::Shared(handle) => handle.finish_reconcile(
1249                path,
1250                started_at,
1251                complete,
1252                listed_incomplete,
1253                failed_paths,
1254                errors,
1255            ),
1256            Self::Controlled { handle, control } => {
1257                control.check_active()?;
1258                handle.finish_reconcile(
1259                    path,
1260                    started_at,
1261                    complete,
1262                    listed_incomplete,
1263                    failed_paths,
1264                    errors,
1265                )
1266            }
1267        }
1268    }
1269}
1270
1271#[cfg(unix)]
1272pub(crate) fn metadata_for_fingerprint(entry: &fs::DirEntry) -> std::io::Result<fs::Metadata> {
1273    crate::counters::bump(|c| c.stats += 1);
1274    entry.metadata()
1275}
1276
1277#[cfg(any(all(windows, test), not(any(unix, windows))))]
1278pub(crate) fn metadata_for_fingerprint(entry: &fs::DirEntry) -> std::io::Result<fs::Metadata> {
1279    crate::counters::bump(|c| c.stats += 1);
1280    // Windows serves DirEntry metadata from directory-enumeration data, which the
1281    // platform permits to be stale. Fingerprints need a fresh non-following query.
1282    fs::symlink_metadata(entry.path())
1283}
1284
1285#[cfg(test)]
1286type WalkHook = std::sync::Arc<dyn Fn(WalkHookPoint<'_>) -> Option<std::io::Error> + Send + Sync>;
1287
1288/// Where a test hook runs in a listing walk.
1289#[cfg(test)]
1290#[derive(Clone, Copy, Debug)]
1291pub(crate) enum WalkHookPoint<'a> {
1292    /// Before the metadata lookup of the listed child at this absolute path; an error
1293    /// stands in for the lookup's.
1294    ChildMetadata(&'a Path),
1295    /// After a reconciliation's listing of a directory returns its last entry; an error is
1296    /// read as one more listing item, which leaves the listing incomplete.
1297    ListingEnd,
1298}
1299
1300/// Hooks run at each [`WalkHookPoint`], each for the paths under its root.
1301///
1302/// Process-wide, because a parallel walk looks children up on its worker threads; keyed
1303/// by root, because tests run in parallel and each walks its own temporary directory.
1304#[cfg(test)]
1305static WALK_HOOKS: std::sync::RwLock<Vec<(Vec<PathBuf>, WalkHook)>> =
1306    std::sync::RwLock::new(Vec::new());
1307
1308/// Removes its hook from [`WALK_HOOKS`] when dropped.
1309#[cfg(test)]
1310#[must_use = "the hook is removed as soon as the guard is dropped"]
1311pub(crate) struct WalkHookGuard(WalkHook);
1312
1313#[cfg(test)]
1314impl Drop for WalkHookGuard {
1315    fn drop(&mut self) {
1316        WALK_HOOKS
1317            .write()
1318            .unwrap_or_else(std::sync::PoisonError::into_inner)
1319            .retain(|(_, hook)| !std::sync::Arc::ptr_eq(hook, &self.0));
1320    }
1321}
1322
1323/// Run `hook` at every [`WalkHookPoint`] under `root`, on any thread, until the guard drops.
1324///
1325/// The hook may also change the tree before it returns. `root` matches as given and
1326/// canonical, since an opened root and a detached scan walk the canonical path.
1327#[cfg(test)]
1328pub(crate) fn install_walk_hook(
1329    root: &Path,
1330    hook: impl Fn(WalkHookPoint<'_>) -> Option<std::io::Error> + Send + Sync + 'static,
1331) -> WalkHookGuard {
1332    let mut roots = vec![root.to_path_buf()];
1333    if let Ok(canonical) = root.canonicalize() {
1334        roots.push(canonical);
1335    }
1336    let hook: WalkHook = std::sync::Arc::new(hook);
1337    WALK_HOOKS
1338        .write()
1339        .unwrap_or_else(std::sync::PoisonError::into_inner)
1340        .push((roots, std::sync::Arc::clone(&hook)));
1341    WalkHookGuard(hook)
1342}
1343
1344/// Run `hook` before every listed child's metadata lookup under `root`, with the child's
1345/// absolute path, until the guard drops.
1346#[cfg(test)]
1347pub(crate) fn install_child_metadata_hook(
1348    root: &Path,
1349    hook: impl Fn(&Path) -> Option<std::io::Error> + Send + Sync + 'static,
1350) -> WalkHookGuard {
1351    install_walk_hook(root, move |point| match point {
1352        WalkHookPoint::ChildMetadata(path) => hook(path),
1353        WalkHookPoint::ListingEnd => None,
1354    })
1355}
1356
1357/// The hook installed for a root containing `path`, if any.
1358#[cfg(test)]
1359fn walk_hook(path: &Path) -> Option<WalkHook> {
1360    WALK_HOOKS
1361        .read()
1362        .unwrap_or_else(std::sync::PoisonError::into_inner)
1363        .iter()
1364        .find(|(roots, _)| roots.iter().any(|root| path.starts_with(root)))
1365        .map(|(_, hook)| std::sync::Arc::clone(hook))
1366}
1367
1368/// A reconciliation's listing of `dir`, followed by any error a test hook injects.
1369///
1370/// Callers bind the result to `listing` and iterate it as `for … in listing`, because the
1371/// admission audit (`scripts/check-admission-sites.mjs`) counts routed listing loops by that
1372/// shape. Keep the binding when editing a call site; dropping it silently removes the loop
1373/// from the audit, whose expected count would then look too high rather than wrong.
1374#[cfg(test)]
1375fn reconcile_listing(
1376    listing: fs::ReadDir,
1377    dir: &Path,
1378) -> impl Iterator<Item = std::io::Result<fs::DirEntry>> {
1379    let injected = walk_hook(dir).and_then(|hook| hook(WalkHookPoint::ListingEnd));
1380    listing.chain(injected.map(Err))
1381}
1382
1383/// A reconciliation's listing of `dir`.
1384#[cfg(not(test))]
1385fn reconcile_listing(listing: fs::ReadDir, _dir: &Path) -> fs::ReadDir {
1386    listing
1387}
1388
1389/// Metadata for one entry a directory listing returned, or `None` when it is gone.
1390///
1391/// `NotFound` for a name the listing just returned means the entry was deleted in
1392/// between, and every walk records it as it records a name the listing never returned: a
1393/// cold walk has nothing to record, and a reconciliation removes what its baseline held.
1394/// Reported as an error, it would make a walk over a tree being cleaned partial, and in a
1395/// reconciliation it would settle as a phantom entry with permanent partial freshness.
1396/// Any other error means the entry is present but unreadable.
1397#[cfg(not(windows))]
1398pub(crate) fn listed_child_metadata(entry: &fs::DirEntry) -> std::io::Result<Option<fs::Metadata>> {
1399    #[cfg(test)]
1400    {
1401        let path = entry.path();
1402        if let Some(error) =
1403            walk_hook(&path).and_then(|hook| hook(WalkHookPoint::ChildMetadata(&path)))
1404        {
1405            return missing_as_none(Err(error));
1406        }
1407    }
1408    missing_as_none(metadata_for_fingerprint(entry))
1409}
1410
1411/// Kind and attributes for one listed child.
1412///
1413/// The transient summary fold counts directories and ignores symlink attributes, so a
1414/// listing `file_type` (`d_type` on Linux) is enough for those kinds when the walk is
1415/// not bound to one filesystem. Files and specials still need a metadata lookup for
1416/// size, allocated bytes, and mtime. `one_filesystem` still stats directories because
1417/// descent compares `attrs.dev` to the root device, and `dev == 0` would otherwise
1418/// cross a mount.
1419///
1420/// Where `d_type` is `DT_UNKNOWN` (XFS without `ftype`, some FUSE/NFS mounts, older
1421/// ext3), std's `file_type` performs the non-following stat itself. The skip is a
1422/// no-op there, and the `stats` counter does not see that fallback.
1423///
1424/// Windows never takes the skip: its observation contract reads every listed entry
1425/// through a fresh non-following handle ([`observe_dir_entry`]), so the transient fold
1426/// there performs exactly the observations the retained walk performs.
1427fn listed_child_kind_and_attrs(
1428    entry: &fs::DirEntry,
1429    skip_dir_symlink_stat: bool,
1430    one_filesystem: bool,
1431) -> std::io::Result<Option<(EntryKind, Attrs)>> {
1432    #[cfg(not(windows))]
1433    {
1434        if skip_dir_symlink_stat {
1435            if let Ok(file_type) = entry.file_type() {
1436                if file_type.is_dir() && !one_filesystem {
1437                    return Ok(Some((EntryKind::Dir, Attrs::default())));
1438                }
1439                if file_type.is_symlink() {
1440                    return Ok(Some((EntryKind::Symlink, Attrs::default())));
1441                }
1442            }
1443        }
1444    }
1445    #[cfg(windows)]
1446    {
1447        let _ = (skip_dir_symlink_stat, one_filesystem);
1448    }
1449    observe_dir_entry(entry)
1450}
1451
1452fn missing_as_none<T>(lookup: std::io::Result<T>) -> std::io::Result<Option<T>> {
1453    match lookup {
1454        Ok(metadata) => Ok(Some(metadata)),
1455        Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(None),
1456        Err(error) => Err(error),
1457    }
1458}
1459
1460/// Whether a test hook observes lookups or listings under `path`, which a bulk read would
1461/// not make.
1462#[cfg(all(test, target_os = "macos"))]
1463fn walk_hook_covers(path: &Path) -> bool {
1464    walk_hook(path).is_some()
1465}
1466
1467#[cfg(all(not(test), target_os = "macos"))]
1468const fn walk_hook_covers(_path: &Path) -> bool {
1469    false
1470}
1471
1472/// Owned output from the filesystem walker before it crosses a public mutation boundary.
1473///
1474/// Only the scan and opened-discovery producers construct this type. Their admission,
1475/// depth, filesystem, and symlink checks have already selected every operation, and the
1476/// index consumes the owned paths while proving their parent identities under its write
1477/// boundary. Public scan callers receive an [`Observation`] instead and therefore keep
1478/// the full public normalization and atomic-validation contract.
1479#[derive(Debug)]
1480pub(crate) struct ScannerBatch {
1481    ops: Vec<ObservationOp>,
1482    /// When set, the consumer must return `ops` through this sender instead of dropping
1483    /// them. Workers allocate the `PathBuf`s; returning the drained vec lets glibc free
1484    /// those arenas on the producing thread. The public [`scan`] path leaves this unset.
1485    recycle: Option<std::sync::mpsc::Sender<Vec<ObservationOp>>>,
1486}
1487
1488impl ScannerBatch {
1489    pub(crate) const fn new(ops: Vec<ObservationOp>) -> Self {
1490        Self { ops, recycle: None }
1491    }
1492
1493    fn with_recycle(self, recycle: std::sync::mpsc::Sender<Vec<ObservationOp>>) -> Self {
1494        Self { recycle: Some(recycle), ..self }
1495    }
1496
1497    #[cfg(test)]
1498    pub(crate) fn from_ops(ops: Vec<Op>) -> Self {
1499        Self { ops: ops.into_iter().map(ObservationOp::unconditional).collect(), recycle: None }
1500    }
1501
1502    pub(crate) fn len(&self) -> usize {
1503        self.ops.len()
1504    }
1505
1506    pub(crate) fn ops(&self) -> &[ObservationOp] {
1507        &self.ops
1508    }
1509
1510    pub(crate) fn into_ops(self) -> Vec<ObservationOp> {
1511        self.ops
1512    }
1513
1514    fn into_observation(self) -> Observation {
1515        Observation::from_ops(self.ops)
1516    }
1517
1518    fn recycle(self) {
1519        if let Some(recycle) = self.recycle {
1520            let _ = recycle.send(self.ops);
1521        }
1522    }
1523}
1524
1525/// One direct child retained by the private detached cold-bootstrap builder.
1526///
1527/// The worker owns the component once. Unlike [`ScannerBatch`], this record does not
1528/// manufacture a full relative path or a public observation for every entry.
1529#[derive(Debug)]
1530pub(crate) struct DetachedChild {
1531    pub(crate) name: OsString,
1532    pub(crate) kind: EntryKind,
1533    pub(crate) attrs: Attrs,
1534    /// Enumeration order within the listing. An enumerator can repeat a name while its
1535    /// directory is modified, and the builder keeps the later observation, as a
1536    /// streaming re-upsert does.
1537    pub(crate) position: u32,
1538}
1539
1540/// One directory listing retained by a worker for detached bootstrap consolidation.
1541///
1542/// `path` is paid once per directory. Its children remain grouped exactly as the
1543/// filesystem enumerator produced them, so consolidation resolves the parent once and
1544/// never reconstructs a child path for nondirectories. A fixed control is retained
1545/// separately so the consumer can install the directory's complete control state
1546/// before it classifies any sibling or makes descendants visible.
1547#[derive(Debug)]
1548pub(crate) struct DetachedDirectory {
1549    pub(crate) path: PathBuf,
1550    pub(crate) children: Vec<DetachedChild>,
1551    pub(crate) control: Option<Op>,
1552}
1553
1554/// Walk `root` and emit observations describing everything found.
1555pub fn scan(
1556    root: &Path,
1557    config: &ScanConfig,
1558    sink: &mut dyn FnMut(Observation),
1559) -> Result<ScanReport> {
1560    let mut public_sink = |batch: ScannerBatch| sink(batch.into_observation());
1561    let (mut report, _diagnostics) = scan_internal(
1562        root,
1563        config,
1564        &mut public_sink,
1565        false,
1566        WorkerPolicyExperiment::ShippedOneShot,
1567        SinkMode::Retained,
1568    )?;
1569    normalize_walk_errors(root, &mut report.errors);
1570    Ok(report)
1571}
1572
1573/// Walk `root` for the transient summary tier, folding each op without retaining it.
1574///
1575/// The public [`scan`] path hands each batch to the caller as an [`Observation`], so
1576/// worker-allocated `PathBuf`s are freed on the consumer thread. This path returns
1577/// drained batches to the producing worker so each arena is allocated and freed on one
1578/// thread. Tallies must match [`scan`], and so must the normalized error set: the
1579/// summary report's status is built from these errors exactly as a retained walk's is.
1580pub(crate) fn scan_summary_fold(
1581    root: &Path,
1582    config: &ScanConfig,
1583    fold: &mut dyn FnMut(&ObservationOp),
1584) -> Result<ScanReport> {
1585    let mut sink = |batch: ScannerBatch| {
1586        for op in batch.ops() {
1587            fold(op);
1588        }
1589        batch.recycle();
1590    };
1591    let (mut report, _diagnostics) = scan_internal(
1592        root,
1593        config,
1594        &mut sink,
1595        false,
1596        WorkerPolicyExperiment::ShippedOneShot,
1597        SinkMode::TransientFold,
1598    )?;
1599    normalize_walk_errors(root, &mut report.errors);
1600    Ok(report)
1601}
1602
1603/// [`scan_summary_fold`] plus the diagnostic trace [`scan_with_diagnostics`] collects.
1604pub(crate) fn scan_summary_fold_with_diagnostics(
1605    root: &Path,
1606    config: &ScanConfig,
1607    fold: &mut dyn FnMut(&ObservationOp),
1608) -> Result<(ScanReport, ScanDiagnostics)> {
1609    let mut sink = |batch: ScannerBatch| {
1610        for op in batch.ops() {
1611            fold(op);
1612        }
1613        batch.recycle();
1614    };
1615    let (mut report, diagnostics) = scan_internal(
1616        root,
1617        config,
1618        &mut sink,
1619        true,
1620        WorkerPolicyExperiment::ShippedOneShot,
1621        SinkMode::TransientFold,
1622    )?;
1623    normalize_walk_errors(root, &mut report.errors);
1624    Ok((report, diagnostics.expect("diagnostic scan creates a recorder")))
1625}
1626
1627/// Walk `root`, emitting observations and a bounded run-scoped diagnostic trace.
1628///
1629/// This is the measurement counterpart to [`scan`]. It produces the same observation
1630/// stream and report while recording controller and backend evidence that ordinary
1631/// scans intentionally do not collect.
1632pub fn scan_with_diagnostics(
1633    root: &Path,
1634    config: &ScanConfig,
1635    sink: &mut dyn FnMut(Observation),
1636) -> Result<(ScanReport, ScanDiagnostics)> {
1637    scan_with_policy_diagnostics(root, config, sink, WorkerPolicyExperiment::ShippedOneShot)
1638}
1639
1640/// Exercise a repository-only worker-controller candidate and retain its trace.
1641#[doc(hidden)]
1642pub fn scan_with_policy_diagnostics(
1643    root: &Path,
1644    config: &ScanConfig,
1645    sink: &mut dyn FnMut(Observation),
1646    policy: WorkerPolicyExperiment,
1647) -> Result<(ScanReport, ScanDiagnostics)> {
1648    let mut public_sink = |batch: ScannerBatch| sink(batch.into_observation());
1649    let (mut report, diagnostics) =
1650        scan_internal(root, config, &mut public_sink, true, policy, SinkMode::Retained)?;
1651    normalize_walk_errors(root, &mut report.errors);
1652    Ok((report, diagnostics.expect("diagnostic scan creates a recorder")))
1653}
1654
1655/// What the caller does with each batch of observations.
1656///
1657/// Two measured keeps hang off this one concept, and both were measured on the
1658/// transient fold alone: returning drained batches to the producing worker (H147,
1659/// exp-151) and taking directory and symlink kind from the listing without a stat
1660/// (H72, exp-153). They are named here as properties of the mode rather than passed as
1661/// one flag under one of their names, so a measurement on another platform can move
1662/// one without silently moving the other.
1663#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1664enum SinkMode {
1665    /// The consumer keeps the observations: the public [`scan`] and the index.
1666    Retained,
1667    /// The consumer folds each batch and drops it: the transient summary tier.
1668    TransientFold,
1669}
1670
1671impl SinkMode {
1672    /// Drained batches go back to the worker that allocated them (H147).
1673    fn recycles_batches(self) -> bool {
1674        self == Self::TransientFold
1675    }
1676
1677    /// Directory and symlink kind come from the listing without a stat (H72).
1678    fn skips_dir_symlink_stat(self) -> bool {
1679        self == Self::TransientFold
1680    }
1681}
1682
1683fn scan_internal(
1684    root: &Path,
1685    config: &ScanConfig,
1686    sink: &mut dyn FnMut(ScannerBatch),
1687    collect_diagnostics: bool,
1688    policy: WorkerPolicyExperiment,
1689    sink_mode: SinkMode,
1690) -> Result<(ScanReport, Option<ScanDiagnostics>)> {
1691    config.validate()?;
1692    if let Some(progress) = &config.progress {
1693        progress.enter(crate::ProgressPhase::Scanning);
1694    }
1695    let root_meta = {
1696        crate::counters::bump(|c| c.stats += 1);
1697        fs::symlink_metadata(root)
1698    }
1699    .map_err(|e| Error::io(root, e))?;
1700    if !root_meta.is_dir() {
1701        return Err(Error::io(
1702            root,
1703            std::io::Error::new(std::io::ErrorKind::NotADirectory, "scan root is not a directory"),
1704        ));
1705    }
1706    let root_dev = root_device(root, &root_meta).map_err(|error| Error::io(root, error))?;
1707    let available_parallelism =
1708        std::thread::available_parallelism().map_or(1, std::num::NonZero::get);
1709    // Narrow populations need control admission before child enumeration. Keep one ordered
1710    // producer and control table so a bounded budget makes the same decisions as the
1711    // index that consumes the observations.
1712    let pool = if config.population == crate::query::IgnoredEntries::Include {
1713        config.worker_pool_for(available_parallelism)
1714    } else {
1715        WorkerPool::fixed(1)
1716    };
1717    let diagnostics = collect_diagnostics
1718        .then(|| ScanDiagnosticsRecorder::new(pool, available_parallelism, policy));
1719
1720    if config.max_depth != Some(0) && pool.initial > 1 {
1721        let report = scan_concurrent(
1722            root,
1723            config,
1724            root_dev,
1725            sink,
1726            pool,
1727            diagnostics.as_ref(),
1728            policy,
1729            sink_mode,
1730        );
1731        return Ok((report, diagnostics.as_ref().map(|value| value.finish())));
1732    }
1733
1734    let mut report = ScanReport::default();
1735    if config.max_depth == Some(0) {
1736        if let Some(diagnostics) = &diagnostics {
1737            diagnostics.mark_not_run();
1738            diagnostics.record_queue_finish(0, 0);
1739        }
1740        return Ok((report, diagnostics.as_ref().map(|value| value.finish())));
1741    }
1742    let worker_guard = diagnostics.as_ref().map(ScanDiagnosticsRecorder::worker_guard);
1743    let walk_started = std::time::Instant::now();
1744    let mut batch: Vec<ObservationOp> = Vec::with_capacity(config.batch_size);
1745    let mut queue: VecDeque<(PathBuf, usize)> = VecDeque::from(vec![(PathBuf::new(), 0)]);
1746    let mut controls = (config.population != crate::query::IgnoredEntries::Include)
1747        .then(|| crate::control::ControlTable::with_limits(config.control_limits));
1748    let mut unreadable_controls = std::collections::BTreeSet::new();
1749    let mut tally = ProgressTally::new(config.progress.as_ref());
1750    // Every batch leaves through here, so the batch is where the serial walk reports
1751    // its progress: the handoff the consumer already pays for, never the entry.
1752    let mut emit = |ops: Vec<ObservationOp>, report: &mut ScanReport| {
1753        let send_started = std::time::Instant::now();
1754        sink(ScannerBatch::new(ops));
1755        report.attribution.send_ns += elapsed_ns(send_started);
1756        tally.flush(report);
1757    };
1758
1759    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
1760        let abs_dir = root.join(&rel_dir);
1761        if let Some(controls) = controls.as_mut() {
1762            let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
1763            match read_directory_control(config, root, &control_path) {
1764                Ok(Some(op)) => {
1765                    apply_discovery_control(controls, &op)?;
1766                    batch.push(ObservationOp::unconditional(op));
1767                }
1768                Ok(None) => {}
1769                Err(error) => {
1770                    unreadable_controls.insert(rel_dir.clone());
1771                    report.errors.push(error);
1772                }
1773            }
1774        }
1775        crate::counters::bump(|c| c.dir_opens += 1);
1776        if let Some(diagnostics) = &diagnostics {
1777            diagnostics.portable_attempted();
1778        }
1779        let listing = match fs::read_dir(&abs_dir) {
1780            Ok(listing) => {
1781                if let Some(diagnostics) = &diagnostics {
1782                    diagnostics.portable_succeeded();
1783                }
1784                listing
1785            }
1786            Err(e) => {
1787                report.errors.push(Error::io(abs_dir, e));
1788                continue;
1789            }
1790        };
1791        report.dirs_read += 1;
1792
1793        for item in listing {
1794            let item = match item {
1795                Ok(item) => item,
1796                Err(e) => {
1797                    report.errors.push(Error::io(&abs_dir, e));
1798                    continue;
1799                }
1800            };
1801            crate::counters::bump(|c| c.dir_entries += 1);
1802            let name = item.file_name();
1803            let rel_path = rel_dir.join(&name);
1804            let (kind, attrs) = match listed_child_kind_and_attrs(
1805                &item,
1806                sink_mode.skips_dir_symlink_stat(),
1807                config.one_filesystem,
1808            ) {
1809                Ok(Some(observed)) => observed,
1810                Ok(None) => continue,
1811                Err(error) => {
1812                    report.errors.push(Error::io(item.path(), error));
1813                    continue;
1814                }
1815            };
1816            let disposition =
1817                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
1818            if disposition == crate::admission::Disposition::Reject {
1819                continue;
1820            }
1821            if population_prunes(
1822                config.population,
1823                &rel_path,
1824                kind,
1825                disposition,
1826                controls.as_ref(),
1827                &unreadable_controls,
1828            ) {
1829                continue;
1830            }
1831            let control = match if controls.is_some() && name == crate::control::CONTROL_FILE_NAME {
1832                Ok(None)
1833            } else {
1834                read_control_op(config, root, &rel_path, kind)
1835            } {
1836                Ok(control) => control,
1837                Err(error) => {
1838                    report.errors.push(error);
1839                    None
1840                }
1841            };
1842            if disposition == crate::admission::Disposition::ControlOnly {
1843                if let Some(control) = control {
1844                    batch.push(ObservationOp::unconditional(control));
1845                    if batch.len() >= config.batch_size {
1846                        emit(std::mem::take(&mut batch), &mut report);
1847                        batch.reserve(config.batch_size);
1848                    }
1849                }
1850                continue;
1851            }
1852            report.observe(kind, attrs);
1853            batch.push(ObservationOp::unconditional(Op::Upsert {
1854                path: rel_path.clone(),
1855                kind,
1856                attrs,
1857            }));
1858            if batch.len() >= config.batch_size {
1859                emit(std::mem::take(&mut batch), &mut report);
1860                batch.reserve(config.batch_size);
1861            }
1862            if let Some(control) = control {
1863                batch.push(ObservationOp::unconditional(control));
1864                if batch.len() >= config.batch_size {
1865                    emit(std::mem::take(&mut batch), &mut report);
1866                    batch.reserve(config.batch_size);
1867                }
1868            }
1869
1870            if should_descend(kind, attrs, depth, root_dev, config) {
1871                queue.push_back((rel_path, depth + 1));
1872            }
1873        }
1874    }
1875
1876    if !batch.is_empty() {
1877        emit(batch, &mut report);
1878    }
1879    // A walk whose last directories filled no batch has counted them and sent nothing.
1880    tally.flush(&report);
1881    // A serial walk has no coordination to attribute: wall is the loop, "send" is the
1882    // inline sink — which is the consumer actually running — and work is the rest.
1883    report.attribution.wall_ns = elapsed_ns(walk_started);
1884    report.attribution.work_ns =
1885        report.attribution.wall_ns.saturating_sub(report.attribution.send_ns);
1886    drop(worker_guard);
1887    if let Some(diagnostics) = &diagnostics {
1888        diagnostics.record_queue_finish(0, 0);
1889    }
1890    Ok((report, diagnostics.as_ref().map(|value| value.finish())))
1891}
1892
1893/// Take the next directory in the configured order.
1894///
1895/// Both orders push to the back; only the end they are taken from differs, which is
1896/// what keeps this a one-line policy rather than two walkers.
1897fn take_next(queue: &mut VecDeque<(PathBuf, usize)>, order: ScanOrder) -> Option<(PathBuf, usize)> {
1898    match order {
1899        ScanOrder::BreadthFirst => queue.pop_front(),
1900        ScanOrder::DepthFirst => queue.pop_back(),
1901    }
1902}
1903
1904/// Largest worker pool a caller may ask for explicitly.
1905///
1906/// Well past anything measured to help. It exists so a caller that computes a thread
1907/// count from something silly cannot spawn thousands of threads.
1908const MAX_SCAN_THREADS: usize = 32;
1909
1910/// Ceiling on the workers active at the start of an automatic scan.
1911///
1912/// Measured, not guessed. On a 10-core machine walking a 60k-entry `node_modules`
1913/// tree, wall time fell 37% at two workers and 50% at four, then stopped improving:
1914/// six matched four within noise and eight was 4% worse than four. The walk becomes
1915/// bound by the single index consumer, so past this point extra workers buy queue
1916/// contention and efficiency-core scheduling rather than throughput. See
1917/// `docs/project/reports/report-2026-08-10-fdu-performance-experiments.md`.
1918///
1919/// That measurement was taken on macOS, and every constant in this group now reads its
1920/// value from [`crate::platform_tuning`], which records per platform whether the number
1921/// was measured there or inherited. On Linux these are inherited.
1922const DEFAULT_SCAN_THREADS_CAP: usize = crate::platform_tuning::tuning().scan_threads_cap.get();
1923
1924/// Ceiling on automatic workers for an immutable-baseline reconciliation wave.
1925///
1926/// Reconciliation reads both filesystem and index state. Its measured knee arrives
1927/// before the cold producer's because additional metadata calls amplify kernel work
1928/// after the index comparisons already saturate the performance cores.
1929const DEFAULT_RECONCILE_THREADS_CAP: usize =
1930    crate::platform_tuning::tuning().reconcile_threads_cap.get();
1931
1932/// Ceiling an automatic scan may unlock after it establishes that the tree is large.
1933///
1934/// Sixteen was the knee on the 720k-entry cache-pressure corpus in exp-015. Thirty-two
1935/// did not improve on it and spent substantially more worker time waiting at the end.
1936const ADAPTIVE_SCAN_THREADS_CAP: usize =
1937    crate::platform_tuning::tuning().adaptive_scan_threads_cap.get();
1938
1939/// Maximum reserve depth relative to the host's reported parallelism.
1940const ADAPTIVE_SCAN_PARALLELISM_MULTIPLIER: usize =
1941    crate::platform_tuning::tuning().adaptive_scan_parallelism_multiplier.get();
1942
1943/// Entries used to calibrate the initial workers' filesystem service time.
1944const ADAPTIVE_SCAN_CALIBRATION_ENTRIES: u64 =
1945    crate::platform_tuning::tuning().adaptive_scan_calibration_entries.get();
1946
1947/// Average worker time per observed entry that identifies a latency-bound scan.
1948///
1949/// Whole-run attribution separated the measured regimes: roughly 18 microseconds on
1950/// the 60k tree, 22 on the 120k boundary, and 42 or more on the 720k cache-pressure
1951/// tree. Thirty leaves margin between them. The calibration uses the same chunk timing
1952/// already collected for attribution, so it adds no per-entry clock reads.
1953///
1954/// Those are APFS regimes. The Linux warm floor is about 1.5 µs per entry, twenty times
1955/// below this threshold, so the trigger may never fire there — which is exactly the kind
1956/// of inherited constant [`crate::platform_tuning`] exists to make visible (H84).
1957const ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY: u64 =
1958    crate::platform_tuning::tuning().adaptive_scan_slow_work_ns_per_entry.get();
1959
1960#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1961struct WorkerPool {
1962    initial: usize,
1963    maximum: usize,
1964    calibration: Option<WorkerCalibration>,
1965}
1966
1967#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1968struct WorkerCalibration {
1969    minimum_entries: u64,
1970    slow_work_ns_per_entry: u64,
1971    entries: u64,
1972    work_ns: u64,
1973    chunks: u64,
1974}
1975
1976#[derive(Clone, Copy, Debug)]
1977struct PolicyWindowSnapshot {
1978    sequence: u64,
1979    start_entry_ordinal: u64,
1980    end_entry_ordinal: u64,
1981    observed_entries: u64,
1982    observed_chunks: u64,
1983    observed_work_ns: u64,
1984    ready_directories: usize,
1985    in_flight_directories: usize,
1986    active_workers: usize,
1987    handoff_backlog: usize,
1988    requested_workers: Option<usize>,
1989    decision: WorkerPolicyDecision,
1990}
1991
1992struct PolicyTraceState {
1993    outcome: WorkerPolicyOutcome,
1994    outcome_reason: Option<&'static str>,
1995    outcome_sequence: Option<u64>,
1996    windows: Vec<WorkerPolicyWindow>,
1997    events_truncated: bool,
1998    ready_directories_at_finish: usize,
1999    in_flight_directories_at_finish: usize,
2000}
2001
2002/// Shared state used only by the opt-in diagnostic scan APIs.
2003///
2004/// Normal scans pass no recorder and therefore never touch these atomics or locks. The
2005/// trace mutex is deliberately separate from the directory queue: recording a policy
2006/// window may add diagnostic cost, but it cannot alter the queue's synchronization or
2007/// the controller's decision.
2008struct ScanDiagnosticsRecorder {
2009    available_parallelism: usize,
2010    pool: WorkerPool,
2011    policy: WorkerPolicyExperiment,
2012    trace: std::sync::Mutex<PolicyTraceState>,
2013    workers_spawned: std::sync::atomic::AtomicUsize,
2014    active_workers: std::sync::atomic::AtomicUsize,
2015    peak_active_workers: std::sync::atomic::AtomicUsize,
2016    handoff_backlog: std::sync::atomic::AtomicUsize,
2017    handoff_backlog_high_water: std::sync::atomic::AtomicUsize,
2018    calibration_chunks: std::sync::atomic::AtomicU64,
2019    calibration_entries: std::sync::atomic::AtomicU64,
2020    calibration_work_ns: std::sync::atomic::AtomicU64,
2021    worker_expansions: std::sync::atomic::AtomicU64,
2022    portable_attempts: std::sync::atomic::AtomicU64,
2023    portable_successes: std::sync::atomic::AtomicU64,
2024    #[cfg(target_os = "macos")]
2025    macos_bulk_attempts: std::sync::atomic::AtomicU64,
2026    #[cfg(target_os = "macos")]
2027    macos_bulk_successes: std::sync::atomic::AtomicU64,
2028    #[cfg(target_os = "macos")]
2029    macos_bulk_fallbacks: std::sync::atomic::AtomicU64,
2030}
2031
2032impl ScanDiagnosticsRecorder {
2033    fn new(
2034        pool: WorkerPool,
2035        available_parallelism: usize,
2036        policy: WorkerPolicyExperiment,
2037    ) -> std::sync::Arc<Self> {
2038        let (outcome, outcome_reason) = if pool.calibration.is_some() {
2039            (
2040                WorkerPolicyOutcome::Undecided,
2041                Some("the adaptive calibration window has not completed"),
2042            )
2043        } else {
2044            (WorkerPolicyOutcome::Fixed, Some("this worker pool has no adaptive reserve"))
2045        };
2046        std::sync::Arc::new(Self {
2047            available_parallelism,
2048            pool,
2049            policy,
2050            trace: std::sync::Mutex::new(PolicyTraceState {
2051                outcome,
2052                outcome_reason,
2053                outcome_sequence: None,
2054                windows: Vec::new(),
2055                events_truncated: false,
2056                ready_directories_at_finish: 0,
2057                in_flight_directories_at_finish: 0,
2058            }),
2059            workers_spawned: std::sync::atomic::AtomicUsize::new(0),
2060            active_workers: std::sync::atomic::AtomicUsize::new(0),
2061            peak_active_workers: std::sync::atomic::AtomicUsize::new(0),
2062            handoff_backlog: std::sync::atomic::AtomicUsize::new(0),
2063            handoff_backlog_high_water: std::sync::atomic::AtomicUsize::new(0),
2064            calibration_chunks: std::sync::atomic::AtomicU64::new(0),
2065            calibration_entries: std::sync::atomic::AtomicU64::new(0),
2066            calibration_work_ns: std::sync::atomic::AtomicU64::new(0),
2067            worker_expansions: std::sync::atomic::AtomicU64::new(0),
2068            portable_attempts: std::sync::atomic::AtomicU64::new(0),
2069            portable_successes: std::sync::atomic::AtomicU64::new(0),
2070            #[cfg(target_os = "macos")]
2071            macos_bulk_attempts: std::sync::atomic::AtomicU64::new(0),
2072            #[cfg(target_os = "macos")]
2073            macos_bulk_successes: std::sync::atomic::AtomicU64::new(0),
2074            #[cfg(target_os = "macos")]
2075            macos_bulk_fallbacks: std::sync::atomic::AtomicU64::new(0),
2076        })
2077    }
2078
2079    fn worker_guard(self: &std::sync::Arc<Self>) -> ScanWorkerGuard {
2080        let active = self
2081            .active_workers
2082            .fetch_add(1, std::sync::atomic::Ordering::Relaxed)
2083            .saturating_add(1);
2084        self.workers_spawned.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2085        atomic_update_max(&self.peak_active_workers, active);
2086        ScanWorkerGuard { recorder: self.clone() }
2087    }
2088
2089    fn record_policy_window(&self, snapshot: PolicyWindowSnapshot) {
2090        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2091        let sequence = snapshot.sequence;
2092        let supersedes = trace.outcome_sequence.is_none_or(|current| sequence >= current);
2093        match snapshot.decision {
2094            WorkerPolicyDecision::Undecided if supersedes => {
2095                trace.outcome = WorkerPolicyOutcome::Undecided;
2096                trace.outcome_reason =
2097                    Some("the walk ended before the adaptive calibration window completed");
2098                trace.outcome_sequence = Some(sequence);
2099            }
2100            WorkerPolicyDecision::Hold
2101                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2102            {
2103                trace.outcome = WorkerPolicyOutcome::Held;
2104                trace.outcome_reason = None;
2105                trace.outcome_sequence = Some(sequence);
2106            }
2107            WorkerPolicyDecision::ScaleUp => {
2108                trace.outcome = WorkerPolicyOutcome::ScaledUp;
2109                trace.outcome_reason = None;
2110                trace.outcome_sequence = Some(sequence);
2111            }
2112            WorkerPolicyDecision::HoldNoUsefulWork
2113                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2114            {
2115                trace.outcome = WorkerPolicyOutcome::HeldNoUsefulWork;
2116                trace.outcome_reason =
2117                    Some("the slow trigger fired only after ready and in-flight work had drained");
2118                trace.outcome_sequence = Some(sequence);
2119            }
2120            WorkerPolicyDecision::HoldInsufficientFrontier
2121                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2122            {
2123                trace.outcome = WorkerPolicyOutcome::Held;
2124                trace.outcome_reason =
2125                    Some("the observed frontier could not use additional workers");
2126                trace.outcome_sequence = Some(sequence);
2127            }
2128            WorkerPolicyDecision::HoldHandoffBacklog
2129                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2130            {
2131                trace.outcome = WorkerPolicyOutcome::Held;
2132                trace.outcome_reason =
2133                    Some("the observation handoff backlog was already at the controller limit");
2134                trace.outcome_sequence = Some(sequence);
2135            }
2136            WorkerPolicyDecision::Incomplete
2137                if supersedes && trace.outcome == WorkerPolicyOutcome::Undecided =>
2138            {
2139                trace.outcome_reason =
2140                    Some("the walk ended before any adaptive calibration window completed");
2141                trace.outcome_sequence = Some(sequence);
2142            }
2143            _ => {}
2144        }
2145        if sequence >= MAX_POLICY_TRACE_EVENTS as u64 {
2146            trace.events_truncated = true;
2147            return;
2148        }
2149        let work_ns_per_entry = (snapshot.observed_entries > 0)
2150            .then(|| snapshot.observed_work_ns / snapshot.observed_entries);
2151        trace.windows.push(WorkerPolicyWindow {
2152            sequence,
2153            start_entry_ordinal: snapshot.start_entry_ordinal,
2154            end_entry_ordinal: snapshot.end_entry_ordinal,
2155            observed_entries: snapshot.observed_entries,
2156            observed_chunks: snapshot.observed_chunks,
2157            observed_work_ns: snapshot.observed_work_ns,
2158            work_ns_per_entry,
2159            work_ns_per_entry_unavailable_reason: work_ns_per_entry
2160                .is_none()
2161                .then_some("the window observed no entries"),
2162            ready_directories: snapshot.ready_directories,
2163            in_flight_directories: snapshot.in_flight_directories,
2164            active_workers: snapshot.active_workers,
2165            handoff_backlog: snapshot.handoff_backlog,
2166            requested_workers: snapshot.requested_workers,
2167            decision: snapshot.decision,
2168        });
2169    }
2170
2171    fn mark_not_run(&self) {
2172        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2173        trace.outcome = WorkerPolicyOutcome::NotRun;
2174        trace.outcome_reason = Some("max_depth zero requested no directory walk");
2175    }
2176
2177    fn record_queue_finish(&self, ready_directories: usize, in_flight_directories: usize) {
2178        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2179        trace.ready_directories_at_finish = ready_directories;
2180        trace.in_flight_directories_at_finish = in_flight_directories;
2181    }
2182
2183    fn handoff_sent(&self) {
2184        let backlog = self
2185            .handoff_backlog
2186            .fetch_add(1, std::sync::atomic::Ordering::Relaxed)
2187            .saturating_add(1);
2188        atomic_update_max(&self.handoff_backlog_high_water, backlog);
2189    }
2190
2191    fn handoff_received(&self) {
2192        let previous = self.handoff_backlog.fetch_sub(1, std::sync::atomic::Ordering::Relaxed);
2193        debug_assert!(previous > 0, "received handoff must have been sent");
2194    }
2195
2196    fn calibration_chunk(&self, entries: u64, work_ns: u64) {
2197        self.calibration_chunks.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2198        self.calibration_entries.fetch_add(entries, std::sync::atomic::Ordering::Relaxed);
2199        self.calibration_work_ns.fetch_add(work_ns, std::sync::atomic::Ordering::Relaxed);
2200    }
2201
2202    fn worker_expanded(&self) {
2203        self.worker_expansions.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2204    }
2205
2206    fn portable_attempted(&self) {
2207        self.portable_attempts.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2208    }
2209
2210    fn portable_succeeded(&self) {
2211        self.portable_successes.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2212    }
2213
2214    #[cfg(target_os = "macos")]
2215    fn macos_bulk_attempted(&self) {
2216        self.macos_bulk_attempts.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2217    }
2218
2219    #[cfg(target_os = "macos")]
2220    fn macos_bulk_succeeded(&self) {
2221        self.macos_bulk_successes.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2222    }
2223
2224    #[cfg(target_os = "macos")]
2225    fn macos_bulk_fell_back(&self) {
2226        self.macos_bulk_fallbacks.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2227    }
2228
2229    fn finish(&self) -> ScanDiagnostics {
2230        let trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2231        let calibration = self.pool.calibration;
2232        let backend = ScanBackendDiagnostics {
2233            portable_attempts: self.portable_attempts.load(std::sync::atomic::Ordering::Relaxed),
2234            portable_directory_reads: self
2235                .portable_successes
2236                .load(std::sync::atomic::Ordering::Relaxed),
2237            #[cfg(target_os = "macos")]
2238            macos_bulk_attempts: Some(
2239                self.macos_bulk_attempts.load(std::sync::atomic::Ordering::Relaxed),
2240            ),
2241            #[cfg(not(target_os = "macos"))]
2242            macos_bulk_attempts: None,
2243            #[cfg(target_os = "macos")]
2244            macos_bulk_successes: Some(
2245                self.macos_bulk_successes.load(std::sync::atomic::Ordering::Relaxed),
2246            ),
2247            #[cfg(not(target_os = "macos"))]
2248            macos_bulk_successes: None,
2249            #[cfg(target_os = "macos")]
2250            macos_bulk_fallbacks: Some(
2251                self.macos_bulk_fallbacks.load(std::sync::atomic::Ordering::Relaxed),
2252            ),
2253            #[cfg(not(target_os = "macos"))]
2254            macos_bulk_fallbacks: None,
2255            #[cfg(target_os = "macos")]
2256            unavailable_reason: None,
2257            #[cfg(not(target_os = "macos"))]
2258            unavailable_reason: Some(
2259                "macOS bulk directory enumeration is unavailable on this platform",
2260            ),
2261        };
2262        ScanDiagnostics {
2263            schema: SCAN_DIAGNOSTICS_SCHEMA,
2264            worker_policy: WorkerPolicyDiagnostics {
2265                controller: worker_policy_experiment_name(self.policy),
2266                available_parallelism: self.available_parallelism,
2267                initial_workers: self.pool.initial,
2268                maximum_workers: self.pool.maximum,
2269                calibration_window_entries: calibration.map(|value| value.minimum_entries),
2270                slow_threshold_ns_per_entry: calibration.map(|value| value.slow_work_ns_per_entry),
2271                calibration_chunks: self
2272                    .calibration_chunks
2273                    .load(std::sync::atomic::Ordering::Relaxed),
2274                calibration_entries: self
2275                    .calibration_entries
2276                    .load(std::sync::atomic::Ordering::Relaxed),
2277                calibration_work_ns: self
2278                    .calibration_work_ns
2279                    .load(std::sync::atomic::Ordering::Relaxed),
2280                worker_expansions: self
2281                    .worker_expansions
2282                    .load(std::sync::atomic::Ordering::Relaxed),
2283                outcome: trace.outcome,
2284                outcome_reason: trace.outcome_reason,
2285                workers_spawned: self.workers_spawned.load(std::sync::atomic::Ordering::Relaxed),
2286                peak_active_workers: self
2287                    .peak_active_workers
2288                    .load(std::sync::atomic::Ordering::Relaxed),
2289                ready_directories_at_finish: trace.ready_directories_at_finish,
2290                in_flight_directories_at_finish: trace.in_flight_directories_at_finish,
2291                handoff_backlog_at_finish: self
2292                    .handoff_backlog
2293                    .load(std::sync::atomic::Ordering::Relaxed),
2294                handoff_backlog_high_water: self
2295                    .handoff_backlog_high_water
2296                    .load(std::sync::atomic::Ordering::Relaxed),
2297                windows: {
2298                    let mut windows = trace.windows.clone();
2299                    windows.sort_by_key(|window| window.sequence);
2300                    windows
2301                },
2302                events_truncated: trace.events_truncated,
2303            },
2304            backend,
2305        }
2306    }
2307}
2308
2309const fn worker_policy_experiment_name(value: WorkerPolicyExperiment) -> &'static str {
2310    match value {
2311        WorkerPolicyExperiment::ShippedOneShot => "shipped_one_shot",
2312        WorkerPolicyExperiment::RepeatedWindows => "repeated_windows",
2313        WorkerPolicyExperiment::StagedGatedWindows => "staged_gated_windows",
2314    }
2315}
2316
2317struct ScanWorkerGuard {
2318    recorder: std::sync::Arc<ScanDiagnosticsRecorder>,
2319}
2320
2321impl Drop for ScanWorkerGuard {
2322    fn drop(&mut self) {
2323        let previous =
2324            self.recorder.active_workers.fetch_sub(1, std::sync::atomic::Ordering::Relaxed);
2325        debug_assert!(previous > 0, "worker guard must balance worker start");
2326    }
2327}
2328
2329fn atomic_update_max(target: &std::sync::atomic::AtomicUsize, value: usize) {
2330    let mut observed = target.load(std::sync::atomic::Ordering::Relaxed);
2331    while value > observed {
2332        match target.compare_exchange_weak(
2333            observed,
2334            value,
2335            std::sync::atomic::Ordering::Relaxed,
2336            std::sync::atomic::Ordering::Relaxed,
2337        ) {
2338            Ok(_) => break,
2339            Err(actual) => observed = actual,
2340        }
2341    }
2342}
2343
2344enum WalkMessage {
2345    Batch(ScannerBatch),
2346    DetachedDirectories(Vec<DetachedDirectory>),
2347    ScaleUp { sender: std::sync::mpsc::Sender<Self>, target_workers: usize },
2348}
2349
2350impl WorkerPool {
2351    const fn fixed(workers: usize) -> Self {
2352        Self { initial: workers, maximum: workers, calibration: None }
2353    }
2354}
2355
2356impl WorkerCalibration {
2357    const fn new(minimum_entries: u64, slow_work_ns_per_entry: u64) -> Self {
2358        Self { minimum_entries, slow_work_ns_per_entry, entries: 0, work_ns: 0, chunks: 0 }
2359    }
2360
2361    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<bool> {
2362        self.chunks = self.chunks.saturating_add(1);
2363        self.entries = self.entries.saturating_add(entries);
2364        self.work_ns = self.work_ns.saturating_add(work_ns);
2365        (self.entries >= self.minimum_entries)
2366            .then(|| self.work_ns / self.entries >= self.slow_work_ns_per_entry)
2367    }
2368}
2369
2370#[derive(Clone, Copy, Debug)]
2371struct CalibrationWindow {
2372    start_entry_ordinal: u64,
2373    end_entry_ordinal: u64,
2374    entries: u64,
2375    chunks: u64,
2376    work_ns: u64,
2377    slow: bool,
2378}
2379
2380#[derive(Debug)]
2381struct RepeatedCalibration {
2382    minimum_entries: u64,
2383    slow_work_ns_per_entry: u64,
2384    window_start: u64,
2385    entries: u64,
2386    chunks: u64,
2387    work_ns: u64,
2388    completed_windows: u64,
2389}
2390
2391impl RepeatedCalibration {
2392    const fn new(calibration: WorkerCalibration) -> Self {
2393        Self {
2394            minimum_entries: calibration.minimum_entries,
2395            slow_work_ns_per_entry: calibration.slow_work_ns_per_entry,
2396            window_start: 0,
2397            entries: 0,
2398            chunks: 0,
2399            work_ns: 0,
2400            completed_windows: 0,
2401        }
2402    }
2403
2404    const fn starting_at(calibration: WorkerCalibration, window_start: u64) -> Self {
2405        let mut repeated = Self::new(calibration);
2406        repeated.window_start = window_start;
2407        repeated
2408    }
2409
2410    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<CalibrationWindow> {
2411        self.chunks = self.chunks.saturating_add(1);
2412        self.entries = self.entries.saturating_add(entries);
2413        self.work_ns = self.work_ns.saturating_add(work_ns);
2414        if self.entries < self.minimum_entries {
2415            return None;
2416        }
2417        let end_entry_ordinal = self.window_start.saturating_add(self.entries);
2418        let window = CalibrationWindow {
2419            start_entry_ordinal: self.window_start,
2420            end_entry_ordinal,
2421            entries: self.entries,
2422            chunks: self.chunks,
2423            work_ns: self.work_ns,
2424            slow: self.work_ns / self.entries >= self.slow_work_ns_per_entry,
2425        };
2426        self.window_start = end_entry_ordinal;
2427        self.entries = 0;
2428        self.chunks = 0;
2429        self.work_ns = 0;
2430        self.completed_windows = self.completed_windows.saturating_add(1);
2431        Some(window)
2432    }
2433}
2434
2435#[derive(Debug)]
2436enum WorkerController {
2437    OneShot(WorkerCalibration),
2438    Repeated { calibration: RepeatedCalibration, staged_gated: bool },
2439}
2440
2441impl WorkerController {
2442    fn new(calibration: WorkerCalibration, policy: WorkerPolicyExperiment) -> Self {
2443        match policy {
2444            WorkerPolicyExperiment::ShippedOneShot => Self::OneShot(calibration),
2445            WorkerPolicyExperiment::RepeatedWindows => Self::Repeated {
2446                calibration: RepeatedCalibration::new(calibration),
2447                staged_gated: false,
2448            },
2449            WorkerPolicyExperiment::StagedGatedWindows => Self::Repeated {
2450                calibration: RepeatedCalibration::new(calibration),
2451                staged_gated: true,
2452            },
2453        }
2454    }
2455
2456    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<CalibrationWindow> {
2457        match self {
2458            Self::OneShot(calibration) => {
2459                let slow = calibration.observe(entries, work_ns)?;
2460                Some(CalibrationWindow {
2461                    start_entry_ordinal: 0,
2462                    end_entry_ordinal: calibration.entries,
2463                    entries: calibration.entries,
2464                    chunks: calibration.chunks,
2465                    work_ns: calibration.work_ns,
2466                    slow,
2467                })
2468            }
2469            Self::Repeated { calibration, .. } => calibration.observe(entries, work_ns),
2470        }
2471    }
2472
2473    fn partial_window(&self) -> (CalibrationWindow, WorkerPolicyDecision) {
2474        match self {
2475            Self::OneShot(calibration) => (
2476                CalibrationWindow {
2477                    start_entry_ordinal: 0,
2478                    end_entry_ordinal: calibration.entries,
2479                    entries: calibration.entries,
2480                    chunks: calibration.chunks,
2481                    work_ns: calibration.work_ns,
2482                    slow: false,
2483                },
2484                WorkerPolicyDecision::Undecided,
2485            ),
2486            Self::Repeated { calibration, .. } => (
2487                CalibrationWindow {
2488                    start_entry_ordinal: calibration.window_start,
2489                    end_entry_ordinal: calibration.window_start.saturating_add(calibration.entries),
2490                    entries: calibration.entries,
2491                    chunks: calibration.chunks,
2492                    work_ns: calibration.work_ns,
2493                    slow: false,
2494                },
2495                if calibration.completed_windows == 0 {
2496                    WorkerPolicyDecision::Undecided
2497                } else {
2498                    WorkerPolicyDecision::Incomplete
2499                },
2500            ),
2501        }
2502    }
2503
2504    const fn is_staged_gated(&self) -> bool {
2505        matches!(self, Self::Repeated { staged_gated: true, .. })
2506    }
2507
2508    const fn is_one_shot(&self) -> bool {
2509        matches!(self, Self::OneShot(_))
2510    }
2511
2512    const fn calibration_spec(&self) -> WorkerCalibration {
2513        match self {
2514            Self::OneShot(calibration) => WorkerCalibration::new(
2515                calibration.minimum_entries,
2516                calibration.slow_work_ns_per_entry,
2517            ),
2518            Self::Repeated { calibration, .. } => WorkerCalibration::new(
2519                calibration.minimum_entries,
2520                calibration.slow_work_ns_per_entry,
2521            ),
2522        }
2523    }
2524}
2525
2526fn automatic_worker_pool(available: usize) -> WorkerPool {
2527    let initial = available.clamp(1, DEFAULT_SCAN_THREADS_CAP);
2528    // Preserve the serial fallback when the platform cannot report more than one
2529    // available processor. There is no measured basis for inventing parallelism there.
2530    if initial == 1 {
2531        return WorkerPool::fixed(1);
2532    }
2533    let maximum = available
2534        .saturating_mul(ADAPTIVE_SCAN_PARALLELISM_MULTIPLIER)
2535        .clamp(initial, ADAPTIVE_SCAN_THREADS_CAP);
2536    WorkerPool {
2537        initial,
2538        maximum,
2539        calibration: (maximum > initial).then_some(WorkerCalibration::new(
2540            ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
2541            ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
2542        )),
2543    }
2544}
2545
2546/// Directories handed to a worker in one go.
2547///
2548/// Popping one directory at a time makes the queue lock the bottleneck on a wide,
2549/// shallow tree; taking a small run amortizes the lock without letting one worker
2550/// starve the others by hoarding the queue.
2551const DIR_CLAIM: usize = 4;
2552
2553/// A parallel directory walk that produces exactly the observations the serial walk does.
2554///
2555/// The shape is deliberate. Workers read directories and *produce* observations; they
2556/// never touch an index. A single consumer — the caller's sink, on this thread —
2557/// applies them. That keeps the crate's one mutation contract intact: parallelism is a
2558/// property of the producer, and the index still sees one ordered stream of observations.
2559///
2560/// Ordering across independent subtrees is not fixed, but a directory observation is
2561/// published before that directory becomes claimable. The index therefore sees a
2562/// parent-first causal stream without imposing a global level barrier or serializing
2563/// filesystem work. The resulting index is byte-identical to the serial walker's,
2564/// which the benchmark harness re-proves on every trial by comparing engine digests
2565/// against an independent oracle.
2566#[allow(clippy::too_many_arguments)]
2567fn scan_concurrent(
2568    root: &Path,
2569    config: &ScanConfig,
2570    root_dev: u64,
2571    sink: &mut dyn FnMut(ScannerBatch),
2572    pool: WorkerPool,
2573    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2574    policy: WorkerPolicyExperiment,
2575    sink_mode: SinkMode,
2576) -> ScanReport {
2577    let mut consume = |message| match message {
2578        WalkMessage::Batch(batch) => {
2579            if let Some(diagnostics) = diagnostics {
2580                diagnostics.handoff_received();
2581            }
2582            sink(batch);
2583        }
2584        WalkMessage::DetachedDirectories(_) => {
2585            unreachable!("the streaming walker never publishes detached directories")
2586        }
2587        WalkMessage::ScaleUp { .. } => {
2588            unreachable!("the shared runner consumes scale-up messages")
2589        }
2590    };
2591    run_concurrent_walk(
2592        root,
2593        config,
2594        root_dev,
2595        pool,
2596        diagnostics,
2597        policy,
2598        match sink_mode {
2599            SinkMode::Retained => walk_worker,
2600            SinkMode::TransientFold => walk_worker_transient_fold,
2601        },
2602        &mut consume,
2603    )
2604}
2605
2606/// Parallel cold walk for a detached index that has no streaming consumer.
2607///
2608/// Workers publish directory-shaped facts before making their children claimable. The
2609/// caller consumes those groups into a private builder while filesystem work continues,
2610/// preserving parent-first causality and pipeline overlap without sending one full path
2611/// or public observation per entry.
2612fn scan_concurrent_detached(
2613    root: &Path,
2614    config: &ScanConfig,
2615    root_dev: u64,
2616    pool: WorkerPool,
2617    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2618    policy: WorkerPolicyExperiment,
2619) -> Result<(ScanReport, DetachedIndexBuilder)> {
2620    let mut builder = DetachedIndexBuilder::new(root, config.scope(), config.types_shared())
2621        .with_control_limits(config.control_limits);
2622    let mut build_error = None;
2623    let output = {
2624        let mut consume = |message| match message {
2625            WalkMessage::Batch(_) => {
2626                unreachable!("the detached walker never publishes scanner batches")
2627            }
2628            WalkMessage::DetachedDirectories(directories) => {
2629                if let Some(diagnostics) = diagnostics {
2630                    diagnostics.handoff_received();
2631                }
2632                if build_error.is_none() {
2633                    for directory in directories {
2634                        if let Err(error) = builder.push_directory(directory) {
2635                            build_error = Some(error);
2636                            break;
2637                        }
2638                    }
2639                }
2640            }
2641            WalkMessage::ScaleUp { .. } => {
2642                unreachable!("the shared runner consumes scale-up messages")
2643            }
2644        };
2645        run_concurrent_walk(
2646            root,
2647            config,
2648            root_dev,
2649            pool,
2650            diagnostics,
2651            policy,
2652            walk_detached_worker,
2653            &mut consume,
2654        )
2655    };
2656    if let Some(error) = build_error {
2657        return Err(error);
2658    }
2659    Ok((output, builder))
2660}
2661
2662type WalkWorker = fn(
2663    &Path,
2664    &ScanConfig,
2665    u64,
2666    &DirectoryQueue,
2667    &std::sync::mpsc::Sender<WalkMessage>,
2668    Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2669) -> ScanReport;
2670
2671/// Run the shared pool, scaling controller, diagnostics, and report reduction.
2672///
2673/// Streaming and detached scans differ only in their worker emission and main-thread
2674/// consumer. Keeping orchestration here prevents fixes to termination, diagnostics, or
2675/// panic handling from diverging between the two cold paths.
2676#[allow(clippy::too_many_arguments)]
2677fn run_concurrent_walk<C>(
2678    root: &Path,
2679    config: &ScanConfig,
2680    root_dev: u64,
2681    pool: WorkerPool,
2682    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2683    policy: WorkerPolicyExperiment,
2684    worker: WalkWorker,
2685    consume: &mut C,
2686) -> ScanReport
2687where
2688    C: FnMut(WalkMessage),
2689{
2690    let diagnostics = diagnostics.cloned();
2691    let queue = DirectoryQueue::new_with_policy(
2692        (PathBuf::new(), 0),
2693        config.order,
2694        pool.calibration,
2695        diagnostics.clone(),
2696        pool.initial,
2697        pool.maximum,
2698        policy,
2699    );
2700    let (sender, receiver) = std::sync::mpsc::channel::<WalkMessage>();
2701
2702    let mut report = std::thread::scope(|scope| {
2703        let mut handles: Vec<_> = (0..pool.initial)
2704            .map(|_| {
2705                let sender = sender.clone();
2706                let queue = &queue;
2707                let diagnostics = diagnostics.clone();
2708                scope.spawn(move || {
2709                    worker(root, config, root_dev, queue, &sender, diagnostics.as_ref())
2710                })
2711            })
2712            .collect();
2713        // The loop below ends when every sender is gone, so this one must go first.
2714        drop(sender);
2715
2716        let mut spawned_workers = pool.initial;
2717        for message in receiver {
2718            match message {
2719                WalkMessage::ScaleUp { sender, target_workers }
2720                    if target_workers > spawned_workers =>
2721                {
2722                    let target_workers = target_workers.min(pool.maximum);
2723                    record_adaptive_worker_expansion(diagnostics.as_ref());
2724                    for _ in spawned_workers..target_workers {
2725                        let sender = sender.clone();
2726                        let queue = &queue;
2727                        let diagnostics = diagnostics.clone();
2728                        handles.push(scope.spawn(move || {
2729                            worker(root, config, root_dev, queue, &sender, diagnostics.as_ref())
2730                        }));
2731                    }
2732                    spawned_workers = target_workers;
2733                }
2734                WalkMessage::ScaleUp { .. } => {}
2735                output => consume(output),
2736            }
2737        }
2738
2739        // A walk that ends before its calibration window fills never observed enough to
2740        // decide anything. That is an *unobservable* policy, not a decision to hold the
2741        // initial pool, and an artifact that conflated the two would report a held pool
2742        // as if the walk had measured one and chosen it.
2743        let queue_finish = {
2744            let mut state = queue.lock();
2745            let mut trailing_window = None;
2746            if let Some(controller) = &state.controller {
2747                let (window, decision) = controller.partial_window();
2748                if decision == WorkerPolicyDecision::Undecided {
2749                    crate::counters::bump(|counts| {
2750                        counts.adaptive_policy_undecided =
2751                            counts.adaptive_policy_undecided.saturating_add(1);
2752                    });
2753                }
2754                if diagnostics.is_some() {
2755                    let sequence = state.allocate_policy_sequence();
2756                    trailing_window = Some(PolicyWindowSnapshot {
2757                        sequence,
2758                        start_entry_ordinal: window.start_entry_ordinal,
2759                        end_entry_ordinal: window.end_entry_ordinal,
2760                        observed_entries: window.entries,
2761                        observed_chunks: window.chunks,
2762                        observed_work_ns: window.work_ns,
2763                        ready_directories: state.ready_directories,
2764                        in_flight_directories: state.in_flight_directories,
2765                        active_workers: diagnostics.as_ref().map_or(0, |diagnostics| {
2766                            diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
2767                        }),
2768                        handoff_backlog: diagnostics.as_ref().map_or(0, |diagnostics| {
2769                            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
2770                        }),
2771                        requested_workers: None,
2772                        decision,
2773                    });
2774                }
2775            } else if diagnostics.is_some() {
2776                let shadow = state.shadow_calibration.as_ref().and_then(|shadow| {
2777                    (shadow.entries > 0).then_some((
2778                        shadow.window_start,
2779                        shadow.entries,
2780                        shadow.chunks,
2781                        shadow.work_ns,
2782                    ))
2783                });
2784                if let Some((window_start, entries, chunks, work_ns)) = shadow {
2785                    let sequence = state.allocate_policy_sequence();
2786                    trailing_window = Some(PolicyWindowSnapshot {
2787                        sequence,
2788                        start_entry_ordinal: window_start,
2789                        end_entry_ordinal: window_start.saturating_add(entries),
2790                        observed_entries: entries,
2791                        observed_chunks: chunks,
2792                        observed_work_ns: work_ns,
2793                        ready_directories: state.ready_directories,
2794                        in_flight_directories: state.in_flight_directories,
2795                        active_workers: diagnostics.as_ref().map_or(0, |diagnostics| {
2796                            diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
2797                        }),
2798                        handoff_backlog: diagnostics.as_ref().map_or(0, |diagnostics| {
2799                            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
2800                        }),
2801                        requested_workers: None,
2802                        decision: WorkerPolicyDecision::ObserveIncomplete,
2803                    });
2804                }
2805            }
2806            (state.ready_directories, state.in_flight_directories, trailing_window)
2807        };
2808        if let Some(diagnostics) = &diagnostics {
2809            diagnostics.record_queue_finish(queue_finish.0, queue_finish.1);
2810            if let Some(window) = queue_finish.2 {
2811                diagnostics.record_policy_window(window);
2812            }
2813        }
2814
2815        let mut report = ScanReport::default();
2816        for handle in handles {
2817            match handle.join() {
2818                Ok(worker) => report.absorb(worker),
2819                Err(_) => {
2820                    // A worker panic leaves directories unaccounted for. Preserve that
2821                    // as a partial scan instead of reporting a short tree as complete.
2822                    report.errors.push(Error::io(
2823                        root,
2824                        std::io::Error::other("a scan worker thread panicked"),
2825                    ));
2826                }
2827            }
2828        }
2829        report
2830    });
2831
2832    // Workers finish in filesystem order, so normalize errors before they escape.
2833    report.errors.sort_by_cached_key(ToString::to_string);
2834    report
2835}
2836
2837/// Compile-time adapter for the one directory walker.
2838///
2839/// The filesystem, queue, admission, and diagnostics logic stays singular. Generic
2840/// emission keeps the public streaming path and private detached path branch-free in
2841/// their per-entry loops after monomorphization.
2842trait WalkEmission {
2843    type Directory;
2844
2845    fn begin_directory(&mut self, path: &Path) -> Self::Directory;
2846
2847    #[allow(clippy::too_many_arguments)]
2848    fn record_entry(
2849        &mut self,
2850        root: &Path,
2851        rel_dir: &Path,
2852        depth: usize,
2853        region: RegionId,
2854        name: &OsStr,
2855        kind: EntryKind,
2856        attrs: Attrs,
2857        root_dev: u64,
2858        config: &ScanConfig,
2859        directory: &mut Self::Directory,
2860        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
2861        report: &mut ScanReport,
2862        sender: &std::sync::mpsc::Sender<WalkMessage>,
2863        chunk_send_ns: &mut u64,
2864        diagnostics: Option<&ScanDiagnosticsRecorder>,
2865    ) -> bool;
2866
2867    fn finish_directory(&mut self, directory: Self::Directory);
2868
2869    fn publish_before_discovery(
2870        &mut self,
2871        has_discovered: bool,
2872        sender: &std::sync::mpsc::Sender<WalkMessage>,
2873        chunk_send_ns: &mut u64,
2874        diagnostics: Option<&ScanDiagnosticsRecorder>,
2875    ) -> bool;
2876
2877    fn finish(
2878        &mut self,
2879        sender: &std::sync::mpsc::Sender<WalkMessage>,
2880        report: &mut ScanReport,
2881        diagnostics: Option<&ScanDiagnosticsRecorder>,
2882    );
2883
2884    /// Transient summary can take directory and symlink kind from the listing.
2885    fn skip_dir_symlink_stat(&self) -> bool {
2886        false
2887    }
2888}
2889
2890struct StreamingEmission {
2891    batch: Vec<ObservationOp>,
2892    batch_size: usize,
2893    recycle_tx: Option<std::sync::mpsc::Sender<Vec<ObservationOp>>>,
2894    recycle_rx: Option<std::sync::mpsc::Receiver<Vec<ObservationOp>>>,
2895    skip_dir_symlink_stat: bool,
2896}
2897
2898impl StreamingEmission {
2899    /// An emission with the properties [`SinkMode`] names for `mode`.
2900    ///
2901    /// The recycle channel returns drained `PathBuf` arenas to this worker so glibc
2902    /// frees them on the thread that allocated them.
2903    fn for_sink(batch_size: usize, mode: SinkMode) -> Self {
2904        let (recycle_tx, recycle_rx) = if mode.recycles_batches() {
2905            let (tx, rx) = std::sync::mpsc::channel();
2906            (Some(tx), Some(rx))
2907        } else {
2908            (None, None)
2909        };
2910        Self {
2911            batch: Vec::with_capacity(batch_size),
2912            batch_size,
2913            recycle_tx,
2914            recycle_rx,
2915            skip_dir_symlink_stat: mode.skips_dir_symlink_stat(),
2916        }
2917    }
2918
2919    fn wrap(&self, ops: Vec<ObservationOp>) -> ScannerBatch {
2920        match &self.recycle_tx {
2921            Some(recycle) => ScannerBatch::new(ops).with_recycle(recycle.clone()),
2922            None => ScannerBatch::new(ops),
2923        }
2924    }
2925
2926    fn next_vec(&self) -> Vec<ObservationOp> {
2927        // The retained path keeps its pre-H147 shape: an empty vec that grows by
2928        // doubling. Pre-sizing every batch there was never measured, and the public
2929        // `scan` is what a library caller pays for.
2930        let Some(recycle_rx) = &self.recycle_rx else {
2931            return Vec::new();
2932        };
2933        let mut kept = None;
2934        while let Ok(mut recycled) = recycle_rx.try_recv() {
2935            recycled.clear();
2936            kept = Some(recycled);
2937        }
2938        kept.unwrap_or_else(|| Vec::with_capacity(self.batch_size))
2939    }
2940
2941    fn send_full(
2942        &mut self,
2943        sender: &std::sync::mpsc::Sender<WalkMessage>,
2944        diagnostics: Option<&ScanDiagnosticsRecorder>,
2945    ) -> bool {
2946        let ops = std::mem::take(&mut self.batch);
2947        let sent = send_scanner_batch(sender, self.wrap(ops), diagnostics);
2948        self.batch = self.next_vec();
2949        sent
2950    }
2951}
2952
2953impl WalkEmission for StreamingEmission {
2954    type Directory = ();
2955
2956    fn begin_directory(&mut self, _path: &Path) {}
2957
2958    #[allow(clippy::too_many_arguments)]
2959    fn record_entry(
2960        &mut self,
2961        root: &Path,
2962        rel_dir: &Path,
2963        depth: usize,
2964        region: RegionId,
2965        name: &OsStr,
2966        kind: EntryKind,
2967        attrs: Attrs,
2968        root_dev: u64,
2969        config: &ScanConfig,
2970        _directory: &mut Self::Directory,
2971        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
2972        report: &mut ScanReport,
2973        sender: &std::sync::mpsc::Sender<WalkMessage>,
2974        chunk_send_ns: &mut u64,
2975        diagnostics: Option<&ScanDiagnosticsRecorder>,
2976    ) -> bool {
2977        record_walk_entry(
2978            root,
2979            rel_dir,
2980            depth,
2981            region,
2982            name,
2983            kind,
2984            attrs,
2985            root_dev,
2986            config,
2987            self,
2988            discovered,
2989            report,
2990            sender,
2991            chunk_send_ns,
2992            diagnostics,
2993        )
2994    }
2995
2996    fn finish_directory(&mut self, _directory: Self::Directory) {}
2997
2998    fn publish_before_discovery(
2999        &mut self,
3000        has_discovered: bool,
3001        sender: &std::sync::mpsc::Sender<WalkMessage>,
3002        chunk_send_ns: &mut u64,
3003        diagnostics: Option<&ScanDiagnosticsRecorder>,
3004    ) -> bool {
3005        if self.batch.is_empty() || !has_discovered {
3006            return true;
3007        }
3008        let send_started = std::time::Instant::now();
3009        let sent = self.send_full(sender, diagnostics);
3010        *chunk_send_ns += elapsed_ns(send_started);
3011        sent
3012    }
3013
3014    fn finish(
3015        &mut self,
3016        sender: &std::sync::mpsc::Sender<WalkMessage>,
3017        report: &mut ScanReport,
3018        diagnostics: Option<&ScanDiagnosticsRecorder>,
3019    ) {
3020        if self.batch.is_empty() {
3021            return;
3022        }
3023        let send_started = std::time::Instant::now();
3024        // The walk is over: do not ask `send_full` for a replacement vec that no
3025        // later `record_entry` would use.
3026        let ops = std::mem::take(&mut self.batch);
3027        let _ = send_scanner_batch(sender, self.wrap(ops), diagnostics);
3028        self.batch = Vec::new();
3029        report.attribution.send_ns += elapsed_ns(send_started);
3030    }
3031
3032    fn skip_dir_symlink_stat(&self) -> bool {
3033        self.skip_dir_symlink_stat
3034    }
3035}
3036
3037#[derive(Default)]
3038struct DetachedEmission {
3039    directories: Vec<DetachedDirectory>,
3040}
3041
3042impl WalkEmission for DetachedEmission {
3043    type Directory = DetachedDirectory;
3044
3045    fn begin_directory(&mut self, path: &Path) -> Self::Directory {
3046        DetachedDirectory { path: path.to_path_buf(), children: Vec::new(), control: None }
3047    }
3048
3049    #[allow(clippy::too_many_arguments)]
3050    fn record_entry(
3051        &mut self,
3052        root: &Path,
3053        rel_dir: &Path,
3054        depth: usize,
3055        region: RegionId,
3056        name: &OsStr,
3057        kind: EntryKind,
3058        attrs: Attrs,
3059        root_dev: u64,
3060        config: &ScanConfig,
3061        directory: &mut Self::Directory,
3062        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3063        report: &mut ScanReport,
3064        _sender: &std::sync::mpsc::Sender<WalkMessage>,
3065        _chunk_send_ns: &mut u64,
3066        _diagnostics: Option<&ScanDiagnosticsRecorder>,
3067    ) -> bool {
3068        record_detached_entry(
3069            root,
3070            rel_dir,
3071            depth,
3072            region,
3073            name,
3074            kind,
3075            attrs,
3076            root_dev,
3077            config,
3078            &mut directory.children,
3079            &mut directory.control,
3080            discovered,
3081            report,
3082        );
3083        true
3084    }
3085
3086    fn finish_directory(&mut self, directory: Self::Directory) {
3087        self.directories.push(directory);
3088    }
3089
3090    fn publish_before_discovery(
3091        &mut self,
3092        _has_discovered: bool,
3093        sender: &std::sync::mpsc::Sender<WalkMessage>,
3094        chunk_send_ns: &mut u64,
3095        diagnostics: Option<&ScanDiagnosticsRecorder>,
3096    ) -> bool {
3097        if self.directories.is_empty() {
3098            return true;
3099        }
3100        let send_started = std::time::Instant::now();
3101        let sent =
3102            send_detached_directories(sender, std::mem::take(&mut self.directories), diagnostics);
3103        *chunk_send_ns += elapsed_ns(send_started);
3104        sent
3105    }
3106
3107    fn finish(
3108        &mut self,
3109        _sender: &std::sync::mpsc::Sender<WalkMessage>,
3110        _report: &mut ScanReport,
3111        _diagnostics: Option<&ScanDiagnosticsRecorder>,
3112    ) {
3113    }
3114}
3115
3116fn walk_detached_worker(
3117    root: &Path,
3118    config: &ScanConfig,
3119    root_dev: u64,
3120    queue: &DirectoryQueue,
3121    sender: &std::sync::mpsc::Sender<WalkMessage>,
3122    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3123) -> ScanReport {
3124    let report = walk_worker_with(
3125        root,
3126        config,
3127        root_dev,
3128        queue,
3129        sender,
3130        diagnostics,
3131        DetachedEmission::default(),
3132    );
3133    // A walker leaves only when the queue is empty with nothing in flight, or when its
3134    // consumer is gone, so the walk is over. The index may still be assembling the
3135    // listings already sent; the counters have stopped, and the phase says why.
3136    if let Some(progress) = &config.progress {
3137        progress.enter(crate::ProgressPhase::Indexing);
3138    }
3139    report
3140}
3141
3142/// One worker's share of the public observation walk.
3143fn walk_worker(
3144    root: &Path,
3145    config: &ScanConfig,
3146    root_dev: u64,
3147    queue: &DirectoryQueue,
3148    sender: &std::sync::mpsc::Sender<WalkMessage>,
3149    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3150) -> ScanReport {
3151    walk_worker_with(
3152        root,
3153        config,
3154        root_dev,
3155        queue,
3156        sender,
3157        diagnostics,
3158        StreamingEmission::for_sink(config.batch_size, SinkMode::Retained),
3159    )
3160}
3161
3162/// One worker's share of the transient summary walk.
3163fn walk_worker_transient_fold(
3164    root: &Path,
3165    config: &ScanConfig,
3166    root_dev: u64,
3167    queue: &DirectoryQueue,
3168    sender: &std::sync::mpsc::Sender<WalkMessage>,
3169    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3170) -> ScanReport {
3171    walk_worker_with(
3172        root,
3173        config,
3174        root_dev,
3175        queue,
3176        sender,
3177        diagnostics,
3178        StreamingEmission::for_sink(config.batch_size, SinkMode::TransientFold),
3179    )
3180}
3181
3182fn walk_worker_with<E: WalkEmission>(
3183    root: &Path,
3184    config: &ScanConfig,
3185    root_dev: u64,
3186    queue: &DirectoryQueue,
3187    sender: &std::sync::mpsc::Sender<WalkMessage>,
3188    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3189    mut emission: E,
3190) -> ScanReport {
3191    let _counter_guard = crate::counters::thread_flush_guard();
3192    let _worker_guard = diagnostics.map(ScanDiagnosticsRecorder::worker_guard);
3193    let worker_started = std::time::Instant::now();
3194    let mut report = ScanReport::default();
3195    let mut tally = ProgressTally::new(config.progress.as_ref());
3196    let mut claimed: Vec<(PathBuf, usize, RegionId)> = Vec::with_capacity(DIR_CLAIM);
3197    let mut discovered: Vec<(PathBuf, usize, RegionId)> = Vec::new();
3198    let mut consumer_gone = false;
3199    #[cfg(target_os = "macos")]
3200    let mut bulk_reader = macos_bulk::Reader::new();
3201
3202    'walk: while let Some(claim) = queue.claim(&mut claimed, &mut report.attribution) {
3203        // One timing pair per claimed chunk, never per entry: the chunk is the unit
3204        // the amortization argument is made in, so it is the unit the evidence is
3205        // collected in.
3206        let chunk_started = std::time::Instant::now();
3207        let mut chunk_send_ns: u64 = 0;
3208        let entries_before = report.entries;
3209        for (rel_dir, depth, region) in claimed.drain(..) {
3210            let abs_dir = root.join(&rel_dir);
3211            let mut directory = emission.begin_directory(&rel_dir);
3212            #[cfg(target_os = "macos")]
3213            {
3214                if let Some(diagnostics) = diagnostics {
3215                    diagnostics.macos_bulk_attempted();
3216                }
3217                if let Some(entries) =
3218                    (!walk_hook_covers(&abs_dir)).then(|| bulk_reader.read(&abs_dir)).flatten()
3219                {
3220                    if let Some(diagnostics) = diagnostics {
3221                        diagnostics.macos_bulk_succeeded();
3222                    }
3223                    report.dirs_read += 1;
3224                    for entry in entries {
3225                        if !emission.record_entry(
3226                            root,
3227                            &rel_dir,
3228                            depth,
3229                            region,
3230                            &entry.name,
3231                            entry.kind,
3232                            entry.attrs,
3233                            root_dev,
3234                            config,
3235                            &mut directory,
3236                            &mut discovered,
3237                            &mut report,
3238                            sender,
3239                            &mut chunk_send_ns,
3240                            diagnostics.map(AsRef::as_ref),
3241                        ) {
3242                            consumer_gone = true;
3243                            break 'walk;
3244                        }
3245                    }
3246                    emission.finish_directory(directory);
3247                    continue;
3248                }
3249                if let Some(diagnostics) = diagnostics {
3250                    diagnostics.macos_bulk_fell_back();
3251                }
3252            }
3253
3254            crate::counters::bump(|c| c.dir_opens += 1);
3255            if let Some(diagnostics) = diagnostics {
3256                diagnostics.portable_attempted();
3257            }
3258
3259            let listing = match fs::read_dir(&abs_dir) {
3260                Ok(listing) => {
3261                    if let Some(diagnostics) = diagnostics {
3262                        diagnostics.portable_succeeded();
3263                    }
3264                    listing
3265                }
3266                Err(e) => {
3267                    report.errors.push(Error::io(abs_dir, e));
3268                    continue;
3269                }
3270            };
3271            report.dirs_read += 1;
3272
3273            for item in listing {
3274                let item = match item {
3275                    Ok(item) => item,
3276                    Err(e) => {
3277                        report.errors.push(Error::io(&abs_dir, e));
3278                        continue;
3279                    }
3280                };
3281                crate::counters::bump(|c| c.dir_entries += 1);
3282                let name = item.file_name();
3283                let (kind, attrs) = match listed_child_kind_and_attrs(
3284                    &item,
3285                    emission.skip_dir_symlink_stat(),
3286                    config.one_filesystem,
3287                ) {
3288                    Ok(Some(observed)) => observed,
3289                    Ok(None) => continue,
3290                    Err(error) => {
3291                        report.errors.push(Error::io(item.path(), error));
3292                        continue;
3293                    }
3294                };
3295                if !emission.record_entry(
3296                    root,
3297                    &rel_dir,
3298                    depth,
3299                    region,
3300                    &name,
3301                    kind,
3302                    attrs,
3303                    root_dev,
3304                    config,
3305                    &mut directory,
3306                    &mut discovered,
3307                    &mut report,
3308                    sender,
3309                    &mut chunk_send_ns,
3310                    diagnostics.map(AsRef::as_ref),
3311                ) {
3312                    // The consumer is gone; nothing further will be read.
3313                    consumer_gone = true;
3314                    break 'walk;
3315                }
3316            }
3317            emission.finish_directory(directory);
3318        }
3319        // Publish facts that authorize newly discovered directories before making
3320        // those directories claimable. Both emission modes preserve this boundary.
3321        if !emission.publish_before_discovery(
3322            !discovered.is_empty(),
3323            sender,
3324            &mut chunk_send_ns,
3325            diagnostics.map(AsRef::as_ref),
3326        ) {
3327            report.attribution.send_ns += chunk_send_ns;
3328            report.attribution.work_ns += elapsed_ns(chunk_started).saturating_sub(chunk_send_ns);
3329            consumer_gone = true;
3330            break 'walk;
3331        }
3332        report.attribution.send_ns += chunk_send_ns;
3333        let chunk_work_ns = elapsed_ns(chunk_started).saturating_sub(chunk_send_ns);
3334        report.attribution.work_ns += chunk_work_ns;
3335        // Progress is reported per chunk for the same reason timing is: the chunk is
3336        // the unit of handoff, so it is the unit the shared counters are touched in.
3337        tally.flush(&report);
3338
3339        // Publish new work before releasing the claim so a worker that finds nothing
3340        // new does not hold work that others could be doing.
3341        if !discovered.is_empty() {
3342            queue.extend(discovered.drain(..), &mut report.attribution);
3343        }
3344        if let Some(target_workers) = claim.release(
3345            report.entries.saturating_sub(entries_before),
3346            chunk_work_ns,
3347            &mut report.attribution,
3348        ) {
3349            // Carry a sender in-band so the consumer can create the reserve workers
3350            // without retaining a channel endpoint that would keep a small scan alive.
3351            // Only the release that completes a slow calibration returns true, so one
3352            // message expands the pool exactly once.
3353            let _ = sender.send(WalkMessage::ScaleUp { sender: sender.clone(), target_workers });
3354        }
3355    }
3356
3357    if !consumer_gone {
3358        emission.finish(sender, &mut report, diagnostics.map(AsRef::as_ref));
3359    }
3360    // A worker that left mid-chunk because its consumer was gone still read what it
3361    // read, and the report it returns says so.
3362    tally.flush(&report);
3363    report.attribution.wall_ns = elapsed_ns(worker_started);
3364    report
3365}
3366
3367#[allow(clippy::too_many_arguments)]
3368fn record_detached_entry(
3369    root: &Path,
3370    rel_dir: &Path,
3371    depth: usize,
3372    region: RegionId,
3373    name: &OsStr,
3374    kind: EntryKind,
3375    attrs: Attrs,
3376    root_dev: u64,
3377    config: &ScanConfig,
3378    children: &mut Vec<DetachedChild>,
3379    control: &mut Option<Op>,
3380    discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3381    report: &mut ScanReport,
3382) {
3383    let disposition = crate::admission::decide(name, kind, config.hidden(), config.exclude_special);
3384    if disposition == crate::admission::Disposition::Reject {
3385        return;
3386    }
3387    // Construct a full path only for the one fixed control name. The scanner's public
3388    // preparation builds one for every retained entry because that path escapes in an
3389    // observation; this private builder keeps ordinary children component-only.
3390    if config.read_controls && name == OsStr::new(crate::control::CONTROL_FILE_NAME) {
3391        let path = rel_dir.join(name);
3392        match read_control_op(config, root, &path, kind) {
3393            // A listing can repeat the control name while the directory changes. The
3394            // later read wins, as the builder keeps the later observation of the entry.
3395            Ok(observed) => *control = observed,
3396            Err(error) => report.errors.push(error),
3397        }
3398    }
3399    if disposition != crate::admission::Disposition::Retain {
3400        return;
3401    }
3402    report.observe(kind, attrs);
3403    // Positions only order repeated names, and no real listing reaches `u32::MAX` entries.
3404    let position = u32::try_from(children.len()).unwrap_or(u32::MAX);
3405    children.push(DetachedChild { name: name.to_os_string(), kind, attrs, position });
3406    if should_descend(kind, attrs, depth, root_dev, config) {
3407        let child_region = if depth == 0 { RegionId::UNASSIGNED } else { region };
3408        discovered.push((rel_dir.join(name), depth + 1, child_region));
3409    }
3410}
3411
3412/// One filesystem entry after the scan's shared admission, control, and descent rules.
3413pub(crate) struct PreparedWalkEntry {
3414    pub(crate) path: PathBuf,
3415    pub(crate) kind: EntryKind,
3416    pub(crate) attrs: Attrs,
3417    pub(crate) retained: bool,
3418    pub(crate) control: Option<Op>,
3419    pub(crate) descend: bool,
3420    pub(crate) control_error: Option<Error>,
3421}
3422
3423/// Apply the producer-independent part of a directory walk to one verified entry.
3424///
3425/// Both blocking and opened-root scans call this after obtaining non-following metadata,
3426/// which keeps admission, fixed controls, and traversal boundaries from drifting.
3427#[allow(clippy::too_many_arguments)]
3428pub(crate) fn prepare_walk_entry(
3429    root: &Path,
3430    rel_dir: &Path,
3431    depth: usize,
3432    name: &OsStr,
3433    kind: EntryKind,
3434    attrs: Attrs,
3435    root_dev: u64,
3436    config: &ScanConfig,
3437) -> Option<PreparedWalkEntry> {
3438    let disposition = crate::admission::decide(name, kind, config.hidden(), config.exclude_special);
3439    if disposition == crate::admission::Disposition::Reject {
3440        return None;
3441    }
3442    let path = rel_dir.join(name);
3443    let (control, control_error) = match read_control_op(config, root, &path, kind) {
3444        Ok(control) => (control, None),
3445        Err(error) => (None, Some(error)),
3446    };
3447    Some(PreparedWalkEntry {
3448        path,
3449        kind,
3450        attrs,
3451        retained: disposition == crate::admission::Disposition::Retain,
3452        control,
3453        descend: should_descend(kind, attrs, depth, root_dev, config),
3454        control_error,
3455    })
3456}
3457
3458#[allow(clippy::too_many_arguments)]
3459fn record_walk_entry(
3460    root: &Path,
3461    rel_dir: &Path,
3462    depth: usize,
3463    region: RegionId,
3464    name: &OsStr,
3465    kind: EntryKind,
3466    attrs: Attrs,
3467    root_dev: u64,
3468    config: &ScanConfig,
3469    emission: &mut StreamingEmission,
3470    discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3471    report: &mut ScanReport,
3472    sender: &std::sync::mpsc::Sender<WalkMessage>,
3473    chunk_send_ns: &mut u64,
3474    diagnostics: Option<&ScanDiagnosticsRecorder>,
3475) -> bool {
3476    let Some(prepared) =
3477        prepare_walk_entry(root, rel_dir, depth, name, kind, attrs, root_dev, config)
3478    else {
3479        return true;
3480    };
3481    if let Some(error) = prepared.control_error {
3482        report.errors.push(error);
3483    }
3484    if !prepared.retained {
3485        if let Some(control) = prepared.control {
3486            emission.batch.push(ObservationOp::unconditional(control));
3487            if emission.batch.len() >= config.batch_size {
3488                let send_started = std::time::Instant::now();
3489                let sent = emission.send_full(sender, diagnostics);
3490                *chunk_send_ns += elapsed_ns(send_started);
3491                return sent;
3492            }
3493        }
3494        return true;
3495    }
3496    report.observe(kind, attrs);
3497    emission.batch.push(ObservationOp::unconditional(Op::Upsert {
3498        path: prepared.path.clone(),
3499        kind,
3500        attrs,
3501    }));
3502    if emission.batch.len() >= config.batch_size {
3503        let send_started = std::time::Instant::now();
3504        let sent = emission.send_full(sender, diagnostics);
3505        *chunk_send_ns += elapsed_ns(send_started);
3506        if !sent {
3507            return false;
3508        }
3509    }
3510    if let Some(control) = prepared.control {
3511        emission.batch.push(ObservationOp::unconditional(control));
3512        if emission.batch.len() >= config.batch_size {
3513            let send_started = std::time::Instant::now();
3514            let sent = emission.send_full(sender, diagnostics);
3515            *chunk_send_ns += elapsed_ns(send_started);
3516            if !sent {
3517                return false;
3518            }
3519        }
3520    }
3521    if prepared.descend {
3522        // A child of the root seeds a new region; everything deeper inherits its
3523        // parent's. Region membership therefore costs one integer copy and never
3524        // inspects a path.
3525        let child_region = if depth == 0 { RegionId::UNASSIGNED } else { region };
3526        discovered.push((prepared.path, depth + 1, child_region));
3527    }
3528    true
3529}
3530
3531/// Observe one control file if the scan's policy asks for control state at all.
3532///
3533/// Every control observation goes through here -- each walk and reconcile site, and the
3534/// watch layer's verification -- so the policy cannot be forgotten at one of them. A
3535/// watch must honor it like a scan does: its scope has to equal the index's, the scope
3536/// carries this bit, and a verifier that read control files regardless would grow a
3537/// partial rule set, from whichever sources events touched, under a scope that says
3538/// there is none.
3539pub(crate) fn read_control_op(
3540    config: &ScanConfig,
3541    root: &Path,
3542    path: &Path,
3543    kind: EntryKind,
3544) -> Result<Option<Op>> {
3545    if !config.read_controls {
3546        return Ok(None);
3547    }
3548    read_control_op_unconditional(root, path, kind, config.control_limits.budget)
3549}
3550
3551/// Read a directory's fixed control before admitting any of its other children.
3552///
3553/// The extra metadata probe is paid only by an exclusion scan; it permits a streaming
3554/// directory listing while preserving control-first ordering for safe pruning.
3555fn read_directory_control(
3556    config: &ScanConfig,
3557    root: &Path,
3558    control_path: &Path,
3559) -> Result<Option<Op>> {
3560    let absolute = root.join(control_path);
3561    crate::counters::bump(|counts| counts.stats = counts.stats.saturating_add(1));
3562    let kind = match fs::symlink_metadata(&absolute) {
3563        Ok(metadata) if metadata.file_type().is_file() => EntryKind::File,
3564        Ok(_) => EntryKind::Other,
3565        Err(error) if error.kind() == std::io::ErrorKind::NotFound => return Ok(None),
3566        Err(error) => return Err(Error::io(&absolute, error)),
3567    };
3568    read_control_op(config, root, control_path, kind)
3569}
3570
3571fn read_listed_control_op(
3572    config: &ScanConfig,
3573    root: &Path,
3574    path: &Path,
3575    kind: EntryKind,
3576) -> Result<Option<Op>> {
3577    if config.population != crate::query::IgnoredEntries::Include
3578        && crate::control::is_control_file(path)
3579    {
3580        // The exclusion walk already read this source before any sibling.
3581        Ok(None)
3582    } else {
3583        read_control_op(config, root, path, kind)
3584    }
3585}
3586
3587fn population_prunes(
3588    population: crate::query::IgnoredEntries,
3589    path: &Path,
3590    kind: EntryKind,
3591    disposition: crate::admission::Disposition,
3592    controls: Option<&crate::control::ControlTable>,
3593    unreadable_controls: &std::collections::BTreeSet<PathBuf>,
3594) -> bool {
3595    // The fixed control source stays in the entry tier even for `only`: removing its
3596    // entry would also remove the retained rule source during a watch reconciliation.
3597    if crate::control::is_control_file(path) {
3598        return false;
3599    }
3600    let Some(table) = controls else { return false };
3601    if !table.classification_known(path)
3602        || path.ancestors().skip(1).any(|ancestor| unreadable_controls.contains(ancestor))
3603    {
3604        return false;
3605    }
3606    let ignored = table.is_ignored(path, kind.is_dir());
3607    match population {
3608        crate::query::IgnoredEntries::Include => false,
3609        crate::query::IgnoredEntries::Exclude => ignored,
3610        crate::query::IgnoredEntries::Only => {
3611            !ignored && !kind.is_dir() && disposition != crate::admission::Disposition::ControlOnly
3612        }
3613    }
3614}
3615
3616fn apply_discovery_control(table: &mut crate::control::ControlTable, op: &Op) -> Result<()> {
3617    match op {
3618        Op::ControlUpsert { path, source } => {
3619            table.upsert(path, source.clone())?;
3620        }
3621        Op::ControlRemove { path } => {
3622            table.remove(path)?;
3623        }
3624        _ => unreachable!("directory control probe emits only control operations"),
3625    }
3626    Ok(())
3627}
3628
3629/// Read one fixed control source without allowing a raced or hostile file to allocate
3630/// beyond the index-wide control budget.
3631///
3632/// A file longer than the budget is read only to one byte past it. No table under that
3633/// budget can admit a source that long, since every retained byte is charged at least
3634/// once, so the truncated source it sends is refused for the budget rather than parsed.
3635///
3636/// Private to this module, so no caller elsewhere can step around the policy gate in
3637/// `read_control_op`.
3638fn read_control_op_unconditional(
3639    root: &Path,
3640    path: &Path,
3641    kind: EntryKind,
3642    budget: Option<usize>,
3643) -> Result<Option<Op>> {
3644    if !crate::control::is_control_file(path) {
3645        return Ok(None);
3646    }
3647    if kind != EntryKind::File {
3648        return Ok(Some(Op::ControlRemove { path: path.to_path_buf() }));
3649    }
3650    let absolute = root.join(path);
3651    let file = open_control_file(&absolute).map_err(|error| Error::io(&absolute, error))?;
3652    if !file.metadata().map_err(|error| Error::io(&absolute, error))?.file_type().is_file() {
3653        return Ok(Some(Op::ControlRemove { path: path.to_path_buf() }));
3654    }
3655    let read_limit = budget
3656        .map_or(u64::MAX, |budget| u64::try_from(budget).unwrap_or(u64::MAX).saturating_add(1));
3657    let mut source = Vec::new();
3658    file.take(read_limit).read_to_end(&mut source).map_err(|error| Error::io(&absolute, error))?;
3659    crate::counters::bump(|counts| counts.control_reads = counts.control_reads.saturating_add(1));
3660    Ok(Some(Op::ControlUpsert { path: path.to_path_buf(), source }))
3661}
3662
3663#[cfg(unix)]
3664fn open_control_file(path: &Path) -> std::io::Result<fs::File> {
3665    use std::os::unix::fs::OpenOptionsExt as _;
3666
3667    fs::OpenOptions::new().read(true).custom_flags(libc::O_NONBLOCK | libc::O_NOFOLLOW).open(path)
3668}
3669
3670#[cfg(not(unix))]
3671fn open_control_file(path: &Path) -> std::io::Result<fs::File> {
3672    fs::File::open(path)
3673}
3674
3675fn send_scanner_batch(
3676    sender: &std::sync::mpsc::Sender<WalkMessage>,
3677    batch: ScannerBatch,
3678    diagnostics: Option<&ScanDiagnosticsRecorder>,
3679) -> bool {
3680    if let Some(diagnostics) = diagnostics {
3681        diagnostics.handoff_sent();
3682    }
3683    let sent = sender.send(WalkMessage::Batch(batch)).is_ok();
3684    if !sent {
3685        // Balance the reservation when the receiver disappeared before accepting it.
3686        if let Some(diagnostics) = diagnostics {
3687            diagnostics.handoff_received();
3688        }
3689    }
3690    sent
3691}
3692
3693fn send_detached_directories(
3694    sender: &std::sync::mpsc::Sender<WalkMessage>,
3695    directories: Vec<DetachedDirectory>,
3696    diagnostics: Option<&ScanDiagnosticsRecorder>,
3697) -> bool {
3698    if let Some(diagnostics) = diagnostics {
3699        diagnostics.handoff_sent();
3700    }
3701    let sent = sender.send(WalkMessage::DetachedDirectories(directories)).is_ok();
3702    if !sent {
3703        if let Some(diagnostics) = diagnostics {
3704            diagnostics.handoff_received();
3705        }
3706    }
3707    sent
3708}
3709
3710/// Nanoseconds since `started`, saturating rather than panicking on the absurd.
3711fn elapsed_ns(started: std::time::Instant) -> u64 {
3712    u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)
3713}
3714
3715/// A top-level subtree, used to spread workers across the breadth of the tree.
3716///
3717/// Every directory below the root belongs to the region seeded by its depth-1
3718/// ancestor, inherited from its parent rather than recomputed from its path. The root
3719/// itself is [`RegionId::ROOT`], which exists only to bootstrap.
3720#[derive(Clone, Copy, PartialEq, Eq, Debug)]
3721struct RegionId(usize);
3722
3723impl RegionId {
3724    /// The root's own region, which exists only to bootstrap the walk.
3725    const ROOT: Self = Self(0);
3726    /// "Allocate a fresh region for this directory." Resolved by
3727    /// [`DirectoryQueueState::push`], which is the only place holding the lock that
3728    /// owns the region table.
3729    const UNASSIGNED: Self = Self(usize::MAX);
3730}
3731
3732/// Directories still to read, plus enough state to know when the walk is finished.
3733///
3734/// The termination condition is the only subtle part: the queue being empty does not
3735/// mean the walk is done, because a worker that is mid-directory may be about to push
3736/// its children. So a worker holds a claim from the moment it takes work until the
3737/// moment it has published everything that work produced, and the walk ends only when
3738/// the queue is empty *and* no claim is outstanding.
3739///
3740/// # Why breadth-first is region-scheduled rather than a global FIFO
3741///
3742/// A global FIFO orders the *queue*, but claims are unordered: workers take whatever
3743/// is at the front, which on a real tree means several workers grinding through the
3744/// same top-level subtree while others sit untouched. Measured on the branching
3745/// fixture, that left a global-FIFO walk starting the same 7–8 of 12 subtrees at the
3746/// halfway mark as depth-first did — the ordering bought nothing a consumer could see.
3747/// It also made the pending set hold a whole level of the tree, which is where the
3748/// +1.5–3.7% peak RSS in exp-012 came from.
3749///
3750/// So breadth-first keeps work in per-region buckets and hands each free worker a
3751/// *different* region, round-robin. Within a region the bucket is LIFO, which restores
3752/// depth-first's locality and spine-bounded memory. Nothing waits on a level boundary:
3753/// if only one region has work, every worker takes it. The result is a scheduler whose
3754/// shallow preference is expressed in *which subtree a worker picks up*, not in the
3755/// order a single queue drains — which is the property progressive consumers actually
3756/// need.
3757struct DirectoryQueue {
3758    state: std::sync::Mutex<DirectoryQueueState>,
3759    ready: std::sync::Condvar,
3760    order: ScanOrder,
3761    diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
3762}
3763
3764/// One outstanding claim, held for exactly as long as the worker owes the queue the
3765/// work it took.
3766///
3767/// Giving the claim back is the queue's liveness condition, not a courtesy: [`claim`]
3768/// parks every other worker on the condvar while `outstanding` is nonzero, so a single
3769/// claim that is never returned stops the whole walk and the scoped join that waits on
3770/// it. That makes `Drop` the only safe place to put the release, because the paths that
3771/// skip a hand-written call are exactly the ones that matter — the `break` taken when
3772/// the consumer disconnects, and an unwinding panic inside a directory read.
3773///
3774/// [`claim`]: DirectoryQueue::claim
3775struct DirectoryClaim<'a> {
3776    queue: &'a DirectoryQueue,
3777    /// Directories represented by the claim, for exact in-flight accounting.
3778    directories: usize,
3779    /// Whether the worker already returned this claim through [`Self::release`].
3780    released: bool,
3781}
3782
3783impl DirectoryClaim<'_> {
3784    /// Return the claim at the end of a completed chunk, feeding the chunk's own
3785    /// measurements to the shared calibration.
3786    ///
3787    /// Returns the queue's scale-up decision, which is why the normal path cannot be
3788    /// `Drop`: a destructor has neither the chunk's timing nor anywhere to put an
3789    /// answer.
3790    fn release(
3791        mut self,
3792        entries: u64,
3793        work_ns: u64,
3794        timing: &mut WalkAttribution,
3795    ) -> Option<usize> {
3796        self.released = true;
3797        self.queue.release(self.directories, entries, work_ns, timing)
3798    }
3799}
3800
3801impl Drop for DirectoryClaim<'_> {
3802    fn drop(&mut self) {
3803        if self.released {
3804            return;
3805        }
3806        // An abandoned chunk: the consumer went away mid-directory, or a read panicked.
3807        // Either way the partial timing describes an aborted chunk rather than the cost
3808        // of reading directories, so it must not reach the calibration that sizes the
3809        // worker pool. Returning the claim is the whole job.
3810        self.queue.abandon(self.directories);
3811    }
3812}
3813
3814struct DirectoryQueueState {
3815    /// Depth-first's single stack. Unused under breadth-first.
3816    pending: VecDeque<(PathBuf, usize, RegionId)>,
3817    /// Breadth-first's per-region work, indexed by [`RegionId`]. Each is a LIFO stack.
3818    regions: Vec<Vec<(PathBuf, usize, RegionId)>>,
3819    /// Regions with work, in round-robin order. A region appears at most once; the
3820    /// flag array is what keeps that true without scanning the ring.
3821    ready_ring: VecDeque<RegionId>,
3822    /// Whether each region is currently in `ready_ring`.
3823    enqueued: Vec<bool>,
3824    /// Directories currently available for a future claim.
3825    ready_directories: usize,
3826    /// Directories held by outstanding claims.
3827    in_flight_directories: usize,
3828    outstanding: usize,
3829    finished: bool,
3830    controller: Option<WorkerController>,
3831    /// Observation-only windows retained after the shipped one-shot decision.
3832    shadow_calibration: Option<RepeatedCalibration>,
3833    /// Completion-order sequence assigned under the queue lock.
3834    next_policy_sequence: u64,
3835    worker_target: usize,
3836    maximum_workers: usize,
3837}
3838
3839impl DirectoryQueueState {
3840    fn seeded(
3841        root: (PathBuf, usize),
3842        order: ScanOrder,
3843        calibration: Option<WorkerCalibration>,
3844        initial_workers: usize,
3845        maximum_workers: usize,
3846        policy: WorkerPolicyExperiment,
3847    ) -> Self {
3848        let mut state = Self {
3849            pending: VecDeque::new(),
3850            regions: Vec::new(),
3851            ready_ring: VecDeque::new(),
3852            enqueued: Vec::new(),
3853            ready_directories: 0,
3854            in_flight_directories: 0,
3855            outstanding: 0,
3856            finished: false,
3857            controller: calibration.map(|value| WorkerController::new(value, policy)),
3858            shadow_calibration: None,
3859            next_policy_sequence: 0,
3860            worker_target: initial_workers,
3861            maximum_workers,
3862        };
3863        state.push((root.0, root.1, RegionId::ROOT), order);
3864        state
3865    }
3866
3867    /// Push one directory into the structure the order uses.
3868    fn push(&mut self, item: (PathBuf, usize, RegionId), order: ScanOrder) {
3869        self.ready_directories = self.ready_directories.saturating_add(1);
3870        match order {
3871            ScanOrder::DepthFirst => self.pending.push_back(item),
3872            ScanOrder::BreadthFirst => {
3873                let region = if item.2 == RegionId::UNASSIGNED {
3874                    // One region per top-level subtree, numbered as they are found.
3875                    self.regions.len().max(1)
3876                } else {
3877                    item.2.0
3878                };
3879                if region >= self.regions.len() {
3880                    self.regions.resize_with(region + 1, Vec::new);
3881                    self.enqueued.resize(region + 1, false);
3882                }
3883                // Resolve the id *into* the item, so every directory discovered beneath
3884                // this one inherits a concrete region instead of the sentinel. Without
3885                // this the sentinel propagates and each directory allocates a region of
3886                // its own, degenerating the scheduler into round-robin over the whole
3887                // frontier.
3888                let mut item = item;
3889                item.2 = RegionId(region);
3890                self.regions[region].push(item);
3891                if !self.enqueued[region] {
3892                    self.enqueued[region] = true;
3893                    self.ready_ring.push_back(RegionId(region));
3894                }
3895            }
3896        }
3897    }
3898
3899    /// Whether any work is available.
3900    fn is_empty(&self, order: ScanOrder) -> bool {
3901        match order {
3902            ScanOrder::DepthFirst => self.pending.is_empty(),
3903            ScanOrder::BreadthFirst => self.ready_ring.is_empty(),
3904        }
3905    }
3906
3907    /// Take up to `limit` directories from the next region in the round-robin ring.
3908    ///
3909    /// Every region holding work is in the ring exactly once, so popping it always
3910    /// finds work and always moves to a *different* subtree than the previous claim.
3911    /// An earlier version preferred the caller's previous region for locality, which
3912    /// pinned each worker to one subtree: with twelve deep chains and six workers only
3913    /// six subtrees ever advanced, and depth-first — whose four-directory claims
3914    /// happen to fan across the root's children — spread wider than breadth-first did.
3915    /// Locality still comes from the claim being a run of directories out of one
3916    /// region; it must not come from a worker refusing to leave.
3917    fn take(
3918        &mut self,
3919        limit: usize,
3920        order: ScanOrder,
3921        into: &mut Vec<(PathBuf, usize, RegionId)>,
3922    ) -> usize {
3923        let before = into.len();
3924        match order {
3925            ScanOrder::DepthFirst => {
3926                let take = self.pending.len().min(limit);
3927                let start = self.pending.len() - take;
3928                into.extend(self.pending.drain(start..));
3929            }
3930            ScanOrder::BreadthFirst => {
3931                let Some(region) = self.ready_ring.pop_front() else { return 0 };
3932                self.enqueued[region.0] = false;
3933                let bucket = &mut self.regions[region.0];
3934                let take = bucket.len().min(limit);
3935                let start = bucket.len() - take;
3936                into.extend(bucket.drain(start..));
3937                // Re-arm the region only if work remains and it is not already queued,
3938                // so a busy region cannot appear twice and starve the others.
3939                if !bucket.is_empty() && !self.enqueued[region.0] {
3940                    self.enqueued[region.0] = true;
3941                    self.ready_ring.push_back(region);
3942                }
3943            }
3944        }
3945        into.len().saturating_sub(before)
3946    }
3947
3948    fn allocate_policy_sequence(&mut self) -> u64 {
3949        let sequence = self.next_policy_sequence;
3950        self.next_policy_sequence = self.next_policy_sequence.saturating_add(1);
3951        sequence
3952    }
3953}
3954
3955impl DirectoryQueue {
3956    /// Seed the queue with the root, which is region zero until its children fan out.
3957    #[cfg(test)]
3958    fn new(
3959        root: (PathBuf, usize),
3960        order: ScanOrder,
3961        calibration: Option<WorkerCalibration>,
3962        diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
3963    ) -> Self {
3964        Self::new_with_policy(
3965            root,
3966            order,
3967            calibration,
3968            diagnostics,
3969            1,
3970            2,
3971            WorkerPolicyExperiment::ShippedOneShot,
3972        )
3973    }
3974
3975    #[allow(clippy::too_many_arguments)]
3976    fn new_with_policy(
3977        root: (PathBuf, usize),
3978        order: ScanOrder,
3979        calibration: Option<WorkerCalibration>,
3980        diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
3981        initial_workers: usize,
3982        maximum_workers: usize,
3983        policy: WorkerPolicyExperiment,
3984    ) -> Self {
3985        let state = DirectoryQueueState::seeded(
3986            root,
3987            order,
3988            calibration,
3989            initial_workers,
3990            maximum_workers,
3991            policy,
3992        );
3993        Self {
3994            state: std::sync::Mutex::new(state),
3995            ready: std::sync::Condvar::new(),
3996            order,
3997            diagnostics,
3998        }
3999    }
4000
4001    /// Take up to [`DIR_CLAIM`] directories, blocking until there is work or the walk
4002    /// is over. Returns `None` once no more work will ever arrive.
4003    ///
4004    /// Time spent waiting is charged to `timing`: lock acquisition to `lock_wait_ns`
4005    /// when contended, condvar waits to `starved_ns`. The condvar span includes the
4006    /// lock re-acquisition on wake, which slightly overstates starvation rather than
4007    /// understating contention — the fail-honest direction for the number that is
4008    /// supposed to stay near zero.
4009    fn claim<'a>(
4010        &'a self,
4011        into: &mut Vec<(PathBuf, usize, RegionId)>,
4012        timing: &mut WalkAttribution,
4013    ) -> Option<DirectoryClaim<'a>> {
4014        let mut state = self.lock_timed(timing);
4015        loop {
4016            if !state.is_empty(self.order) {
4017                let directories = state.take(DIR_CLAIM, self.order, into);
4018                state.ready_directories = state.ready_directories.saturating_sub(directories);
4019                state.in_flight_directories =
4020                    state.in_flight_directories.saturating_add(directories);
4021                state.outstanding += 1;
4022                timing.claims += 1;
4023                return Some(DirectoryClaim { queue: self, directories, released: false });
4024            }
4025            if state.finished {
4026                return None;
4027            }
4028            if state.outstanding == 0 {
4029                state.finished = true;
4030                self.ready.notify_all();
4031                return None;
4032            }
4033            let started = std::time::Instant::now();
4034            state = self.ready.wait(state).unwrap_or_else(std::sync::PoisonError::into_inner);
4035            timing.starved_ns += elapsed_ns(started);
4036        }
4037    }
4038
4039    fn extend(
4040        &self,
4041        directories: impl Iterator<Item = (PathBuf, usize, RegionId)>,
4042        timing: &mut WalkAttribution,
4043    ) {
4044        let mut state = self.lock_timed(timing);
4045        for item in directories {
4046            state.push(item, self.order);
4047        }
4048        drop(state);
4049        self.ready.notify_all();
4050    }
4051
4052    /// Give up a claim whose chunk never finished. Wakes everyone if it was the last.
4053    ///
4054    /// Reached only from [`DirectoryClaim::drop`], where there is no `WalkAttribution`
4055    /// to charge and nothing worth charging: an abandoned chunk read some unknown
4056    /// fraction of its directories, so its lock wait says nothing about contention
4057    /// during the walk.
4058    fn abandon(&self, directories: usize) {
4059        let mut state = self.lock();
4060        state.outstanding -= 1;
4061        state.in_flight_directories = state.in_flight_directories.saturating_sub(directories);
4062        if state.outstanding == 0 && state.is_empty(self.order) {
4063            state.finished = true;
4064            drop(state);
4065            self.ready.notify_all();
4066        }
4067    }
4068
4069    /// Give up a claim taken by [`claim`]. Wakes everyone if this was the last one.
4070    ///
4071    /// The chunk's own entry count and work time feed the shared service-time
4072    /// calibration, so the decision uses the timing the walk already collects for
4073    /// attribution rather than a second clock. Returns the new worker target when a
4074    /// controller requests expansion. A release that ends the walk returns no target,
4075    /// because there is no longer useful work for a reserve worker to take.
4076    fn release(
4077        &self,
4078        directories: usize,
4079        observed_entries: u64,
4080        observed_work_ns: u64,
4081        timing: &mut WalkAttribution,
4082    ) -> Option<usize> {
4083        let mut state = self.lock_timed(timing);
4084        // Whether this chunk reached the calibration at all. Only chunks released while
4085        // it is still live are policy history; later ones are ordinary walk work.
4086        let calibrating = state.controller.is_some();
4087        let one_shot = state.controller.as_ref().is_some_and(WorkerController::is_one_shot);
4088        let controller_spec = state.controller.as_ref().map(WorkerController::calibration_spec);
4089        let staged_gated = state.controller.as_ref().is_some_and(WorkerController::is_staged_gated);
4090        let completed_window = state
4091            .controller
4092            .as_mut()
4093            .and_then(|value| value.observe(observed_entries, observed_work_ns));
4094        let observed_window = state
4095            .shadow_calibration
4096            .as_mut()
4097            .and_then(|value| value.observe(observed_entries, observed_work_ns));
4098        let window_completed = completed_window.is_some();
4099        state.outstanding -= 1;
4100        state.in_flight_directories = state.in_flight_directories.saturating_sub(directories);
4101        let finished = state.outstanding == 0 && state.is_empty(self.order);
4102        if finished {
4103            state.finished = true;
4104        }
4105        let handoff_backlog = self.diagnostics.as_ref().map_or(0, |diagnostics| {
4106            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
4107        });
4108        let mut requested_workers = None;
4109        let policy_window = completed_window.map(|window| {
4110            let useful_frontier =
4111                state.ready_directories.saturating_add(state.in_flight_directories);
4112            let decision = if !window.slow {
4113                WorkerPolicyDecision::Hold
4114            } else if finished {
4115                WorkerPolicyDecision::HoldNoUsefulWork
4116            } else if staged_gated && useful_frontier <= state.worker_target {
4117                WorkerPolicyDecision::HoldInsufficientFrontier
4118            } else if staged_gated && handoff_backlog >= state.worker_target {
4119                WorkerPolicyDecision::HoldHandoffBacklog
4120            } else {
4121                let target = if staged_gated {
4122                    state.worker_target.saturating_mul(2).min(state.maximum_workers)
4123                } else {
4124                    state.maximum_workers
4125                };
4126                if target > state.worker_target {
4127                    state.worker_target = target;
4128                    requested_workers = Some(target);
4129                    WorkerPolicyDecision::ScaleUp
4130                } else {
4131                    WorkerPolicyDecision::HoldInsufficientFrontier
4132                }
4133            };
4134            let sequence = state.allocate_policy_sequence();
4135            PolicyWindowSnapshot {
4136                sequence,
4137                start_entry_ordinal: window.start_entry_ordinal,
4138                end_entry_ordinal: window.end_entry_ordinal,
4139                observed_entries: window.entries,
4140                observed_chunks: window.chunks,
4141                observed_work_ns: window.work_ns,
4142                ready_directories: state.ready_directories,
4143                in_flight_directories: state.in_flight_directories,
4144                active_workers: self.diagnostics.as_ref().map_or(0, |diagnostics| {
4145                    diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
4146                }),
4147                handoff_backlog,
4148                requested_workers,
4149                decision,
4150            }
4151        });
4152        let shadow_window = observed_window.map(|window| {
4153            let sequence = state.allocate_policy_sequence();
4154            PolicyWindowSnapshot {
4155                sequence,
4156                start_entry_ordinal: window.start_entry_ordinal,
4157                end_entry_ordinal: window.end_entry_ordinal,
4158                observed_entries: window.entries,
4159                observed_chunks: window.chunks,
4160                observed_work_ns: window.work_ns,
4161                ready_directories: state.ready_directories,
4162                in_flight_directories: state.in_flight_directories,
4163                active_workers: self.diagnostics.as_ref().map_or(0, |diagnostics| {
4164                    diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
4165                }),
4166                handoff_backlog,
4167                requested_workers: None,
4168                decision: if window.slow {
4169                    WorkerPolicyDecision::ObserveSlow
4170                } else {
4171                    WorkerPolicyDecision::ObserveFast
4172                },
4173            }
4174        });
4175        let controller_terminates = (one_shot || finished) && window_completed
4176            || requested_workers.is_some_and(|target| target == state.maximum_workers);
4177        if controller_terminates {
4178            // Continue observing after every terminal decision, including a candidate
4179            // that reached the maximum pool. Without this shadow history a slow-prefix
4180            // expansion makes a later fast phase unobservable, precisely the
4181            // irreversible over-expansion case the evidence matrix must detect.
4182            if let (Some(spec), Some(window)) = (controller_spec, completed_window) {
4183                if self.diagnostics.is_some() && !finished {
4184                    state.shadow_calibration =
4185                        Some(RepeatedCalibration::starting_at(spec, window.end_entry_ordinal));
4186                }
4187            }
4188            state.controller = None;
4189        }
4190        drop(state);
4191
4192        // The sequence was assigned while the queue was locked, but the trace lock and
4193        // bounded-vector update stay outside that critical section. Recorder arrival
4194        // may differ from completion order; `finish` sorts the retained prefix.
4195        if let (Some(diagnostics), Some(window)) = (&self.diagnostics, policy_window) {
4196            diagnostics.record_policy_window(window);
4197        }
4198        if let (Some(diagnostics), Some(window)) = (&self.diagnostics, shadow_window) {
4199            diagnostics.record_policy_window(window);
4200        }
4201
4202        // Recorded outside the lock: the sampling is off by default, and a disabled
4203        // counter must not lengthen the critical section it observes.
4204        if calibrating {
4205            record_adaptive_calibration_chunk(
4206                self.diagnostics.as_ref(),
4207                observed_entries,
4208                observed_work_ns,
4209            );
4210        }
4211
4212        if finished {
4213            self.ready.notify_all();
4214        }
4215        requested_workers
4216    }
4217
4218    /// Acquire the state lock, charging any contention to `timing`.
4219    ///
4220    /// The fast path is a `try_lock` that succeeds and costs one counter increment;
4221    /// only the contended path pays for reading the clock. Poisoning is tolerated for
4222    /// the same reason as [`Self::lock`].
4223    fn lock_timed(
4224        &self,
4225        timing: &mut WalkAttribution,
4226    ) -> std::sync::MutexGuard<'_, DirectoryQueueState> {
4227        timing.lock_ops += 1;
4228        match self.state.try_lock() {
4229            Ok(guard) => guard,
4230            Err(std::sync::TryLockError::Poisoned(poisoned)) => poisoned.into_inner(),
4231            Err(std::sync::TryLockError::WouldBlock) => {
4232                timing.lock_contended += 1;
4233                let started = std::time::Instant::now();
4234                let guard = self.lock();
4235                timing.lock_wait_ns += elapsed_ns(started);
4236                guard
4237            }
4238        }
4239    }
4240
4241    /// A poisoned queue means a worker panicked mid-walk. The data behind the lock is
4242    /// a plain work list with no invariant that a panic could have broken, and the
4243    /// caller already reports the panic as a scan error, so recovering the list is
4244    /// strictly better than propagating a second panic into every other worker.
4245    fn lock(&self) -> std::sync::MutexGuard<'_, DirectoryQueueState> {
4246        self.state.lock().unwrap_or_else(std::sync::PoisonError::into_inner)
4247    }
4248}
4249
4250fn record_adaptive_calibration_chunk(
4251    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
4252    entries: u64,
4253    work_ns: u64,
4254) {
4255    if let Some(diagnostics) = diagnostics {
4256        diagnostics.calibration_chunk(entries, work_ns);
4257    }
4258    crate::counters::bump(|counts| {
4259        counts.adaptive_calibration_chunks = counts.adaptive_calibration_chunks.saturating_add(1);
4260        counts.adaptive_calibration_entries =
4261            counts.adaptive_calibration_entries.saturating_add(entries);
4262        counts.adaptive_calibration_work_us =
4263            counts.adaptive_calibration_work_us.saturating_add(work_ns / 1_000);
4264    });
4265}
4266
4267fn record_adaptive_worker_expansion(diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>) {
4268    if let Some(diagnostics) = diagnostics {
4269        diagnostics.worker_expanded();
4270    }
4271    crate::counters::bump(|counts| {
4272        counts.adaptive_scale_ups = counts.adaptive_scale_ups.saturating_add(1);
4273    });
4274}
4275
4276fn scan_detached_directories(
4277    root: &Path,
4278    config: &ScanConfig,
4279    collect_diagnostics: bool,
4280    policy: WorkerPolicyExperiment,
4281) -> Result<(ScanReport, DetachedIndexBuilder, Option<ScanDiagnostics>)> {
4282    if let Some(progress) = &config.progress {
4283        progress.enter(crate::ProgressPhase::Scanning);
4284    }
4285    let root_metadata = {
4286        crate::counters::bump(|counts| counts.stats += 1);
4287        fs::symlink_metadata(root)
4288    }
4289    .map_err(|error| Error::io(root, error))?;
4290    if !root_metadata.is_dir() {
4291        return Err(Error::io(
4292            root,
4293            std::io::Error::new(std::io::ErrorKind::NotADirectory, "scan root is not a directory"),
4294        ));
4295    }
4296    let root_dev = root_device(root, &root_metadata).map_err(|error| Error::io(root, error))?;
4297    let available_parallelism =
4298        std::thread::available_parallelism().map_or(1, std::num::NonZero::get);
4299    let pool = config.worker_pool_for(available_parallelism);
4300    let diagnostics = collect_diagnostics
4301        .then(|| ScanDiagnosticsRecorder::new(pool, available_parallelism, policy));
4302
4303    if config.max_depth == Some(0) {
4304        if let Some(diagnostics) = &diagnostics {
4305            diagnostics.mark_not_run();
4306            diagnostics.record_queue_finish(0, 0);
4307        }
4308        return Ok((
4309            ScanReport::default(),
4310            DetachedIndexBuilder::new(root, config.scope(), config.types_shared())
4311                .with_control_limits(config.control_limits),
4312            diagnostics.as_ref().map(|value| value.finish()),
4313        ));
4314    }
4315
4316    let walk_started = crate::counters::enabled().then(std::time::Instant::now);
4317    let (output, builder) =
4318        scan_concurrent_detached(root, config, root_dev, pool, diagnostics.as_ref(), policy)?;
4319    // Also reached by a walk no worker left, such as a single-threaded one, so every
4320    // cold index ends its walk in the same phase.
4321    if let Some(progress) = &config.progress {
4322        progress.enter(crate::ProgressPhase::Indexing);
4323    }
4324    if let Some(started) = walk_started {
4325        let elapsed = elapsed_ns(started) / 1_000;
4326        crate::counters::bump(|counts| {
4327            counts.detached_walk_us = counts.detached_walk_us.saturating_add(elapsed);
4328        });
4329    }
4330    Ok((output, builder, diagnostics.as_ref().map(|value| value.finish())))
4331}
4332
4333fn consolidate_detached_index(
4334    mut output: ScanReport,
4335    builder: DetachedIndexBuilder,
4336) -> (Index, ScanReport) {
4337    let entries = output.entries;
4338    let consolidate_started = crate::counters::enabled().then(std::time::Instant::now);
4339    let mut index = builder.finish();
4340    if let Some(started) = consolidate_started {
4341        let elapsed = elapsed_ns(started) / 1_000;
4342        crate::counters::bump(|counts| {
4343            counts.detached_builds = counts.detached_builds.saturating_add(1);
4344            counts.detached_entries = counts.detached_entries.saturating_add(entries);
4345            counts.detached_finish_us = counts.detached_finish_us.saturating_add(elapsed);
4346        });
4347    }
4348    index.record_walk_errors(&mut output.errors);
4349    index.set_initial_scan_freshness(&output.errors);
4350    (index, output)
4351}
4352
4353/// Walk `root` and return a fully populated index.
4354pub fn scan_into_index(root: &Path, config: &ScanConfig) -> Result<(Index, ScanReport)> {
4355    config.validate()?;
4356    let root = root.canonicalize().map_err(|error| Error::io(root, error))?;
4357    if config.population != crate::query::IgnoredEntries::Include {
4358        let (index, report, _) = scan_into_index_with_scanner(
4359            &root,
4360            config,
4361            false,
4362            WorkerPolicyExperiment::ShippedOneShot,
4363        )?;
4364        return Ok((index, report));
4365    }
4366    let (output, builder, _diagnostics) =
4367        scan_detached_directories(&root, config, false, WorkerPolicyExperiment::ShippedOneShot)?;
4368    Ok(consolidate_detached_index(output, builder))
4369}
4370
4371fn scan_into_index_with_scanner(
4372    root: &Path,
4373    config: &ScanConfig,
4374    collect_diagnostics: bool,
4375    policy: WorkerPolicyExperiment,
4376) -> Result<(Index, ScanReport, Option<ScanDiagnostics>)> {
4377    let mut index = Index::new_with_scope_and_types(root, config.scope(), config.types_shared());
4378    index.set_control_limits(config.control_limits);
4379    let mut apply_error: Option<Error> = None;
4380    let (mut report, diagnostics) = scan_internal(
4381        root,
4382        config,
4383        &mut |batch| {
4384            if apply_error.is_none() {
4385                if let Err(error) = index.apply_scanner_baseline(batch) {
4386                    apply_error = Some(error);
4387                }
4388            }
4389        },
4390        collect_diagnostics,
4391        policy,
4392        SinkMode::Retained,
4393    )?;
4394    if let Some(error) = apply_error {
4395        return Err(error);
4396    }
4397    index.record_walk_errors(&mut report.errors);
4398    index.set_initial_scan_freshness(&report.errors);
4399    Ok((index, report, diagnostics))
4400}
4401
4402#[cfg(test)]
4403fn scan_into_index_via_scanner(root: &Path, config: &ScanConfig) -> Result<(Index, ScanReport)> {
4404    let (index, report, _) =
4405        scan_into_index_with_scanner(root, config, false, WorkerPolicyExperiment::ShippedOneShot)?;
4406    Ok((index, report))
4407}
4408
4409/// Walk `root` into an index and retain the opt-in diagnostic trace for that run.
4410///
4411/// The index and [`ScanReport`] have the same semantics as [`scan_into_index`]. The
4412/// additional trace is versioned independently so evidence tooling can fail closed on
4413/// changes without coupling cache or engine behavior to measurement details.
4414pub fn scan_into_index_with_diagnostics(
4415    root: &Path,
4416    config: &ScanConfig,
4417) -> Result<(Index, ScanReport, ScanDiagnostics)> {
4418    scan_into_index_with_policy_diagnostics(root, config, WorkerPolicyExperiment::ShippedOneShot)
4419}
4420
4421/// Exercise a repository-only worker-controller candidate while building an index.
4422#[doc(hidden)]
4423pub fn scan_into_index_with_policy_diagnostics(
4424    root: &Path,
4425    config: &ScanConfig,
4426    policy: WorkerPolicyExperiment,
4427) -> Result<(Index, ScanReport, ScanDiagnostics)> {
4428    config.validate()?;
4429    let root = root.canonicalize().map_err(|error| Error::io(root, error))?;
4430    if config.population != crate::query::IgnoredEntries::Include {
4431        let (index, report, diagnostics) =
4432            scan_into_index_with_scanner(&root, config, true, policy)?;
4433        return Ok((index, report, diagnostics.expect("diagnostic scanner creates a recorder")));
4434    }
4435    let (output, builder, diagnostics) = scan_detached_directories(&root, config, true, policy)?;
4436    let (index, report) = consolidate_detached_index(output, builder);
4437    Ok((index, report, diagnostics.expect("diagnostic detached scan creates a recorder")))
4438}
4439
4440/// Diff the filesystem against an existing index and emit conditional observations.
4441///
4442/// This is cache tier 2: after a snapshot is loaded, a sweep like this is what makes the
4443/// answer trustworthy rather than merely fast. Unchanged entries produce upserts whose
4444/// fingerprints already match, which the index discards as no-ops, so the caller can
4445/// apply the whole stream without filtering it first.
4446///
4447/// Entries the index holds but the filesystem no longer has become [`Op::Remove`],
4448/// detected per directory rather than by accumulating every visited path in memory.
4449///
4450/// This observation-only reference API assumes its emitted stream is applied to the same
4451/// unchanged baseline after the borrow ends. Use [`reconcile`] or [`reconcile_handle`]
4452/// when other producers can write concurrently; those paths capture the stronger
4453/// generation/revision/absence expectations returned by [`Index::expectation`].
4454pub fn revalidate(
4455    index: &Index,
4456    config: &ScanConfig,
4457    sink: &mut dyn FnMut(Observation),
4458) -> Result<ScanReport> {
4459    config.validate_for_scope(index.scope())?;
4460    let root = index.root_path().to_path_buf();
4461    let root_meta = {
4462        crate::counters::bump(|c| c.stats += 1);
4463        fs::symlink_metadata(&root)
4464    }
4465    .map_err(|error| Error::io(&root, error))?;
4466    if !root_meta.is_dir() {
4467        return Err(Error::io(
4468            &root,
4469            std::io::Error::new(
4470                std::io::ErrorKind::NotADirectory,
4471                "revalidation root is not a directory",
4472            ),
4473        ));
4474    }
4475    let root_dev = root_device(&root, &root_meta).map_err(|error| Error::io(&root, error))?;
4476    if let Some(progress) = &config.progress {
4477        progress.enter(crate::ProgressPhase::Revalidating);
4478    }
4479    let mut report = ScanReport::default();
4480    let mut tally = ProgressTally::new(config.progress.as_ref());
4481    let batch_limit = config.batch_size.max(1);
4482    let mut batch: Vec<ObservationOp> = Vec::with_capacity(batch_limit);
4483    if config.max_depth == Some(0) {
4484        if let Some(children) = index.children(Path::new("")) {
4485            for (name, _) in children {
4486                let path = PathBuf::from(name);
4487                batch.push(ObservationOp::if_state(
4488                    Op::Remove { path: path.clone() },
4489                    index.relaxed_expectation(&path),
4490                ));
4491                if batch.len() >= batch_limit {
4492                    sink(Observation::from_ops(std::mem::take(&mut batch)));
4493                    batch.reserve(batch_limit);
4494                }
4495            }
4496        }
4497        if !batch.is_empty() {
4498            sink(Observation::from_ops(batch));
4499        }
4500        return Ok(report);
4501    }
4502    let mut queue: VecDeque<(PathBuf, usize)> = VecDeque::from(vec![(PathBuf::new(), 0)]);
4503    let mut controls = (config.population != crate::query::IgnoredEntries::Include)
4504        .then(|| index.control_table().clone());
4505    let mut unreadable_controls = std::collections::BTreeSet::new();
4506
4507    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
4508        let abs_dir = root.join(&rel_dir);
4509        let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
4510        let mut had_control = index.control_table().contains(&control_path);
4511        let mut control_seen = false;
4512        if let Some(table) = controls.as_mut() {
4513            let baseline = index.relaxed_expectation(&control_path);
4514            match read_directory_control(config, &root, &control_path) {
4515                Ok(Some(op)) => {
4516                    apply_discovery_control(table, &op)?;
4517                    batch.push(ObservationOp::if_state(op, baseline));
4518                    control_seen = true;
4519                }
4520                Ok(None) => {
4521                    table.remove(&control_path)?;
4522                    if had_control {
4523                        batch.push(ObservationOp::if_state(
4524                            Op::ControlRemove { path: control_path.clone() },
4525                            baseline,
4526                        ));
4527                        had_control = false;
4528                    }
4529                }
4530                Err(error) => {
4531                    unreadable_controls.insert(rel_dir.clone());
4532                    report.errors.push(error);
4533                    control_seen = true;
4534                }
4535            }
4536        }
4537        crate::counters::bump(|c| c.dir_opens += 1);
4538        let listing = match fs::read_dir(&abs_dir) {
4539            Ok(listing) => listing,
4540            Err(e) => {
4541                report.errors.push(Error::io(abs_dir, e));
4542                continue;
4543            }
4544        };
4545        report.dirs_read += 1;
4546
4547        let mut seen: BTreeSet<OsString> = BTreeSet::new();
4548        let mut listing_complete = true;
4549        let listing = reconcile_listing(listing, &abs_dir);
4550        for item in listing {
4551            let item = match item {
4552                Ok(item) => item,
4553                Err(e) => {
4554                    listing_complete = false;
4555                    report.errors.push(Error::io(&abs_dir, e));
4556                    continue;
4557                }
4558            };
4559            let name = item.file_name();
4560            // Seeing the name proves it is not absent even when a following metadata
4561            // lookup fails. Record it before any fallible per-entry work so an
4562            // operational error cannot become a false removal in the missing sweep.
4563            seen.insert(name.clone());
4564            let rel_path = rel_dir.join(&name);
4565            let baseline = index.relaxed_expectation(&rel_path);
4566            let (kind, attrs) = match observe_dir_entry(&item) {
4567                Ok(Some(observed)) => observed,
4568                Ok(None) => {
4569                    let entry_held = baseline.state != PathState::Absent;
4570                    for removal in
4571                        vanished_child_removals(&rel_dir, &name, entry_held, &mut had_control)
4572                            .into_iter()
4573                            .flatten()
4574                    {
4575                        batch.push(ObservationOp::if_state(removal, baseline));
4576                    }
4577                    if batch.len() >= batch_limit {
4578                        sink(Observation::from_ops(std::mem::take(&mut batch)));
4579                        batch.reserve(batch_limit);
4580                    }
4581                    continue;
4582                }
4583                Err(e) => {
4584                    control_seen |= name == crate::control::CONTROL_FILE_NAME;
4585                    report.errors.push(Error::io(item.path(), e));
4586                    continue;
4587                }
4588            };
4589            control_seen |= name == crate::control::CONTROL_FILE_NAME;
4590            let disposition =
4591                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
4592            if population_prunes(
4593                config.population,
4594                &rel_path,
4595                kind,
4596                disposition,
4597                controls.as_ref(),
4598                &unreadable_controls,
4599            ) {
4600                if baseline.state != PathState::Absent {
4601                    batch.push(ObservationOp::if_state(Op::Remove { path: rel_path }, baseline));
4602                }
4603                continue;
4604            }
4605            let control = match read_listed_control_op(config, &root, &rel_path, kind) {
4606                Ok(control) => control,
4607                Err(error) => {
4608                    report.errors.push(error);
4609                    None
4610                }
4611            };
4612            if disposition != crate::admission::Disposition::Retain {
4613                if baseline.state != PathState::Absent {
4614                    batch.push(ObservationOp::if_state(
4615                        Op::Remove { path: rel_path.clone() },
4616                        baseline,
4617                    ));
4618                }
4619                if disposition == crate::admission::Disposition::ControlOnly {
4620                    if let Some(control) = control {
4621                        batch.push(ObservationOp::if_state(control, baseline));
4622                    }
4623                }
4624                if batch.len() >= batch_limit {
4625                    sink(Observation::from_ops(std::mem::take(&mut batch)));
4626                    batch.reserve(batch_limit);
4627                }
4628                continue;
4629            }
4630            report.observe(kind, attrs);
4631            batch.push(ObservationOp::if_state(
4632                Op::Upsert { path: rel_path.clone(), kind, attrs },
4633                baseline,
4634            ));
4635            if batch.len() >= batch_limit {
4636                sink(Observation::from_ops(std::mem::take(&mut batch)));
4637                batch.reserve(batch_limit);
4638            }
4639            if let Some(control) = control {
4640                batch.push(ObservationOp::if_state(control, baseline));
4641                if batch.len() >= batch_limit {
4642                    sink(Observation::from_ops(std::mem::take(&mut batch)));
4643                    batch.reserve(batch_limit);
4644                }
4645            }
4646
4647            if should_descend(kind, attrs, depth, root_dev, config) {
4648                queue.push_back((rel_path, depth + 1));
4649            } else if kind.is_dir() {
4650                if let Some(children) = index.children(&rel_path) {
4651                    for (child_name, _) in children {
4652                        let child_path = rel_path.join(child_name);
4653                        batch.push(ObservationOp::if_state(
4654                            Op::Remove { path: child_path.clone() },
4655                            index.relaxed_expectation(&child_path),
4656                        ));
4657                        if batch.len() >= batch_limit {
4658                            sink(Observation::from_ops(std::mem::take(&mut batch)));
4659                            batch.reserve(batch_limit);
4660                        }
4661                    }
4662                }
4663            }
4664        }
4665
4666        // Anything the index still lists here but the filesystem did not return is gone.
4667        if listing_complete {
4668            if let Some(known) = index.children(&rel_dir) {
4669                for (name, _) in known {
4670                    if !seen.contains(name) {
4671                        let path = rel_dir.join(name);
4672                        batch.push(ObservationOp::if_state(
4673                            Op::Remove { path: path.clone() },
4674                            index.relaxed_expectation(&path),
4675                        ));
4676                    }
4677                }
4678            }
4679            if had_control && !control_seen {
4680                batch.push(ObservationOp::if_state(
4681                    Op::ControlRemove { path: control_path.clone() },
4682                    index.relaxed_expectation(&control_path),
4683                ));
4684            }
4685        }
4686        // Per directory: an unchanged tree fills no batch, so the batch cannot be the
4687        // unit here without the counters standing still for the whole walk.
4688        tally.flush(&report);
4689    }
4690
4691    if !batch.is_empty() {
4692        sink(Observation::from_ops(batch));
4693    }
4694    tally.flush(&report);
4695    Ok(report)
4696}
4697
4698/// Reconcile the full index and publish each exact commit as it lands.
4699pub fn reconcile(
4700    index: &mut Index,
4701    config: &ScanConfig,
4702    sink: &mut dyn FnMut(&Commit),
4703) -> Result<ReconcileReport> {
4704    reconcile_subtree(index, Path::new(""), config, sink)
4705}
4706
4707/// Reconcile one relative subtree, applying effective changes during the walk.
4708///
4709/// If an ancestor vanished or became a non-directory, reconciliation widens to that
4710/// ancestor so a child invalidation can converge instead of retrying `ENOTDIR` forever.
4711pub fn reconcile_subtree(
4712    index: &mut Index,
4713    subtree: &Path,
4714    config: &ScanConfig,
4715    sink: &mut dyn FnMut(&Commit),
4716) -> Result<ReconcileReport> {
4717    reconcile_target(&mut ReconcileTarget::Direct(index), subtree, config, sink)
4718}
4719
4720/// Reconcile a shared index while allowing readers between applied batches.
4721pub fn reconcile_handle(
4722    handle: &IndexHandle,
4723    config: &ScanConfig,
4724    sink: &mut dyn FnMut(&Commit),
4725) -> Result<ReconcileReport> {
4726    reconcile_subtree_handle(handle, Path::new(""), config, sink)
4727}
4728
4729/// Reconcile one subtree of a shared index, widening to a missing/non-directory ancestor
4730/// when necessary.
4731pub fn reconcile_subtree_handle(
4732    handle: &IndexHandle,
4733    subtree: &Path,
4734    config: &ScanConfig,
4735    sink: &mut dyn FnMut(&Commit),
4736) -> Result<ReconcileReport> {
4737    reconcile_target(&mut ReconcileTarget::Shared(handle), subtree, config, sink)
4738}
4739
4740/// Internal effects of one opened-root multi-path reconciliation.
4741#[derive(Debug, Default)]
4742pub(crate) struct ReconcilePathsReport {
4743    pub(crate) reconciliation: ReconcileReport,
4744    pub(crate) accepted: Vec<PathBuf>,
4745    pub(crate) rejected: Vec<crate::RejectedRefreshPath>,
4746}
4747
4748/// Reconcile one bounded path set under an opened-root lifecycle controller.
4749///
4750/// Classification precedes I/O, overlapping descendants fold into one walk, and all
4751/// surviving scopes enter `Reconciling` before the first is read. `forbid_expansion`
4752/// is the conservative resource-stop rule: removals and same-file verification remain
4753/// legal, while work that could retain another file or discover children is refused.
4754pub(crate) fn reconcile_paths_handle_controlled(
4755    handle: &IndexHandle,
4756    paths: &[PathBuf],
4757    config: &ScanConfig,
4758    forbid_expansion: bool,
4759    control: &dyn ReconcileControl,
4760    sink: &mut dyn FnMut(&Commit),
4761) -> Result<ReconcilePathsReport> {
4762    let mut target = ReconcileTarget::Controlled { handle, control };
4763    reconcile_paths_target(&mut target, paths, config, forbid_expansion, sink)
4764}
4765
4766fn reconcile_paths_target(
4767    target: &mut ReconcileTarget<'_>,
4768    paths: &[PathBuf],
4769    config: &ScanConfig,
4770    forbid_expansion: bool,
4771    sink: &mut dyn FnMut(&Commit),
4772) -> Result<ReconcilePathsReport> {
4773    config.validate_for_scope(target.scope()?)?;
4774    let mut report = ReconcilePathsReport::default();
4775    let mut accepted = BTreeSet::new();
4776
4777    for requested in paths {
4778        let reject = |reason| crate::RejectedRefreshPath { path: requested.clone(), reason };
4779        let Ok(path) = normalize_subtree(requested) else {
4780            report.rejected.push(reject(crate::RefreshRejection::OutsideRoot));
4781            continue;
4782        };
4783        if config.max_depth.is_some_and(|maximum| path.components().count() > maximum) {
4784            report.rejected.push(reject(crate::RefreshRejection::BeyondDepth));
4785            continue;
4786        }
4787        // This is lexical admission before the final kind is observed. Treating the
4788        // boundary as a file preserves the fixed hidden `.gitignore` control exception;
4789        // the verified walk still applies the real kind and special-object policy.
4790        if crate::admission::decide_path(&path, EntryKind::File, config.hidden(), false)
4791            == crate::admission::Disposition::Reject
4792        {
4793            report.rejected.push(reject(crate::RefreshRejection::NotAdmitted));
4794            continue;
4795        }
4796        if forbid_expansion && refresh_may_expand(target, &path, &mut report.reconciliation.scan)? {
4797            report.rejected.push(reject(crate::RefreshRejection::ResourceBudget));
4798            continue;
4799        }
4800        accepted.insert(path);
4801    }
4802
4803    report.accepted = accepted.into_iter().collect();
4804    let mut resolved = Vec::new();
4805    let mut unsafe_roots = Vec::new();
4806    for requested_root in covering_roots(report.accepted.clone()) {
4807        match resolve_subtree_root(target, &requested_root, config) {
4808            Ok(root) => resolved.push(root),
4809            Err(Error::SubtreeOutsideScanScope { .. }) => unsafe_roots.push(requested_root),
4810            Err(error) => return Err(error),
4811        }
4812    }
4813    if !unsafe_roots.is_empty() {
4814        let mut retained = Vec::with_capacity(report.accepted.len());
4815        for path in std::mem::take(&mut report.accepted) {
4816            if unsafe_roots.iter().any(|root| path.starts_with(root)) {
4817                report.rejected.push(crate::RejectedRefreshPath {
4818                    path,
4819                    reason: crate::RefreshRejection::UnsafeAncestry,
4820                });
4821            } else {
4822                retained.push(path);
4823            }
4824        }
4825        report.accepted = retained;
4826    }
4827    let walked = covering_roots(resolved);
4828    if walked.is_empty() {
4829        return Ok(report);
4830    }
4831
4832    let mut opened = Vec::with_capacity(walked.len());
4833    for subtree in walked {
4834        let (started_at, commit) = target.begin_reconcile(&subtree)?;
4835        if let Some(commit) = commit.as_ref() {
4836            sink(commit);
4837        }
4838        opened.push((subtree, started_at));
4839    }
4840
4841    // Each subtree closes on its own walk's outcome, so a subtree that could not be read
4842    // neither marks a verified sibling partial nor withholds the completeness its listing
4843    // earned.
4844    let mut failure = None;
4845    let mut outcomes = Vec::with_capacity(opened.len());
4846    for (subtree, started_at) in &opened {
4847        if failure.is_some() {
4848            outcomes.push((false, false));
4849            continue;
4850        }
4851        match reconcile_target_inner(
4852            target,
4853            subtree,
4854            *started_at,
4855            config,
4856            MAX_DEFERRED_RECONCILE_OPS,
4857            sink,
4858        ) {
4859            Ok(mut reconciliation) => {
4860                outcomes.push((
4861                    reconciliation.is_complete(),
4862                    reconciliation.apply.stale == 0 && reconciliation.apply.resource_refused == 0,
4863                ));
4864                reconciliation.listed_incomplete = reconciliation.take_recordable_completeness();
4865                merge_reconcile_report(&mut report.reconciliation, reconciliation);
4866            }
4867            Err(error) => {
4868                outcomes.push((false, false));
4869                failure = Some(error);
4870            }
4871        }
4872    }
4873
4874    let listed_incomplete = std::mem::take(&mut report.reconciliation.listed_incomplete);
4875    let root = target.root_path()?;
4876    normalize_walk_errors(&root, &mut report.reconciliation.scan.errors);
4877    let failed_paths = failure_paths(target, &report.reconciliation.scan.errors)?;
4878    for ((subtree, started_at), (complete, disproves_old)) in opened.into_iter().zip(outcomes) {
4879        let commit = target.finish_reconcile(
4880            &subtree,
4881            started_at,
4882            complete,
4883            &listed_incomplete,
4884            &failed_paths,
4885            ReconcileErrors {
4886                errors: &report.reconciliation.scan.errors,
4887                terminal: failure.as_ref(),
4888                disproves_old,
4889            },
4890        )?;
4891        if let Some(commit) = commit.commit.as_ref() {
4892            sink(commit);
4893        }
4894        report.reconciliation.retry_required |= commit.retry;
4895    }
4896
4897    match failure {
4898        Some(error) => Err(error),
4899        None => Ok(report),
4900    }
4901}
4902
4903/// Root-relative paths whose filesystem facts a failed reconciliation could not verify.
4904///
4905/// An unscoped error returns an empty set, which makes the closer conservatively mark the
4906/// whole requested subtree partial. Precise I/O paths let verified siblings remain fresh.
4907fn failure_paths(target: &ReconcileTarget<'_>, errors: &[Error]) -> Result<Vec<PathBuf>> {
4908    if errors.is_empty() {
4909        return Ok(Vec::new());
4910    }
4911    let root = target.root_path()?;
4912    let mut paths = Vec::with_capacity(errors.len());
4913    for error in errors {
4914        let Some(path) = crate::Issue::from_error_under(&root, error).path else {
4915            return Ok(Vec::new());
4916        };
4917        if path.is_absolute() {
4918            return Ok(Vec::new());
4919        }
4920        paths.push(path);
4921    }
4922    paths.sort();
4923    paths.dedup();
4924    Ok(paths)
4925}
4926
4927/// Whether verification could increase the retained-file set.
4928///
4929/// This deliberately recognizes only cases that prove non-expansion. At a resource
4930/// boundary, uncertainty is a refusal rather than permission to exceed the bound.
4931fn refresh_may_expand(
4932    target: &ReconcileTarget<'_>,
4933    path: &Path,
4934    work: &mut ScanReport,
4935) -> Result<bool> {
4936    let current = target.expectation(path)?.state;
4937    let absolute = target.root_path()?.join(path);
4938    let observed = match fs::symlink_metadata(&absolute) {
4939        Ok(metadata) => {
4940            let Ok((kind, attrs)) = observe(&absolute, &metadata) else {
4941                return Ok(true);
4942            };
4943            work.observe(kind, attrs);
4944            Some(kind)
4945        }
4946        Err(error)
4947            if matches!(
4948                error.kind(),
4949                std::io::ErrorKind::NotFound | std::io::ErrorKind::NotADirectory
4950            ) =>
4951        {
4952            None
4953        }
4954        Err(_) => return Ok(true),
4955    };
4956    Ok(!matches!(
4957        (current, observed),
4958        (PathState::Present { kind: EntryKind::File, .. }, Some(EntryKind::File)) | (_, None)
4959    ))
4960}
4961
4962/// Drop every path covered by a shallower member of the same sorted set.
4963fn covering_roots(mut paths: Vec<PathBuf>) -> Vec<PathBuf> {
4964    paths.sort();
4965    paths.dedup();
4966    if paths.first().is_some_and(|first| first.as_os_str().is_empty()) {
4967        return vec![PathBuf::new()];
4968    }
4969    let mut roots: Vec<PathBuf> = Vec::with_capacity(paths.len());
4970    for path in paths {
4971        if roots.last().is_some_and(|kept| path.starts_with(kept)) {
4972            continue;
4973        }
4974        roots.push(path);
4975    }
4976    roots
4977}
4978
4979fn reconcile_target(
4980    target: &mut ReconcileTarget<'_>,
4981    subtree: &Path,
4982    config: &ScanConfig,
4983    sink: &mut dyn FnMut(&Commit),
4984) -> Result<ReconcileReport> {
4985    config.validate_for_scope(target.scope()?)?;
4986    if let Some(progress) = &config.progress {
4987        progress.enter(crate::ProgressPhase::Revalidating);
4988    }
4989    let subtree = normalize_subtree(subtree)?;
4990    if config.max_depth.is_some_and(|maximum| subtree.components().count() > maximum) {
4991        return Err(Error::SubtreeOutsideScanScope { path: subtree, scope: config.scope() });
4992    }
4993    let subtree = if config.population != crate::query::IgnoredEntries::Include
4994        && !subtree.as_os_str().is_empty()
4995    {
4996        // A narrowed tier may have pruned the requested entry or an ancestor. Start
4997        // from its nearest retained parent so the governing control is read before the
4998        // directory listing decides whether the boundary itself belongs in the tier.
4999        // A control-file edit also needs this parent listing to discover siblings that
5000        // were absent under the previous rule.
5001        let mut parent = subtree.parent().map_or_else(PathBuf::new, Path::to_path_buf);
5002        while !parent.as_os_str().is_empty()
5003            && (target.expectation(&parent)?.state == PathState::Absent
5004                || !target.control_classification_known(&parent)?)
5005        {
5006            parent = parent.parent().map_or_else(PathBuf::new, Path::to_path_buf);
5007        }
5008        parent
5009    } else {
5010        subtree
5011    };
5012    let subtree = resolve_subtree_root(target, &subtree, config)?;
5013    let (started_at, started) = target.begin_reconcile(&subtree)?;
5014    if let Some(commit) = started.as_ref() {
5015        sink(commit);
5016    }
5017    match reconcile_target_inner(
5018        target,
5019        &subtree,
5020        started_at,
5021        config,
5022        MAX_DEFERRED_RECONCILE_OPS,
5023        sink,
5024    ) {
5025        Ok(mut report) => {
5026            let root = target.root_path()?;
5027            normalize_walk_errors(&root, &mut report.scan.errors);
5028            let listed_incomplete = report.take_recordable_completeness();
5029            let failed_paths = failure_paths(target, &report.scan.errors)?;
5030            let finished = target.finish_reconcile(
5031                &subtree,
5032                started_at,
5033                report.is_complete(),
5034                &listed_incomplete,
5035                &failed_paths,
5036                ReconcileErrors {
5037                    errors: &report.scan.errors,
5038                    terminal: None,
5039                    disproves_old: report.apply.stale == 0 && report.apply.resource_refused == 0,
5040                },
5041            )?;
5042            if let Some(commit) = finished.commit.as_ref() {
5043                sink(commit);
5044            }
5045            report.retry_required |= finished.retry;
5046            Ok(report)
5047        }
5048        Err(error) => {
5049            let finished = target.finish_reconcile(
5050                &subtree,
5051                started_at,
5052                false,
5053                &[],
5054                &[],
5055                ReconcileErrors { errors: &[], terminal: Some(&error), disproves_old: false },
5056            )?;
5057            if let Some(commit) = finished.commit.as_ref() {
5058                sink(commit);
5059            }
5060            Err(error)
5061        }
5062    }
5063}
5064
5065fn reconcile_target_inner(
5066    target: &mut ReconcileTarget<'_>,
5067    subtree: &Path,
5068    started_at: u64,
5069    config: &ScanConfig,
5070    max_deferred_ops: usize,
5071    sink: &mut dyn FnMut(&Commit),
5072) -> Result<ReconcileReport> {
5073    let root = target.root_path()?;
5074    let root_meta = {
5075        crate::counters::bump(|c| c.stats += 1);
5076        fs::symlink_metadata(&root)
5077    }
5078    .map_err(|error| Error::io(&root, error))?;
5079    if !root_meta.is_dir() {
5080        return Err(Error::io(
5081            &root,
5082            std::io::Error::new(
5083                std::io::ErrorKind::NotADirectory,
5084                "reconciliation root is not a directory",
5085            ),
5086        ));
5087    }
5088    let root_dev = root_device(&root, &root_meta).map_err(|error| Error::io(&root, error))?;
5089    let start_depth = subtree.components().count();
5090    let mut report =
5091        ReconcileReport { reconcile_epoch: Some(started_at), ..ReconcileReport::default() };
5092    let mut tally = ProgressTally::new(config.progress.as_ref());
5093    let mut retry_frontier = None;
5094    let mut batch: Vec<ObservationOp> = Vec::with_capacity(config.batch_size.max(1));
5095
5096    if config.max_depth == Some(0) {
5097        remove_known_children(target, Path::new(""), config, &mut batch, sink, &mut report)?;
5098        return Ok(report);
5099    }
5100
5101    if !subtree.as_os_str().is_empty() {
5102        let baseline = target.expectation(subtree)?;
5103        let absolute = root.join(subtree);
5104        let meta = match fs::symlink_metadata(&absolute) {
5105            Ok(meta) => meta,
5106            Err(error) if error.kind() == std::io::ErrorKind::NotFound => {
5107                batch.push(ObservationOp::if_state(
5108                    Op::Remove { path: subtree.to_path_buf() },
5109                    baseline,
5110                ));
5111                if target.has_control(subtree)? {
5112                    batch.push(ObservationOp::if_state(
5113                        Op::ControlRemove { path: subtree.to_path_buf() },
5114                        baseline,
5115                    ));
5116                }
5117                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5118                return Ok(report);
5119            }
5120            Err(error) => {
5121                report.scan.errors.push(Error::io(&absolute, error));
5122                if baseline.state != PathState::Absent {
5123                    batch.push(ObservationOp::if_state(
5124                        Op::Remove { path: subtree.to_path_buf() },
5125                        baseline,
5126                    ));
5127                }
5128                if target.has_control(subtree)? {
5129                    batch.push(ObservationOp::if_state(
5130                        Op::ControlRemove { path: subtree.to_path_buf() },
5131                        baseline,
5132                    ));
5133                }
5134                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5135                return Ok(report);
5136            }
5137        };
5138        let (kind, attrs) = match observe(&absolute, &meta) {
5139            Ok(observed) => observed,
5140            Err(error) => {
5141                report.scan.errors.push(Error::io(absolute, error));
5142                return Ok(report);
5143            }
5144        };
5145        let disposition =
5146            crate::admission::decide_path(subtree, kind, config.hidden(), config.exclude_special);
5147        if disposition != crate::admission::Disposition::Retain {
5148            if baseline.state != PathState::Absent {
5149                batch.push(ObservationOp::if_state(
5150                    Op::Remove { path: subtree.to_path_buf() },
5151                    baseline,
5152                ));
5153            }
5154            if disposition == crate::admission::Disposition::ControlOnly {
5155                match read_control_op(config, &root, subtree, kind) {
5156                    Ok(Some(control)) => {
5157                        batch.push(ObservationOp::if_state(control, baseline));
5158                    }
5159                    Ok(None) => {}
5160                    Err(error) => {
5161                        if target.has_control(subtree)? {
5162                            batch.push(ObservationOp::if_state(
5163                                Op::ControlRemove { path: subtree.to_path_buf() },
5164                                baseline,
5165                            ));
5166                        }
5167                        report.scan.errors.push(error);
5168                    }
5169                }
5170            }
5171            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5172            return Ok(report);
5173        }
5174        report.scan.observe(kind, attrs);
5175        push_reconcile_upsert(target, subtree, kind, attrs, baseline, &mut batch, &mut report);
5176        // A retained control file at the root of the walk reads its rules here, as the
5177        // listing walk does for every retained entry it lists: a file does not descend,
5178        // so nothing below would read them, and the table kept the old source while the
5179        // pass reported complete and marked the path fresh. In the same batch as the
5180        // upsert, so both are arbitrated against one baseline.
5181        match read_control_op(config, &root, subtree, kind) {
5182            Ok(Some(control)) => batch.push(ObservationOp::if_state(control, baseline)),
5183            Ok(None) => {}
5184            Err(error) => {
5185                if target.has_control(subtree)? {
5186                    batch.push(ObservationOp::if_state(
5187                        Op::ControlRemove { path: subtree.to_path_buf() },
5188                        baseline,
5189                    ));
5190                }
5191                report.scan.errors.push(error);
5192            }
5193        }
5194        flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5195        if !should_descend(kind, attrs, start_depth.saturating_sub(1), root_dev, config) {
5196            if kind.is_dir() {
5197                remove_known_children(target, subtree, config, &mut batch, sink, &mut report)?;
5198            }
5199            tally.flush(&report.scan);
5200            return Ok(report);
5201        }
5202    }
5203
5204    if subtree.as_os_str().is_empty()
5205        && config.population == crate::query::IgnoredEntries::Include
5206        && config.reconciliation_worker_threads() > 1
5207    {
5208        if let ReconcileTarget::Direct(index) = target {
5209            match reconcile_direct_parallel(index, &root, root_dev, config, max_deferred_ops, sink)?
5210            {
5211                DirectParallelOutcome::Complete(parallel) => return Ok(parallel),
5212                DirectParallelOutcome::RetrySerial { prefix, remaining } => {
5213                    report = prefix;
5214                    report.reconcile_epoch = Some(started_at);
5215                    retry_frontier = Some(remaining);
5216                    // The wave workers reported the prefix themselves, and the wave
5217                    // that overflowed as well: progress counts that wave's reads twice,
5218                    // once there and once as the serial retry rereads it, while this
5219                    // report counts each directory once. Work done, not the answer.
5220                    tally.skip_to(&report.scan);
5221                }
5222            }
5223        }
5224    }
5225
5226    let mut queue: VecDeque<(PathBuf, usize)> = retry_frontier
5227        .unwrap_or_else(|| VecDeque::from(vec![(subtree.to_path_buf(), start_depth)]));
5228    let mut controls = if config.population == crate::query::IgnoredEntries::Include {
5229        None
5230    } else {
5231        Some(target.control_table()?)
5232    };
5233    let mut unreadable_controls = std::collections::BTreeSet::new();
5234    #[cfg(target_os = "macos")]
5235    let mut bulk_reader = (config.worker_threads() > 1).then(macos_bulk::Reader::new);
5236    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
5237        let errors_before = report.scan.errors.len();
5238        let abs_dir = root.join(&rel_dir);
5239        let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
5240        let mut had_control = target.has_control(&control_path)?;
5241        let mut control_seen = false;
5242        if let Some(table) = controls.as_mut() {
5243            match read_directory_control(config, &root, &control_path) {
5244                Ok(Some(op)) => {
5245                    let baseline = target.expectation(&control_path)?;
5246                    apply_discovery_control(table, &op)?;
5247                    batch.push(ObservationOp::if_state(op, baseline));
5248                    control_seen = true;
5249                }
5250                Ok(None) => {
5251                    table.remove(&control_path)?;
5252                    if had_control {
5253                        let baseline = target.expectation(&control_path)?;
5254                        batch.push(ObservationOp::if_state(
5255                            Op::ControlRemove { path: control_path.clone() },
5256                            baseline,
5257                        ));
5258                        had_control = false;
5259                    }
5260                }
5261                Err(error) => {
5262                    unreadable_controls.insert(rel_dir.clone());
5263                    report.scan.errors.push(error);
5264                }
5265            }
5266            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5267            if report.apply.stale > 0 || report.apply.resource_refused > 0 {
5268                // This directory's population decisions use the source just read.
5269                // If its conditional control commit lost ownership, walking siblings
5270                // against the local copy could prune under rules the index rejected.
5271                report.retry_required = true;
5272                return Ok(report);
5273            }
5274        }
5275        let (mut known, records_completeness) = target.listing_baseline(&rel_dir)?;
5276        let mut listing_complete = true;
5277        let process_entry = |name: OsString,
5278                             kind: EntryKind,
5279                             attrs: Attrs,
5280                             baseline: PathExpectation,
5281                             control_seen: &mut bool,
5282                             target: &mut ReconcileTarget<'_>,
5283                             queue: &mut VecDeque<(PathBuf, usize)>,
5284                             batch: &mut Vec<ObservationOp>,
5285                             sink: &mut dyn FnMut(&Commit),
5286                             report: &mut ReconcileReport|
5287         -> Result<()> {
5288            let rel_path = rel_dir.join(&name);
5289            *control_seen |= name == crate::control::CONTROL_FILE_NAME;
5290            let disposition =
5291                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
5292            if population_prunes(
5293                config.population,
5294                &rel_path,
5295                kind,
5296                disposition,
5297                controls.as_ref(),
5298                &unreadable_controls,
5299            ) {
5300                if baseline.state != PathState::Absent {
5301                    batch.push(ObservationOp::if_state(Op::Remove { path: rel_path }, baseline));
5302                }
5303                return Ok(());
5304            }
5305            if disposition != crate::admission::Disposition::Retain {
5306                if baseline.state != PathState::Absent {
5307                    batch.push(ObservationOp::if_state(
5308                        Op::Remove { path: rel_path.clone() },
5309                        baseline,
5310                    ));
5311                }
5312                if disposition == crate::admission::Disposition::ControlOnly {
5313                    match read_listed_control_op(config, &root, &rel_path, kind) {
5314                        Ok(Some(control)) => {
5315                            batch.push(ObservationOp::if_state(control, baseline));
5316                        }
5317                        Ok(None) => {}
5318                        Err(error) => {
5319                            if target.has_control(&rel_path)? {
5320                                batch.push(ObservationOp::if_state(
5321                                    Op::ControlRemove { path: rel_path.clone() },
5322                                    baseline,
5323                                ));
5324                            }
5325                            report.scan.errors.push(error);
5326                        }
5327                    }
5328                }
5329                if batch.len() >= config.batch_size.max(1) {
5330                    flush_reconcile_batch(target, batch, sink, report)?;
5331                }
5332                return Ok(());
5333            }
5334            report.scan.observe(kind, attrs);
5335            push_reconcile_upsert(target, &rel_path, kind, attrs, baseline, batch, report);
5336            if batch.len() >= config.batch_size.max(1) {
5337                flush_reconcile_batch(target, batch, sink, report)?;
5338            }
5339            match read_listed_control_op(config, &root, &rel_path, kind) {
5340                Ok(Some(control)) => {
5341                    batch.push(ObservationOp::if_state(control, baseline));
5342                    if batch.len() >= config.batch_size.max(1) {
5343                        flush_reconcile_batch(target, batch, sink, report)?;
5344                    }
5345                }
5346                Ok(None) => {}
5347                Err(error) => {
5348                    if target.has_control(&rel_path)? {
5349                        batch.push(ObservationOp::if_state(
5350                            Op::ControlRemove { path: rel_path.clone() },
5351                            baseline,
5352                        ));
5353                    }
5354                    report.scan.errors.push(error);
5355                }
5356            }
5357
5358            if should_descend(kind, attrs, depth, root_dev, config) {
5359                queue.push_back((rel_path, depth + 1));
5360            } else if kind.is_dir() {
5361                remove_known_children(target, &rel_path, config, batch, sink, report)?;
5362            }
5363            Ok(())
5364        };
5365
5366        #[cfg(target_os = "macos")]
5367        let used_bulk = if walk_hook_covers(&abs_dir) {
5368            false
5369        } else if let Some(entries) = bulk_reader.as_mut().and_then(|reader| reader.read(&abs_dir))
5370        {
5371            report.scan.dirs_read += 1;
5372            for entry in entries {
5373                let baseline = match known.remove(&entry.name) {
5374                    Some(baseline) => baseline,
5375                    None => target.expectation(&rel_dir.join(&entry.name))?,
5376                };
5377                process_entry(
5378                    entry.name,
5379                    entry.kind,
5380                    entry.attrs,
5381                    baseline,
5382                    &mut control_seen,
5383                    target,
5384                    &mut queue,
5385                    &mut batch,
5386                    sink,
5387                    &mut report,
5388                )?;
5389            }
5390            true
5391        } else {
5392            false
5393        };
5394        #[cfg(not(target_os = "macos"))]
5395        let used_bulk = false;
5396
5397        if !used_bulk {
5398            crate::counters::bump(|c| c.dir_opens += 1);
5399            let listing = match fs::read_dir(&abs_dir) {
5400                Ok(listing) => listing,
5401                Err(error) => {
5402                    report.scan.errors.push(Error::io(&abs_dir, error));
5403                    remove_known_children(target, &rel_dir, config, &mut batch, sink, &mut report)?;
5404                    if had_control {
5405                        let baseline = target.expectation(&control_path)?;
5406                        batch.push(ObservationOp::if_state(
5407                            Op::ControlRemove { path: control_path },
5408                            baseline,
5409                        ));
5410                        flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5411                    }
5412                    continue;
5413                }
5414            };
5415            report.scan.dirs_read += 1;
5416            let listing = reconcile_listing(listing, &abs_dir);
5417            for item in listing {
5418                let item = match item {
5419                    Ok(item) => item,
5420                    Err(error) => {
5421                        listing_complete = false;
5422                        report.scan.errors.push(Error::io(&abs_dir, error));
5423                        continue;
5424                    }
5425                };
5426                let name = item.file_name();
5427                // Seeing the name proves it is not absent even if the following
5428                // metadata lookup fails. Remove it from the missing set before that
5429                // fallible lookup so an operational error cannot turn an existing
5430                // entry into a deletion.
5431                let baseline = match known.remove(&name) {
5432                    Some(baseline) => baseline,
5433                    None => target.expectation(&rel_dir.join(&name))?,
5434                };
5435                let (kind, attrs) = match observe_dir_entry(&item) {
5436                    Ok(Some(observed)) => observed,
5437                    Ok(None) => {
5438                        let entry_held = baseline.state != PathState::Absent;
5439                        for removal in
5440                            vanished_child_removals(&rel_dir, &name, entry_held, &mut had_control)
5441                                .into_iter()
5442                                .flatten()
5443                        {
5444                            batch.push(ObservationOp::if_state(removal, baseline));
5445                        }
5446                        if batch.len() >= config.batch_size.max(1) {
5447                            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5448                        }
5449                        continue;
5450                    }
5451                    Err(error) => {
5452                        control_seen |= name == crate::control::CONTROL_FILE_NAME;
5453                        report.scan.errors.push(Error::io(item.path(), error));
5454                        if baseline.state != PathState::Absent {
5455                            batch.push(ObservationOp::if_state(
5456                                Op::Remove { path: rel_dir.join(&name) },
5457                                baseline,
5458                            ));
5459                        }
5460                        if name == crate::control::CONTROL_FILE_NAME && had_control {
5461                            batch.push(ObservationOp::if_state(
5462                                Op::ControlRemove { path: rel_dir.join(&name) },
5463                                baseline,
5464                            ));
5465                            had_control = false;
5466                        }
5467                        if batch.len() >= config.batch_size.max(1) {
5468                            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5469                        }
5470                        continue;
5471                    }
5472                };
5473                process_entry(
5474                    name,
5475                    kind,
5476                    attrs,
5477                    baseline,
5478                    &mut control_seen,
5479                    target,
5480                    &mut queue,
5481                    &mut batch,
5482                    sink,
5483                    &mut report,
5484                )?;
5485            }
5486        }
5487
5488        for (name, baseline) in known {
5489            batch.push(ObservationOp::if_state(Op::Remove { path: rel_dir.join(name) }, baseline));
5490            if batch.len() >= config.batch_size.max(1) {
5491                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5492            }
5493        }
5494        if had_control && !control_seen {
5495            let baseline = target.expectation(&control_path)?;
5496            batch.push(ObservationOp::if_state(Op::ControlRemove { path: control_path }, baseline));
5497        }
5498        if listing_complete {
5499            // Only a directory with no error inside its own processing vouches for its
5500            // child set; an error under a sibling or a descendant is that directory's
5501            // to answer for, as discovery decides completeness per directory.
5502            if records_completeness && report.scan.errors.len() == errors_before {
5503                report.listed_incomplete.push(rel_dir);
5504            }
5505        }
5506        // Per directory, as in `revalidate`: an unchanged tree hands the sink nothing.
5507        tally.flush(&report.scan);
5508    }
5509
5510    flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5511    report.scan.errors.sort_by_cached_key(ToString::to_string);
5512    tally.flush(&report.scan);
5513    Ok(report)
5514}
5515
5516#[derive(Debug, Default)]
5517struct DeferredReconcile {
5518    scan: ScanReport,
5519    unchanged: u64,
5520    operations: Vec<Op>,
5521    discovered: Vec<(PathBuf, usize, RegionId)>,
5522    listed_incomplete: Vec<PathBuf>,
5523}
5524
5525enum DirectParallelOutcome {
5526    Complete(ReconcileReport),
5527    RetrySerial { prefix: ReconcileReport, remaining: VecDeque<(PathBuf, usize)> },
5528}
5529
5530/// Reconcile an exclusive full tree in bounded immutable-baseline waves.
5531///
5532/// No index write occurs while a wave's workers hold shared baseline references. That
5533/// lets each worker discard exact no-ops where they are observed instead of funnelling
5534/// every entry through one consumer. Effective changes still enter through ordinary
5535/// observations between waves, preserving both the index's sole mutation contract and
5536/// progressive delta delivery.
5537///
5538/// Unlike every other reconciliation path, the operations a wave defers are
5539/// unconditional: they carry no [`ObservationOp::if_state`] guard. Three properties
5540/// have to hold together for that to be safe, and a change to any one of them puts the
5541/// guards back. The target is an exclusive `&mut Index`, so no other producer can
5542/// commit between a worker's read and the wave's write. Nothing is applied while
5543/// workers run, so no baseline a worker compared against can go stale beneath it. And
5544/// a directory is only reconciled in a wave after the wave that discovered it has
5545/// committed, so a parent is never absent when its children arrive.
5546fn reconcile_direct_parallel(
5547    index: &mut Index,
5548    root: &Path,
5549    root_dev: u64,
5550    config: &ScanConfig,
5551    max_deferred_ops: usize,
5552    sink: &mut dyn FnMut(&Commit),
5553) -> Result<DirectParallelOutcome> {
5554    let mut frontier = DirectoryQueueState::seeded(
5555        (PathBuf::new(), 0),
5556        config.order,
5557        None,
5558        1,
5559        1,
5560        WorkerPolicyExperiment::ShippedOneShot,
5561    );
5562    let mut report = ReconcileReport::default();
5563    while !frontier.is_empty(config.order) {
5564        let mut wave = Vec::with_capacity(RECONCILE_WAVE_DIRECTORIES);
5565        while wave.len() < RECONCILE_WAVE_DIRECTORIES && !frontier.is_empty(config.order) {
5566            let remaining = RECONCILE_WAVE_DIRECTORIES - wave.len();
5567            frontier.take(remaining.min(DIR_CLAIM), config.order, &mut wave);
5568        }
5569
5570        let next = std::sync::atomic::AtomicUsize::new(0);
5571        let deferred_count = std::sync::atomic::AtomicUsize::new(0);
5572        let overflowed = std::sync::atomic::AtomicBool::new(false);
5573        let workers = config.reconciliation_worker_threads().min(wave.len());
5574        let baseline: &Index = index;
5575        let results: Vec<DeferredReconcile> = std::thread::scope(|scope| {
5576            let handles: Vec<_> = (0..workers)
5577                .map(|_| {
5578                    scope.spawn(|| {
5579                        reconcile_wave_worker(
5580                            baseline,
5581                            root,
5582                            root_dev,
5583                            config,
5584                            &wave,
5585                            &next,
5586                            &deferred_count,
5587                            &overflowed,
5588                            max_deferred_ops,
5589                        )
5590                    })
5591                })
5592                .collect();
5593
5594            handles
5595                .into_iter()
5596                .map(|handle| {
5597                    if let Ok(worker) = handle.join() {
5598                        return worker;
5599                    }
5600                    let mut worker = DeferredReconcile::default();
5601                    worker.scan.errors.push(Error::io(
5602                        root,
5603                        std::io::Error::other("a reconciliation worker thread panicked"),
5604                    ));
5605                    worker
5606                })
5607                .collect()
5608        });
5609
5610        // Nothing from an overflowing wave was applied, so the ordinary incremental
5611        // reconciler can resume at that wave. Completed waves and their statistics are
5612        // retained exactly once; restarting from the root would count their unchanged
5613        // entries again and misreport the logical reconciliation pass.
5614        if overflowed.load(std::sync::atomic::Ordering::Relaxed) {
5615            let mut remaining: VecDeque<_> =
5616                wave.into_iter().map(|(path, depth, _region)| (path, depth)).collect();
5617            let mut deferred = Vec::with_capacity(DIR_CLAIM);
5618            while !frontier.is_empty(config.order) {
5619                frontier.take(DIR_CLAIM, config.order, &mut deferred);
5620                remaining.extend(deferred.drain(..).map(|(path, depth, _region)| (path, depth)));
5621            }
5622            return Ok(DirectParallelOutcome::RetrySerial { prefix: report, remaining });
5623        }
5624
5625        let operation_count = deferred_count.load(std::sync::atomic::Ordering::Relaxed);
5626        let mut operations = Vec::with_capacity(operation_count);
5627        for worker in results {
5628            report.listed_incomplete.extend(worker.listed_incomplete);
5629            report.scan.absorb(worker.scan);
5630            report.apply.unchanged += worker.unchanged;
5631            report.observations = report.observations.saturating_add(worker.unchanged);
5632            operations.extend(worker.operations);
5633            for directory in worker.discovered {
5634                frontier.push(directory, config.order);
5635            }
5636        }
5637        apply_deferred_reconcile(
5638            index,
5639            &mut operations,
5640            config,
5641            sink,
5642            &mut report.apply,
5643            &mut report.observations,
5644        )?;
5645    }
5646    report.scan.errors.sort_by_cached_key(ToString::to_string);
5647    Ok(DirectParallelOutcome::Complete(report))
5648}
5649
5650fn apply_deferred_reconcile(
5651    index: &mut Index,
5652    operations: &mut Vec<Op>,
5653    config: &ScanConfig,
5654    sink: &mut dyn FnMut(&Commit),
5655    stats: &mut ApplyStats,
5656    observations: &mut u64,
5657) -> Result<()> {
5658    // Parent upserts establish real directory attributes before children arrive.
5659    // Removals run deepest first so a parent removal never precedes an independently
5660    // observed descendant operation. Deterministic causal order also makes emitted
5661    // commits stable for callers.
5662    operations.sort_by(|left, right| {
5663        let left_remove = matches!(left, Op::Remove { .. });
5664        let right_remove = matches!(right, Op::Remove { .. });
5665        left_remove.cmp(&right_remove).then_with(|| {
5666            let left_depth = left.path().components().count();
5667            let right_depth = right.path().components().count();
5668            if left_remove {
5669                right_depth.cmp(&left_depth).then_with(|| left.path().cmp(right.path()))
5670            } else {
5671                left_depth.cmp(&right_depth).then_with(|| left.path().cmp(right.path()))
5672            }
5673        })
5674    });
5675
5676    let batch_limit = config.batch_size.max(1);
5677    let mut batch = Vec::with_capacity(batch_limit.min(operations.len()));
5678    for operation in operations.drain(..) {
5679        batch.push(operation);
5680        if batch.len() >= batch_limit {
5681            flush_direct_reconcile_batch(index, &mut batch, sink, stats, observations)?;
5682        }
5683    }
5684    flush_direct_reconcile_batch(index, &mut batch, sink, stats, observations)
5685}
5686
5687#[allow(clippy::too_many_arguments)]
5688fn reconcile_wave_worker(
5689    index: &Index,
5690    root: &Path,
5691    root_dev: u64,
5692    config: &ScanConfig,
5693    wave: &[(PathBuf, usize, RegionId)],
5694    next: &std::sync::atomic::AtomicUsize,
5695    deferred_count: &std::sync::atomic::AtomicUsize,
5696    overflowed: &std::sync::atomic::AtomicBool,
5697    max_deferred_ops: usize,
5698) -> DeferredReconcile {
5699    let _counter_guard = crate::counters::thread_flush_guard();
5700    let mut result = DeferredReconcile::default();
5701    let mut tally = ProgressTally::new(config.progress.as_ref());
5702    #[cfg(target_os = "macos")]
5703    let mut bulk_reader = macos_bulk::Reader::new();
5704
5705    loop {
5706        let start = next.fetch_add(DIR_CLAIM, std::sync::atomic::Ordering::Relaxed);
5707        if start >= wave.len() {
5708            break;
5709        }
5710        let end = start.saturating_add(DIR_CLAIM).min(wave.len());
5711        for (rel_dir, depth, region) in &wave[start..end] {
5712            let errors_before = result.scan.errors.len();
5713            let mut known = collect_child_expectations(index, rel_dir);
5714            let abs_dir = root.join(rel_dir);
5715            let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
5716            let mut had_control = index.control_table().contains(&control_path);
5717            let mut control_seen = false;
5718            let mut control_errors = Vec::new();
5719            let mut vanished = Vec::new();
5720            let mut unverified = Vec::new();
5721            let mut control_read_failed = false;
5722            let mut listing_open_failed = false;
5723
5724            {
5725                let mut process_entry =
5726                    |name: OsString,
5727                     kind: EntryKind,
5728                     attrs: Attrs,
5729                     baseline: PathExpectation,
5730                     control_seen: &mut bool| {
5731                        let rel_path = rel_dir.join(&name);
5732                        *control_seen |= name == crate::control::CONTROL_FILE_NAME;
5733                        let disposition = crate::admission::decide(
5734                            &name,
5735                            kind,
5736                            config.hidden(),
5737                            config.exclude_special,
5738                        );
5739                        if disposition == crate::admission::Disposition::Reject {
5740                            if baseline.state != PathState::Absent {
5741                                defer_reconcile_op(
5742                                    Op::Remove { path: rel_path },
5743                                    &mut result.operations,
5744                                    deferred_count,
5745                                    overflowed,
5746                                    max_deferred_ops,
5747                                );
5748                            }
5749                            return;
5750                        }
5751                        if disposition == crate::admission::Disposition::ControlOnly {
5752                            match read_control_op(config, root, &rel_path, kind) {
5753                                Ok(Some(Op::ControlUpsert { path, source })) => {
5754                                    if !index.control_table().source_is(&path, &source) {
5755                                        defer_reconcile_op(
5756                                            Op::ControlUpsert { path, source },
5757                                            &mut result.operations,
5758                                            deferred_count,
5759                                            overflowed,
5760                                            max_deferred_ops,
5761                                        );
5762                                    }
5763                                }
5764                                Ok(Some(Op::ControlRemove { path })) => {
5765                                    if index.control_table().contains(&path) {
5766                                        defer_reconcile_op(
5767                                            Op::ControlRemove { path },
5768                                            &mut result.operations,
5769                                            deferred_count,
5770                                            overflowed,
5771                                            max_deferred_ops,
5772                                        );
5773                                    }
5774                                }
5775                                Ok(Some(_) | None) => {}
5776                                Err(error) => {
5777                                    control_errors.push(error);
5778                                    control_read_failed = true;
5779                                }
5780                            }
5781                            return;
5782                        }
5783                        result.scan.entries += 1;
5784                        if kind == EntryKind::File {
5785                            result.scan.files_walked += 1;
5786                            result.scan.bytes_walked += attrs.size;
5787                            result.scan.allocated_walked += attrs.allocated;
5788                        }
5789                        if baseline.state == (PathState::Present { kind, attrs }) {
5790                            result.unchanged += 1;
5791                        } else {
5792                            defer_reconcile_op(
5793                                Op::Upsert { path: rel_path.clone(), kind, attrs },
5794                                &mut result.operations,
5795                                deferred_count,
5796                                overflowed,
5797                                max_deferred_ops,
5798                            );
5799                        }
5800                        match read_control_op(config, root, &rel_path, kind) {
5801                            Ok(Some(Op::ControlUpsert { path, source })) => {
5802                                if !index.control_table().source_is(&path, &source) {
5803                                    defer_reconcile_op(
5804                                        Op::ControlUpsert { path, source },
5805                                        &mut result.operations,
5806                                        deferred_count,
5807                                        overflowed,
5808                                        max_deferred_ops,
5809                                    );
5810                                }
5811                            }
5812                            Ok(Some(_) | None) => {}
5813                            Err(error) => {
5814                                control_errors.push(error);
5815                                control_read_failed |= name == crate::control::CONTROL_FILE_NAME;
5816                            }
5817                        }
5818
5819                        if should_descend(kind, attrs, *depth, root_dev, config) {
5820                            let child_region =
5821                                if *depth == 0 { RegionId::UNASSIGNED } else { *region };
5822                            result.discovered.push((rel_path, depth + 1, child_region));
5823                        } else if kind.is_dir() {
5824                            for name in collect_child_expectations(index, &rel_path).into_keys() {
5825                                defer_reconcile_op(
5826                                    Op::Remove { path: rel_path.join(name) },
5827                                    &mut result.operations,
5828                                    deferred_count,
5829                                    overflowed,
5830                                    max_deferred_ops,
5831                                );
5832                            }
5833                        }
5834                    };
5835
5836                #[cfg(target_os = "macos")]
5837                let used_bulk = if let Some(entries) =
5838                    (!walk_hook_covers(&abs_dir)).then(|| bulk_reader.read(&abs_dir)).flatten()
5839                {
5840                    result.scan.dirs_read += 1;
5841                    for entry in entries {
5842                        let baseline = known
5843                            .remove(&entry.name)
5844                            .unwrap_or_else(|| index.expectation(&rel_dir.join(&entry.name)));
5845                        process_entry(
5846                            entry.name,
5847                            entry.kind,
5848                            entry.attrs,
5849                            baseline,
5850                            &mut control_seen,
5851                        );
5852                    }
5853                    true
5854                } else {
5855                    false
5856                };
5857                #[cfg(not(target_os = "macos"))]
5858                let used_bulk = false;
5859
5860                if !used_bulk {
5861                    crate::counters::bump(|c| c.dir_opens += 1);
5862                    let listing = match fs::read_dir(&abs_dir) {
5863                        Ok(listing) => Some(listing),
5864                        Err(error) => {
5865                            result.scan.errors.push(Error::io(&abs_dir, error));
5866                            listing_open_failed = true;
5867                            None
5868                        }
5869                    };
5870                    if let Some(listing) = listing {
5871                        result.scan.dirs_read += 1;
5872                        let listing = reconcile_listing(listing, &abs_dir);
5873                        for item in listing {
5874                            let item = match item {
5875                                Ok(item) => item,
5876                                Err(error) => {
5877                                    result.scan.errors.push(Error::io(&abs_dir, error));
5878                                    continue;
5879                                }
5880                            };
5881                            let name = item.file_name();
5882                            // Match the serial path: an entry whose name was enumerated is
5883                            // not missing merely because its metadata could not be read.
5884                            let baseline = known
5885                                .remove(&name)
5886                                .unwrap_or_else(|| index.expectation(&rel_dir.join(&name)));
5887                            let (kind, attrs) = match observe_dir_entry(&item) {
5888                                Ok(Some(observed)) => observed,
5889                                Ok(None) => {
5890                                    // Removed once this directory's listing is done.
5891                                    vanished.push((name, baseline.state != PathState::Absent));
5892                                    continue;
5893                                }
5894                                Err(error) => {
5895                                    control_seen |= name == crate::control::CONTROL_FILE_NAME;
5896                                    result.scan.errors.push(Error::io(item.path(), error));
5897                                    unverified.push((name, baseline.state != PathState::Absent));
5898                                    continue;
5899                                }
5900                            };
5901                            process_entry(name, kind, attrs, baseline, &mut control_seen);
5902                        }
5903                    }
5904                }
5905            }
5906            control_read_failed |= listing_open_failed && had_control;
5907            result.scan.errors.append(&mut control_errors);
5908            if index.directory_complete(rel_dir) != Some(true)
5909                && result.scan.errors.len() == errors_before
5910            {
5911                result.listed_incomplete.push(rel_dir.clone());
5912            }
5913            for (name, entry_held) in vanished {
5914                for removal in vanished_child_removals(rel_dir, &name, entry_held, &mut had_control)
5915                    .into_iter()
5916                    .flatten()
5917                {
5918                    defer_reconcile_op(
5919                        removal,
5920                        &mut result.operations,
5921                        deferred_count,
5922                        overflowed,
5923                        max_deferred_ops,
5924                    );
5925                }
5926            }
5927            for (name, entry_held) in unverified {
5928                if entry_held {
5929                    defer_reconcile_op(
5930                        Op::Remove { path: rel_dir.join(&name) },
5931                        &mut result.operations,
5932                        deferred_count,
5933                        overflowed,
5934                        max_deferred_ops,
5935                    );
5936                }
5937                if name == crate::control::CONTROL_FILE_NAME {
5938                    control_read_failed = true;
5939                }
5940            }
5941            for (name, _) in known {
5942                defer_reconcile_op(
5943                    Op::Remove { path: rel_dir.join(name) },
5944                    &mut result.operations,
5945                    deferred_count,
5946                    overflowed,
5947                    max_deferred_ops,
5948                );
5949            }
5950            if had_control && (!control_seen || control_read_failed) {
5951                defer_reconcile_op(
5952                    Op::ControlRemove { path: control_path },
5953                    &mut result.operations,
5954                    deferred_count,
5955                    overflowed,
5956                    max_deferred_ops,
5957                );
5958            }
5959        }
5960        // Once per claimed chunk, as the cold walker reports, so a long wave on a slow
5961        // filesystem moves the counters while it runs rather than when it lands.
5962        tally.flush(&result.scan);
5963    }
5964    result
5965}
5966
5967/// What a listed child gone at its stat removes: its entry, if the baseline holds one, and
5968/// its rules, if it is the directory's control file and the table holds them.
5969///
5970/// The stat's `NotFound` is positive evidence that both are gone, so neither waits for a
5971/// complete listing; only a name the listing never returned has to, because it may merely
5972/// be unread. Removing a retained control file's entry drops its rules too, but a
5973/// hidden-pruned one has no entry, and without its own removal one unreadable sibling would
5974/// leave its rules applied with no file behind them. Clears `had_control` once the rules
5975/// are removed, so the listing's closing removals do not repeat it.
5976fn vanished_child_removals(
5977    dir: &Path,
5978    name: &OsStr,
5979    entry_held: bool,
5980    had_control: &mut bool,
5981) -> [Option<Op>; 2] {
5982    let path = dir.join(name);
5983    let rules = (*had_control && name == crate::control::CONTROL_FILE_NAME).then(|| {
5984        *had_control = false;
5985        Op::ControlRemove { path: path.clone() }
5986    });
5987    [entry_held.then_some(Op::Remove { path }), rules]
5988}
5989
5990fn defer_reconcile_op(
5991    operation: Op,
5992    operations: &mut Vec<Op>,
5993    deferred_count: &std::sync::atomic::AtomicUsize,
5994    overflowed: &std::sync::atomic::AtomicBool,
5995    max_deferred_ops: usize,
5996) {
5997    let position = deferred_count.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
5998    if position < max_deferred_ops {
5999        operations.push(operation);
6000    } else {
6001        overflowed.store(true, std::sync::atomic::Ordering::Relaxed);
6002    }
6003}
6004
6005fn flush_direct_reconcile_batch(
6006    index: &mut Index,
6007    batch: &mut Vec<Op>,
6008    sink: &mut dyn FnMut(&Commit),
6009    stats: &mut ApplyStats,
6010    observations: &mut u64,
6011) -> Result<()> {
6012    if batch.is_empty() {
6013        return Ok(());
6014    }
6015    *observations = observations.saturating_add(u64::try_from(batch.len()).unwrap_or(u64::MAX));
6016    let outcome = index.apply(&Observation::new(std::mem::take(batch)))?;
6017    merge_apply_stats(stats, outcome.stats);
6018    if let Some(commit) = outcome.commit.as_ref() {
6019        sink(commit);
6020    }
6021    Ok(())
6022}
6023
6024/// Drain and reconcile every pending invalidation, collapsing nested requests.
6025///
6026/// An invalidation whose reconciliation comes back incomplete -- a subtree that could not
6027/// be read, or a conditional commit that lost a race -- is queued again, so the next call
6028/// retries it. That suits a caller that drains when it chooses. A caller that drains after
6029/// every event would re-walk an unreadable subtree each time, and belongs on
6030/// [`reconcile_pending_handle`], which settles it instead.
6031pub fn reconcile_pending(
6032    index: &mut Index,
6033    config: &ScanConfig,
6034    sink: &mut dyn FnMut(&Commit),
6035) -> Result<ReconcileReport> {
6036    let mut target = ReconcileTarget::Direct(index);
6037    reconcile_pending_target(&mut target, config, sink)
6038}
6039
6040/// Drain and reconcile invalidations on a shared index.
6041///
6042/// Unlike [`reconcile_pending`], a subtree that could not be read is not queued again.
6043/// `Watcher::apply_next` drains after every event, and a retry there re-walks the same
6044/// unreadable subtree on each unrelated one, for the life of the watch. The error is a
6045/// settled boundary instead: the subtree stays [`crate::Freshness::Partial`], the returned
6046/// report names the error once, and the watcher retains its cause as an issue. Only a lost
6047/// race -- a stale conditional commit -- is queued for the next call. Invalidate the subtree
6048/// again to retry it deliberately.
6049pub fn reconcile_pending_handle(
6050    handle: &IndexHandle,
6051    config: &ScanConfig,
6052    sink: &mut dyn FnMut(&Commit),
6053) -> Result<ReconcileReport> {
6054    let mut target = ReconcileTarget::Shared(handle);
6055    reconcile_pending_target(&mut target, config, sink)
6056}
6057
6058/// Drain and reconcile invalidations under an opened-root lifecycle and resource bound.
6059#[cfg(feature = "watch")]
6060pub(crate) fn reconcile_pending_handle_controlled(
6061    handle: &IndexHandle,
6062    config: &ScanConfig,
6063    control: &dyn ReconcileControl,
6064    sink: &mut dyn FnMut(&Commit),
6065) -> Result<ReconcileReport> {
6066    let mut target = ReconcileTarget::Controlled { handle, control };
6067    reconcile_pending_target(&mut target, config, sink)
6068}
6069
6070fn reconcile_pending_target(
6071    target: &mut ReconcileTarget<'_>,
6072    config: &ScanConfig,
6073    sink: &mut dyn FnMut(&Commit),
6074) -> Result<ReconcileReport> {
6075    config.validate_for_scope(target.scope()?)?;
6076    let roots = take_invalidation_roots(target)?;
6077    let mut combined = ReconcileReport::default();
6078    for (position, (root, reason)) in roots.iter().enumerate() {
6079        match reconcile_target(target, root, config, sink) {
6080            Ok(report) => {
6081                if target.retries_incomplete(&report) {
6082                    target.restore_pending_invalidations(vec![(root.clone(), *reason)])?;
6083                }
6084                merge_reconcile_report(&mut combined, report);
6085            }
6086            Err(error) => {
6087                target.restore_pending_invalidations(roots[position..].to_vec())?;
6088                return Err(error);
6089            }
6090        }
6091    }
6092    Ok(combined)
6093}
6094
6095fn take_invalidation_roots(
6096    target: &mut ReconcileTarget<'_>,
6097) -> Result<Vec<(PathBuf, crate::InvalidateReason)>> {
6098    let mut pending = target.take_pending_invalidations()?;
6099    pending.sort_by(|(left, _), (right, _)| {
6100        left.components().count().cmp(&right.components().count()).then_with(|| left.cmp(right))
6101    });
6102    let mut roots: Vec<(PathBuf, crate::InvalidateReason)> = Vec::new();
6103    for (path, reason) in pending {
6104        if roots.iter().any(|(root, _)| path.starts_with(root)) {
6105            continue;
6106        }
6107        roots.push((path, reason));
6108    }
6109
6110    Ok(roots)
6111}
6112
6113fn remove_known_children(
6114    target: &mut ReconcileTarget<'_>,
6115    path: &Path,
6116    config: &ScanConfig,
6117    batch: &mut Vec<ObservationOp>,
6118    sink: &mut dyn FnMut(&Commit),
6119    report: &mut ReconcileReport,
6120) -> Result<()> {
6121    for (name, baseline) in target.child_states(path)? {
6122        batch.push(ObservationOp::if_state(Op::Remove { path: path.join(name) }, baseline));
6123        if batch.len() >= config.batch_size.max(1) {
6124            flush_reconcile_batch(target, batch, sink, report)?;
6125        }
6126    }
6127    flush_reconcile_batch(target, batch, sink, report)
6128}
6129
6130fn push_reconcile_upsert(
6131    target: &ReconcileTarget<'_>,
6132    path: &Path,
6133    kind: EntryKind,
6134    attrs: Attrs,
6135    baseline: PathExpectation,
6136    batch: &mut Vec<ObservationOp>,
6137    report: &mut ReconcileReport,
6138) {
6139    // An exclusive Index borrow cannot race another index producer. If filesystem
6140    // metadata exactly matches the captured state, applying this upsert can only be a
6141    // no-op, so avoid allocating an owned op and walking the index again. Shared
6142    // reconciliation keeps the conditional observation so ABA arbitration remains
6143    // authoritative between its read and write lock boundaries.
6144    if target.direct_upsert_is_unchanged(baseline, kind, attrs) {
6145        report.observations = report.observations.saturating_add(1);
6146        report.apply.unchanged = report.apply.unchanged.saturating_add(1);
6147        return;
6148    }
6149    batch.push(ObservationOp::if_state(
6150        Op::Upsert { path: path.to_path_buf(), kind, attrs },
6151        baseline,
6152    ));
6153}
6154
6155fn flush_reconcile_batch(
6156    target: &mut ReconcileTarget<'_>,
6157    batch: &mut Vec<ObservationOp>,
6158    sink: &mut dyn FnMut(&Commit),
6159    report: &mut ReconcileReport,
6160) -> Result<()> {
6161    if batch.is_empty() {
6162        return Ok(());
6163    }
6164    report.observations =
6165        report.observations.saturating_add(u64::try_from(batch.len()).unwrap_or(u64::MAX));
6166    let started_at = report.reconcile_epoch.expect("reconciliation report has an owner");
6167    let outcome = target.apply(started_at, &Observation::from_ops(std::mem::take(batch)))?;
6168    merge_apply_stats(&mut report.apply, outcome.stats);
6169    if let Some(commit) = outcome.commit.as_ref() {
6170        sink(commit);
6171    }
6172    Ok(())
6173}
6174
6175fn merge_apply_stats(total: &mut ApplyStats, addition: ApplyStats) {
6176    total.inserted += addition.inserted;
6177    total.updated += addition.updated;
6178    total.removed += addition.removed;
6179    total.unchanged += addition.unchanged;
6180    total.invalidated += addition.invalidated;
6181    total.controls += addition.controls;
6182    total.reclassified += addition.reclassified;
6183    total.stale += addition.stale;
6184    total.resource_refused += addition.resource_refused;
6185}
6186
6187fn merge_reconcile_report(total: &mut ReconcileReport, addition: ReconcileReport) {
6188    total.retry_required |= addition.retry_required;
6189    total.scan.dirs_read += addition.scan.dirs_read;
6190    total.scan.entries += addition.scan.entries;
6191    total.scan.files_walked += addition.scan.files_walked;
6192    total.scan.bytes_walked += addition.scan.bytes_walked;
6193    total.scan.allocated_walked += addition.scan.allocated_walked;
6194    total.scan.errors.extend(addition.scan.errors);
6195    total.observations = total.observations.saturating_add(addition.observations);
6196    merge_apply_stats(&mut total.apply, addition.apply);
6197    total.listed_incomplete.extend(addition.listed_incomplete);
6198}
6199
6200fn should_descend(
6201    kind: EntryKind,
6202    attrs: Attrs,
6203    parent_depth: usize,
6204    root_dev: u64,
6205    config: &ScanConfig,
6206) -> bool {
6207    crate::admission::should_descend(
6208        kind,
6209        attrs,
6210        parent_depth,
6211        root_dev,
6212        config.max_depth,
6213        config.one_filesystem,
6214    )
6215}
6216
6217pub(crate) fn normalize_subtree(path: &Path) -> Result<PathBuf> {
6218    let mut normalized = PathBuf::new();
6219    for component in path.components() {
6220        match component {
6221            Component::Normal(part) => normalized.push(part),
6222            Component::CurDir => {}
6223            Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
6224                return Err(Error::PathEscapesRoot(path.to_path_buf()));
6225            }
6226        }
6227    }
6228    Ok(normalized)
6229}
6230
6231fn resolve_subtree_root(
6232    target: &ReconcileTarget<'_>,
6233    subtree: &Path,
6234    config: &ScanConfig,
6235) -> Result<PathBuf> {
6236    if subtree.as_os_str().is_empty() {
6237        return Ok(PathBuf::new());
6238    }
6239    let root = target.root_path()?;
6240    crate::counters::bump(|c| c.stats += 1);
6241    let Ok(root_metadata) = fs::symlink_metadata(&root) else {
6242        // The applying pass reports operational root failures as partial.
6243        return Ok(subtree.to_path_buf());
6244    };
6245    if !root_metadata.is_dir() {
6246        return Ok(subtree.to_path_buf());
6247    }
6248    let Ok(root_dev) = root_device(&root, &root_metadata) else {
6249        return Ok(subtree.to_path_buf());
6250    };
6251    let mut prefix = PathBuf::new();
6252    let mut components = subtree.components().peekable();
6253    while let Some(component) = components.next() {
6254        if components.peek().is_none() {
6255            break; // The boundary entry itself remains visible even when descent stops.
6256        }
6257        prefix.push(component.as_os_str());
6258        let metadata = match fs::symlink_metadata(root.join(&prefix)) {
6259            Ok(metadata) => metadata,
6260            Err(error)
6261                if matches!(
6262                    error.kind(),
6263                    std::io::ErrorKind::NotFound | std::io::ErrorKind::NotADirectory
6264                ) =>
6265            {
6266                return Ok(prefix);
6267            }
6268            Err(_) => break, // The applying pass records operational failures as partial.
6269        };
6270        if metadata.file_type().is_symlink() {
6271            return Err(Error::SubtreeOutsideScanScope {
6272                path: subtree.to_path_buf(),
6273                scope: config.scope(),
6274            });
6275        }
6276        if !metadata.is_dir() {
6277            return Ok(prefix);
6278        }
6279        let Ok(attrs) = attrs_from(&root.join(&prefix), &metadata) else {
6280            break;
6281        };
6282        if config.one_filesystem && root_dev != 0 && attrs.dev != 0 && attrs.dev != root_dev {
6283            return Err(Error::SubtreeOutsideScanScope {
6284                path: subtree.to_path_buf(),
6285                scope: config.scope(),
6286            });
6287        }
6288    }
6289    Ok(subtree.to_path_buf())
6290}
6291
6292/// Read an entry's kind and roll-up attributes out of its metadata.
6293///
6294/// Exposed so the watch layer verifies entries exactly the way the walker records them —
6295/// two stat interpretations that could drift would show up as an index that disagrees
6296/// with itself depending on which producer last touched a path.
6297///
6298/// On Windows the observation comes from a fresh non-following handle, and `meta` is
6299/// what answers for an entry whose handle cannot be opened because it is locked or
6300/// access is denied — the same fallback std's `metadata` makes, with identity and change
6301/// time unavailable for that entry.
6302pub fn observe(path: &Path, meta: &fs::Metadata) -> std::io::Result<(EntryKind, Attrs)> {
6303    #[cfg(windows)]
6304    {
6305        windows_metadata::observe(path, || Ok(meta.clone()))
6306    }
6307    #[cfg(not(windows))]
6308    {
6309        Ok((kind_from(meta), attrs_from(path, meta)?))
6310    }
6311}
6312
6313pub(crate) fn observe_dir_entry(
6314    entry: &fs::DirEntry,
6315) -> std::io::Result<Option<(EntryKind, Attrs)>> {
6316    #[cfg(windows)]
6317    {
6318        crate::counters::bump(|c| c.stats += 1);
6319        #[cfg(test)]
6320        {
6321            let path = entry.path();
6322            if let Some(error) =
6323                walk_hook(&path).and_then(|hook| hook(WalkHookPoint::ChildMetadata(&path)))
6324            {
6325                return missing_as_none(Err(error));
6326            }
6327        }
6328        // The listing already holds the entry's enumeration data; it is read only when
6329        // the handle cannot be opened, so the ordinary path allocates nothing more.
6330        missing_as_none(windows_metadata::observe(&entry.path(), || entry.metadata()))
6331    }
6332    #[cfg(not(windows))]
6333    {
6334        let Some(meta) = listed_child_metadata(entry)? else {
6335            return Ok(None);
6336        };
6337        Ok(Some((kind_from(&meta), attrs_from(Path::new(""), &meta)?)))
6338    }
6339}
6340
6341#[cfg(not(windows))]
6342fn kind_from(meta: &fs::Metadata) -> EntryKind {
6343    let file_type = meta.file_type();
6344    if file_type.is_symlink() {
6345        EntryKind::Symlink
6346    } else if file_type.is_dir() {
6347        EntryKind::Dir
6348    } else if file_type.is_file() {
6349        EntryKind::File
6350    } else {
6351        EntryKind::Other
6352    }
6353}
6354
6355#[cfg(unix)]
6356#[allow(clippy::unnecessary_wraps)] // Windows observation is fallible; keep one call contract.
6357pub(crate) fn attrs_from(_path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
6358    use std::os::unix::fs::MetadataExt;
6359    Ok(Attrs {
6360        size: meta.size(),
6361        // st_blocks is in 512-byte units by POSIX convention regardless of the
6362        // filesystem's own block size.
6363        allocated: meta.blocks().saturating_mul(512),
6364        mtime_ns: compose_ns(meta.mtime(), meta.mtime_nsec()),
6365        ctime_ns: compose_ns(meta.ctime(), meta.ctime_nsec()),
6366        inode: meta.ino(),
6367        dev: meta.dev(),
6368    })
6369}
6370
6371#[cfg(unix)]
6372fn compose_ns(secs: i64, nanos: i64) -> i64 {
6373    secs.saturating_mul(1_000_000_000).saturating_add(nanos)
6374}
6375
6376#[cfg(windows)]
6377pub(crate) fn attrs_from(path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
6378    windows_metadata::observe(path, || Ok(meta.clone())).map(|(_, attrs)| attrs)
6379}
6380
6381#[cfg(not(any(unix, windows)))]
6382#[allow(clippy::unnecessary_wraps)] // Windows observation is fallible; keep one call contract.
6383pub(crate) fn attrs_from(_path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
6384    let mtime_ns = meta.modified().map_or(0, system_time_ns);
6385    Ok(Attrs {
6386        size: meta.len(),
6387        // No allocated size without platform-specific calls; apparent size is the
6388        // honest fallback rather than a guess at block rounding.
6389        allocated: meta.len(),
6390        mtime_ns,
6391        // Windows has no ctime in the Unix sense. Leaving it zero means the fingerprint
6392        // degrades to size + mtime there, which is what every portable tool does.
6393        ctime_ns: 0,
6394        inode: 0,
6395        dev: 0,
6396    })
6397}
6398
6399/// The device a walk's root is on, which bounds a one-filesystem walk.
6400///
6401/// Only the device is needed, and on Windows it is read without demanding a consistent
6402/// observation of the root's times, which change whenever a child is created or removed.
6403pub(crate) fn root_device(root: &Path, meta: &fs::Metadata) -> std::io::Result<u64> {
6404    #[cfg(windows)]
6405    {
6406        let _ = meta;
6407        windows_metadata::volume_serial(root)
6408    }
6409    #[cfg(not(windows))]
6410    {
6411        attrs_from(root, meta).map(|attrs| attrs.dev)
6412    }
6413}
6414
6415pub(crate) fn attrs_from_file(file: &fs::File, meta: &fs::Metadata) -> std::io::Result<Attrs> {
6416    #[cfg(windows)]
6417    {
6418        let _ = meta;
6419        windows_metadata::attrs_from_file(file)
6420    }
6421    #[cfg(not(windows))]
6422    {
6423        let _ = file;
6424        attrs_from(Path::new(""), meta)
6425    }
6426}
6427
6428#[cfg(any(not(any(unix, windows)), test))]
6429fn system_time_ns(time: std::time::SystemTime) -> i64 {
6430    match time.duration_since(std::time::UNIX_EPOCH) {
6431        Ok(duration) => i64::try_from(duration.as_nanos()).unwrap_or(i64::MAX),
6432        Err(error) => {
6433            i64::try_from(error.duration().as_nanos()).map_or(i64::MIN, i64::saturating_neg)
6434        }
6435    }
6436}
6437
6438#[cfg(test)]
6439mod tests {
6440    use super::*;
6441    use std::fs::File;
6442    use std::io::Write;
6443
6444    fn write_file(path: &Path, contents: &[u8]) {
6445        if let Some(parent) = path.parent() {
6446            fs::create_dir_all(parent).expect("create parent");
6447        }
6448        let mut f = File::create(path).expect("create file");
6449        f.write_all(contents).expect("write");
6450    }
6451
6452    fn sample_tree() -> tempfile::TempDir {
6453        let dir = tempfile::tempdir().expect("tempdir");
6454        write_file(&dir.path().join("a.txt"), b"hello");
6455        write_file(&dir.path().join("src/main.rs"), b"fn main() {}");
6456        write_file(&dir.path().join("src/deep/nested.rs"), b"// nested");
6457        dir
6458    }
6459
6460    /// A counter that silently reads zero is worse than a missing one, because a report
6461    /// full of zeroes invites the conclusion that the work did not happen.
6462    ///
6463    /// This has already gone wrong twice: once when the per-entry counter was added to
6464    /// the serial walk while the parallel walk went uninstrumented, and once when a
6465    /// clippy fix hoisted a `read_dir` out of a match scrutinee and took the counter
6466    /// with it. Both builds compiled, passed every other test, and reported zero. This
6467    /// asserts the relationships a real walk must satisfy, so the next such edit fails
6468    /// here instead of in a report someone believes.
6469    #[test]
6470    fn a_walk_moves_every_counter_it_should() {
6471        // Both walkers, because they are separate loops with separate call sites. The
6472        // first version of this test only exercised the parallel one, and deleting the
6473        // serial walker's counter still passed — a guard covering one path gives false
6474        // confidence about the other.
6475        let _serial = crate::counters::test_serial();
6476        crate::counters::enable(true);
6477        for threads in [Some(1), Some(4)] {
6478            let dir = sample_tree();
6479            let config = ScanConfig { threads, ..ScanConfig::default() };
6480
6481            // Deltas around the scan, not absolute totals. The counters are
6482            // process-global, so a test running beside this one can add to them — and
6483            // `test_serial` cannot prevent that, since it only serializes tests that
6484            // take it, not every test that happens to walk a tree.
6485            let before = crate::counters::snapshot();
6486            let report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
6487            crate::counters::flush_thread();
6488            let after = crate::counters::snapshot();
6489            let observed_entries = after.dir_entries - before.dir_entries;
6490            let observed_opens = after.dir_opens - before.dir_opens;
6491            let observed_stats = after.stats - before.stats;
6492
6493            // `>=` rather than `==`, and the direction is the whole point: concurrent
6494            // work can only inflate these, never deflate them. So a counter that is too
6495            // low means a path ran uninstrumented, which is the failure worth catching
6496            // and the one that has actually happened — the macOS bulk reader reported
6497            // zero opens against three real ones. Equality would catch double-counting
6498            // too, and would be flaky for it.
6499            assert!(
6500                observed_entries >= report.entries,
6501                "every enumerated entry is counted at {threads:?}: {observed_entries} < {}",
6502                report.entries
6503            );
6504            assert!(
6505                observed_opens >= report.dirs_read,
6506                "every directory open is counted at {threads:?}: {observed_opens} < {}",
6507                report.dirs_read
6508            );
6509            assert!(
6510                observed_stats >= report.entries,
6511                "every entry is stated at {threads:?}: {observed_stats} < {}",
6512                report.entries
6513            );
6514            #[cfg(target_os = "macos")]
6515            {
6516                let observed_enum = after.dir_enumeration_calls - before.dir_enumeration_calls;
6517                // The serial walker is the portable `read_dir` path, which cannot see
6518                // getdents multiplicity. Enumeration calls are a bulk-backend fact.
6519                if threads != Some(1) {
6520                    assert!(
6521                        observed_enum >= report.dirs_read,
6522                        "every successful bulk directory issues at least one enumeration \
6523                         call at {threads:?}: {observed_enum} < {}",
6524                        report.dirs_read
6525                    );
6526                }
6527            }
6528
6529            // Deliberately not asserted: `allocs` stays zero in a library test, because
6530            // allocation counting needs a binary to install `CountingAlloc` as its
6531            // global allocator and a test harness installs its own. The probe covers
6532            // that half; this covers the counters the library itself drives.
6533        }
6534        crate::counters::enable(false);
6535    }
6536
6537    #[test]
6538    fn summary_fold_skips_stat_on_directories_and_symlinks() {
6539        let _serial = crate::counters::test_serial();
6540        crate::counters::enable(true);
6541        let dir = tempfile::tempdir().expect("tempdir");
6542        fs::create_dir(dir.path().join("src")).expect("directory");
6543        write_file(&dir.path().join("a.txt"), b"hi");
6544        #[cfg(unix)]
6545        std::os::unix::fs::symlink("a.txt", dir.path().join("link")).expect("symlink");
6546        let config = ScanConfig { threads: Some(1), read_controls: false, ..ScanConfig::default() };
6547
6548        crate::counters::test_thread_reset();
6549        let scan_report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
6550        let scan_stats = crate::counters::test_thread_snapshot().stats;
6551
6552        crate::counters::test_thread_reset();
6553        let fold_report = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
6554        let fold_stats = crate::counters::test_thread_snapshot().stats;
6555        crate::counters::enable(false);
6556
6557        assert_eq!(fold_report.entries, scan_report.entries);
6558        assert_eq!(fold_report.files_walked, scan_report.files_walked);
6559        assert_eq!(fold_report.bytes_walked, scan_report.bytes_walked);
6560        // Windows observes every listed entry through a fresh handle on both paths, so the
6561        // fold performs exactly the retained walk's observations there; the skip is a
6562        // non-Windows saving.
6563        #[cfg(not(windows))]
6564        assert!(
6565            fold_stats < scan_stats,
6566            "fold {fold_stats} should skip directory/symlink stats versus scan {scan_stats}"
6567        );
6568        #[cfg(unix)]
6569        assert_eq!(scan_stats.saturating_sub(fold_stats), 2);
6570        #[cfg(not(any(unix, windows)))]
6571        assert_eq!(scan_stats.saturating_sub(fold_stats), 1);
6572        #[cfg(windows)]
6573        assert_eq!(fold_stats, scan_stats);
6574    }
6575
6576    #[cfg(unix)]
6577    #[test]
6578    fn summary_fold_still_stats_directories_when_bound_to_one_filesystem() {
6579        let _serial = crate::counters::test_serial();
6580        crate::counters::enable(true);
6581        let dir = tempfile::tempdir().expect("tempdir");
6582        fs::create_dir(dir.path().join("src")).expect("directory");
6583        write_file(&dir.path().join("a.txt"), b"hi");
6584        std::os::unix::fs::symlink("a.txt", dir.path().join("link")).expect("symlink");
6585        let config = ScanConfig {
6586            threads: Some(1),
6587            read_controls: false,
6588            one_filesystem: true,
6589            ..ScanConfig::default()
6590        };
6591
6592        crate::counters::test_thread_reset();
6593        let scan_report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
6594        let scan_stats = crate::counters::test_thread_snapshot().stats;
6595
6596        crate::counters::test_thread_reset();
6597        let fold_report = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
6598        let fold_stats = crate::counters::test_thread_snapshot().stats;
6599        crate::counters::enable(false);
6600
6601        assert_eq!(fold_report.entries, scan_report.entries);
6602        assert_eq!(scan_stats.saturating_sub(fold_stats), 1);
6603    }
6604
6605    #[test]
6606    fn summary_fold_reuses_cleared_recycled_batches() {
6607        // Four workers and a batch of three force StreamingEmission to send more than
6608        // once per worker on this tree. Without `recycled.clear()`, the next send
6609        // re-folds the previous ops and files/bytes/dirs double-count.
6610        const DIRS: usize = 16;
6611        const FILES_PER_DIR: usize = 40;
6612        let dir = tempfile::tempdir().expect("tempdir");
6613        let mut expected_bytes = 0u64;
6614        for directory in 0..DIRS {
6615            let child = dir.path().join(format!("d{directory:02}"));
6616            fs::create_dir(&child).expect("directory");
6617            for file in 0..FILES_PER_DIR {
6618                let size = directory * FILES_PER_DIR + file + 1;
6619                expected_bytes += size as u64;
6620                write_file(&child.join(format!("f{file:02}.dat")), &vec![b'x'; size]);
6621            }
6622        }
6623        let expected_files = (DIRS * FILES_PER_DIR) as u64;
6624        let expected_dirs = DIRS as u64;
6625        let expected_entries = expected_files + expected_dirs;
6626        let config = ScanConfig {
6627            threads: Some(4),
6628            batch_size: 3,
6629            read_controls: false,
6630            ..ScanConfig::default()
6631        };
6632        let mut files = 0u64;
6633        let mut bytes = 0u64;
6634        let mut dirs = 0u64;
6635        let mut ops = 0u64;
6636        let report = scan_summary_fold(dir.path(), &config, &mut |observed| {
6637            ops += 1;
6638            let Op::Upsert { kind, attrs, .. } = &observed.op else {
6639                return;
6640            };
6641            match kind {
6642                EntryKind::File => {
6643                    files += 1;
6644                    bytes += attrs.size;
6645                }
6646                EntryKind::Dir => dirs += 1,
6647                EntryKind::Symlink | EntryKind::Other => {}
6648            }
6649        })
6650        .expect("fold");
6651        assert_eq!(files, expected_files);
6652        assert_eq!(bytes, expected_bytes);
6653        assert_eq!(dirs, expected_dirs);
6654        assert_eq!(ops, report.entries);
6655        assert_eq!(report.entries, expected_entries);
6656        assert_eq!(report.files_walked, expected_files);
6657        assert_eq!(report.bytes_walked, expected_bytes);
6658    }
6659
6660    /// An automatic walk too short to fill its calibration window must say so.
6661    ///
6662    /// The failure this guards is quiet: such a walk runs on its initial pool, which is
6663    /// indistinguishable in the artifacts from a walk that measured the filesystem and
6664    /// chose to hold — unless the undecided case is recorded separately. Reading the
6665    /// first as the second is how a policy with no evidence behind it comes to look
6666    /// like a policy with evidence behind it.
6667    #[test]
6668    fn a_short_automatic_walk_records_an_undecided_policy() {
6669        let _serial = crate::counters::test_serial();
6670        let available = std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get);
6671        if automatic_worker_pool(available).calibration.is_none() {
6672            // A host reporting one processor has no reserve to unlock, so there is no
6673            // policy here to leave undecided.
6674            return;
6675        }
6676
6677        crate::counters::enable(true);
6678        let dir = sample_tree();
6679        let config = ScanConfig { threads: None, ..ScanConfig::default() };
6680        let before = crate::counters::snapshot();
6681        scan(dir.path(), &config, &mut |_| {}).expect("scan");
6682        crate::counters::flush_thread();
6683        let after = crate::counters::snapshot();
6684        crate::counters::enable(false);
6685
6686        // A strict increase, so a counter inflated by a test running beside this one
6687        // cannot turn the assertion into a false pass.
6688        assert!(
6689            after.adaptive_policy_undecided > before.adaptive_policy_undecided,
6690            "a three-file tree cannot fill a {ADAPTIVE_SCAN_CALIBRATION_ENTRIES}-entry window"
6691        );
6692    }
6693
6694    #[test]
6695    fn diagnostics_make_a_fixed_pool_and_backend_choice_explicit() {
6696        let dir = sample_tree();
6697        let config = ScanConfig { threads: Some(1), ..ScanConfig::default() };
6698
6699        let (report, diagnostics) =
6700            scan_with_diagnostics(dir.path(), &config, &mut |_| {}).expect("diagnostic scan");
6701
6702        assert_eq!(diagnostics.schema, SCAN_DIAGNOSTICS_SCHEMA);
6703        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Fixed);
6704        assert_eq!(diagnostics.worker_policy.initial_workers, 1);
6705        assert_eq!(diagnostics.worker_policy.maximum_workers, 1);
6706        assert_eq!(diagnostics.worker_policy.peak_active_workers, 1);
6707        assert!(diagnostics.worker_policy.windows.is_empty());
6708        assert!(!diagnostics.worker_policy.events_truncated);
6709        assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
6710        assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
6711        assert_eq!(diagnostics.backend.portable_directory_reads, report.dirs_read);
6712
6713        #[cfg(target_os = "macos")]
6714        {
6715            assert_eq!(diagnostics.backend.macos_bulk_attempts, Some(0));
6716            assert_eq!(diagnostics.backend.macos_bulk_successes, Some(0));
6717            assert_eq!(diagnostics.backend.macos_bulk_fallbacks, Some(0));
6718            assert!(diagnostics.backend.unavailable_reason.is_none());
6719        }
6720        #[cfg(not(target_os = "macos"))]
6721        {
6722            assert_eq!(diagnostics.backend.macos_bulk_attempts, None);
6723            assert_eq!(diagnostics.backend.macos_bulk_successes, None);
6724            assert_eq!(diagnostics.backend.macos_bulk_fallbacks, None);
6725            assert_eq!(
6726                diagnostics.backend.unavailable_reason,
6727                Some("macOS bulk directory enumeration is unavailable on this platform")
6728            );
6729        }
6730    }
6731
6732    #[test]
6733    fn diagnostics_fail_closed_when_an_automatic_window_is_incomplete() {
6734        let available = std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get);
6735        let pool = automatic_worker_pool(available);
6736        if pool.calibration.is_none() {
6737            return;
6738        }
6739        let dir = sample_tree();
6740        let config = ScanConfig { threads: None, ..ScanConfig::default() };
6741
6742        let (report, diagnostics) =
6743            scan_with_diagnostics(dir.path(), &config, &mut |_| {}).expect("diagnostic scan");
6744
6745        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Undecided);
6746        assert_eq!(diagnostics.worker_policy.available_parallelism, available);
6747        assert_eq!(diagnostics.worker_policy.initial_workers, pool.initial);
6748        assert_eq!(diagnostics.worker_policy.maximum_workers, pool.maximum);
6749        assert_eq!(
6750            diagnostics.worker_policy.calibration_window_entries,
6751            Some(ADAPTIVE_SCAN_CALIBRATION_ENTRIES)
6752        );
6753        assert_eq!(
6754            diagnostics.worker_policy.slow_threshold_ns_per_entry,
6755            Some(ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY)
6756        );
6757        assert_eq!(diagnostics.worker_policy.windows.len(), 1);
6758        let window = &diagnostics.worker_policy.windows[0];
6759        assert_eq!(window.sequence, 0);
6760        assert_eq!(window.start_entry_ordinal, 0);
6761        assert_eq!(window.end_entry_ordinal, report.entries);
6762        assert_eq!(window.observed_entries, report.entries);
6763        assert_eq!(window.decision, WorkerPolicyDecision::Undecided);
6764        assert!(window.end_entry_ordinal < ADAPTIVE_SCAN_CALIBRATION_ENTRIES);
6765        assert!(window.active_workers <= diagnostics.worker_policy.peak_active_workers);
6766        assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
6767        assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
6768        assert!(diagnostics.worker_policy.handoff_backlog_high_water >= 1);
6769    }
6770
6771    #[test]
6772    fn diagnostic_trace_is_bounded_and_marks_truncation() {
6773        let recorder = ScanDiagnosticsRecorder::new(
6774            WorkerPool::fixed(2),
6775            2,
6776            WorkerPolicyExperiment::ShippedOneShot,
6777        );
6778        for sequence in 0..=MAX_POLICY_TRACE_EVENTS {
6779            recorder.record_policy_window(PolicyWindowSnapshot {
6780                sequence: sequence as u64,
6781                start_entry_ordinal: sequence as u64,
6782                end_entry_ordinal: sequence as u64 + 1,
6783                observed_entries: 1,
6784                observed_chunks: 1,
6785                observed_work_ns: 10,
6786                ready_directories: 1,
6787                in_flight_directories: 1,
6788                active_workers: 1,
6789                handoff_backlog: 0,
6790                requested_workers: None,
6791                decision: WorkerPolicyDecision::Hold,
6792            });
6793        }
6794
6795        let diagnostics = recorder.finish();
6796        assert_eq!(diagnostics.worker_policy.windows.len(), MAX_POLICY_TRACE_EVENTS);
6797        assert!(diagnostics.worker_policy.events_truncated);
6798    }
6799
6800    #[test]
6801    fn diagnostic_trace_preserves_queue_order_when_recorders_arrive_out_of_order() {
6802        let pool =
6803            WorkerPool { initial: 2, maximum: 4, calibration: Some(WorkerCalibration::new(1, 1)) };
6804        let recorder =
6805            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::RepeatedWindows);
6806        let snapshot = |sequence, decision| PolicyWindowSnapshot {
6807            sequence,
6808            start_entry_ordinal: sequence,
6809            end_entry_ordinal: sequence + 1,
6810            observed_entries: 1,
6811            observed_chunks: 1,
6812            observed_work_ns: 1,
6813            ready_directories: 0,
6814            in_flight_directories: 0,
6815            active_workers: 1,
6816            handoff_backlog: 0,
6817            requested_workers: None,
6818            decision,
6819        };
6820
6821        recorder.record_policy_window(snapshot(1, WorkerPolicyDecision::HoldNoUsefulWork));
6822        recorder.record_policy_window(snapshot(0, WorkerPolicyDecision::Hold));
6823
6824        let diagnostics = recorder.finish();
6825        assert_eq!(
6826            diagnostics
6827                .worker_policy
6828                .windows
6829                .iter()
6830                .map(|window| window.sequence)
6831                .collect::<Vec<_>>(),
6832            vec![0, 1]
6833        );
6834        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::HeldNoUsefulWork);
6835    }
6836
6837    #[test]
6838    fn diagnostic_policy_aggregates_cross_check_runtime_counters() {
6839        let _serial = crate::counters::test_serial();
6840        let pool = WorkerPool {
6841            initial: 2,
6842            maximum: 4,
6843            calibration: Some(WorkerCalibration::new(17, 100)),
6844        };
6845        let recorder =
6846            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::ShippedOneShot);
6847
6848        crate::counters::enable(true);
6849        let before = crate::counters::snapshot();
6850        record_adaptive_calibration_chunk(Some(&recorder), 17, 2_100);
6851        record_adaptive_worker_expansion(Some(&recorder));
6852        crate::counters::flush_thread();
6853        let after = crate::counters::snapshot();
6854        crate::counters::enable(false);
6855
6856        recorder.record_policy_window(PolicyWindowSnapshot {
6857            sequence: 0,
6858            start_entry_ordinal: 0,
6859            end_entry_ordinal: 17,
6860            observed_entries: 17,
6861            observed_chunks: 1,
6862            observed_work_ns: 2_100,
6863            ready_directories: 2,
6864            in_flight_directories: 2,
6865            active_workers: 2,
6866            handoff_backlog: 0,
6867            requested_workers: Some(4),
6868            decision: WorkerPolicyDecision::ScaleUp,
6869        });
6870        let diagnostics = recorder.finish();
6871        let policy = diagnostics.worker_policy;
6872        assert_eq!(policy.calibration_chunks, 1);
6873        assert_eq!(policy.calibration_entries, 17);
6874        assert_eq!(policy.calibration_work_ns, 2_100);
6875        assert_eq!(policy.worker_expansions, 1);
6876        assert_eq!(policy.windows[0].observed_chunks, policy.calibration_chunks);
6877        assert_eq!(policy.windows[0].observed_entries, policy.calibration_entries);
6878        assert_eq!(policy.windows[0].observed_work_ns, policy.calibration_work_ns);
6879
6880        // Other tests can record while this process-global interval is enabled, so the
6881        // counter delta may be larger but must never be smaller than this run-scoped
6882        // trace. The shared helpers above make the two observations one event.
6883        assert!(
6884            after.adaptive_calibration_chunks - before.adaptive_calibration_chunks
6885                >= policy.calibration_chunks
6886        );
6887        assert!(
6888            after.adaptive_calibration_entries - before.adaptive_calibration_entries
6889                >= policy.calibration_entries
6890        );
6891        assert!(
6892            after.adaptive_calibration_work_us - before.adaptive_calibration_work_us
6893                >= policy.calibration_work_ns / 1_000
6894        );
6895        assert!(after.adaptive_scale_ups - before.adaptive_scale_ups >= policy.worker_expansions);
6896    }
6897
6898    #[test]
6899    fn diagnostic_index_scan_preserves_the_regular_result() {
6900        let dir = branching_tree();
6901        let config = ScanConfig { threads: Some(4), ..ScanConfig::default() };
6902        let (plain, plain_report) = scan_into_index(dir.path(), &config).expect("plain scan");
6903        let (diagnostic, diagnostic_report, diagnostics) =
6904            scan_into_index_with_diagnostics(dir.path(), &config).expect("diagnostic scan");
6905
6906        assert_eq!(index_fingerprint(&plain), index_fingerprint(&diagnostic));
6907        assert_eq!(plain_report.entries, diagnostic_report.entries);
6908        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Fixed);
6909    }
6910
6911    #[test]
6912    fn detached_bootstrap_matches_the_streaming_reducer_for_each_worker_count() {
6913        let dir = branching_tree();
6914        for threads in 1..=4 {
6915            let config = ScanConfig {
6916                read_controls: false,
6917                threads: Some(threads),
6918                ..ScanConfig::default()
6919            };
6920            let _ = detached_and_streaming_indexes(dir.path(), &config);
6921        }
6922    }
6923
6924    #[test]
6925    fn detached_control_bootstrap_matches_the_streaming_reducer_for_each_worker_count() {
6926        let dir = controlled_branching_tree();
6927        for threads in 1..=4 {
6928            let config =
6929                ScanConfig { read_controls: true, threads: Some(threads), ..ScanConfig::default() };
6930            let _ = detached_and_streaming_indexes(dir.path(), &config);
6931        }
6932    }
6933
6934    #[test]
6935    fn detached_bootstrap_preserves_the_exact_first_mutation() {
6936        let dir = branching_tree();
6937        let config = ScanConfig { read_controls: false, threads: Some(4), ..ScanConfig::default() };
6938        let (mut detached, mut streaming) = detached_and_streaming_indexes(dir.path(), &config);
6939        let created = dir.path().join("t3/m2/after-bootstrap.rs");
6940        write_file(&created, b"new fact");
6941        let attrs =
6942            attrs_from(&created, &fs::symlink_metadata(&created).expect("new file metadata"))
6943                .expect("observe new file");
6944        let observation = Observation::new(vec![Op::Upsert {
6945            path: PathBuf::from("t3/m2/after-bootstrap.rs"),
6946            kind: EntryKind::File,
6947            attrs,
6948        }]);
6949
6950        let detached_outcome = detached.apply(&observation).expect("detached mutation");
6951        let streaming_outcome = streaming.apply(&observation).expect("streaming mutation");
6952        assert_eq!(detached_outcome, streaming_outcome);
6953        assert_indexes_equal(&detached, &streaming);
6954    }
6955
6956    #[test]
6957    fn detached_control_bootstrap_preserves_the_exact_first_mutation() {
6958        let dir = controlled_branching_tree();
6959        let config = ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
6960        let (mut detached, mut streaming) = detached_and_streaming_indexes(dir.path(), &config);
6961        let observation = Observation::new(vec![Op::ControlUpsert {
6962            path: PathBuf::from(".gitignore"),
6963            source: b"leaf-2.dat\n".to_vec(),
6964        }]);
6965
6966        let detached_outcome = detached.apply(&observation).expect("detached control mutation");
6967        let streaming_outcome = streaming.apply(&observation).expect("streaming control mutation");
6968        assert_eq!(detached_outcome, streaming_outcome);
6969        assert_indexes_equal(&detached, &streaming);
6970    }
6971
6972    fn observed_coverage(index: &Index) -> crate::control::ControlObservation {
6973        match index.control_coverage() {
6974            crate::control::ControlCoverage::Observed(observation) => observation,
6975            crate::control::ControlCoverage::NotObserved => panic!("controls were observed"),
6976        }
6977    }
6978
6979    /// Both bootstrap lanes refuse a line over the limit and a file over the budget, and
6980    /// neither ends the scan or makes it partial. Both refusals are order-independent, so
6981    /// the lanes agree on exactly which files they refused.
6982    #[test]
6983    fn both_bootstrap_lanes_refuse_over_bound_controls_without_ending_the_scan() {
6984        let dir = tempfile::tempdir().expect("tempdir");
6985        let mut long_line = b"*.log\n".to_vec();
6986        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
6987        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
6988        write_file(&dir.path().join("guarded/kept.log"), b"guarded");
6989        write_file(
6990            &dir.path().join("huge/.gitignore"),
6991            &b"x\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
6992        );
6993        write_file(&dir.path().join("applied/.gitignore"), b"*.log\n");
6994        write_file(&dir.path().join("applied/dropped.log"), b"applied");
6995        let config = ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
6996
6997        let (detached, _) = detached_and_streaming_indexes(dir.path(), &config);
6998        let (_, report) = scan_into_index(dir.path(), &config).expect("scan");
6999
7000        assert!(report.is_complete(), "{:?}", report.errors);
7001        let coverage = observed_coverage(&detached);
7002        assert_eq!((coverage.applied, coverage.refused), (1, 2));
7003        assert_eq!(
7004            coverage.refusals,
7005            vec![
7006                crate::control::RefusedControl {
7007                    path: PathBuf::from("guarded/.gitignore"),
7008                    reason: crate::control::ControlRefusalReason::LineLimit,
7009                },
7010                crate::control::RefusedControl {
7011                    path: PathBuf::from("huge/.gitignore"),
7012                    reason: crate::control::ControlRefusalReason::Budget,
7013                },
7014            ]
7015        );
7016        assert_eq!(detached.is_ignored(Path::new("guarded/kept.log")).expect("observed"), None);
7017        assert_eq!(
7018            detached.is_ignored(Path::new("applied/dropped.log")).expect("observed"),
7019            Some(true)
7020        );
7021    }
7022
7023    /// The control counters attribute what a scan's control state cost: files read, sources
7024    /// refused, and sources that shared a retained content instead of parsing their own.
7025    ///
7026    /// Off by default and compiled in, like every counter, so the numbers a speed check
7027    /// reads come from the shipped path rather than an instrumented build.
7028    #[test]
7029    fn control_counters_attribute_reads_refusals_and_sharing() {
7030        let _serial = crate::counters::test_serial();
7031        let dir = tempfile::tempdir().expect("tempdir");
7032        let shared = b"*.log\n".to_vec();
7033        write_file(&dir.path().join(".gitignore"), &shared);
7034        write_file(&dir.path().join("twin/.gitignore"), &shared);
7035        let mut long_line = b"*.tmp\n".to_vec();
7036        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
7037        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
7038        let config = ScanConfig { read_controls: true, threads: Some(1), ..ScanConfig::default() };
7039
7040        crate::counters::enable(true);
7041        // Deltas around the scan rather than absolute totals, for the reason
7042        // `a_walk_moves_every_counter_it_should` gives: the counters are process-global,
7043        // `test_serial` only serializes the tests that take it, and every report in this
7044        // binary now reads `.gitignore` by default, so a test running beside this one can
7045        // add control reads of its own.
7046        let before = crate::counters::snapshot();
7047        let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
7048        crate::counters::flush_thread();
7049        let after = crate::counters::snapshot();
7050        crate::counters::enable(false);
7051
7052        assert!(report.is_complete(), "{:?}", report.errors);
7053        assert_eq!(observed_coverage(&index).refused, 1);
7054        // `>=` in the one direction concurrency can move them. A count that is too low
7055        // means a path ran uninstrumented, which is the defect worth catching; too high
7056        // is another test's tree, which is not.
7057        for (label, observed, expected) in [
7058            ("one read per .gitignore", after.control_reads - before.control_reads, 3),
7059            ("the line over the limit", after.control_refused - before.control_refused, 1),
7060            (
7061                "the twin shares one parsed content",
7062                after.control_sources_shared - before.control_sources_shared,
7063                1,
7064            ),
7065        ] {
7066            assert!(observed >= expected, "{label}: counted {observed}, expected {expected}");
7067        }
7068    }
7069
7070    /// Both limits are part of the scope, and each lifts only its own refusals: no budget
7071    /// still refuses a long line, and no line limit still refuses a file past the budget.
7072    #[test]
7073    fn each_control_limit_is_scope_and_lifts_only_its_own_refusals() {
7074        use crate::control::{ControlLimits, ControlRefusalReason, RefusedControl};
7075
7076        let with = |limits| ScanConfig { control_limits: limits, ..ScanConfig::default() };
7077        let defaults = ControlLimits::default();
7078        let default = ScanConfig::default();
7079        let no_budget = with(ControlLimits { budget: None, ..defaults });
7080        let no_line_limit = with(ControlLimits { line_limit: None, ..defaults });
7081        let configs = [
7082            default.clone(),
7083            with(ControlLimits { budget: Some(16 * 1024 * 1024), ..defaults }),
7084            no_budget.clone(),
7085            with(ControlLimits { line_limit: Some(64 * 1024), ..defaults }),
7086            no_line_limit.clone(),
7087            with(ControlLimits { budget: None, line_limit: None }),
7088            // The same values in each other's places are a different scope.
7089            with(ControlLimits { budget: defaults.line_limit, line_limit: defaults.budget }),
7090        ];
7091        let scopes: Vec<ScanScope> = configs.iter().map(ScanConfig::scope).collect();
7092        for (index, scope) in scopes.iter().enumerate() {
7093            assert!(scope.observes_controls());
7094            assert!(scopes[index + 1..].iter().all(|other| other != scope), "{scopes:?}");
7095        }
7096        for config in &configs {
7097            let blind = ScanConfig { read_controls: false, ..config.clone() };
7098            assert_eq!(blind.scope().ignore_rules_fingerprint, 0, "unobserved has one scope");
7099        }
7100
7101        let dir = tempfile::tempdir().expect("tempdir");
7102        let mut long_line = b"*.log\n".to_vec();
7103        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
7104        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
7105        write_file(&dir.path().join("guarded/dropped.log"), b"log");
7106        write_file(
7107            &dir.path().join("huge/.gitignore"),
7108            &b"x\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
7109        );
7110        let refused = |path: &str, reason| RefusedControl { path: PathBuf::from(path), reason };
7111        let (bounded, _) = scan_into_index(dir.path(), &default).expect("default scan");
7112        assert_eq!(observed_coverage(&bounded).refused, 2);
7113
7114        let (budget_lifted, _) = detached_and_streaming_indexes(dir.path(), &no_budget);
7115        let coverage = observed_coverage(&budget_lifted);
7116        assert_eq!(coverage.limits, no_budget.control_limits);
7117        assert_eq!(
7118            coverage.refusals,
7119            [refused("guarded/.gitignore", ControlRefusalReason::LineLimit)]
7120        );
7121        assert_eq!(
7122            budget_lifted.is_ignored(Path::new("guarded/dropped.log")).expect("observed"),
7123            None
7124        );
7125
7126        let (line_limit_lifted, _) = detached_and_streaming_indexes(dir.path(), &no_line_limit);
7127        let coverage = observed_coverage(&line_limit_lifted);
7128        assert_eq!(coverage.refusals, [refused("huge/.gitignore", ControlRefusalReason::Budget)]);
7129        assert_eq!(
7130            line_limit_lifted.is_ignored(Path::new("guarded/dropped.log")).expect("observed"),
7131            Some(true)
7132        );
7133        assert_eq!(line_limit_lifted.scope(), no_line_limit.scope());
7134    }
7135
7136    /// The synthetic tree that ended a cold scan (fdu-1onj): 1,105 directories, each with
7137    /// a distinct 510-byte `.gitignore` of short rules, plus one line over the limit. The
7138    /// scan completes with every size exact and names what it refused, on both lanes.
7139    #[test]
7140    fn a_tree_past_both_control_bounds_completes_with_exact_sizes() {
7141        const DIRECTORIES: usize = 1_105;
7142        let dir = tempfile::tempdir().expect("tempdir");
7143        for directory in 0..DIRECTORIES {
7144            let mut source = Vec::new();
7145            for line in 0..63 {
7146                source.extend(format!("p{directory:04}{line:02}\n").bytes());
7147            }
7148            source.extend(format!("q{directory:04}\n").bytes());
7149            assert_eq!(source.len(), 510);
7150            let root = dir.path().join(format!("d{directory:04}"));
7151            write_file(&root.join(".gitignore"), &source);
7152            write_file(&root.join("file.txt"), b"contents");
7153        }
7154        write_file(
7155            &dir.path().join("a-guard/.gitignore"),
7156            &vec![b'x'; crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1],
7157        );
7158        let observing =
7159            ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
7160        let blind = ScanConfig { read_controls: false, ..observing.clone() };
7161
7162        let (unobserved, _) = scan_into_index(dir.path(), &blind).expect("controls-off scan");
7163        let canonical = dir.path().canonicalize().expect("canonical root");
7164        let lanes = [
7165            scan_into_index(dir.path(), &observing).expect("detached scan"),
7166            scan_into_index_via_scanner(&canonical, &observing).expect("streaming scan"),
7167        ];
7168        for (index, report) in &lanes {
7169            assert!(report.is_complete(), "{:?}", report.errors);
7170            assert_eq!(index.total(), unobserved.total(), "sizes do not depend on controls");
7171            let coverage = observed_coverage(index);
7172            assert!(coverage.refused > 1, "the budget refused sources: {coverage:?}");
7173            assert_eq!(
7174                coverage.applied + coverage.refused,
7175                u64::try_from(DIRECTORIES + 1).expect("small")
7176            );
7177            assert_eq!(coverage.refusals.len(), crate::MAX_RETAINED_ISSUES);
7178            assert!(!coverage.lists_every_refusal());
7179            assert_eq!(
7180                coverage.refusals[0],
7181                crate::control::RefusedControl {
7182                    path: PathBuf::from("a-guard/.gitignore"),
7183                    reason: crate::control::ControlRefusalReason::LineLimit,
7184                }
7185            );
7186            assert!(
7187                index.control_table().retained_cost() <= crate::control::DEFAULT_CONTROL_BUDGET
7188            );
7189        }
7190    }
7191
7192    #[test]
7193    fn fingerprint_metadata_observes_mutation_after_directory_enumeration() {
7194        let dir = tempfile::tempdir().expect("tempdir");
7195        let path = dir.path().join("changing.bin");
7196        write_file(&path, b"before");
7197        let entry = fs::read_dir(dir.path())
7198            .expect("read directory")
7199            .next()
7200            .expect("one entry")
7201            .expect("read entry");
7202
7203        write_file(&path, b"after mutation");
7204
7205        let metadata = metadata_for_fingerprint(&entry).expect("fresh metadata");
7206        assert_eq!(metadata.len(), b"after mutation".len() as u64);
7207    }
7208
7209    /// A tree wide and deep enough that workers genuinely interleave.
7210    ///
7211    /// A three-file fixture would pass every one of these tests with a broken queue,
7212    /// because one worker would finish before another started.
7213    fn branching_tree() -> tempfile::TempDir {
7214        let dir = tempfile::tempdir().expect("tempdir");
7215        for top in 0..12 {
7216            for middle in 0..6 {
7217                for leaf in 0..7 {
7218                    write_file(
7219                        &dir.path().join(format!("t{top}/m{middle}/leaf-{leaf}.dat")),
7220                        &vec![b'x'; leaf * 13],
7221                    );
7222                }
7223            }
7224            // A deep chain alongside the wide fan-out, so depth and width are both
7225            // exercised by the same walk.
7226            write_file(&dir.path().join(format!("t{top}/a/b/c/d/e/deep.txt")), b"deep");
7227        }
7228        dir
7229    }
7230
7231    fn controlled_branching_tree() -> tempfile::TempDir {
7232        let dir = branching_tree();
7233        write_file(&dir.path().join(".gitignore"), b"leaf-1.dat\nt7/\n");
7234        write_file(&dir.path().join("t3/.gitignore"), b"!m2/leaf-1.dat\n*.tmp\n");
7235        write_file(&dir.path().join("t3/m2/generated.tmp"), b"ignored by nested control");
7236        write_file(&dir.path().join("t7/.gitignore"), b"!m0/leaf-1.dat\n");
7237        fs::create_dir_all(dir.path().join("t5/.gitignore")).expect("non-file control directory");
7238        write_file(&dir.path().join("t5/.gitignore/ordinary.txt"), b"ordinary child");
7239        dir
7240    }
7241
7242    fn index_fingerprint(index: &Index) -> Vec<(PathBuf, EntryKind, Attrs)> {
7243        let mut entries: Vec<(PathBuf, EntryKind, Attrs)> = Vec::new();
7244        let mut queue = vec![PathBuf::new()];
7245        while let Some(path) = queue.pop() {
7246            let Some(children) = index.children(&path) else {
7247                continue;
7248            };
7249            let names: Vec<PathBuf> = children.map(|(name, _id)| path.join(name)).collect();
7250            for child_path in names {
7251                let kind = index.kind(&child_path).expect("child has a kind");
7252                let attrs = *index.attrs(&child_path).expect("child has attrs");
7253                entries.push((child_path.clone(), kind, attrs));
7254                if kind.is_dir() {
7255                    queue.push(child_path);
7256                }
7257            }
7258        }
7259        entries.sort_by(|left, right| left.0.cmp(&right.0));
7260        entries
7261    }
7262
7263    fn detached_and_streaming_indexes(root: &Path, config: &ScanConfig) -> (Index, Index) {
7264        let canonical = root.canonicalize().expect("canonical test root");
7265        let (streaming, streaming_report) =
7266            scan_into_index_via_scanner(&canonical, config).expect("streaming oracle");
7267        let (detached, detached_report) = scan_into_index(root, config).expect("detached scan");
7268
7269        assert_eq!(detached_report.dirs_read, streaming_report.dirs_read);
7270        assert_eq!(detached_report.entries, streaming_report.entries);
7271        assert_eq!(detached_report.files_walked, streaming_report.files_walked);
7272        assert_eq!(detached_report.bytes_walked, streaming_report.bytes_walked);
7273        assert_eq!(
7274            detached_report.errors.iter().map(ToString::to_string).collect::<Vec<_>>(),
7275            streaming_report.errors.iter().map(ToString::to_string).collect::<Vec<_>>()
7276        );
7277        assert_indexes_equal(&detached, &streaming);
7278        (detached, streaming)
7279    }
7280
7281    fn assert_indexes_equal(left: &Index, right: &Index) {
7282        assert_eq!(index_fingerprint(left), index_fingerprint(right));
7283        assert_eq!(left.total(), right.total());
7284        assert_eq!(left.partition_total().ok(), right.partition_total().ok());
7285        assert_eq!(left.scope(), right.scope());
7286        assert_eq!(left.freshness(), right.freshness());
7287        assert_eq!(left.state(), right.state());
7288        assert_eq!(left.clock(), right.clock());
7289        assert_eq!(left.len(), right.len());
7290        assert_eq!(left.issues(), right.issues());
7291        assert_eq!(left.observes_controls(), right.observes_controls());
7292        assert_eq!(left.control_coverage(), right.control_coverage());
7293        assert_eq!(
7294            left.control_table()
7295                .sources()
7296                .map(|(path, source)| (path, source.to_vec()))
7297                .collect::<Vec<_>>(),
7298            right
7299                .control_table()
7300                .sources()
7301                .map(|(path, source)| (path, source.to_vec()))
7302                .collect::<Vec<_>>()
7303        );
7304        for (path, _, _) in index_fingerprint(left) {
7305            assert_eq!(left.is_ignored(&path).ok(), right.is_ignored(&path).ok(), "{path:?}");
7306        }
7307    }
7308
7309    /// A small tree whose mutation crosses every structural reconciliation boundary.
7310    fn reconciliation_transition_tree() -> tempfile::TempDir {
7311        let dir = tempfile::tempdir().expect("tempdir");
7312        write_file(&dir.path().join("changed.txt"), b"before");
7313        write_file(&dir.path().join("removed.txt"), b"remove me");
7314        write_file(&dir.path().join("directory-to-file/old.rs"), b"old child");
7315        write_file(&dir.path().join("file-to-directory"), b"old file");
7316        write_file(&dir.path().join("removed-tree/nested/gone.md"), b"gone");
7317        write_file(&dir.path().join("stable/deep/kept.rs"), b"kept");
7318        dir
7319    }
7320
7321    fn mutate_reconciliation_transition_tree(root: &Path) {
7322        write_file(&root.join("changed.txt"), b"after, with a distinct size");
7323        fs::remove_file(root.join("removed.txt")).expect("remove root file");
7324
7325        fs::remove_dir_all(root.join("directory-to-file")).expect("remove old directory");
7326        write_file(&root.join("directory-to-file"), b"replacement file");
7327
7328        fs::remove_file(root.join("file-to-directory")).expect("remove old file");
7329        write_file(&root.join("file-to-directory/new.txt"), b"replacement child");
7330
7331        fs::remove_dir_all(root.join("removed-tree")).expect("remove nested tree");
7332        write_file(&root.join("added-tree/nested/new.md"), b"new nested file");
7333    }
7334
7335    fn effective_ops(commits: &[Commit]) -> Vec<Op> {
7336        let mut operations: Vec<_> = commits
7337            .iter()
7338            .flat_map(|commit| commit.changes.iter())
7339            .filter_map(|change| match change {
7340                crate::EffectiveChange::Inserted { path, kind, attrs } => {
7341                    Some(Op::Upsert { path: path.clone(), kind: *kind, attrs: *attrs })
7342                }
7343                crate::EffectiveChange::Updated { path, kind, current, .. } => {
7344                    Some(Op::Upsert { path: path.clone(), kind: *kind, attrs: *current })
7345                }
7346                crate::EffectiveChange::Removed { path, .. } => {
7347                    Some(Op::Remove { path: path.clone() })
7348                }
7349                crate::EffectiveChange::Invalidated { path, reason } => {
7350                    Some(Op::InvalidateSubtree { path: path.clone(), reason: *reason })
7351                }
7352                crate::EffectiveChange::ControlUpdated { .. }
7353                | crate::EffectiveChange::ControlRefusalUpdated { .. }
7354                | crate::EffectiveChange::Reclassified { .. } => None,
7355            })
7356            .collect();
7357        operations.sort_by(|left, right| left.path().cmp(right.path()));
7358        operations
7359    }
7360
7361    fn commit_touches(commit: &Commit, path: &Path) -> bool {
7362        commit.changes.iter().any(|change| change.path() == path)
7363    }
7364
7365    #[test]
7366    fn parallel_and_serial_walks_produce_the_same_index() {
7367        let dir = branching_tree();
7368        let serial_config = ScanConfig { threads: Some(1), ..ScanConfig::default() };
7369        let (serial, serial_report) =
7370            scan_into_index(dir.path(), &serial_config).expect("serial scan");
7371        assert!(serial_report.is_complete());
7372
7373        for threads in [2_usize, 3, 8] {
7374            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
7375            let (parallel, report) = scan_into_index(dir.path(), &config).expect("parallel scan");
7376            assert!(report.is_complete(), "{threads} threads reported errors");
7377            assert_eq!(report.entries, serial_report.entries, "{threads} threads");
7378            assert_eq!(report.dirs_read, serial_report.dirs_read, "{threads} threads");
7379            assert_eq!(report.files_walked, serial_report.files_walked, "{threads} threads");
7380            assert_eq!(report.bytes_walked, serial_report.bytes_walked, "{threads} threads");
7381            // Public roll-ups carry extension names even though the internal merge path
7382            // uses ids whose assignment order differs between serial and parallel walks.
7383            let (serial_total, parallel_total) = (serial.total(), parallel.total());
7384            assert_eq!(
7385                (
7386                    parallel_total.files,
7387                    parallel_total.dirs,
7388                    parallel_total.bytes,
7389                    parallel_total.allocated,
7390                    parallel_total.newest_mtime_ns,
7391                ),
7392                (
7393                    serial_total.files,
7394                    serial_total.dirs,
7395                    serial_total.bytes,
7396                    serial_total.allocated,
7397                    serial_total.newest_mtime_ns,
7398                ),
7399                "{threads} threads roll-up"
7400            );
7401            assert_eq!(
7402                parallel_total.by_ext, serial_total.by_ext,
7403                "{threads} threads per-extension roll-up"
7404            );
7405            assert_eq!(
7406                index_fingerprint(&parallel),
7407                index_fingerprint(&serial),
7408                "{threads} threads produced a different index"
7409            );
7410        }
7411    }
7412
7413    #[test]
7414    fn parallel_walk_emits_every_entry_exactly_once() {
7415        let dir = branching_tree();
7416        let config = ScanConfig { threads: Some(4), batch_size: 16, ..ScanConfig::default() };
7417        let mut seen: BTreeMap<PathBuf, usize> = BTreeMap::new();
7418        let report = scan(dir.path(), &config, &mut |observation| {
7419            for op in &observation.ops {
7420                if let Op::Upsert { path, .. } = &op.op {
7421                    *seen.entry(path.clone()).or_default() += 1;
7422                }
7423            }
7424        })
7425        .expect("parallel scan");
7426
7427        assert!(report.is_complete());
7428        assert_eq!(seen.len() as u64, report.entries, "entry count disagrees with the report");
7429        let duplicated: Vec<_> =
7430            seen.iter().filter(|(_path, count)| **count != 1).map(|(path, _)| path).collect();
7431        assert!(duplicated.is_empty(), "paths emitted more than once: {duplicated:?}");
7432    }
7433
7434    #[test]
7435    fn parallel_walk_honours_max_depth() {
7436        let dir = branching_tree();
7437        for threads in [1_usize, 4] {
7438            let config =
7439                ScanConfig { threads: Some(threads), max_depth: Some(2), ..ScanConfig::default() };
7440            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
7441            assert!(report.is_complete());
7442            for (path, _kind, _attrs) in index_fingerprint(&index) {
7443                assert!(
7444                    path.components().count() <= 2,
7445                    "{threads} threads kept {path:?} past the depth limit"
7446                );
7447            }
7448        }
7449    }
7450
7451    #[test]
7452    fn scan_order_never_changes_the_resulting_index() {
7453        let dir = branching_tree();
7454        let depth_first =
7455            ScanConfig { order: ScanOrder::DepthFirst, threads: Some(1), ..ScanConfig::default() };
7456        let (expected, expected_report) =
7457            scan_into_index(dir.path(), &depth_first).expect("depth-first scan");
7458
7459        for (order, threads) in
7460            [(ScanOrder::BreadthFirst, 1), (ScanOrder::BreadthFirst, 4), (ScanOrder::DepthFirst, 4)]
7461        {
7462            let config = ScanConfig { order, threads: Some(threads), ..ScanConfig::default() };
7463            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
7464            assert_eq!(report.entries, expected_report.entries, "{order:?}/{threads}");
7465            assert_eq!(report.dirs_read, expected_report.dirs_read, "{order:?}/{threads}");
7466            // Public roll-ups resolve internal ids, so their named maps are stable even
7467            // when traversal order changes id assignment.
7468            let (totals, expected_totals) = (index.total(), expected.total());
7469            assert_eq!(
7470                (totals.files, totals.dirs, totals.bytes, totals.allocated),
7471                (
7472                    expected_totals.files,
7473                    expected_totals.dirs,
7474                    expected_totals.bytes,
7475                    expected_totals.allocated
7476                ),
7477                "{order:?}/{threads} roll-up"
7478            );
7479            assert_eq!(
7480                totals.newest_mtime_ns, expected_totals.newest_mtime_ns,
7481                "{order:?}/{threads} newest mtime"
7482            );
7483            assert_eq!(
7484                totals.by_ext, expected_totals.by_ext,
7485                "{order:?}/{threads} extension tallies"
7486            );
7487            assert_eq!(
7488                index_fingerprint(&index),
7489                index_fingerprint(&expected),
7490                "{order:?}/{threads} produced a different index"
7491            );
7492        }
7493    }
7494
7495    #[test]
7496    fn a_single_worker_breadth_first_walk_is_strictly_level_ordered() {
7497        // The strict guarantee, which holds only with one worker. With several, the
7498        // queue is ordered but the claims are not: a fast worker can enqueue and claim
7499        // depth d+2 while a slow worker still holds depth d+1. See
7500        // `breadth_first_starts_every_top_level_subtree_early` for the property the
7501        // default configuration actually provides, which is the one consumers rely on.
7502        let dir = branching_tree();
7503        let config = ScanConfig {
7504            order: ScanOrder::BreadthFirst,
7505            threads: Some(1),
7506            batch_size: 1,
7507            ..ScanConfig::default()
7508        };
7509        let mut depths_in_order: Vec<usize> = Vec::new();
7510        scan(dir.path(), &config, &mut |observation| {
7511            for op in &observation.ops {
7512                if let Op::Upsert { path, kind, .. } = &op.op {
7513                    if kind.is_dir() {
7514                        depths_in_order.push(path.components().count());
7515                    }
7516                }
7517            }
7518        })
7519        .expect("scan");
7520
7521        assert!(depths_in_order.len() > 10, "fixture should have many directories");
7522        assert!(
7523            depths_in_order.windows(2).all(|pair| pair[0] <= pair[1]),
7524            "directory depths were not non-decreasing: {depths_in_order:?}"
7525        );
7526    }
7527
7528    /// How many of the fixture's twelve top-level subtrees have received any file by
7529    /// the time half the files have been emitted.
7530    ///
7531    /// This is the product metric — "is a mid-scan ranking meaningful?" — rather than
7532    /// first-touch, which cannot distinguish the orders at all: reading the root
7533    /// enumerates all twelve children at once either way. What a ranking needs is that
7534    /// the subtrees grow *together*.
7535    fn subtrees_started_at_halfway(order: ScanOrder, threads: usize, dir: &Path) -> usize {
7536        let config =
7537            ScanConfig { order, batch_size: 1, threads: Some(threads), ..ScanConfig::default() };
7538
7539        let mut files: Vec<PathBuf> = Vec::new();
7540        scan(dir, &config, &mut |observation| {
7541            for op in &observation.ops {
7542                if let Op::Upsert { path, kind, .. } = &op.op {
7543                    if !kind.is_dir() {
7544                        files.push(path.clone());
7545                    }
7546                }
7547            }
7548        })
7549        .expect("scan");
7550
7551        let halfway = files.len() / 2;
7552        let mut started: BTreeSet<PathBuf> = BTreeSet::new();
7553        for path in files.iter().take(halfway) {
7554            if let Some(top) = path.components().next() {
7555                started.insert(PathBuf::from(top.as_os_str()));
7556            }
7557        }
7558        started.len()
7559    }
7560
7561    #[test]
7562    fn a_parallel_walk_accounts_for_where_its_time_went() {
7563        // The attribution identity: every named cause is a disjoint slice of worker
7564        // wall time, so the parts can never exceed the whole, and the counters that
7565        // amortization depends on are actually incremented. This is the instrument
7566        // the scheduler experiments will read; if it drifts, they measure noise.
7567        let dir = branching_tree();
7568        let config = ScanConfig { threads: Some(4), batch_size: 64, ..ScanConfig::default() };
7569        let report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
7570        let a = report.attribution;
7571
7572        assert!(a.claims > 0, "a parallel walk claims chunks: {a:?}");
7573        assert!(a.work_ns > 0, "reading directories takes time: {a:?}");
7574        assert!(a.wall_ns > 0);
7575        // claim() locks at least once per successful claim, and release() locks once
7576        // per claim cycle too.
7577        assert!(a.lock_ops >= a.claims * 2, "lock ops out of step with claims: {a:?}");
7578        assert!(
7579            a.accounted_ns() <= a.wall_ns,
7580            "attributed slices are disjoint intervals inside worker wall: {a:?}"
7581        );
7582    }
7583
7584    #[test]
7585    fn a_serial_walk_has_no_coordination_to_attribute() {
7586        // Serial semantics: wall is the loop, "send" is the inline sink (the consumer
7587        // actually running), work is the rest — and the coordination counters stay
7588        // zero because there is no queue lock and no channel.
7589        let dir = branching_tree();
7590        let config = ScanConfig { threads: Some(1), batch_size: 64, ..ScanConfig::default() };
7591        let mut observations = 0usize;
7592        let report = scan(dir.path(), &config, &mut |_| observations += 1).expect("scan");
7593        let a = report.attribution;
7594
7595        assert!(observations > 0, "the sink ran, so send_ns measured something real");
7596        assert!(a.work_ns > 0 && a.wall_ns >= a.work_ns);
7597        assert_eq!(
7598            (a.claims, a.lock_ops, a.lock_contended, a.starved_ns, a.lock_wait_ns),
7599            (0, 0, 0, 0, 0),
7600            "no queue, no lock, nothing to wait on: {a:?}"
7601        );
7602    }
7603
7604    /// Twelve top-level subtrees, each a branching tree several levels deep.
7605    ///
7606    /// Branching matters: an earlier fixture gave every level exactly one child, which
7607    /// pinned the frontier at twelve directories and made both orders behave
7608    /// identically — a LIFO cannot dive when there is nothing to dive into. With two
7609    /// children per level, depth-first pushes siblings and immediately descends into
7610    /// the last one, which is the behaviour that leaves other subtrees behind.
7611    ///
7612    /// It is also deliberately uniform. A version using one deep spur beside shallow
7613    /// siblings made the result depend on whether `readdir` returned the spur early:
7614    /// it passed on APFS and failed on ext4.
7615    fn deep_forest() -> tempfile::TempDir {
7616        let dir = tempfile::tempdir().expect("tempdir");
7617        for top in 0..12 {
7618            let mut level: Vec<PathBuf> = vec![dir.path().join(format!("t{top}"))];
7619            for _ in 0..5 {
7620                let mut next = Vec::new();
7621                for parent in &level {
7622                    for child in 0..2 {
7623                        let path = parent.join(format!("c{child}"));
7624                        for file in 0..3 {
7625                            write_file(&path.join(format!("f{file}.dat")), b"xxxxxxxxxx");
7626                        }
7627                        next.push(path);
7628                    }
7629                }
7630                level = next;
7631            }
7632        }
7633        dir
7634    }
7635
7636    /// Files accumulated by the *least advanced* top-level subtree in the first
7637    /// quarter of the walk.
7638    ///
7639    /// Counting subtrees merely *started* cannot discriminate on a tree whose root
7640    /// fans out twelve ways: every scheduler touches all twelve immediately, because
7641    /// reading the root enumerates them. What differs is whether they then advance
7642    /// together, so the question is how far behind the laggard is.
7643    fn leanest_subtree_early(order: ScanOrder, threads: usize, dir: &Path) -> usize {
7644        let config =
7645            ScanConfig { order, batch_size: 1, threads: Some(threads), ..ScanConfig::default() };
7646        let mut files: Vec<PathBuf> = Vec::new();
7647        scan(dir, &config, &mut |observation| {
7648            for op in &observation.ops {
7649                if let Op::Upsert { path, kind, .. } = &op.op {
7650                    if !kind.is_dir() {
7651                        files.push(path.clone());
7652                    }
7653                }
7654            }
7655        })
7656        .expect("scan");
7657
7658        let quarter = files.len() / 4;
7659        let mut per_top: BTreeMap<PathBuf, usize> = BTreeMap::new();
7660        for path in files.iter().take(quarter) {
7661            if let Some(top) = path.components().next() {
7662                *per_top.entry(PathBuf::from(top.as_os_str())).or_default() += 1;
7663            }
7664        }
7665        (0..12)
7666            .map(|top| per_top.get(&PathBuf::from(format!("t{top}"))).copied().unwrap_or(0))
7667            .min()
7668            .unwrap_or(0)
7669    }
7670
7671    #[test]
7672    fn deep_subtrees_do_not_delay_their_siblings() {
7673        // The orientation property, and the reason breadth-first is the default: when
7674        // every top-level subtree is deep, depth-first pours its early effort down
7675        // whichever ones it picked up and leaves the rest at zero, while the region
7676        // scheduler advances all twelve together. A user watching the top level fill
7677        // in sees a meaningful ranking in the first case and a misleading one in the
7678        // second.
7679        //
7680        // Asserted at one worker only, and that bound is deliberate. This metric reads
7681        // *emission* order, and under several workers emission reflects which worker
7682        // finished first as much as which region was claimed — so it varies with core
7683        // count. Measured on a six-core machine the margin is wide (33-37 files against
7684        // 6); on a CI runner with fewer cores both orders can report zero. That makes it
7685        // a benchmark-grade observation, recorded in exp-013, not a unit-test assertion.
7686        //
7687        // The scheduling property itself *is* asserted deterministically, against the
7688        // queue rather than through a walk, by
7689        // `the_region_scheduler_spreads_workers_over_distinct_subtrees`.
7690        let dir = deep_forest();
7691        let breadth = leanest_subtree_early(ScanOrder::BreadthFirst, 1, dir.path());
7692        let depth = leanest_subtree_early(ScanOrder::DepthFirst, 1, dir.path());
7693        assert!(
7694            breadth > depth,
7695            "breadth-first should leave its least advanced top-level subtree further \
7696             along: {breadth} files against {depth}"
7697        );
7698    }
7699
7700    #[test]
7701    fn the_region_scheduler_spreads_workers_over_distinct_subtrees() {
7702        // The scheduler invariant, checked directly on the queue rather than through a
7703        // walk: consecutive claims by *different* workers must land in different
7704        // regions while several regions have work. This is what the round-robin ready
7705        // ring buys, and it is the thing a global FIFO could not promise.
7706        let queue = DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, None, None);
7707        let mut timing = WalkAttribution::default();
7708
7709        // Bootstrap: drain the root, then seed four top-level regions.
7710        let mut claimed = Vec::new();
7711        let root = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
7712        claimed.clear();
7713        queue.extend(
7714            (0..4).map(|top| (PathBuf::from(format!("t{top}")), 1, RegionId::UNASSIGNED)),
7715            &mut timing,
7716        );
7717        assert!(root.release(0, 0, &mut timing).is_none());
7718
7719        // Four workers with no affinity must each be handed a different region. The
7720        // claims are held for the whole loop, as four concurrent workers would hold
7721        // them, because releasing between them would let one worker take every region.
7722        let mut regions = BTreeSet::new();
7723        let mut held = Vec::new();
7724        for _ in 0..4 {
7725            let mut claimed = Vec::new();
7726            held.push(queue.claim(&mut claimed, &mut timing).expect("a region has work"));
7727            regions.insert(claimed[0].2.0);
7728            assert_eq!(claimed.len(), 1, "one directory per region so far");
7729        }
7730        assert_eq!(regions.len(), 4, "each claim took a distinct region: {regions:?}");
7731    }
7732
7733    #[test]
7734    fn breadth_first_spreads_early_work_across_top_level_subtrees() {
7735        // The justification for making breadth-first the default: at the halfway point
7736        // more of the tree's top-level subtrees have started filling, so a consumer
7737        // ranking by size mid-scan is comparing partial values rather than a mix of
7738        // final values and zeros.
7739        //
7740        // Pinned with one worker, where the ordering guarantee is strict and the result
7741        // is deterministic. The multi-worker case is deliberately NOT asserted here:
7742        // measured on this fixture the advantage disappears under the default worker
7743        // count (both orders start 7-8 subtrees, run to run), because emission order is
7744        // then dominated by worker scheduling rather than by queue order. That is a
7745        // real limitation of the current design, recorded in the plan and tracked
7746        // rather than papered over with a test tuned until it passed.
7747        let dir = branching_tree();
7748        let breadth = subtrees_started_at_halfway(ScanOrder::BreadthFirst, 1, dir.path());
7749        let depth = subtrees_started_at_halfway(ScanOrder::DepthFirst, 1, dir.path());
7750
7751        assert!(
7752            breadth > depth,
7753            "breadth-first should have more top-level subtrees underway at the halfway \
7754             point, but started {breadth} against depth-first's {depth}"
7755        );
7756    }
7757
7758    #[test]
7759    fn scan_order_does_not_change_the_cache_scope() {
7760        // Order is operational, like the worker count: it changes when observations
7761        // appear, never which ones, so it must not be able to invalidate a snapshot.
7762        let breadth = ScanConfig { order: ScanOrder::BreadthFirst, ..ScanConfig::default() };
7763        let depth = ScanConfig { order: ScanOrder::DepthFirst, ..ScanConfig::default() };
7764        assert_eq!(breadth.scope(), depth.scope());
7765    }
7766
7767    #[test]
7768    fn worker_threads_are_bounded_and_never_zero() {
7769        let zero = ScanConfig { threads: Some(0), ..ScanConfig::default() };
7770        assert_eq!(zero.worker_threads(), 1, "zero threads must fall back to the serial walk");
7771        let absurd = ScanConfig { threads: Some(usize::MAX), ..ScanConfig::default() };
7772        assert_eq!(absurd.worker_threads(), MAX_SCAN_THREADS);
7773        // The automatic choice is capped well below what a caller may request, because
7774        // the measured knee is far below the core count on a large machine.
7775        let automatic = ScanConfig { threads: None, ..ScanConfig::default() };
7776        assert!((1..=DEFAULT_SCAN_THREADS_CAP).contains(&automatic.worker_threads()));
7777    }
7778
7779    #[test]
7780    fn automatic_worker_pool_keeps_a_conservative_start_and_bounded_reserve() {
7781        assert_eq!(automatic_worker_pool(1), WorkerPool::fixed(1));
7782        assert_eq!(
7783            automatic_worker_pool(4),
7784            WorkerPool {
7785                initial: 4,
7786                maximum: 8,
7787                calibration: Some(WorkerCalibration::new(
7788                    ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
7789                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
7790                )),
7791            }
7792        );
7793        assert_eq!(
7794            automatic_worker_pool(10),
7795            WorkerPool {
7796                initial: DEFAULT_SCAN_THREADS_CAP,
7797                maximum: ADAPTIVE_SCAN_THREADS_CAP,
7798                calibration: Some(WorkerCalibration::new(
7799                    ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
7800                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
7801                )),
7802            }
7803        );
7804    }
7805
7806    #[test]
7807    fn an_abandoned_claim_does_not_strand_the_other_workers() {
7808        // The liveness property behind `DirectoryClaim`. A worker that stops mid-chunk
7809        // — consumer gone, or a panic unwinding through the directory read — still owes
7810        // the queue its claim, and `claim` parks everyone else until `outstanding`
7811        // reaches zero. Before the claim was an RAII guard both of those exits skipped
7812        // the release, and every remaining worker waited on the condvar forever while
7813        // the scoped join waited on them.
7814        let queue = std::sync::Arc::new(DirectoryQueue::new(
7815            (PathBuf::new(), 0),
7816            ScanOrder::BreadthFirst,
7817            None,
7818            None,
7819        ));
7820        let mut timing = WalkAttribution::default();
7821        let mut claimed = Vec::new();
7822
7823        // One worker takes the root and abandons it without publishing anything.
7824        drop(queue.claim(&mut claimed, &mut timing).expect("root is claimable"));
7825
7826        // A second worker must now be told the walk is over rather than parking.
7827        let waiter = queue.clone();
7828        let (done, finished) = std::sync::mpsc::sync_channel(1);
7829        std::thread::spawn(move || {
7830            let mut timing = WalkAttribution::default();
7831            let mut claimed = Vec::new();
7832            let outcome = waiter.claim(&mut claimed, &mut timing).is_some();
7833            done.send(outcome).expect("publish the claim outcome");
7834        });
7835
7836        assert_eq!(
7837            finished.recv_timeout(std::time::Duration::from_secs(5)),
7838            Ok(false),
7839            "the queue must report the walk finished instead of parking the worker"
7840        );
7841    }
7842
7843    #[test]
7844    fn automatic_queue_activates_its_reserve_only_for_slow_initial_work() {
7845        let slow = WorkerCalibration::new(3, 10);
7846        let queue =
7847            DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, Some(slow), None);
7848        let mut timing = WalkAttribution::default();
7849        let mut claimed = Vec::new();
7850
7851        let claim = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
7852        queue.extend([(PathBuf::from("child"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
7853        assert!(claim.release(2, 20, &mut timing).is_none());
7854
7855        claimed.clear();
7856        let claim = queue.claim(&mut claimed, &mut timing).expect("child is claimable");
7857        queue.extend([(PathBuf::from("grandchild"), 2, claimed[0].2)].into_iter(), &mut timing);
7858        assert_eq!(claim.release(1, 10, &mut timing), Some(2));
7859
7860        let fast = WorkerCalibration::new(3, 11);
7861        let queue =
7862            DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, Some(fast), None);
7863        let mut timing = WalkAttribution::default();
7864        let mut claimed = Vec::new();
7865        let claim = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
7866        assert!(claim.release(3, 30, &mut timing).is_none());
7867        assert!(queue.lock().controller.is_none(), "calibration decides only once");
7868    }
7869
7870    #[test]
7871    fn repeated_windows_reconsider_a_late_slow_phase_in_entry_order() {
7872        let calibration = WorkerCalibration::new(4, 10);
7873        let mut controller =
7874            WorkerController::new(calibration, WorkerPolicyExperiment::RepeatedWindows);
7875
7876        let fast = controller.observe(4, 20).expect("first complete window");
7877        let slow = controller.observe(4, 80).expect("second complete window");
7878
7879        assert!(!fast.slow);
7880        assert!(slow.slow);
7881        assert_eq!((fast.start_entry_ordinal, fast.end_entry_ordinal), (0, 4));
7882        assert_eq!((slow.start_entry_ordinal, slow.end_entry_ordinal), (4, 8));
7883    }
7884
7885    #[test]
7886    fn shipped_trace_retains_post_decision_windows_without_changing_policy() {
7887        let calibration = WorkerCalibration::new(2, 10);
7888        let pool = WorkerPool { initial: 2, maximum: 4, calibration: Some(calibration) };
7889        let recorder =
7890            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::ShippedOneShot);
7891        let queue = DirectoryQueue::new_with_policy(
7892            (PathBuf::new(), 0),
7893            ScanOrder::BreadthFirst,
7894            Some(calibration),
7895            Some(recorder.clone()),
7896            pool.initial,
7897            pool.maximum,
7898            WorkerPolicyExperiment::ShippedOneShot,
7899        );
7900        let mut timing = WalkAttribution::default();
7901        let mut claimed = Vec::new();
7902
7903        let first = queue.claim(&mut claimed, &mut timing).expect("first window");
7904        queue.extend([(PathBuf::from("late"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
7905        assert_eq!(first.release(2, 10, &mut timing), None, "fast prefix holds");
7906
7907        claimed.clear();
7908        let late = queue.claim(&mut claimed, &mut timing).expect("late phase");
7909        queue.extend([(PathBuf::from("tail"), 2, RegionId::UNASSIGNED)].into_iter(), &mut timing);
7910        assert_eq!(late.release(2, 40, &mut timing), None, "shadow cannot scale");
7911
7912        let diagnostics = recorder.finish();
7913        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Held);
7914        assert_eq!(
7915            diagnostics
7916                .worker_policy
7917                .windows
7918                .iter()
7919                .map(|window| window.decision)
7920                .collect::<Vec<_>>(),
7921            vec![WorkerPolicyDecision::Hold, WorkerPolicyDecision::ObserveSlow]
7922        );
7923    }
7924
7925    #[test]
7926    fn staged_controller_requires_a_useful_frontier_then_stays_bounded() {
7927        let calibration = WorkerCalibration::new(1, 10);
7928        let pool = WorkerPool { initial: 2, maximum: 8, calibration: Some(calibration) };
7929        let recorder =
7930            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::StagedGatedWindows);
7931        let queue = DirectoryQueue::new_with_policy(
7932            (PathBuf::new(), 0),
7933            ScanOrder::BreadthFirst,
7934            Some(calibration),
7935            Some(recorder.clone()),
7936            pool.initial,
7937            pool.maximum,
7938            WorkerPolicyExperiment::StagedGatedWindows,
7939        );
7940        let mut timing = WalkAttribution::default();
7941        let mut claimed = Vec::new();
7942
7943        let root = queue.claim(&mut claimed, &mut timing).expect("root");
7944        queue.extend([(PathBuf::from("narrow"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
7945        assert_eq!(root.release(1, 20, &mut timing), None);
7946
7947        claimed.clear();
7948        let narrow = queue.claim(&mut claimed, &mut timing).expect("narrow child");
7949        queue.extend(
7950            (0..9).map(|index| (PathBuf::from(format!("wide-{index}")), 2, RegionId::UNASSIGNED)),
7951            &mut timing,
7952        );
7953        assert_eq!(narrow.release(1, 20, &mut timing), Some(4));
7954
7955        claimed.clear();
7956        let wide = queue.claim(&mut claimed, &mut timing).expect("wide claim");
7957        queue.extend(
7958            (0..9).map(|index| (PathBuf::from(format!("wider-{index}")), 3, RegionId::UNASSIGNED)),
7959            &mut timing,
7960        );
7961        assert_eq!(wide.release(1, 20, &mut timing), Some(8));
7962
7963        let diagnostics = recorder.finish();
7964        let decisions: Vec<_> = diagnostics
7965            .worker_policy
7966            .windows
7967            .iter()
7968            .map(|window| (window.decision, window.requested_workers))
7969            .collect();
7970        assert_eq!(
7971            decisions,
7972            vec![
7973                (WorkerPolicyDecision::HoldInsufficientFrontier, None),
7974                (WorkerPolicyDecision::ScaleUp, Some(4)),
7975                (WorkerPolicyDecision::ScaleUp, Some(8)),
7976            ]
7977        );
7978        assert!(
7979            diagnostics
7980                .worker_policy
7981                .windows
7982                .windows(2)
7983                .all(|pair| pair[0].end_entry_ordinal <= pair[1].start_entry_ordinal)
7984        );
7985        assert!(diagnostics.worker_policy.windows.iter().all(|window| {
7986            window.requested_workers.is_none_or(|workers| workers <= pool.maximum)
7987        }));
7988    }
7989
7990    #[test]
7991    fn staged_controller_does_not_add_producers_to_a_delayed_handoff() {
7992        let calibration = WorkerCalibration::new(1, 10);
7993        let pool = WorkerPool { initial: 2, maximum: 8, calibration: Some(calibration) };
7994        let recorder =
7995            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::StagedGatedWindows);
7996        recorder.handoff_sent();
7997        recorder.handoff_sent();
7998        let queue = DirectoryQueue::new_with_policy(
7999            (PathBuf::new(), 0),
8000            ScanOrder::BreadthFirst,
8001            Some(calibration),
8002            Some(recorder.clone()),
8003            pool.initial,
8004            pool.maximum,
8005            WorkerPolicyExperiment::StagedGatedWindows,
8006        );
8007        let mut timing = WalkAttribution::default();
8008        let mut claimed = Vec::new();
8009        let claim = queue.claim(&mut claimed, &mut timing).expect("root");
8010        queue.extend(
8011            (0..9).map(|index| (PathBuf::from(format!("ready-{index}")), 1, RegionId::UNASSIGNED)),
8012            &mut timing,
8013        );
8014
8015        assert_eq!(claim.release(1, 20, &mut timing), None);
8016        recorder.handoff_received();
8017        recorder.handoff_received();
8018        let diagnostics = recorder.finish();
8019        assert_eq!(
8020            diagnostics.worker_policy.windows[0].decision,
8021            WorkerPolicyDecision::HoldHandoffBacklog
8022        );
8023    }
8024
8025    #[test]
8026    fn candidate_retains_post_expansion_shadow_history() {
8027        let calibration = WorkerCalibration::new(1, 10);
8028        let pool = WorkerPool { initial: 2, maximum: 4, calibration: Some(calibration) };
8029        let recorder =
8030            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::RepeatedWindows);
8031        let queue = DirectoryQueue::new_with_policy(
8032            (PathBuf::new(), 0),
8033            ScanOrder::BreadthFirst,
8034            Some(calibration),
8035            Some(recorder.clone()),
8036            pool.initial,
8037            pool.maximum,
8038            WorkerPolicyExperiment::RepeatedWindows,
8039        );
8040        let mut timing = WalkAttribution::default();
8041        let mut claimed = Vec::new();
8042
8043        let slow = queue.claim(&mut claimed, &mut timing).expect("slow prefix");
8044        queue.extend(
8045            (0..4).map(|index| (PathBuf::from(format!("fast-{index}")), 1, RegionId::UNASSIGNED)),
8046            &mut timing,
8047        );
8048        assert_eq!(slow.release(1, 20, &mut timing), Some(4));
8049
8050        claimed.clear();
8051        let fast = queue.claim(&mut claimed, &mut timing).expect("fast suffix");
8052        queue.extend([(PathBuf::from("tail"), 2, RegionId::UNASSIGNED)].into_iter(), &mut timing);
8053        assert_eq!(fast.release(1, 1, &mut timing), None);
8054
8055        let diagnostics = recorder.finish();
8056        assert_eq!(
8057            diagnostics
8058                .worker_policy
8059                .windows
8060                .iter()
8061                .map(|window| window.decision)
8062                .collect::<Vec<_>>(),
8063            vec![WorkerPolicyDecision::ScaleUp, WorkerPolicyDecision::ObserveFast]
8064        );
8065    }
8066
8067    #[test]
8068    fn every_experimental_controller_preserves_exactness_and_shutdown() {
8069        let dir = branching_tree();
8070        let serial = ScanConfig { threads: Some(1), ..ScanConfig::default() };
8071        let (reference, _) = scan_into_index(dir.path(), &serial).expect("serial reference");
8072        let automatic = ScanConfig { threads: None, ..ScanConfig::default() };
8073
8074        for policy in [
8075            WorkerPolicyExperiment::ShippedOneShot,
8076            WorkerPolicyExperiment::RepeatedWindows,
8077            WorkerPolicyExperiment::StagedGatedWindows,
8078        ] {
8079            let (index, report, diagnostics) =
8080                scan_into_index_with_policy_diagnostics(dir.path(), &automatic, policy)
8081                    .expect("candidate scan finishes");
8082            assert!(report.is_complete(), "{policy:?}: {:?}", report.errors);
8083            assert_eq!(index_fingerprint(&reference), index_fingerprint(&index), "{policy:?}");
8084            assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
8085            assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
8086            assert_eq!(diagnostics.worker_policy.handoff_backlog_at_finish, 0);
8087            assert!(
8088                diagnostics.worker_policy.workers_spawned
8089                    <= diagnostics.worker_policy.maximum_workers
8090            );
8091        }
8092    }
8093
8094    /// A deterministic model of the automatic worker policy under *completion* order.
8095    ///
8096    /// The scaling decision is driven by chunk releases, and chunks complete in whatever
8097    /// order the filesystem and the workers produce them — not in traversal order. On a
8098    /// homogeneous tree that distinction is invisible, because every prefix looks like
8099    /// every other. On a heterogeneous one it decides the answer.
8100    ///
8101    /// These tests exist because the alternative is a stopwatch on a real tree, which
8102    /// measures one host on one day and cannot separate a policy defect from ambient
8103    /// noise. Replaying an explicit completion order through the shipped calibration
8104    /// isolates the policy exactly, and does so identically on every platform.
8105    ///
8106    /// They characterize behavior; they do not endorse a replacement. Which controller
8107    /// is *faster* is a question only the held-out Apple Silicon/APFS matrix can answer.
8108    mod completion_order {
8109        use super::{ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, WorkerCalibration};
8110
8111        /// One chunk release: entries observed and worker time spent observing them.
8112        #[derive(Clone, Copy, Debug)]
8113        struct Chunk {
8114            entries: u64,
8115            work_ns: u64,
8116        }
8117
8118        impl Chunk {
8119            /// A run of `entries` entries costing `per_entry_ns` each.
8120            const fn at(entries: u64, per_entry_ns: u64) -> Self {
8121                Self { entries, work_ns: entries.saturating_mul(per_entry_ns) }
8122            }
8123        }
8124
8125        /// What a policy concluded over one completion order.
8126        #[derive(Debug, PartialEq, Eq)]
8127        enum Outcome {
8128            /// The policy found the filesystem slow and expanded the reserve.
8129            ScaledUp { after_chunks: usize },
8130            /// The policy found the filesystem fast and held the initial pool.
8131            Held { after_chunks: usize },
8132            /// The walk ended before the policy observed enough to conclude anything.
8133            ///
8134            /// Distinct from [`Outcome::Held`] on purpose: nothing was measured, so a
8135            /// held pool here is an absence of evidence rather than a decision.
8136            Undecided,
8137        }
8138
8139        /// Mean cost per entry over a whole trace, which is what the threshold *means*.
8140        fn whole_trace_ns_per_entry(trace: &[Chunk]) -> u64 {
8141            let entries: u64 = trace.iter().map(|chunk| chunk.entries).sum();
8142            let work_ns: u64 = trace.iter().map(|chunk| chunk.work_ns).sum();
8143            assert!(entries > 0, "a trace must observe entries");
8144            work_ns / entries
8145        }
8146
8147        /// Replay a completion order through the *shipped* calibration.
8148        ///
8149        /// This drives [`WorkerCalibration::observe`] itself rather than restating its
8150        /// arithmetic, so the model cannot quietly drift from the policy it is evidence
8151        /// about. The loop mirrors `DirectoryQueue::release`: fold each chunk in, and
8152        /// stop at the first one that produces a verdict.
8153        fn shipped(window: u64, threshold_ns: u64, trace: &[Chunk]) -> Outcome {
8154            let mut calibration = WorkerCalibration::new(window, threshold_ns);
8155            for (index, chunk) in trace.iter().enumerate() {
8156                if let Some(slow) = calibration.observe(chunk.entries, chunk.work_ns) {
8157                    let after_chunks = index + 1;
8158                    return if slow {
8159                        Outcome::ScaledUp { after_chunks }
8160                    } else {
8161                        Outcome::Held { after_chunks }
8162                    };
8163                }
8164            }
8165            Outcome::Undecided
8166        }
8167
8168        /// Entries per chunk in the traces below. Four fill the 16,384-entry window.
8169        const CHUNK: u64 = 4_096;
8170        /// A shallow, cache-warm phase: metadata already resident.
8171        const FAST: Chunk = Chunk::at(CHUNK, 2_000);
8172        /// A deep, cold phase: the latency-bound regime the reserve exists to hide.
8173        const SLOW: Chunk = Chunk::at(CHUNK, 90_000);
8174
8175        #[test]
8176        fn completion_order_alone_flips_the_shipped_decision() {
8177            // The defect, stated as an experiment: hold the *tree* constant and vary
8178            // only the order its chunks complete in. Both traces contain the same four
8179            // fast and four slow chunks, so they describe the same filesystem work.
8180            let fast_phase_first = [FAST, FAST, FAST, FAST, SLOW, SLOW, SLOW, SLOW];
8181            let interleaved = [SLOW, FAST, SLOW, FAST, SLOW, FAST, SLOW, FAST];
8182
8183            let window = 4 * CHUNK;
8184            let threshold = ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY;
8185
8186            // Whole-walk truth is identical, and by the policy's own threshold both
8187            // walks are latency-bound: 46 µs per entry against a 30 µs trigger.
8188            let truth = whole_trace_ns_per_entry(&fast_phase_first);
8189            assert_eq!(truth, whole_trace_ns_per_entry(&interleaved));
8190            assert!(
8191                truth >= threshold,
8192                "both traces are slow walks by the shipped threshold: {truth} < {threshold}"
8193            );
8194
8195            // Yet the decision depends entirely on which chunks happened to finish
8196            // first. One walk hides latency; the other runs the whole slow phase on the
8197            // starting pool, having concluded from an unrepresentative prefix.
8198            assert_eq!(
8199                shipped(window, threshold, &fast_phase_first),
8200                Outcome::Held { after_chunks: 4 },
8201                "a fast prefix holds the pool for a walk that is slow overall"
8202            );
8203            assert_eq!(
8204                shipped(window, threshold, &interleaved),
8205                Outcome::ScaledUp { after_chunks: 4 },
8206                "the same tree scales up when its slow chunks land in the window"
8207            );
8208        }
8209
8210        #[test]
8211        fn a_slow_phase_after_the_window_is_never_reconsidered() {
8212            // The heterogeneous-tree case from the field report. A small fast region
8213            // fills the window, and everything after it is slow — but the calibration
8214            // is already gone, so no amount of later evidence can reopen the decision.
8215            let mut trace = vec![FAST; 4];
8216            trace.extend(std::iter::repeat_n(SLOW, 400));
8217
8218            let observed = whole_trace_ns_per_entry(&trace);
8219            assert!(
8220                observed >= 2 * ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
8221                "the walk is overwhelmingly latency-bound: {observed} ns per entry"
8222            );
8223
8224            assert_eq!(
8225                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
8226                Outcome::Held { after_chunks: 4 },
8227                "1% of the walk decided the worker policy for the other 99%"
8228            );
8229        }
8230
8231        #[test]
8232        fn slow_in_flight_work_is_censored_by_fast_completions() {
8233            // Four slow chunks have already been claimed, but their filesystem calls
8234            // remain in flight while four cache-warm chunks complete. Completion-order
8235            // calibration cannot see owed work: the fast completions close the window
8236            // and permanently hold before any slow claim returns.
8237            let completed_before_slow_returns = [FAST, FAST, FAST, FAST];
8238            assert_eq!(
8239                shipped(
8240                    4 * CHUNK,
8241                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
8242                    &completed_before_slow_returns,
8243                ),
8244                Outcome::Held { after_chunks: 4 }
8245            );
8246
8247            let mut eventual_completions = completed_before_slow_returns.to_vec();
8248            eventual_completions.extend([SLOW, SLOW, SLOW, SLOW]);
8249            assert!(
8250                whole_trace_ns_per_entry(&eventual_completions)
8251                    >= ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY
8252            );
8253            assert_eq!(
8254                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &eventual_completions,),
8255                Outcome::Held { after_chunks: 4 }
8256            );
8257        }
8258
8259        #[test]
8260        fn a_slow_prefix_can_scale_a_walk_that_is_fast_overall() {
8261            // The mirror-image error. A one-way expansion reacts correctly to the
8262            // prefix by its local threshold, but the prefix is under 1% of this walk
8263            // and the whole trace is firmly in the fast regime. A repeated trigger
8264            // alone cannot undo an expansion; staged growth limits exposure but does
8265            // not make reversible parking unnecessary.
8266            let mut trace = vec![SLOW; 4];
8267            trace.extend(std::iter::repeat_n(FAST, 400));
8268            let observed = whole_trace_ns_per_entry(&trace);
8269            assert!(observed < ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY);
8270            assert_eq!(
8271                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
8272                Outcome::ScaledUp { after_chunks: 4 }
8273            );
8274            assert_eq!(
8275                sliding(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
8276                Outcome::ScaledUp { after_chunks: 4 }
8277            );
8278        }
8279
8280        #[test]
8281        fn a_walk_shorter_than_the_window_decides_nothing() {
8282            // Fails closed rather than reporting a held pool: a walk this short never
8283            // observed enough to have an opinion, and an artifact that recorded `Held`
8284            // would claim a measurement that was never taken.
8285            let trace = [FAST, SLOW];
8286            assert_eq!(
8287                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
8288                Outcome::Undecided
8289            );
8290        }
8291
8292        /// A screening-only candidate: a window that slides instead of closing once.
8293        ///
8294        /// Present as *evidence about a design*, not as a proposed change. It keeps the
8295        /// shipped trigger and pool bounds and alters only when the question is asked,
8296        /// which is the narrowest edit that could address the order sensitivity above.
8297        /// Whether it is faster on a real tree is unmeasured here and unmeasurable in a
8298        /// virtualized non-APFS environment; selecting it would need the held-out Apple
8299        /// Silicon matrix that this workstream has not yet been able to run.
8300        struct SlidingWindow {
8301            window_entries: u64,
8302            threshold_ns: u64,
8303            recent: std::collections::VecDeque<Chunk>,
8304            entries: u64,
8305            work_ns: u64,
8306        }
8307
8308        impl SlidingWindow {
8309            fn new(window_entries: u64, threshold_ns: u64) -> Self {
8310                Self {
8311                    window_entries,
8312                    threshold_ns,
8313                    recent: std::collections::VecDeque::new(),
8314                    entries: 0,
8315                    work_ns: 0,
8316                }
8317            }
8318
8319            /// Fold in a chunk and re-ask the question over the trailing window.
8320            fn observe(&mut self, chunk: Chunk) -> Option<bool> {
8321                self.recent.push_back(chunk);
8322                self.entries = self.entries.saturating_add(chunk.entries);
8323                self.work_ns = self.work_ns.saturating_add(chunk.work_ns);
8324
8325                // Drop from the front while the window stays full without the oldest
8326                // chunk, so the answer describes recent work rather than the whole walk.
8327                while let Some(oldest) = self.recent.front().copied() {
8328                    if self.entries - oldest.entries < self.window_entries {
8329                        break;
8330                    }
8331                    self.recent.pop_front();
8332                    self.entries -= oldest.entries;
8333                    self.work_ns -= oldest.work_ns;
8334                }
8335
8336                (self.entries >= self.window_entries)
8337                    .then(|| self.work_ns / self.entries >= self.threshold_ns)
8338            }
8339        }
8340
8341        /// Replay a completion order through the candidate, stopping at its first
8342        /// scale-up. The shipped pool only grows, so a later verdict cannot undo one.
8343        fn sliding(window: u64, threshold_ns: u64, trace: &[Chunk]) -> Outcome {
8344            let mut policy = SlidingWindow::new(window, threshold_ns);
8345            let mut decided = None;
8346            for (index, chunk) in trace.iter().enumerate() {
8347                if let Some(slow) = policy.observe(*chunk) {
8348                    let after_chunks = index + 1;
8349                    if slow {
8350                        return Outcome::ScaledUp { after_chunks };
8351                    }
8352                    decided.get_or_insert(Outcome::Held { after_chunks });
8353                }
8354            }
8355            decided.unwrap_or(Outcome::Undecided)
8356        }
8357
8358        #[test]
8359        fn screening_a_sliding_window_against_the_order_sensitivity() {
8360            let window = 4 * CHUNK;
8361            let threshold = ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY;
8362
8363            // The pair that splits the shipped policy reaches one answer here, and it
8364            // is the answer the whole-trace mean supports in both orders.
8365            let fast_phase_first = [FAST, FAST, FAST, FAST, SLOW, SLOW, SLOW, SLOW];
8366            let interleaved = [SLOW, FAST, SLOW, FAST, SLOW, FAST, SLOW, FAST];
8367            // Both reach the same verdict; they differ only in how long the fast prefix
8368            // delays it, which is the behavior a trailing window is supposed to have.
8369            assert_eq!(
8370                sliding(window, threshold, &fast_phase_first),
8371                Outcome::ScaledUp { after_chunks: 6 }
8372            );
8373            assert_eq!(
8374                sliding(window, threshold, &interleaved),
8375                Outcome::ScaledUp { after_chunks: 4 }
8376            );
8377
8378            // And the late slow phase is reached rather than missed: two slow chunks
8379            // after the window closes are enough to pull the trailing mean over.
8380            let mut late = vec![FAST; 4];
8381            late.extend(std::iter::repeat_n(SLOW, 400));
8382            assert_eq!(sliding(window, threshold, &late), Outcome::ScaledUp { after_chunks: 6 });
8383
8384            // A genuinely fast tree must still hold the pool: the candidate has to keep
8385            // the property the shipped policy gets right, or it is not a candidate.
8386            let uniformly_fast = vec![FAST; 40];
8387            assert_eq!(
8388                sliding(window, threshold, &uniformly_fast),
8389                Outcome::Held { after_chunks: 4 }
8390            );
8391
8392            // A short walk still decides nothing, for the same reason as above.
8393            assert_eq!(sliding(window, threshold, &[FAST, SLOW]), Outcome::Undecided);
8394        }
8395    }
8396
8397    #[test]
8398    fn thread_count_does_not_change_the_cache_scope() {
8399        // Threads are an operational choice. If they leaked into the scope, changing
8400        // the pool size would invalidate every snapshot on disk.
8401        let serial = ScanConfig { threads: Some(1), ..ScanConfig::default() };
8402        let parallel = ScanConfig { threads: Some(8), ..ScanConfig::default() };
8403        assert_eq!(serial.scope(), parallel.scope());
8404    }
8405
8406    #[test]
8407    fn scan_populates_an_index_end_to_end() {
8408        let dir = sample_tree();
8409        let (index, report) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
8410
8411        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8412        let total = index.total();
8413        assert_eq!(total.files, 3);
8414        assert_eq!(total.dirs, 2);
8415        assert_eq!(total.bytes, 5 + 12 + 9);
8416        assert_eq!(total.by_ext[".rs"].files, 2);
8417        assert_eq!(total.by_ext[".txt"].files, 1);
8418
8419        let src = index.rollup(Path::new("src")).expect("src");
8420        assert_eq!(src.files, 2);
8421        assert_eq!(src.dirs, 1);
8422    }
8423
8424    #[test]
8425    fn cold_scan_routes_control_sources_through_both_walkers() {
8426        let dir = tempfile::tempdir().expect("tempdir");
8427        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8428        write_file(&dir.path().join("debug.log"), b"ignored");
8429        write_file(&dir.path().join("keep.rs"), b"visible");
8430
8431        for threads in [1, 4] {
8432            let config =
8433                ScanConfig { read_controls: true, threads: Some(threads), ..ScanConfig::default() };
8434            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
8435
8436            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8437            assert!(
8438                index
8439                    .controls()
8440                    .expect("control state observed")
8441                    .source_is(Path::new(".gitignore"), b"*.log\n")
8442            );
8443            assert_eq!(
8444                index.is_ignored(Path::new("debug.log")).expect("control state observed"),
8445                Some(true)
8446            );
8447            assert_eq!(
8448                index.is_ignored(Path::new("keep.rs")).expect("control state observed"),
8449                Some(false)
8450            );
8451            let partitions = index.partition_total().expect("control state observed");
8452            assert_eq!(partitions.all.files, 3);
8453            assert_eq!(partitions.unignored.files, 2);
8454        }
8455    }
8456
8457    #[cfg(unix)]
8458    #[test]
8459    fn raced_fifo_control_source_is_rejected_without_blocking() {
8460        let dir = tempfile::tempdir().expect("tempdir");
8461        let control = dir.path().join(".gitignore");
8462        let status = match std::process::Command::new("mkfifo").arg(&control).status() {
8463            Ok(status) => status,
8464            Err(error) if error.kind() == std::io::ErrorKind::NotFound => return,
8465            Err(error) => panic!("create fifo: {error}"),
8466        };
8467        assert!(status.success(), "mkfifo exited with {status}");
8468
8469        let root = dir.path().to_path_buf();
8470        let (sender, receiver) = std::sync::mpsc::channel();
8471        std::thread::spawn(move || {
8472            let result = read_control_op_unconditional(
8473                &root,
8474                Path::new(".gitignore"),
8475                EntryKind::File,
8476                Some(crate::control::DEFAULT_CONTROL_BUDGET),
8477            );
8478            sender.send(result).ok();
8479        });
8480        let result = receiver
8481            .recv_timeout(std::time::Duration::from_secs(1))
8482            .expect("a raced FIFO must not block the scan worker")
8483            .expect("the non-regular replacement is a normal control removal");
8484
8485        assert!(matches!(result, Some(Op::ControlRemove { .. })));
8486    }
8487
8488    #[test]
8489    fn hidden_admission_keeps_exact_allowlist_and_control_signals_only() {
8490        let dir = tempfile::tempdir().expect("tempdir");
8491        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8492        write_file(&dir.path().join("debug.log"), b"ignored");
8493        write_file(&dir.path().join(".secret/token"), b"hidden");
8494        write_file(&dir.path().join(".github/workflows/check.yml"), b"visible");
8495        let hidden = std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([".github"]));
8496
8497        for threads in [1, 4] {
8498            let config = ScanConfig {
8499                hidden: Some(std::sync::Arc::clone(&hidden)),
8500                threads: Some(threads),
8501                read_controls: true,
8502                ..ScanConfig::default()
8503            };
8504            let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
8505
8506            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8507            assert!(index.lookup(Path::new(".gitignore")).is_none());
8508            assert!(index.lookup(Path::new(".secret")).is_none());
8509            assert!(index.lookup(Path::new(".secret/token")).is_none());
8510            assert!(index.lookup(Path::new(".github/workflows/check.yml")).is_some());
8511            assert!(
8512                index
8513                    .controls()
8514                    .expect("control state observed")
8515                    .source_is(Path::new(".gitignore"), b"*.log\n")
8516            );
8517            assert_eq!(
8518                index.is_ignored(Path::new("debug.log")).expect("control state observed"),
8519                Some(true)
8520            );
8521
8522            fs::remove_file(dir.path().join(".gitignore")).expect("remove control");
8523            if threads > 1 {
8524                fs::create_dir(dir.path().join(".gitignore")).expect("replace with directory");
8525            }
8526            let reconciled = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
8527            assert!(reconciled.is_complete());
8528            assert!(index.controls().expect("control state observed").is_empty());
8529            if threads > 1 {
8530                fs::remove_dir(dir.path().join(".gitignore")).expect("remove directory");
8531            }
8532            write_file(&dir.path().join(".gitignore"), b"*.log\n");
8533        }
8534    }
8535
8536    #[test]
8537    fn excluded_ignored_directory_is_not_enumerated_and_rule_edits_reconcile_it() {
8538        let root = tempfile::tempdir().expect("root");
8539        write_file(&root.path().join(".gitignore"), b"target/\n");
8540        write_file(&root.path().join("target/deep/file.rs"), b"code");
8541        write_file(&root.path().join("keep.rs"), b"kept");
8542        let config = ScanConfig {
8543            population: crate::query::IgnoredEntries::Exclude,
8544            threads: Some(4),
8545            ..ScanConfig::default()
8546        };
8547
8548        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
8549        assert!(cold.is_complete(), "{:?}", cold.errors);
8550        assert_eq!(cold.dirs_read, 1, "ignored target was not opened");
8551        assert!(index.lookup(Path::new("target")).is_none());
8552        assert!(index.lookup(Path::new("keep.rs")).is_some());
8553
8554        write_file(&root.path().join(".gitignore"), b"");
8555        let exposed = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile exposure");
8556        assert!(exposed.is_complete(), "{:?}", exposed.scan.errors);
8557        assert!(index.lookup(Path::new("target/deep/file.rs")).is_some());
8558        assert!(exposed.scan.dirs_read >= 3);
8559
8560        write_file(&root.path().join(".gitignore"), b"target/\n");
8561        let excluded = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile exclusion");
8562        assert!(excluded.is_complete(), "{:?}", excluded.scan.errors);
8563        assert!(index.lookup(Path::new("target")).is_none());
8564        assert_eq!(excluded.scan.dirs_read, 1, "ignored target was not reopened");
8565    }
8566
8567    #[test]
8568    fn excluded_subtree_refresh_keeps_ignored_file_and_directory_out_of_scope() {
8569        let root = tempfile::tempdir().expect("root");
8570        write_file(&root.path().join(".gitignore"), b"target/\n*.log\n");
8571        write_file(&root.path().join("target/deep/file.rs"), b"code");
8572        write_file(&root.path().join("debug.log"), b"ignored");
8573        let config = ScanConfig {
8574            population: crate::query::IgnoredEntries::Exclude,
8575            ..ScanConfig::default()
8576        };
8577        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
8578        assert!(cold.is_complete());
8579        assert!(index.lookup(Path::new("target")).is_none());
8580        assert!(index.lookup(Path::new("debug.log")).is_none());
8581
8582        for path in ["target", "debug.log"] {
8583            let refresh = reconcile_subtree(&mut index, Path::new(path), &config, &mut |_| {})
8584                .expect("subtree refresh");
8585            assert!(refresh.is_complete(), "{path}: {:?}", refresh.scan.errors);
8586            assert!(index.lookup(Path::new(path)).is_none(), "{path} is outside scope");
8587        }
8588        assert_eq!(index.total().dirs, 0);
8589    }
8590
8591    #[test]
8592    fn handled_subtree_refresh_recovers_pruned_ancestry_after_control_edit() {
8593        let root = tempfile::tempdir().expect("root");
8594        write_file(&root.path().join(".gitignore"), b"target/\n");
8595        write_file(&root.path().join("target/deep/file.rs"), b"code");
8596        let config = ScanConfig {
8597            population: crate::query::IgnoredEntries::Exclude,
8598            ..ScanConfig::default()
8599        };
8600        let (index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
8601        assert!(cold.is_complete());
8602        let handle = IndexHandle::new(index);
8603
8604        let hidden = reconcile_subtree_handle(
8605            &handle,
8606            Path::new("target/deep/file.rs"),
8607            &config,
8608            &mut |_| {},
8609        )
8610        .expect("refresh pruned descendant");
8611        assert!(hidden.is_complete());
8612        assert!(
8613            !handle
8614                .read_with(|index| index.lookup(Path::new("target")).is_some())
8615                .expect("read after hidden refresh")
8616        );
8617
8618        write_file(&root.path().join(".gitignore"), b"");
8619        let exposed = reconcile_subtree_handle(&handle, Path::new("target"), &config, &mut |_| {})
8620            .expect("refresh changed control");
8621        assert!(exposed.is_complete());
8622        assert!(
8623            handle
8624                .read_with(|index| index.lookup(Path::new("target/deep/file.rs")).is_some())
8625                .expect("read after control recovery")
8626        );
8627
8628        write_file(&root.path().join(".gitignore"), b"target/\n");
8629        let hidden_again =
8630            reconcile_subtree_handle(&handle, Path::new("target"), &config, &mut |_| {})
8631                .expect("refresh restored control");
8632        assert!(hidden_again.is_complete());
8633        assert!(
8634            !handle
8635                .read_with(|index| index.lookup(Path::new("target")).is_some())
8636                .expect("read after restored control")
8637        );
8638    }
8639
8640    #[cfg(unix)]
8641    #[test]
8642    fn targeted_refresh_keeps_unknown_population_below_unreadable_ancestor_control() {
8643        use std::os::unix::fs::PermissionsExt;
8644
8645        if !crate::test_support::require_permission_bits() {
8646            return;
8647        }
8648        let root = tempfile::tempdir().expect("root");
8649        let control = root.path().join(".gitignore");
8650        write_file(&control, b"*.log\n");
8651        write_file(&root.path().join("a/keep.rs"), b"unknown membership");
8652        let config =
8653            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
8654        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("unreadable");
8655        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
8656        assert!(!cold.is_complete());
8657        assert!(index.lookup(Path::new("a/keep.rs")).is_some());
8658        assert_eq!(index.ignored_classification(Path::new("a/keep.rs")), None);
8659
8660        let refreshed = reconcile_subtree(&mut index, Path::new("a/keep.rs"), &config, &mut |_| {});
8661        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore control");
8662        let refreshed = refreshed.expect("targeted refresh");
8663        assert!(!refreshed.is_complete(), "unreadable governing control was not visited");
8664        assert!(index.lookup(Path::new("a/keep.rs")).is_some(), "unknown must stay retained");
8665        assert_eq!(index.ignored_classification(Path::new("a/keep.rs")), None);
8666    }
8667
8668    #[test]
8669    fn exclusion_honors_nested_negation_and_refused_rule_changes() {
8670        let root = tempfile::tempdir().expect("root");
8671        write_file(&root.path().join("nested/.gitignore"), b"*.log\n!keep.log\n");
8672        write_file(&root.path().join("nested/keep.log"), b"negated");
8673        write_file(&root.path().join("nested/drop.log"), b"ignored");
8674        let config = ScanConfig {
8675            population: crate::query::IgnoredEntries::Exclude,
8676            control_limits: crate::control::ControlLimits {
8677                line_limit: Some(20),
8678                ..crate::control::ControlLimits::default()
8679            },
8680            ..ScanConfig::default()
8681        };
8682        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold");
8683        assert!(cold.is_complete(), "{:?}", cold.errors);
8684        assert!(index.lookup(Path::new("nested/keep.log")).is_some());
8685        assert!(index.lookup(Path::new("nested/drop.log")).is_none());
8686
8687        write_file(&root.path().join("nested/.gitignore"), b"this-line-is-over-the-limit\n");
8688        let changed =
8689            reconcile(&mut index, &config, &mut |_| {}).expect("reconcile refused source");
8690        assert!(changed.is_complete(), "{:?}", changed.scan.errors);
8691        assert!(index.lookup(Path::new("nested/drop.log")).is_some(), "unknown cannot be pruned");
8692        assert_eq!(index.ignored_classification(Path::new("nested/drop.log")), None);
8693    }
8694
8695    #[test]
8696    fn only_population_skips_nonignored_content_candidates_and_refusals_are_unknown() {
8697        let root = tempfile::tempdir().expect("root");
8698        write_file(&root.path().join(".gitignore"), b"*.log\n");
8699        write_file(&root.path().join("keep.rs"), b"code");
8700        write_file(&root.path().join("debug.log"), b"ignored");
8701        write_file(&root.path().join("nested/keep.rs"), b"kept");
8702        write_file(&root.path().join("nested/debug.log"), b"ignored below nonignored dir");
8703        let only =
8704            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
8705        let (index, report) = scan_into_index(root.path(), &only).expect("scan");
8706        assert!(report.is_complete());
8707        assert!(index.lookup(Path::new("keep.rs")).is_none());
8708        assert!(index.lookup(Path::new("nested")).is_some());
8709        assert!(index.lookup(Path::new("nested/keep.rs")).is_none());
8710        assert!(index.lookup(Path::new("nested/debug.log")).is_some());
8711        let candidates = index.analysis_candidates(crate::content::AnalysisSet::NONE.with_lines());
8712        assert_eq!(candidates.len(), 2);
8713        assert!(
8714            candidates.iter().any(|candidate| candidate.relative_path == Path::new("debug.log"))
8715        );
8716        assert!(
8717            candidates
8718                .iter()
8719                .any(|candidate| candidate.relative_path == Path::new("nested/debug.log"))
8720        );
8721
8722        let refused = ScanConfig {
8723            population: crate::query::IgnoredEntries::Exclude,
8724            control_limits: crate::control::ControlLimits {
8725                line_limit: Some(1),
8726                ..crate::control::ControlLimits::default()
8727            },
8728            ..ScanConfig::default()
8729        };
8730        let (index, report) = scan_into_index(root.path(), &refused).expect("refused scan");
8731        assert!(report.is_complete());
8732        assert!(index.lookup(Path::new("debug.log")).is_some(), "unknown is not pruned");
8733        assert_eq!(index.ignored_classification(Path::new("debug.log")), None);
8734        assert!(
8735            index.analysis_candidates(crate::content::AnalysisSet::NONE.with_lines()).is_empty()
8736        );
8737    }
8738
8739    #[test]
8740    fn only_population_content_is_complete_with_retained_control_file() {
8741        let root = tempfile::tempdir().expect("root");
8742        write_file(&root.path().join(".gitignore"), b"vendor/\n");
8743        write_file(&root.path().join("main.rs"), b"fn main() {}\n");
8744        write_file(&root.path().join("vendor/lib.rs"), b"fn lib() {}\n");
8745        let config =
8746            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
8747        let (mut index, scan) = scan_into_index(root.path(), &config).expect("scan");
8748        assert!(scan.is_complete());
8749        let profile = crate::content::AnalysisSet::NONE.with_lines().with_code();
8750        let analyzed = crate::content::analyze_index(
8751            &mut index,
8752            crate::content::AnalysisRequest {
8753                profile,
8754                ..crate::content::AnalysisRequest::default()
8755            },
8756        );
8757        assert!(analyzed.is_complete(), "{analyzed:?}");
8758        assert_eq!(analyzed.candidates, 1);
8759        assert!(!index.content_has_pending(profile));
8760    }
8761
8762    #[test]
8763    fn only_population_reconciles_rule_changes_without_losing_traversal() {
8764        let root = tempfile::tempdir().expect("root");
8765        write_file(&root.path().join("nested/.gitignore"), b"*.log\n");
8766        write_file(&root.path().join("nested/first.log"), b"first");
8767        write_file(&root.path().join("nested/second.txt"), b"second");
8768        let config =
8769            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
8770        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold");
8771        assert!(cold.is_complete());
8772        assert!(index.lookup(Path::new("nested")).is_some());
8773        assert!(index.lookup(Path::new("nested/first.log")).is_some());
8774        assert!(index.lookup(Path::new("nested/second.txt")).is_none());
8775
8776        write_file(&root.path().join("nested/.gitignore"), b"*.txt\n");
8777        let changed = reconcile(&mut index, &config, &mut |_| {}).expect("rule change");
8778        assert!(changed.is_complete(), "{:?}", changed.scan.errors);
8779        assert!(index.lookup(Path::new("nested/first.log")).is_none());
8780        assert!(index.lookup(Path::new("nested/second.txt")).is_some());
8781    }
8782
8783    #[cfg(unix)]
8784    #[test]
8785    fn excluded_special_objects_never_enter_cold_or_reconciled_facts() {
8786        use std::os::unix::net::UnixListener;
8787
8788        let dir = tempfile::tempdir().expect("tempdir");
8789        let socket_path = dir.path().join("service.sock");
8790        let _listener = UnixListener::bind(&socket_path).expect("bind socket");
8791        write_file(&dir.path().join("replacement"), b"ordinary");
8792        let (kept, kept_report) =
8793            scan_into_index(dir.path(), &ScanConfig::default()).expect("default scan");
8794        assert!(kept_report.is_complete());
8795        assert_eq!(kept.kind(Path::new("service.sock")), Some(EntryKind::Other));
8796
8797        let serial_config =
8798            ScanConfig { exclude_special: true, threads: Some(1), ..ScanConfig::default() };
8799        let parallel_config =
8800            ScanConfig { exclude_special: true, threads: Some(4), ..ScanConfig::default() };
8801        let (mut serial, serial_report) =
8802            scan_into_index(dir.path(), &serial_config).expect("serial scan");
8803        let (mut parallel, parallel_report) =
8804            scan_into_index(dir.path(), &parallel_config).expect("parallel scan");
8805
8806        for (index, report) in [(&serial, &serial_report), (&parallel, &parallel_report)] {
8807            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8808            assert!(index.lookup(Path::new("service.sock")).is_none());
8809            assert!(index.lookup(Path::new("replacement")).is_some());
8810        }
8811
8812        fs::remove_file(dir.path().join("replacement")).expect("remove file");
8813        let _replacement =
8814            UnixListener::bind(dir.path().join("replacement")).expect("bind replacement socket");
8815        let serial_reconciled =
8816            reconcile(&mut serial, &serial_config, &mut |_| {}).expect("serial reconcile");
8817        let parallel_reconciled =
8818            reconcile(&mut parallel, &parallel_config, &mut |_| {}).expect("parallel reconcile");
8819
8820        assert!(serial_reconciled.is_complete());
8821        assert!(parallel_reconciled.is_complete());
8822        assert!(serial.lookup(Path::new("replacement")).is_none());
8823        assert!(parallel.lookup(Path::new("replacement")).is_none());
8824        assert_eq!(index_fingerprint(&serial), index_fingerprint(&parallel));
8825    }
8826
8827    #[test]
8828    fn control_sources_respect_a_single_operation_batch_bound() {
8829        let dir = tempfile::tempdir().expect("tempdir");
8830        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8831        write_file(&dir.path().join("debug.log"), b"ignored");
8832
8833        for threads in [1, 4] {
8834            let config = ScanConfig {
8835                read_controls: true,
8836                threads: Some(threads),
8837                batch_size: 1,
8838                ..ScanConfig::default()
8839            };
8840            let mut largest = 0;
8841            let report = scan(dir.path(), &config, &mut |observation| {
8842                largest = largest.max(observation.len());
8843            })
8844            .expect("scan");
8845
8846            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8847            assert_eq!(largest, 1);
8848        }
8849    }
8850
8851    #[test]
8852    fn cold_scan_matches_the_metabrowser_nested_control_fixture() {
8853        let dir = tempfile::tempdir().expect("tempdir");
8854        write_file(&dir.path().join(".gitignore"), b"node_modules/\n*.pyc\n");
8855        write_file(&dir.path().join("src/app.py"), b"x");
8856        write_file(&dir.path().join("src/thing.pyc"), b"x");
8857        write_file(&dir.path().join("src/generated/.gitignore"), b"*.gen\n");
8858        write_file(&dir.path().join("src/generated/out.gen"), b"x");
8859        write_file(&dir.path().join("node_modules/.gitignore"), b"!keep-me.py\n");
8860        write_file(&dir.path().join("node_modules/keep-me.py"), b"x");
8861
8862        let (index, report) = scan_into_index(
8863            dir.path(),
8864            &ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() },
8865        )
8866        .expect("scan fixture");
8867
8868        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8869        assert_eq!(
8870            index.is_ignored(Path::new("src/app.py")).expect("control state observed"),
8871            Some(false)
8872        );
8873        assert_eq!(
8874            index.is_ignored(Path::new("src/thing.pyc")).expect("control state observed"),
8875            Some(true)
8876        );
8877        assert_eq!(
8878            index.is_ignored(Path::new("src/generated")).expect("control state observed"),
8879            Some(false)
8880        );
8881        assert_eq!(
8882            index.is_ignored(Path::new("src/generated/out.gen")).expect("control state observed"),
8883            Some(true)
8884        );
8885        assert_eq!(
8886            index.is_ignored(Path::new("node_modules")).expect("control state observed"),
8887            Some(true)
8888        );
8889        assert_eq!(
8890            index.is_ignored(Path::new("node_modules/keep-me.py")).expect("control state observed"),
8891            Some(true)
8892        );
8893    }
8894
8895    #[test]
8896    fn reconciliation_observes_same_metadata_control_edits_and_last_deletion() {
8897        let dir = tempfile::tempdir().expect("tempdir");
8898        write_file(&dir.path().join(".gitignore"), b"*.log\n");
8899        write_file(&dir.path().join("debug.log"), b"ignored");
8900        let config = ScanConfig { read_controls: true, threads: Some(1), ..ScanConfig::default() };
8901        let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
8902        assert!(report.is_complete());
8903        assert_eq!(
8904            index.is_ignored(Path::new("debug.log")).expect("control state observed"),
8905            Some(true)
8906        );
8907
8908        // Same-length content proves control identity is not inferred from stat-tier
8909        // metadata, which can remain unchanged on coarse filesystems.
8910        write_file(&dir.path().join(".gitignore"), b"*.tmp\n");
8911        let edited = reconcile(&mut index, &config, &mut |_| {}).expect("edit reconcile");
8912        assert!(edited.is_complete());
8913        assert_eq!(edited.apply.controls, 1);
8914        assert_eq!(edited.apply.reclassified, 1);
8915        assert!(
8916            index
8917                .controls()
8918                .expect("control state observed")
8919                .source_is(Path::new(".gitignore"), b"*.tmp\n")
8920        );
8921        assert_eq!(
8922            index.is_ignored(Path::new("debug.log")).expect("control state observed"),
8923            Some(false)
8924        );
8925
8926        fs::remove_file(dir.path().join(".gitignore")).expect("remove control");
8927        let removed = reconcile(&mut index, &config, &mut |_| {}).expect("remove reconcile");
8928        assert!(removed.is_complete());
8929        assert_eq!(removed.apply.controls, 1);
8930        assert!(index.controls().expect("control state observed").is_empty());
8931        let partitions = index.partition_total().expect("control state observed");
8932        assert_eq!(partitions.all, partitions.unignored);
8933    }
8934
8935    #[cfg(unix)]
8936    #[test]
8937    fn directory_entry_metadata_does_not_follow_symlinks() {
8938        use std::os::unix::fs::symlink;
8939
8940        let root = tempfile::tempdir().expect("root");
8941        let outside = tempfile::tempdir().expect("outside");
8942        write_file(&outside.path().join("must-not-be-scanned.txt"), b"outside");
8943        symlink(outside.path(), root.path().join("link")).expect("symlink");
8944
8945        let (index, report) = scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
8946
8947        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
8948        assert_eq!(index.kind(Path::new("link")), Some(EntryKind::Symlink));
8949        assert!(index.lookup(Path::new("link/must-not-be-scanned.txt")).is_none());
8950        assert_eq!(index.total().files, 0);
8951    }
8952
8953    #[test]
8954    fn cold_scan_establishes_a_baseline_without_change_history() {
8955        let dir = sample_tree();
8956        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
8957
8958        assert_eq!(index.clock(), crate::Clock::ZERO);
8959        assert!(index.since(crate::Clock::ZERO).commits.is_empty());
8960    }
8961
8962    #[test]
8963    fn max_depth_stops_descent() {
8964        let dir = sample_tree();
8965        let config = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
8966        let (index, _) = scan_into_index(dir.path(), &config).expect("scan");
8967
8968        assert!(index.lookup(Path::new("src")).is_some());
8969        assert!(index.lookup(Path::new("src/main.rs")).is_none());
8970    }
8971
8972    #[test]
8973    fn zero_max_depth_keeps_only_the_index_root() {
8974        let dir = sample_tree();
8975        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
8976        let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
8977
8978        assert!(index.is_empty());
8979        assert_eq!(report.entries, 0);
8980        assert_eq!(report.dirs_read, 0);
8981    }
8982
8983    #[test]
8984    fn direct_scan_records_the_canonical_root() {
8985        let dir = sample_tree();
8986        let aliased = dir.path().join(".");
8987        let (index, _) = scan_into_index(&aliased, &ScanConfig::default()).expect("scan");
8988
8989        assert_eq!(index.root_path(), dir.path().canonicalize().expect("canonical root"));
8990    }
8991
8992    #[test]
8993    fn unsupported_symlink_following_is_rejected_on_cold_and_warm_paths() {
8994        let dir = sample_tree();
8995        let unsupported = ScanConfig { follow_symlinks: true, ..ScanConfig::default() };
8996        assert!(matches!(
8997            scan_into_index(dir.path(), &unsupported),
8998            Err(Error::UnsupportedScanConfig(_))
8999        ));
9000
9001        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9002        assert!(matches!(
9003            revalidate(&index, &unsupported, &mut |_| {}),
9004            Err(Error::UnsupportedScanConfig(_))
9005        ));
9006    }
9007
9008    #[test]
9009    fn revalidation_uses_the_same_depth_boundary_as_cold_scan() {
9010        let dir = sample_tree();
9011        let config = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
9012        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
9013        write_file(&dir.path().join("src/added-after-scan.txt"), b"new");
9014
9015        let mut observations = Vec::new();
9016        revalidate(&index, &config, &mut |observation| observations.push(observation))
9017            .expect("revalidate");
9018        for observation in &observations {
9019            index.apply_ok(observation);
9020        }
9021
9022        assert!(index.lookup(Path::new("src/added-after-scan.txt")).is_none());
9023    }
9024
9025    #[test]
9026    fn zero_depth_revalidation_prunes_cached_root_children() {
9027        let dir = tempfile::tempdir().expect("tempdir");
9028        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
9029        let mut index = Index::new_with_scope(dir.path(), config.scope());
9030        index.apply_baseline_ok(&Observation::new(vec![Op::Upsert {
9031            path: PathBuf::from("stale.txt"),
9032            kind: EntryKind::File,
9033            attrs: Attrs::default(),
9034        }]));
9035
9036        let mut observations = Vec::new();
9037        let report = revalidate(&index, &config, &mut |observation| {
9038            observations.push(observation);
9039        })
9040        .expect("revalidate");
9041        for observation in &observations {
9042            index.apply_ok(observation);
9043        }
9044
9045        assert!(index.is_empty());
9046        assert_eq!(report.dirs_read, 0);
9047    }
9048
9049    #[test]
9050    fn zero_depth_applying_reconciliation_prunes_cached_root_children() {
9051        let dir = tempfile::tempdir().expect("tempdir");
9052        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
9053        let mut index = Index::new_with_scope(dir.path(), config.scope());
9054        index.apply_baseline_ok(&Observation::new(vec![Op::Upsert {
9055            path: PathBuf::from("stale.txt"),
9056            kind: EntryKind::File,
9057            attrs: Attrs::default(),
9058        }]));
9059
9060        let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
9061
9062        assert!(index.is_empty());
9063        assert_eq!(report.scan.dirs_read, 0);
9064    }
9065
9066    #[test]
9067    fn filesystem_boundary_is_part_of_the_shared_descent_policy() {
9068        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
9069        let attrs = Attrs { dev: 22, ..Attrs::default() };
9070        assert!(!should_descend(EntryKind::Dir, attrs, 0, 11, &config));
9071        assert!(should_descend(EntryKind::Dir, Attrs { dev: 11, ..attrs }, 0, 11, &config,));
9072    }
9073
9074    /// A cold scan's index records its own pass start, the stamp a snapshot of it writes:
9075    /// never earlier than an instant taken before the scan, so it is not a stale or zero
9076    /// stamp, and never later than one taken after it. The builder constructs the index,
9077    /// and so takes the stamp, before the walk begins.
9078    #[test]
9079    fn a_cold_scan_stamps_its_own_pass_start() {
9080        let nanos = || {
9081            i64::try_from(
9082                std::time::SystemTime::now()
9083                    .duration_since(std::time::UNIX_EPOCH)
9084                    .expect("after the epoch")
9085                    .as_nanos(),
9086            )
9087            .expect("nanoseconds")
9088        };
9089        let dir = sample_tree();
9090        let before = nanos();
9091        let (index, report) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9092        let after = nanos();
9093        assert!(report.is_complete() && report.entries > 0, "{report:?}");
9094        let stamp = index.writing_pass_started_at_ns();
9095        assert!(before <= stamp && stamp <= after, "{before} <= {stamp} <= {after}");
9096    }
9097
9098    #[test]
9099    fn scanning_a_file_is_an_error_not_a_panic() {
9100        let dir = sample_tree();
9101        let err = scan_into_index(&dir.path().join("a.txt"), &ScanConfig::default());
9102        assert!(err.is_err());
9103    }
9104
9105    #[test]
9106    fn deltas_arrive_in_batches_of_the_configured_size() {
9107        let dir = tempfile::tempdir().expect("tempdir");
9108        for i in 0..25 {
9109            write_file(&dir.path().join(format!("f{i}.txt")), b"x");
9110        }
9111        let config = ScanConfig { batch_size: 10, ..ScanConfig::default() };
9112        let mut sizes = Vec::new();
9113        scan(dir.path(), &config, &mut |d| sizes.push(d.len())).expect("scan");
9114
9115        assert!(sizes.len() >= 3, "expected several batches, got {sizes:?}");
9116        assert!(sizes.iter().all(|&n| n <= 10));
9117        assert_eq!(sizes.iter().sum::<usize>(), 25);
9118    }
9119
9120    #[test]
9121    fn invalid_batch_sizes_are_rejected_before_allocation() {
9122        let zero = ScanConfig { batch_size: 0, ..ScanConfig::default() };
9123        let unbounded = ScanConfig { batch_size: usize::MAX, ..ScanConfig::default() };
9124
9125        assert!(matches!(zero.validate(), Err(Error::UnsupportedScanConfig(_))));
9126        assert!(matches!(unbounded.validate(), Err(Error::UnsupportedScanConfig(_))));
9127    }
9128
9129    #[test]
9130    fn reconciliation_scope_budget_publishes_before_returning_retry() {
9131        let directory = tempfile::tempdir().expect("root");
9132        let config = ScanConfig::default();
9133        let (index, _) = scan_into_index(directory.path(), &config).expect("scan root-only tree");
9134        let handle = IndexHandle::new(index);
9135        let mut started = false;
9136        let mut published_partial = false;
9137        let report = reconcile_handle(&handle, &config, &mut |commit| {
9138            if !started {
9139                started = true;
9140                // No entries are added: distinct absent children must not grow history
9141                // for the paused root pass without bound.
9142                for child in ["missing-a", "missing-b", "missing-c"] {
9143                    let nested = reconcile_subtree_handle(&handle, Path::new(child), &config, &mut |_| {}).expect("newer absent scope");
9144                    assert!(nested.is_complete());
9145                }
9146            }
9147            if commit.state.iter().any(|state| matches!(state,
9148                crate::StateTransition::IndexState { current, .. }
9149                    if current.coverage == crate::Coverage::Partial(crate::CoverageReason::Inaccessible))) {
9150                assert_eq!(handle.read_with(Index::state).expect("coherent state").coverage,
9151                    crate::Coverage::Partial(crate::CoverageReason::Inaccessible));
9152                published_partial = true;
9153            }
9154        }).expect("interrupted pass returns retryable report");
9155        assert!(!report.is_complete());
9156        assert!(report.retry_required);
9157        assert!(published_partial, "the transition precedes the caller's retry result");
9158        let recovered = reconcile_handle(&handle, &config, &mut |_| {}).expect("retry");
9159        assert!(recovered.is_complete());
9160        assert_eq!(
9161            handle.read_with(Index::state).expect("recovered state").coverage,
9162            crate::Coverage::Complete
9163        );
9164    }
9165
9166    #[test]
9167    fn stale_arbitration_keeps_a_reconciliation_incomplete() {
9168        let report = ReconcileReport {
9169            scan: ScanReport::default(),
9170            apply: ApplyStats { stale: 1, ..ApplyStats::default() },
9171            observations: 1,
9172            ..ReconcileReport::default()
9173        };
9174
9175        assert!(!report.is_complete());
9176    }
9177
9178    #[test]
9179    fn portable_system_time_conversion_preserves_pre_epoch_values() {
9180        let before_epoch = std::time::UNIX_EPOCH
9181            // Windows timestamps have 100 ns granularity, so use a duration that every
9182            // supported platform can represent without rounding back to the epoch.
9183            .checked_sub(std::time::Duration::from_secs(1))
9184            .expect("represent pre-epoch fixture");
9185
9186        assert_eq!(system_time_ns(before_epoch), -1_000_000_000);
9187        assert_eq!(system_time_ns(std::time::UNIX_EPOCH), 0);
9188    }
9189
9190    #[cfg(not(unix))]
9191    #[test]
9192    fn one_filesystem_fails_when_device_identity_is_unavailable() {
9193        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
9194
9195        assert!(matches!(config.validate(), Err(Error::UnsupportedScanConfig(_))));
9196    }
9197
9198    #[test]
9199    fn revalidate_is_a_no_op_against_an_unchanged_tree() {
9200        let dir = sample_tree();
9201        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9202        let before = index.total();
9203
9204        let mut deltas = Vec::new();
9205        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
9206        let mut unchanged = 0;
9207        for delta in &deltas {
9208            unchanged += index.apply_ok(delta).unchanged;
9209        }
9210
9211        assert_eq!(unchanged, 5, "3 files + 2 dirs all already known");
9212        assert_eq!(index.total(), before);
9213    }
9214
9215    #[cfg(windows)]
9216    #[test]
9217    fn windows_reconcile_detects_same_size_rewrite_with_preserved_mtime() {
9218        let root = tempfile::tempdir().expect("tempdir");
9219        let path = root.path().join("same.txt");
9220        write_file(&path, b"first");
9221        let modified = fs::metadata(&path).expect("metadata").modified().expect("mtime");
9222        let (mut index, _) =
9223            scan_into_index(root.path(), &ScanConfig::default()).expect("initial scan");
9224        let before = *index.attrs(Path::new("same.txt")).expect("initial attrs");
9225
9226        // NTFS stamps change time from the system clock, which advances in ticks of up to
9227        // 15.625 ms, so a rewrite stamped in the same tick as the first write is
9228        // indistinguishable from it. Wait for the clock to leave that tick rather than for
9229        // a fixed interval; the precise clock `SystemTime` reads runs at most one tick ahead.
9230        let stamped = std::time::UNIX_EPOCH
9231            + std::time::Duration::from_nanos(
9232                u64::try_from(before.ctime_ns).expect("change time after the epoch"),
9233            );
9234        while std::time::SystemTime::now() <= stamped + std::time::Duration::from_millis(20) {
9235            std::thread::sleep(std::time::Duration::from_millis(5));
9236        }
9237        write_file(&path, b"other");
9238        File::options()
9239            .write(true)
9240            .open(&path)
9241            .expect("open rewritten file")
9242            .set_times(std::fs::FileTimes::new().set_modified(modified))
9243            .expect("restore mtime");
9244        reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
9245
9246        let after = *index.attrs(Path::new("same.txt")).expect("rewritten attrs");
9247        assert_eq!((after.size, after.mtime_ns), (before.size, before.mtime_ns));
9248        assert_ne!(after.ctime_ns, before.ctime_ns, "change time detects the rewrite");
9249        assert_ne!(after.fingerprint(), before.fingerprint());
9250    }
9251
9252    #[cfg(windows)]
9253    #[test]
9254    fn windows_reconcile_detects_path_identity_replacement() {
9255        let root = tempfile::tempdir().expect("tempdir");
9256        let path = root.path().join("replace.txt");
9257        let displaced = root.path().join("displaced.txt");
9258        write_file(&path, b"first");
9259        let modified = fs::metadata(&path).expect("metadata").modified().expect("mtime");
9260        let (mut index, _) =
9261            scan_into_index(root.path(), &ScanConfig::default()).expect("initial scan");
9262        let before = *index.attrs(Path::new("replace.txt")).expect("initial attrs");
9263
9264        fs::rename(&path, &displaced).expect("retain old file identity");
9265        write_file(&path, b"other");
9266        File::options()
9267            .write(true)
9268            .open(&path)
9269            .expect("open replacement")
9270            .set_times(std::fs::FileTimes::new().set_modified(modified))
9271            .expect("restore mtime");
9272        reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
9273
9274        let after = *index.attrs(Path::new("replace.txt")).expect("replacement attrs");
9275        assert_eq!((after.size, after.mtime_ns), (before.size, before.mtime_ns));
9276        assert_ne!(
9277            (after.dev, after.inode),
9278            (before.dev, before.inode),
9279            "volume serial and file index identify the replacement"
9280        );
9281        assert_ne!(after.fingerprint(), before.fingerprint());
9282    }
9283
9284    #[test]
9285    fn direct_reconciliation_counts_unchanged_entries_and_publishes_state_commits() {
9286        let dir = sample_tree();
9287        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9288        let before_total = index.total();
9289        let before_clock = index.clock();
9290        let mut commits = Vec::new();
9291
9292        let report = reconcile(&mut index, &ScanConfig::default(), &mut |commit| {
9293            commits.push(commit.clone());
9294        })
9295        .expect("reconcile");
9296
9297        assert!(report.is_complete());
9298        assert_eq!(report.apply.unchanged, 5, "3 files + 2 dirs all already known");
9299        assert_eq!(commits.len(), 2);
9300        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
9301        assert_eq!(
9302            index.clock(),
9303            crate::Clock(before_clock.0 + 2),
9304            "start and finish are state commits"
9305        );
9306        let commits = index.since(before_clock).commits;
9307        assert_eq!(commits.len(), 2);
9308        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
9309        assert_eq!(index.total(), before_total);
9310    }
9311
9312    #[test]
9313    fn parallel_and_serial_reconciliation_produce_the_same_index() {
9314        let dir = sample_tree();
9315        let portable_config =
9316            ScanConfig { threads: Some(1), batch_size: 2, ..ScanConfig::default() };
9317        let (mut portable, _) =
9318            scan_into_index(dir.path(), &portable_config).expect("portable baseline");
9319        let mut bulk = portable.clone();
9320
9321        fs::remove_file(dir.path().join("a.txt")).expect("remove file");
9322        write_file(&dir.path().join("src/main.rs"), b"fn main() { much longer }");
9323        write_file(&dir.path().join("src/added.md"), b"new file");
9324
9325        let portable_report = reconcile(&mut portable, &portable_config, &mut |_| {})
9326            .expect("portable reconciliation");
9327        let bulk_config = ScanConfig { threads: Some(2), ..portable_config };
9328        let bulk_report =
9329            reconcile(&mut bulk, &bulk_config, &mut |_| {}).expect("bulk reconciliation");
9330
9331        assert!(portable_report.is_complete());
9332        assert!(bulk_report.is_complete());
9333        assert_eq!(bulk_report.scan.attribution, WalkAttribution::default());
9334        assert_eq!(bulk_report.scan.entries, portable_report.scan.entries);
9335        assert_eq!(bulk_report.scan.dirs_read, portable_report.scan.dirs_read);
9336        assert_eq!(bulk_report.apply, portable_report.apply);
9337        assert_eq!(index_fingerprint(&bulk), index_fingerprint(&portable));
9338        assert_eq!(bulk.total(), portable.total());
9339    }
9340
9341    #[test]
9342    fn parallel_reconciliation_workers_publish_directory_counters() {
9343        let _serial = crate::counters::test_serial();
9344        let dir = sample_tree();
9345        let baseline = ScanConfig { threads: Some(1), ..ScanConfig::default() };
9346        let (mut index, scan) = scan_into_index(dir.path(), &baseline).expect("baseline scan");
9347        assert!(scan.is_complete());
9348
9349        crate::counters::enable(true);
9350        crate::counters::reset();
9351        let config = ScanConfig { threads: Some(4), ..baseline };
9352        let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconciliation");
9353        crate::counters::flush_thread();
9354        let counts = crate::counters::snapshot();
9355        crate::counters::reset();
9356        crate::counters::enable(false);
9357
9358        assert!(report.is_complete());
9359        assert!(
9360            counts.dir_opens >= report.scan.dirs_read,
9361            "parallel worker directory opens were folded: {counts:?}, report={report:?}"
9362        );
9363    }
9364
9365    #[test]
9366    fn parallel_reconciliation_matches_serial_across_structural_transitions() {
9367        for max_depth in [None, Some(1), Some(2)] {
9368            for order in [ScanOrder::BreadthFirst, ScanOrder::DepthFirst] {
9369                let dir = reconciliation_transition_tree();
9370                let reference_config = ScanConfig {
9371                    order,
9372                    max_depth,
9373                    threads: Some(1),
9374                    batch_size: 2,
9375                    ..ScanConfig::default()
9376                };
9377                let (baseline, baseline_report) =
9378                    scan_into_index(dir.path(), &reference_config).expect("baseline scan");
9379                assert!(baseline_report.is_complete());
9380                mutate_reconciliation_transition_tree(dir.path());
9381
9382                let mut serial = baseline.clone();
9383                let mut serial_commits = Vec::new();
9384                let serial_report = reconcile(&mut serial, &reference_config, &mut |commit| {
9385                    serial_commits.push(commit.clone());
9386                })
9387                .expect("serial reconciliation");
9388                let (fresh, fresh_report) =
9389                    scan_into_index(dir.path(), &reference_config).expect("fresh oracle");
9390                assert!(serial_report.is_complete(), "serial {order:?}/{max_depth:?}");
9391                assert!(fresh_report.is_complete(), "fresh {order:?}/{max_depth:?}");
9392                assert_eq!(
9393                    index_fingerprint(&serial),
9394                    index_fingerprint(&fresh),
9395                    "serial did not converge to a fresh scan for {order:?}/{max_depth:?}"
9396                );
9397
9398                for workers in [2, 4] {
9399                    let mut parallel = baseline.clone();
9400                    let config = ScanConfig { threads: Some(workers), ..reference_config.clone() };
9401                    let mut parallel_commits = Vec::new();
9402                    let report = reconcile(&mut parallel, &config, &mut |commit| {
9403                        parallel_commits.push(commit.clone());
9404                    })
9405                    .expect("parallel reconciliation");
9406                    let context = format!("{order:?}/{max_depth:?}/{workers} workers");
9407
9408                    assert!(report.is_complete(), "{context}: unexpected partial report");
9409                    assert_eq!(report.scan.entries, serial_report.scan.entries, "{context}");
9410                    assert_eq!(report.scan.dirs_read, serial_report.scan.dirs_read, "{context}");
9411                    assert_eq!(report.apply, serial_report.apply, "{context}");
9412                    assert_eq!(
9413                        effective_ops(&parallel_commits),
9414                        effective_ops(&serial_commits),
9415                        "{context}: effective delta differs"
9416                    );
9417                    assert_eq!(
9418                        index_fingerprint(&parallel),
9419                        index_fingerprint(&serial),
9420                        "{context}: final index differs"
9421                    );
9422                    let (parallel_total, serial_total) = (parallel.total(), serial.total());
9423                    assert_eq!(
9424                        (
9425                            parallel_total.files,
9426                            parallel_total.dirs,
9427                            parallel_total.bytes,
9428                            parallel_total.allocated,
9429                            parallel_total.newest_mtime_ns,
9430                        ),
9431                        (
9432                            serial_total.files,
9433                            serial_total.dirs,
9434                            serial_total.bytes,
9435                            serial_total.allocated,
9436                            serial_total.newest_mtime_ns,
9437                        ),
9438                        "{context}: roll-up differs"
9439                    );
9440                    assert_eq!(
9441                        parallel_total.by_ext, serial_total.by_ext,
9442                        "{context}: extension roll-up differs"
9443                    );
9444                }
9445            }
9446        }
9447    }
9448
9449    #[cfg(unix)]
9450    #[test]
9451    fn revalidation_metadata_errors_do_not_delete_enumerated_entries() {
9452        use std::os::unix::fs::PermissionsExt;
9453
9454        if !crate::test_support::require_permission_bits() {
9455            return;
9456        }
9457
9458        let dir = sample_tree();
9459        let config = ScanConfig::default();
9460        let (mut index, baseline_report) =
9461            scan_into_index(dir.path(), &config).expect("baseline scan");
9462        assert!(baseline_report.is_complete());
9463        let before = index_fingerprint(&index);
9464        let original_permissions = fs::metadata(dir.path()).expect("root metadata").permissions();
9465
9466        fs::set_permissions(dir.path(), fs::Permissions::from_mode(0o400))
9467            .expect("remove search permission");
9468        let mut observations = Vec::new();
9469        let outcome = revalidate(&index, &config, &mut |observation| {
9470            observations.push(observation);
9471        });
9472        fs::set_permissions(dir.path(), original_permissions).expect("restore permissions");
9473
9474        let report = outcome.expect("operational metadata errors are a partial report");
9475        assert!(!report.errors.is_empty(), "the fixture did not induce metadata errors");
9476        for observation in &observations {
9477            index.apply_ok(observation);
9478        }
9479        assert_eq!(index_fingerprint(&index), before);
9480        assert!(index.attrs(Path::new("a.txt")).is_some(), "existing entry was removed");
9481    }
9482
9483    #[cfg(unix)]
9484    #[test]
9485    fn reconciliation_metadata_errors_drop_unverified_entries_like_a_cold_scan() {
9486        use std::os::unix::fs::PermissionsExt;
9487
9488        if !crate::test_support::require_permission_bits() {
9489            return;
9490        }
9491
9492        // macOS's parallel path may satisfy the whole directory through
9493        // getattrlistbulk even without search permission. One worker pins the portable
9494        // fallback there; Linux also exercises the parallel portable worker.
9495        let worker_counts = if cfg!(target_os = "macos") { vec![1] } else { vec![1, 2] };
9496        for workers in worker_counts {
9497            let dir = sample_tree();
9498            let config =
9499                ScanConfig { threads: Some(workers), batch_size: 2, ..ScanConfig::default() };
9500            let (mut index, baseline_report) =
9501                scan_into_index(dir.path(), &config).expect("baseline scan");
9502            assert!(baseline_report.is_complete());
9503            let before = index_fingerprint(&index);
9504            let original_permissions =
9505                fs::metadata(dir.path()).expect("root metadata").permissions();
9506
9507            // Reading names requires read permission; looking up their metadata also
9508            // requires search permission. This makes enumeration succeed and each
9509            // metadata lookup fail, the boundary where an encountered name used to be
9510            // misclassified as a deletion.
9511            fs::set_permissions(dir.path(), fs::Permissions::from_mode(0o400))
9512                .expect("remove search permission");
9513            let outcome = reconcile(&mut index, &config, &mut |_| {});
9514            let cold = scan_into_index(dir.path(), &config);
9515            fs::set_permissions(dir.path(), original_permissions).expect("restore permissions");
9516
9517            let report = outcome.expect("operational metadata errors are a partial report");
9518            let (cold, cold_report) = cold.expect("cold partial scan");
9519            assert!(!report.scan.errors.is_empty(), "the fixture did not induce metadata errors");
9520            assert!(!report.is_complete());
9521            assert!(!cold_report.is_complete());
9522            assert_eq!(index_fingerprint(&index), index_fingerprint(&cold));
9523            assert!(
9524                index_fingerprint(&index).is_empty(),
9525                "neither warm nor cold may retain attributes it could not verify"
9526            );
9527            assert_eq!(index.directory_complete(Path::new("")), Some(false));
9528            assert_eq!(cold.directory_complete(Path::new("")), Some(false));
9529            assert!(!before.is_empty(), "the fixture began with retained facts");
9530        }
9531    }
9532
9533    #[cfg(unix)]
9534    #[test]
9535    fn failed_listing_withdraws_retained_completeness_and_recovers() {
9536        use std::os::unix::fs::PermissionsExt;
9537
9538        if !crate::test_support::require_permission_bits() {
9539            return;
9540        }
9541        for workers in [1, 2] {
9542            let dir = tempfile::tempdir().expect("root");
9543            write_file(&dir.path().join("ancestor/blocked/unknown.txt"), b"unknown");
9544            write_file(&dir.path().join("healthy/known.txt"), b"known");
9545            let config = ScanConfig { threads: Some(workers), ..ScanConfig::default() };
9546            let (mut warm, baseline) = scan_into_index(dir.path(), &config).expect("baseline");
9547            assert!(baseline.is_complete());
9548            let blocked = dir.path().join("ancestor/blocked");
9549            fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny listing");
9550            let probe = fs::read_dir(&blocked);
9551            let mut commits = Vec::new();
9552            let refreshed =
9553                reconcile(&mut warm, &config, &mut |commit| commits.push(commit.clone()));
9554            let cold = scan_into_index(dir.path(), &config);
9555            fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore");
9556            assert_eq!(
9557                probe.expect_err("real denied listing").kind(),
9558                std::io::ErrorKind::PermissionDenied
9559            );
9560            assert!(!refreshed.expect("partial refresh").is_complete());
9561            let (cold, report) = cold.expect("partial cold scan");
9562            assert!(!report.is_complete());
9563            for index in [&warm, &cold] {
9564                for path in ["", "ancestor", "healthy"] {
9565                    assert_eq!(
9566                        index.directory_complete(Path::new(path)),
9567                        Some(true),
9568                        "workers={workers}: {path}"
9569                    );
9570                }
9571                assert_eq!(index.directory_complete(Path::new("ancestor/blocked")), Some(false));
9572                assert_eq!(index.freshness_at(Path::new("ancestor")), crate::Freshness::Partial);
9573            }
9574            assert!(commits.iter().flat_map(|commit| &commit.state).any(|state| matches!(state,
9575                crate::StateTransition::IndexState { current, .. } if current.coverage != crate::Coverage::Complete
9576            )), "failure is published");
9577            assert!(reconcile(&mut warm, &config, &mut |_| {}).expect("recovery").is_complete());
9578            assert_eq!(warm.directory_complete(Path::new("ancestor/blocked")), Some(true));
9579            assert_eq!(warm.state().coverage, crate::Coverage::Complete);
9580        }
9581    }
9582
9583    #[test]
9584    fn unreadable_directory_warm_answer_matches_cold_verified_tree() {
9585        for workers in [1, 2] {
9586            let dir = tempfile::tempdir().expect("tempdir");
9587            write_file(&dir.path().join("blocked/old.txt"), b"old");
9588            write_file(&dir.path().join("verified.txt"), b"verified");
9589            let config = ScanConfig { threads: Some(workers), ..ScanConfig::default() };
9590            let (mut warm, baseline) = scan_into_index(dir.path(), &config).expect("baseline");
9591            assert!(baseline.is_complete());
9592
9593            let blocked = dir.path().join("blocked").canonicalize().expect("blocked path");
9594            let hook = install_child_metadata_hook(dir.path(), move |path| {
9595                (path == blocked)
9596                    .then(|| std::io::Error::from(std::io::ErrorKind::PermissionDenied))
9597            });
9598            let warm_report = reconcile(&mut warm, &config, &mut |_| {}).expect("warm partial");
9599            let (cold, cold_report) = scan_into_index(dir.path(), &config).expect("cold partial");
9600            drop(hook);
9601
9602            assert!(!warm_report.is_complete(), "workers={workers}");
9603            assert!(!cold_report.is_complete(), "workers={workers}");
9604            assert!(warm.lookup(Path::new("blocked")).is_none(), "workers={workers}");
9605            assert!(warm.lookup(Path::new("blocked/old.txt")).is_none(), "workers={workers}");
9606            assert_eq!(index_fingerprint(&warm), index_fingerprint(&cold), "workers={workers}");
9607            assert!(warm.lookup(Path::new("verified.txt")).is_some(), "workers={workers}");
9608        }
9609    }
9610
9611    #[test]
9612    fn deferred_change_overflow_retries_without_applying_a_partial_wave() {
9613        let dir = sample_tree();
9614        let config = ScanConfig { threads: Some(2), batch_size: 2, ..ScanConfig::default() };
9615        let (mut index, _) = scan_into_index(dir.path(), &config).expect("baseline");
9616        let before = index_fingerprint(&index);
9617
9618        fs::remove_file(dir.path().join("a.txt")).expect("remove file");
9619        write_file(&dir.path().join("added.md"), b"new file");
9620        write_file(&dir.path().join("src/main.rs"), b"fn main() { much longer }");
9621
9622        let root = index.root_path().to_path_buf();
9623        let root_meta = {
9624            crate::counters::bump(|c| c.stats += 1);
9625            fs::symlink_metadata(&root)
9626        }
9627        .expect("root metadata");
9628        let mut commits = Vec::new();
9629        let outcome = reconcile_direct_parallel(
9630            &mut index,
9631            &root,
9632            root_device(&root, &root_meta).expect("root device"),
9633            &config,
9634            1,
9635            &mut |commit| commits.push(commit.clone()),
9636        )
9637        .expect("parallel attempt");
9638        let DirectParallelOutcome::RetrySerial { prefix, remaining } = outcome else {
9639            panic!("the deliberately tiny deferred budget must trigger the retry");
9640        };
9641
9642        assert_eq!(prefix.apply, ApplyStats::default());
9643        assert_eq!(remaining, VecDeque::from([(PathBuf::new(), 0)]));
9644        assert!(commits.is_empty());
9645        assert_eq!(index_fingerprint(&index), before);
9646
9647        let serial = ScanConfig { threads: Some(1), ..config };
9648        let report = reconcile(&mut index, &serial, &mut |_| {}).expect("serial retry");
9649        let (expected, expected_report) = scan_into_index(dir.path(), &serial).expect("oracle");
9650        assert!(report.is_complete());
9651        assert!(expected_report.is_complete());
9652        assert_eq!(index_fingerprint(&index), index_fingerprint(&expected));
9653        assert_eq!(index.total().files, expected.total().files);
9654        assert_eq!(index.total().dirs, expected.total().dirs);
9655        assert_eq!(index.total().bytes, expected.total().bytes);
9656        assert_eq!(index.total().allocated, expected.total().allocated);
9657        assert_eq!(index.total().newest_mtime_ns, expected.total().newest_mtime_ns);
9658        assert_eq!(index.total().by_ext, expected.total().by_ext);
9659    }
9660
9661    #[test]
9662    fn late_overflow_resumes_without_double_counting_completed_waves() {
9663        let dir = tempfile::tempdir().expect("tempdir");
9664        // The root wave discovers more than one full wave of directories. A change in
9665        // the second wave then forces the serial fallback only after the first wave's
9666        // unchanged entries have already been counted.
9667        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
9668            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), b"unchanged");
9669        }
9670        let parallel = ScanConfig { threads: Some(2), ..ScanConfig::default() };
9671        let (baseline, _) = scan_into_index(dir.path(), &parallel).expect("baseline");
9672        let mut candidate = baseline.clone();
9673        let mut serial_oracle = baseline;
9674
9675        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
9676            write_file(
9677                &dir.path().join(format!("d{directory:04}/file.txt")),
9678                b"changed after the first wave",
9679            );
9680        }
9681
9682        let candidate_report = reconcile_target_inner(
9683            &mut ReconcileTarget::Direct(&mut candidate),
9684            Path::new(""),
9685            0,
9686            &parallel,
9687            0,
9688            &mut |_| {},
9689        )
9690        .expect("late-overflow reconciliation");
9691        let serial = ScanConfig { threads: Some(1), ..parallel };
9692        let oracle_report =
9693            reconcile(&mut serial_oracle, &serial, &mut |_| {}).expect("serial oracle");
9694
9695        assert_eq!(candidate_report.apply, oracle_report.apply);
9696        assert_eq!(candidate_report.scan.entries, oracle_report.scan.entries);
9697        assert_eq!(candidate_report.scan.dirs_read, oracle_report.scan.dirs_read);
9698        assert_eq!(index_fingerprint(&candidate), index_fingerprint(&serial_oracle));
9699    }
9700
9701    /// The four counts a walk reports, in the order [`crate::ProgressSnapshot`] shows them.
9702    fn walked(report: &ScanReport) -> (u64, u64, u64, u64) {
9703        (report.dirs_read, report.files_walked, report.bytes_walked, report.allocated_walked)
9704    }
9705
9706    fn reported(progress: &crate::Progress) -> (u64, u64, u64, u64) {
9707        let snapshot = progress.snapshot();
9708        (snapshot.directories, snapshot.files, snapshot.bytes, snapshot.allocated)
9709    }
9710
9711    /// Each walker is a separate loop with its own reporting sites, so each is checked:
9712    /// the detached cold walk, the streaming walk, the transient summary fold, the
9713    /// reference revalidation, exclusive reconciliation serial and in parallel waves,
9714    /// and shared-handle reconciliation. The tree is wide enough that every parallel
9715    /// walker claims several chunks and the small batch size fills several batches, so a
9716    /// walker that reported only its final state would still fail on the counts a
9717    /// mid-walk chunk added twice or not at all.
9718    #[test]
9719    fn every_walker_reports_exactly_what_its_report_counts() {
9720        let dir = tempfile::tempdir().expect("tempdir");
9721        for directory in 0..12 {
9722            for file in 0..5 {
9723                write_file(
9724                    &dir.path().join(format!("d{directory}/f{file}.txt")),
9725                    &vec![b'x'; directory * 5 + file + 1],
9726                );
9727            }
9728        }
9729        for threads in [1, 4] {
9730            let context = format!("threads={threads}");
9731            let progress = crate::Progress::new();
9732            let cold_config = ScanConfig {
9733                threads: Some(threads),
9734                batch_size: 4,
9735                progress: Some(progress.clone()),
9736                ..ScanConfig::default()
9737            };
9738            let (mut index, cold) = scan_into_index(dir.path(), &cold_config).expect("cold scan");
9739            assert_eq!(cold.dirs_read, 13, "{context}: the root and twelve children");
9740            assert_eq!(
9741                progress.snapshot().phase,
9742                crate::ProgressPhase::Indexing,
9743                "{context}: the detached walk ends by assembling the index"
9744            );
9745            assert_eq!(reported(&progress), walked(&cold), "{context}: detached cold walk");
9746
9747            let progress = crate::Progress::new();
9748            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
9749            let streamed = scan(dir.path(), &config, &mut |_| {}).expect("streaming scan");
9750            assert_eq!(walked(&streamed), walked(&cold), "{context}");
9751            assert_eq!(reported(&progress), walked(&streamed), "{context}: streaming walk");
9752
9753            let progress = crate::Progress::new();
9754            let config = ScanConfig {
9755                read_controls: false,
9756                progress: Some(progress.clone()),
9757                ..cold_config.clone()
9758            };
9759            let folded = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
9760            assert_eq!(walked(&folded), walked(&cold), "{context}");
9761            assert_eq!(reported(&progress), walked(&folded), "{context}: summary fold");
9762
9763            let progress = crate::Progress::new();
9764            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
9765            let revalidated = revalidate(&index, &config, &mut |_| {}).expect("revalidate");
9766            assert_eq!(progress.snapshot().phase, crate::ProgressPhase::Revalidating, "{context}");
9767            assert_eq!(reported(&progress), walked(&revalidated), "{context}: revalidate");
9768
9769            // Changes, so reconciliation defers and applies operations rather than
9770            // discarding every entry as unchanged.
9771            write_file(&dir.path().join(format!("d0/new{threads}.txt")), b"added");
9772            fs::remove_file(dir.path().join(format!("d1/f{}.txt", threads - 1))).expect("remove");
9773            write_file(&dir.path().join("d2/f0.txt"), &vec![b'y'; 40 + threads]);
9774            // An independent walk of the changed tree. The handle and the reconcile's own
9775            // report both come from the walker's counts, so agreeing with each other
9776            // would not show that the walker counted anything; agreeing with this does.
9777            let (_, fresh) = scan_into_index(dir.path(), &ScanConfig::default()).expect("fresh");
9778            let progress = crate::Progress::new();
9779            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
9780            let reconciled = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
9781            assert!(reconciled.apply.mutated(), "{context}: the changes were applied");
9782            assert_eq!(progress.snapshot().phase, crate::ProgressPhase::Revalidating, "{context}");
9783            assert_eq!(reported(&progress), walked(&reconciled.scan), "{context}: reconcile");
9784            assert_eq!(walked(&reconciled.scan), walked(&fresh), "{context}: the whole tree");
9785
9786            let handle = crate::IndexHandle::new(index);
9787            let progress = crate::Progress::new();
9788            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config };
9789            let shared = reconcile_handle(&handle, &config, &mut |_| {}).expect("shared");
9790            assert_eq!(reported(&progress), walked(&shared.scan), "{context}: shared handle");
9791        }
9792    }
9793
9794    /// A wave that overflows its deferred-operation budget is thrown away and rewalked
9795    /// serially. The report counts each directory once, as the logical pass does, and
9796    /// progress counts the wave's reads both times, as the filesystem did them: work
9797    /// done, not the answer. This is the one walker relation that is not equality, and
9798    /// the difference is exactly the rewalked wave.
9799    #[test]
9800    fn progress_counts_a_rewalked_wave_twice_where_the_report_counts_it_once() {
9801        let dir = tempfile::tempdir().expect("tempdir");
9802        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
9803            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), b"unchanged");
9804        }
9805        // A file in the wave that completes, so the serial rewalk has counts of that wave
9806        // to carry forward as already added rather than add again.
9807        write_file(&dir.path().join("root.txt"), b"counted by the wave that completes");
9808        let parallel = ScanConfig { threads: Some(2), ..ScanConfig::default() };
9809        let (mut index, _) = scan_into_index(dir.path(), &parallel).expect("baseline");
9810        // Larger than any filesystem stores inline in the inode, so each copy occupies
9811        // blocks of its own wherever the test runs.
9812        let changed = &vec![b'c'; 8_193];
9813        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
9814            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), changed);
9815        }
9816
9817        let progress = crate::Progress::new();
9818        let observed = ScanConfig { progress: Some(progress.clone()), ..parallel };
9819        let report = reconcile_target_inner(
9820            &mut ReconcileTarget::Direct(&mut index),
9821            Path::new(""),
9822            0,
9823            &observed,
9824            0,
9825            &mut |_| {},
9826        )
9827        .expect("late-overflow reconciliation");
9828
9829        // The root wave changes nothing and completes; the second wave holds exactly
9830        // one full wave of changed directories, overflows, and is rewalked with the one
9831        // directory the wave left behind.
9832        let rewalked = u64::try_from(RECONCILE_WAVE_DIRECTORIES).expect("fits");
9833        let snapshot = progress.snapshot();
9834        assert_eq!(snapshot.directories, report.scan.dirs_read + rewalked);
9835        assert_eq!(snapshot.files, report.scan.files_walked + rewalked);
9836        assert_eq!(
9837            snapshot.bytes,
9838            report.scan.bytes_walked + rewalked * u64::try_from(changed.len()).expect("fits")
9839        );
9840        let allocated = index.attrs(Path::new("d0000/file.txt")).expect("indexed").allocated;
9841        assert!(allocated > 0, "a file with content occupies blocks");
9842        assert_eq!(snapshot.allocated, report.scan.allocated_walked + rewalked * allocated);
9843    }
9844
9845    /// Several invalidated roots are reconciled one at a time and their reports summed,
9846    /// so every walked count has to survive the sum, allocated bytes included.
9847    #[test]
9848    fn a_multi_root_reconcile_sums_every_walked_count() {
9849        let dir = tempfile::tempdir().expect("tempdir");
9850        for directory in ["a", "b", "c"] {
9851            for file in 0..3 {
9852                write_file(&dir.path().join(format!("{directory}/f{file}.txt")), b"before");
9853            }
9854        }
9855        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9856        for directory in ["a", "b"] {
9857            write_file(&dir.path().join(format!("{directory}/f0.txt")), &vec![b'x'; 5_000]);
9858        }
9859        index.apply_ok(&Observation::new(
9860            ["a", "b"]
9861                .into_iter()
9862                .map(|directory| Op::InvalidateSubtree {
9863                    path: PathBuf::from(directory),
9864                    reason: crate::InvalidateReason::Requested,
9865                })
9866                .collect(),
9867        ));
9868        let report =
9869            reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
9870
9871        let attrs: Vec<Attrs> = ["a", "b"]
9872            .into_iter()
9873            .flat_map(|directory| (0..3).map(move |file| format!("{directory}/f{file}.txt")))
9874            .map(|path| *index.attrs(Path::new(&path)).expect("indexed"))
9875            .collect();
9876        assert_eq!(report.scan.files_walked, 6, "the two roots' files, and not c's");
9877        assert_eq!(report.scan.bytes_walked, attrs.iter().map(|attrs| attrs.size).sum::<u64>());
9878        assert_eq!(
9879            report.scan.allocated_walked,
9880            attrs.iter().map(|attrs| attrs.allocated).sum::<u64>()
9881        );
9882    }
9883
9884    #[test]
9885    fn shared_reconciliation_retains_conditional_no_op_arbitration() {
9886        let dir = sample_tree();
9887        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9888        let handle = crate::IndexHandle::new(index);
9889        let before_clock = handle.clock().expect("clock");
9890        let mut commits = Vec::new();
9891
9892        let report = reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
9893            commits.push(commit.clone());
9894        })
9895        .expect("reconcile");
9896
9897        assert!(report.is_complete());
9898        assert_eq!(report.apply.unchanged, 5, "3 files + 2 dirs all already known");
9899        assert_eq!(commits.len(), 2);
9900        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
9901        assert_eq!(
9902            handle.clock().expect("clock"),
9903            crate::Clock(before_clock.0 + 2),
9904            "start and finish are state commits"
9905        );
9906        let commits = handle.since(before_clock).expect("state commits").commits;
9907        assert_eq!(commits.len(), 2);
9908        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
9909    }
9910
9911    #[test]
9912    fn revalidate_detects_additions_edits_and_deletions() {
9913        let dir = sample_tree();
9914        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9915
9916        fs::remove_file(dir.path().join("a.txt")).expect("remove");
9917        write_file(&dir.path().join("src/main.rs"), b"fn main() { longer }");
9918        write_file(&dir.path().join("added.md"), b"new");
9919
9920        let mut deltas = Vec::new();
9921        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
9922        let mut stats = crate::index::ApplyStats::default();
9923        for delta in &deltas {
9924            let s = index.apply_ok(delta);
9925            stats.inserted += s.inserted;
9926            stats.updated += s.updated;
9927            stats.removed += s.removed;
9928        }
9929
9930        assert_eq!(stats.inserted, 1, "added.md");
9931        assert_eq!(stats.updated, 1, "main.rs grew");
9932        assert_eq!(stats.removed, 1, "a.txt is gone");
9933
9934        let total = index.total();
9935        assert_eq!(total.files, 3);
9936        assert_eq!(total.bytes, 20 + 9 + 3);
9937        assert!(!total.by_ext.contains_key(".txt"));
9938        assert_eq!(total.by_ext[".md"].files, 1);
9939    }
9940
9941    #[test]
9942    fn revalidate_removes_a_whole_vanished_directory() {
9943        let dir = sample_tree();
9944        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9945        fs::remove_dir_all(dir.path().join("src")).expect("remove dir");
9946
9947        let mut deltas = Vec::new();
9948        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
9949        for delta in &deltas {
9950            index.apply_ok(delta);
9951        }
9952
9953        let total = index.total();
9954        assert_eq!(total.files, 1);
9955        assert_eq!(total.dirs, 0);
9956        assert!(index.lookup(Path::new("src")).is_none());
9957    }
9958
9959    #[test]
9960    fn pending_invalidation_reconciles_the_requested_subtree() {
9961        let dir = sample_tree();
9962        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9963        write_file(&dir.path().join("src/added.rs"), b"new");
9964        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
9965            path: PathBuf::from("src"),
9966            reason: crate::InvalidateReason::Requested,
9967        }]));
9968        assert_eq!(index.freshness_at(Path::new("src")), crate::Freshness::Stale);
9969
9970        let mut applied = Vec::new();
9971        let report = reconcile_pending(&mut index, &ScanConfig::default(), &mut |delta| {
9972            applied.push(delta.clone());
9973        })
9974        .expect("reconcile pending");
9975
9976        assert!(report.is_complete());
9977        assert!(index.lookup(Path::new("src/added.rs")).is_some());
9978        assert_eq!(index.freshness_at(Path::new("src")), crate::Freshness::Fresh);
9979        assert!(index.take_pending_invalidations().is_empty());
9980        assert!(applied.iter().any(|commit| commit_touches(commit, Path::new("src/added.rs"))));
9981    }
9982
9983    /// A retained `.gitignore` reconciled as the root of its own walk re-reads its rules.
9984    /// A file does not descend, so the subtree-root branch was the only place that could
9985    /// read them, and it did not: the table kept `*.log` while the pass reported complete.
9986    #[test]
9987    fn reconciling_a_retained_control_file_as_the_subtree_root_rereads_its_rules() {
9988        let dir = tempfile::tempdir().expect("tempdir");
9989        write_file(&dir.path().join(".gitignore"), b"*.log\n");
9990        write_file(&dir.path().join("a.log"), b"log");
9991        let config = ScanConfig { read_controls: true, ..ScanConfig::default() };
9992        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
9993        assert_eq!(
9994            index.is_ignored(Path::new("a.log")).expect("control state observed"),
9995            Some(true)
9996        );
9997
9998        write_file(&dir.path().join(".gitignore"), b"# nothing is ignored now\n");
9999        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
10000            path: PathBuf::from(".gitignore"),
10001            reason: crate::InvalidateReason::Requested,
10002        }]));
10003        let report = reconcile_pending(&mut index, &config, &mut |_| {}).expect("reconcile");
10004
10005        assert!(report.is_complete(), "{:?}", report.scan.errors);
10006        assert_eq!(index.freshness_at(Path::new(".gitignore")), crate::Freshness::Fresh);
10007        assert_eq!(
10008            index.is_ignored(Path::new("a.log")).expect("control state observed"),
10009            Some(false)
10010        );
10011    }
10012
10013    /// A control file the pass cannot verify contributes neither stale rules nor an entry.
10014    #[cfg(unix)]
10015    #[test]
10016    fn reconciling_an_unreadable_control_file_root_drops_its_rules_and_stays_partial() {
10017        use std::os::unix::fs::PermissionsExt;
10018
10019        if !crate::test_support::require_permission_bits() {
10020            return;
10021        }
10022
10023        let dir = tempfile::tempdir().expect("tempdir");
10024        let control = dir.path().join(".gitignore");
10025        write_file(&control, b"*.log\n");
10026        write_file(&dir.path().join("a.log"), b"log");
10027        let config = ScanConfig { read_controls: true, ..ScanConfig::default() };
10028        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
10029
10030        write_file(&control, b"# rewritten, then made unreadable\n");
10031        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("chmod");
10032        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
10033            path: PathBuf::from(".gitignore"),
10034            reason: crate::InvalidateReason::Requested,
10035        }]));
10036        let report = reconcile_pending(&mut index, &config, &mut |_| {});
10037        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore");
10038        let report = report.expect("reconcile");
10039
10040        assert!(!report.is_complete());
10041        assert_eq!(report.scan.errors.len(), 1, "{:?}", report.scan.errors);
10042        assert_eq!(index.freshness_at(Path::new(".gitignore")), crate::Freshness::Partial);
10043        assert!(
10044            !index.controls().expect("control state observed").contains(Path::new(".gitignore"))
10045        );
10046        assert_eq!(index.is_ignored(Path::new("a.log")).expect("control state observed"), None);
10047    }
10048
10049    #[cfg(unix)]
10050    #[test]
10051    fn unreadable_control_keeps_new_excluded_file_unknown_and_out_of_analysis() {
10052        use std::os::unix::fs::PermissionsExt;
10053
10054        if !crate::test_support::require_permission_bits() {
10055            return;
10056        }
10057        let dir = tempfile::tempdir().expect("tempdir");
10058        let control = dir.path().join(".gitignore");
10059        write_file(&control, b"*.log\n");
10060        write_file(&dir.path().join("keep.rs"), b"code");
10061        let config = ScanConfig {
10062            population: crate::query::IgnoredEntries::Exclude,
10063            ..ScanConfig::default()
10064        };
10065        let (mut index, cold) = scan_into_index(dir.path(), &config).expect("cold");
10066        assert!(cold.is_complete());
10067        write_file(&dir.path().join("debug.log"), b"must not analyze");
10068        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("chmod");
10069        let report = reconcile(&mut index, &config, &mut |_| {});
10070        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore");
10071        let report = report.expect("reconcile");
10072        assert!(!report.is_complete());
10073        assert!(index.lookup(Path::new("debug.log")).is_some(), "unknown is retained");
10074        assert_eq!(index.ignored_classification(Path::new("debug.log")), None);
10075        assert!(!index.ignored_classification_complete_below(Path::new("")));
10076        let candidates = index.analysis_candidates(crate::content::AnalysisSet::NONE.with_lines());
10077        assert!(
10078            candidates.iter().all(|candidate| candidate.relative_path != Path::new("debug.log"))
10079        );
10080        let repaired =
10081            reconcile(&mut index, &config, &mut |_| {}).expect("reconcile repaired control");
10082        assert!(repaired.is_complete(), "{:?}", repaired.scan.errors);
10083        assert!(index.ignored_classification_complete_below(Path::new("")));
10084        assert!(index.lookup(Path::new("debug.log")).is_none(), "known ignored file is pruned");
10085    }
10086
10087    #[test]
10088    fn handle_reconciliation_publishes_after_each_delta_is_applied() {
10089        let dir = sample_tree();
10090        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10091        let handle = crate::IndexHandle::new(index);
10092        let reader = handle.clone();
10093        write_file(&dir.path().join("added.md"), b"new");
10094
10095        let mut observed_after_apply = false;
10096        reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
10097            if commit_touches(commit, Path::new("added.md")) {
10098                observed_after_apply =
10099                    reader.kind(Path::new("added.md")).expect("query index").is_some();
10100            }
10101        })
10102        .expect("reconcile handle");
10103
10104        assert!(observed_after_apply);
10105    }
10106
10107    /// Delete `name` under `root` from inside its own metadata lookup, after the listing
10108    /// returned it, on whichever thread performs the lookup.
10109    fn delete_between_listing_and_stat(root: &Path, name: &'static str) -> WalkHookGuard {
10110        install_child_metadata_hook(root, move |path| {
10111            delete_if_named(path, name);
10112            None
10113        })
10114    }
10115
10116    /// As [`delete_between_listing_and_stat`], and every reconciliation listing under `root`
10117    /// also ends in an error, so none of them is complete.
10118    fn delete_between_listing_and_stat_in_a_failing_listing(
10119        root: &Path,
10120        name: &'static str,
10121    ) -> WalkHookGuard {
10122        install_walk_hook(root, move |point| match point {
10123            WalkHookPoint::ChildMetadata(path) => {
10124                delete_if_named(path, name);
10125                None
10126            }
10127            WalkHookPoint::ListingEnd => Some(std::io::Error::other("injected listing error")),
10128        })
10129    }
10130
10131    fn delete_if_named(path: &Path, name: &str) {
10132        if path.file_name() == Some(OsStr::new(name)) {
10133            fs::remove_file(path).expect("delete between listing and stat");
10134        }
10135    }
10136
10137    /// A name the listing returned that is gone by the time it is stat'd was deleted, on
10138    /// the serial path and on the parallel waves a full-root pass takes by default. The
10139    /// walk removes it; recorded as an error, it would settle as a phantom entry with
10140    /// permanent partial freshness.
10141    #[test]
10142    fn a_child_deleted_between_listing_and_stat_is_removed_rather_than_reported() {
10143        for threads in [1, 4] {
10144            let dir = tempfile::tempdir().expect("tempdir");
10145            write_file(&dir.path().join("keep.txt"), b"keep");
10146            write_file(&dir.path().join("gone.txt"), b"gone");
10147            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
10148            let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
10149            assert!(index.lookup(Path::new("gone.txt")).is_some());
10150
10151            index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
10152                path: PathBuf::new(),
10153                reason: crate::InvalidateReason::Requested,
10154            }]));
10155            let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
10156            let report = reconcile_pending(&mut index, &config, &mut |_| {});
10157            drop(hook);
10158            let report = report.expect("reconcile");
10159
10160            assert!(report.is_complete(), "threads {threads}: {:?}", report.scan.errors);
10161            assert_eq!(report.apply.removed, 1, "threads {threads}");
10162            assert!(index.lookup(Path::new("gone.txt")).is_none(), "threads {threads}");
10163            assert!(index.lookup(Path::new("keep.txt")).is_some(), "threads {threads}");
10164            assert_eq!(index.freshness(), crate::Freshness::Fresh, "threads {threads}");
10165        }
10166    }
10167
10168    /// A vanished control file takes its rules with it on both reconcile paths, even when the
10169    /// rest of its listing fails: the stat's `NotFound` is the evidence. A retained file's
10170    /// rules go with its entry's removal; a hidden-pruned one has no entry to remove, so
10171    /// without its own removal its rules would go on ignoring its siblings.
10172    #[test]
10173    fn a_control_file_deleted_between_listing_and_stat_takes_its_rules_with_it() {
10174        for prune_hidden in [false, true] {
10175            for (threads, listing_fails) in [(1, false), (4, false), (1, true), (4, true)] {
10176                let case = format!(
10177                    "prune hidden {prune_hidden}, threads {threads}, listing fails {listing_fails}"
10178                );
10179                let dir = tempfile::tempdir().expect("tempdir");
10180                write_file(&dir.path().join(".gitignore"), b"*.log\n");
10181                write_file(&dir.path().join("a.log"), b"log");
10182                let config = ScanConfig {
10183                    read_controls: true,
10184                    threads: Some(threads),
10185                    hidden: prune_hidden
10186                        .then(|| std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([""; 0]))),
10187                    ..ScanConfig::default()
10188                };
10189                let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
10190                assert_eq!(
10191                    index.is_ignored(Path::new("a.log")).expect("control state observed"),
10192                    Some(true),
10193                    "{case}"
10194                );
10195
10196                index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
10197                    path: PathBuf::new(),
10198                    reason: crate::InvalidateReason::Requested,
10199                }]));
10200                let hook = if listing_fails {
10201                    delete_between_listing_and_stat_in_a_failing_listing(dir.path(), ".gitignore")
10202                } else {
10203                    delete_between_listing_and_stat(dir.path(), ".gitignore")
10204                };
10205                let report = reconcile_pending(&mut index, &config, &mut |_| {});
10206                drop(hook);
10207                let report = report.expect("reconcile");
10208
10209                assert_eq!(
10210                    report.is_complete(),
10211                    !listing_fails,
10212                    "{case}: {:?}",
10213                    report.scan.errors
10214                );
10215                assert!(index.lookup(Path::new(".gitignore")).is_none(), "{case}");
10216                assert!(
10217                    !index
10218                        .controls()
10219                        .expect("control state observed")
10220                        .contains(Path::new(".gitignore")),
10221                    "{case}"
10222                );
10223                assert_eq!(
10224                    index.is_ignored(Path::new("a.log")).expect("control state observed"),
10225                    Some(false),
10226                    "{case}"
10227                );
10228            }
10229        }
10230    }
10231
10232    /// A cold walk records a name gone by its stat as it records a name the listing never
10233    /// returned: not at all, and without an error that would make the walk partial.
10234    #[test]
10235    fn a_cold_walk_omits_a_child_deleted_between_listing_and_stat() {
10236        for threads in [1, 4] {
10237            let dir = tempfile::tempdir().expect("tempdir");
10238            write_file(&dir.path().join("keep.txt"), b"keep");
10239            write_file(&dir.path().join("gone.txt"), b"gone");
10240            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
10241
10242            let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
10243            let scanned = scan_into_index_via_scanner(dir.path(), &config);
10244            drop(hook);
10245            let (index, report) = scanned.expect("scan");
10246
10247            assert!(report.is_complete(), "threads {threads}: {:?}", report.errors);
10248            assert!(index.lookup(Path::new("gone.txt")).is_none(), "threads {threads}");
10249            assert!(index.lookup(Path::new("keep.txt")).is_some(), "threads {threads}");
10250        }
10251    }
10252
10253    /// Revalidation emits the removal a reconciliation would apply.
10254    #[test]
10255    fn revalidation_removes_a_child_deleted_between_listing_and_stat() {
10256        let dir = tempfile::tempdir().expect("tempdir");
10257        write_file(&dir.path().join("keep.txt"), b"keep");
10258        write_file(&dir.path().join("gone.txt"), b"gone");
10259        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10260
10261        let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
10262        let mut observations = Vec::new();
10263        let report = revalidate(&index, &ScanConfig::default(), &mut |observation| {
10264            observations.push(observation);
10265        });
10266        drop(hook);
10267        let report = report.expect("revalidate");
10268        for observation in &observations {
10269            index.apply_ok(observation);
10270        }
10271
10272        assert!(report.is_complete(), "{:?}", report.errors);
10273        assert!(index.lookup(Path::new("gone.txt")).is_none());
10274        assert!(index.lookup(Path::new("keep.txt")).is_some());
10275    }
10276
10277    /// Revalidation emits the same rules removal when the rest of the listing fails.
10278    #[test]
10279    fn revalidation_removes_the_rules_of_a_control_file_deleted_in_a_failing_listing() {
10280        for prune_hidden in [false, true] {
10281            let dir = tempfile::tempdir().expect("tempdir");
10282            write_file(&dir.path().join(".gitignore"), b"*.log\n");
10283            write_file(&dir.path().join("a.log"), b"log");
10284            let config = ScanConfig {
10285                read_controls: true,
10286                hidden: prune_hidden
10287                    .then(|| std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([""; 0]))),
10288                ..ScanConfig::default()
10289            };
10290            let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
10291            assert_eq!(
10292                index.is_ignored(Path::new("a.log")).expect("control state observed"),
10293                Some(true)
10294            );
10295
10296            let hook =
10297                delete_between_listing_and_stat_in_a_failing_listing(dir.path(), ".gitignore");
10298            let mut observations = Vec::new();
10299            let report = revalidate(&index, &config, &mut |observation| {
10300                observations.push(observation);
10301            });
10302            drop(hook);
10303            let report = report.expect("revalidate");
10304            for observation in &observations {
10305                index.apply_ok(observation);
10306            }
10307
10308            assert!(!report.is_complete(), "prune hidden {prune_hidden}");
10309            assert!(index.lookup(Path::new(".gitignore")).is_none(), "prune hidden {prune_hidden}");
10310            assert!(
10311                !index
10312                    .controls()
10313                    .expect("control state observed")
10314                    .contains(Path::new(".gitignore")),
10315                "prune hidden {prune_hidden}"
10316            );
10317            assert_eq!(
10318                index.is_ignored(Path::new("a.log")).expect("control state observed"),
10319                Some(false),
10320                "prune hidden {prune_hidden}"
10321            );
10322        }
10323    }
10324
10325    #[test]
10326    fn reconciliation_does_not_clear_a_newer_invalidation() {
10327        let dir = sample_tree();
10328        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10329        let handle = crate::IndexHandle::new(index);
10330        write_file(&dir.path().join("added.md"), b"new");
10331
10332        let invalidator = handle.clone();
10333        let mut saw_reconciling = false;
10334        reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
10335            if commit_touches(commit, Path::new("added.md")) {
10336                saw_reconciling =
10337                    invalidator.freshness().expect("query") == crate::Freshness::Reconciling;
10338                invalidator
10339                    .apply(&Observation::new(vec![Op::InvalidateSubtree {
10340                        path: PathBuf::new(),
10341                        reason: crate::InvalidateReason::WatchOverflow,
10342                    }]))
10343                    .expect("new invalidation");
10344            }
10345        })
10346        .expect("reconcile handle");
10347
10348        assert!(saw_reconciling);
10349        assert_eq!(handle.freshness().expect("query"), crate::Freshness::Stale);
10350    }
10351
10352    #[test]
10353    fn failed_reconciliation_marks_the_scope_partial() {
10354        let dir = sample_tree();
10355        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10356        fs::remove_dir_all(dir.path()).expect("remove root");
10357
10358        assert!(reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).is_err());
10359        assert_eq!(index.freshness(), crate::Freshness::Partial);
10360    }
10361
10362    #[test]
10363    fn successful_subtree_retry_restores_complete_root_coverage() {
10364        let dir = tempfile::tempdir().expect("tempdir");
10365        write_file(&dir.path().join("blocked/known.txt"), b"known");
10366        let config = ScanConfig::default();
10367        let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
10368        assert!(report.is_complete());
10369        let blocked = dir.path().join("blocked");
10370        let fault = install_walk_hook(&blocked, |_| {
10371            Some(std::io::Error::new(
10372                std::io::ErrorKind::PermissionDenied,
10373                "deterministic subtree refusal",
10374            ))
10375        });
10376
10377        let failed = reconcile_subtree(&mut index, Path::new("blocked"), &config, &mut |_| {})
10378            .expect("partial");
10379        assert!(!failed.scan.is_complete());
10380        assert_eq!(
10381            index.state().coverage,
10382            crate::Coverage::Partial(crate::CoverageReason::Inaccessible)
10383        );
10384        drop(fault);
10385
10386        let recovered = reconcile_subtree(&mut index, Path::new("blocked"), &config, &mut |_| {})
10387            .expect("retry");
10388
10389        assert!(recovered.scan.is_complete());
10390        assert_eq!(index.state().coverage, crate::Coverage::Complete);
10391    }
10392
10393    #[test]
10394    fn failed_pending_reconciliation_remains_queued_for_retry() {
10395        let dir = tempfile::tempdir().expect("tempdir");
10396        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10397        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
10398            path: PathBuf::new(),
10399            reason: crate::InvalidateReason::Requested,
10400        }]));
10401        fs::remove_dir_all(dir.path()).expect("remove root");
10402
10403        assert!(reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {}).is_err());
10404        assert_eq!(
10405            index.take_pending_invalidations(),
10406            vec![(PathBuf::new(), crate::InvalidateReason::Requested)]
10407        );
10408        assert_eq!(index.freshness(), crate::Freshness::Partial);
10409    }
10410
10411    #[cfg(unix)]
10412    #[test]
10413    fn partial_cold_scan_keeps_verified_siblings_complete() {
10414        use std::os::unix::fs::PermissionsExt;
10415        if !crate::test_support::require_permission_bits() {
10416            return;
10417        }
10418        let root = tempfile::tempdir().expect("root");
10419        write_file(&root.path().join("blocked/unknown.txt"), b"unread");
10420        write_file(&root.path().join("healthy/nested/known.txt"), b"known");
10421        let blocked = root.path().join("blocked");
10422        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
10423        let scan = ScanConfig::default();
10424        let detached = scan_into_index(root.path(), &scan);
10425        let streamed = scan_into_index_via_scanner(root.path(), &scan);
10426        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
10427        for result in [detached, streamed] {
10428            let (index, report) = result.expect("partial scan still returns its facts");
10429            assert!(!report.is_complete(), "permission fixture must fail the blocked listing");
10430            assert!(
10431                index
10432                    .issues()
10433                    .iter()
10434                    .any(|issue| issue.path.as_deref() == Some(Path::new("blocked")))
10435            );
10436            assert_eq!(index.freshness_at(Path::new("")), crate::Freshness::Partial);
10437            assert_eq!(index.directory_complete(Path::new("")), Some(true));
10438            assert_eq!(index.directory_complete(Path::new("blocked")), Some(false));
10439            assert_eq!(index.freshness_at(Path::new("blocked")), crate::Freshness::Partial);
10440            assert!(index.lookup(Path::new("blocked/unknown.txt")).is_none());
10441            for sibling in ["healthy", "healthy/nested"] {
10442                assert_eq!(index.directory_complete(Path::new(sibling)), Some(true), "{sibling}");
10443                assert_eq!(
10444                    index.freshness_at(Path::new(sibling)),
10445                    crate::Freshness::Fresh,
10446                    "{sibling}"
10447                );
10448            }
10449            assert!(index.lookup(Path::new("healthy/nested/known.txt")).is_some());
10450            assert!(
10451                !crate::stored_state::entries_writable(&index),
10452                "partial root cannot persist metadata"
10453            );
10454        }
10455    }
10456
10457    #[cfg(unix)]
10458    #[test]
10459    fn partial_pending_reconciliation_remains_queued_for_retry() {
10460        use std::os::unix::fs::PermissionsExt;
10461
10462        if !crate::test_support::require_permission_bits() {
10463            return;
10464        }
10465        let dir = tempfile::tempdir().expect("tempdir");
10466        write_file(&dir.path().join("blocked/known.txt"), b"known");
10467        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10468        let blocked = dir.path().join("blocked");
10469        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
10470        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
10471            path: PathBuf::from("blocked"),
10472            reason: crate::InvalidateReason::VerificationFailed,
10473        }]));
10474
10475        let report = reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {})
10476            .expect("permission failure is a partial report");
10477        let pending = index.take_pending_invalidations();
10478        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
10479        assert!(!report.is_complete(), "permission fixture must make reconciliation partial");
10480        assert_eq!(
10481            pending,
10482            vec![(PathBuf::from("blocked"), crate::InvalidateReason::VerificationFailed)]
10483        );
10484        assert_eq!(index.freshness_at(Path::new("blocked")), crate::Freshness::Partial);
10485    }
10486
10487    /// The shared API settles an unreadable subtree instead of queueing it again.
10488    ///
10489    /// Its per-event driver, `Watcher::apply_next`, drains after every event, so a retry
10490    /// re-walked the same unreadable subtree on each unrelated event, forever. The subtree
10491    /// stays partial and the report still names the error, once.
10492    #[cfg(unix)]
10493    #[test]
10494    fn partial_shared_pending_reconciliation_settles_instead_of_retrying() {
10495        use std::os::unix::fs::PermissionsExt;
10496
10497        if !crate::test_support::require_permission_bits() {
10498            return;
10499        }
10500        let dir = tempfile::tempdir().expect("tempdir");
10501        write_file(&dir.path().join("blocked/known.txt"), b"known");
10502        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10503        let handle = crate::IndexHandle::new(index);
10504        let blocked = dir.path().join("blocked");
10505        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
10506        handle
10507            .apply(&Observation::new(vec![Op::InvalidateSubtree {
10508                path: PathBuf::from("blocked"),
10509                reason: crate::InvalidateReason::VerificationFailed,
10510            }]))
10511            .expect("invalidate");
10512
10513        let report = reconcile_pending_handle(&handle, &ScanConfig::default(), &mut |_| {})
10514            .expect("permission failure is a partial report");
10515        let pending = handle.take_pending_invalidations().expect("pending");
10516        let freshness = handle.freshness_at(Path::new("blocked")).expect("freshness");
10517        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
10518        assert!(!report.is_complete(), "permission fixture must make reconciliation partial");
10519        assert!(pending.is_empty(), "{pending:?}");
10520        assert_eq!(freshness, crate::Freshness::Partial);
10521        assert!(!report.scan.errors.is_empty());
10522    }
10523
10524    #[test]
10525    fn pending_scope_mismatch_does_not_drain_the_retry_queue() {
10526        let dir = tempfile::tempdir().expect("tempdir");
10527        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
10528        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
10529        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
10530            path: PathBuf::new(),
10531            reason: crate::InvalidateReason::Requested,
10532        }]));
10533
10534        let error = reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {})
10535            .expect_err("mismatched scope must fail");
10536
10537        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
10538        assert_eq!(
10539            index.take_pending_invalidations(),
10540            vec![(PathBuf::new(), crate::InvalidateReason::Requested)]
10541        );
10542    }
10543
10544    #[test]
10545    fn reconciliation_rejects_a_scope_mismatch_before_mutating() {
10546        let dir = tempfile::tempdir().expect("tempdir");
10547        write_file(&dir.path().join("deep/nested.txt"), b"nested");
10548        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
10549        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
10550        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
10551
10552        let error = reconcile(&mut index, &ScanConfig::default(), &mut |_| {})
10553            .expect_err("mismatched scope must fail");
10554
10555        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
10556        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
10557        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10558    }
10559
10560    #[test]
10561    fn subtree_reconciliation_rejects_a_path_beyond_the_depth_scope() {
10562        let dir = tempfile::tempdir().expect("tempdir");
10563        write_file(&dir.path().join("deep/nested.txt"), b"nested");
10564        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
10565        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
10566
10567        let result =
10568            reconcile_subtree(&mut index, Path::new("deep/nested.txt"), &shallow, &mut |_| {});
10569
10570        assert!(matches!(result, Err(Error::SubtreeOutsideScanScope { .. })));
10571        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
10572        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10573    }
10574
10575    #[cfg(unix)]
10576    #[test]
10577    fn subtree_reconciliation_does_not_follow_an_ancestor_symlink() {
10578        use std::os::unix::fs::symlink;
10579
10580        let root = tempfile::tempdir().expect("root");
10581        let outside = tempfile::tempdir().expect("outside");
10582        write_file(&outside.path().join("secret.txt"), b"secret");
10583        symlink(outside.path(), root.path().join("link")).expect("symlink");
10584        let config = ScanConfig::default();
10585        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
10586
10587        let result =
10588            reconcile_subtree(&mut index, Path::new("link/secret.txt"), &config, &mut |_| {});
10589
10590        assert!(matches!(result, Err(Error::SubtreeOutsideScanScope { .. })));
10591        assert!(index.lookup(Path::new("link/secret.txt")).is_none());
10592        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10593    }
10594
10595    #[test]
10596    fn subtree_reconciliation_widens_to_a_non_directory_ancestor() {
10597        let root = tempfile::tempdir().expect("root");
10598        write_file(&root.path().join("parent/child.txt"), b"old");
10599        let config = ScanConfig::default();
10600        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
10601        fs::remove_dir_all(root.path().join("parent")).expect("remove directory");
10602        write_file(&root.path().join("parent"), b"replacement");
10603
10604        let report =
10605            reconcile_subtree(&mut index, Path::new("parent/child.txt"), &config, &mut |_| {})
10606                .expect("reconcile widened ancestor");
10607
10608        assert!(report.is_complete());
10609        assert_eq!(index.kind(Path::new("parent")), Some(EntryKind::File));
10610        assert!(index.lookup(Path::new("parent/child.txt")).is_none());
10611        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10612    }
10613
10614    #[test]
10615    fn subtree_reconciliation_widens_to_a_missing_ancestor() {
10616        let root = tempfile::tempdir().expect("root");
10617        write_file(&root.path().join("parent/child.txt"), b"old");
10618        let config = ScanConfig::default();
10619        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
10620        fs::remove_dir_all(root.path().join("parent")).expect("remove directory");
10621
10622        let report =
10623            reconcile_subtree(&mut index, Path::new("parent/child.txt"), &config, &mut |_| {})
10624                .expect("reconcile widened ancestor");
10625
10626        assert!(report.is_complete());
10627        assert!(index.lookup(Path::new("parent")).is_none());
10628        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10629    }
10630
10631    #[test]
10632    fn observation_only_revalidation_rejects_a_scope_mismatch() {
10633        let dir = tempfile::tempdir().expect("tempdir");
10634        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
10635        let (index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
10636        let mut observations = Vec::new();
10637
10638        let error = revalidate(&index, &ScanConfig::default(), &mut |observation| {
10639            observations.push(observation);
10640        })
10641        .expect_err("mismatched scope must fail");
10642
10643        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
10644        assert!(observations.is_empty());
10645    }
10646
10647    #[cfg(unix)]
10648    #[test]
10649    fn a_new_filesystem_boundary_prunes_cached_descendants() {
10650        use std::os::unix::fs::MetadataExt;
10651
10652        let root = Path::new("/");
10653        let root_dev = {
10654            crate::counters::bump(|c| c.stats += 1);
10655            fs::symlink_metadata(root)
10656        }
10657        .expect("stat root")
10658        .dev();
10659        let Some(mount) = [Path::new("/dev"), Path::new("/proc"), Path::new("/sys")]
10660            .into_iter()
10661            .find(|candidate| {
10662                fs::symlink_metadata(candidate)
10663                    .is_ok_and(|metadata| metadata.is_dir() && metadata.dev() != root_dev)
10664            })
10665        else {
10666            return; // This host exposes no convenient cross-device directory.
10667        };
10668        let relative = mount.strip_prefix(root).expect("mount is below root");
10669        let stale_child = relative.join(".fdu-stale-snapshot-entry");
10670        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
10671        let mount_meta = fs::symlink_metadata(mount).expect("stat mount");
10672        let mut index = Index::new_with_scope(root, config.scope());
10673        index.apply_baseline_ok(&Observation::new(vec![
10674            Op::Upsert {
10675                path: relative.to_path_buf(),
10676                kind: EntryKind::Dir,
10677                attrs: attrs_from(mount, &mount_meta).expect("mount attrs"),
10678            },
10679            Op::Upsert {
10680                path: stale_child.clone(),
10681                kind: EntryKind::File,
10682                attrs: Attrs { size: 10, allocated: 10, ..Attrs::default() },
10683            },
10684        ]));
10685
10686        let error = reconcile_subtree(&mut index, &stale_child, &config, &mut |_| {})
10687            .expect_err("a descendant below the mount boundary is outside scope");
10688        assert!(matches!(error, Error::SubtreeOutsideScanScope { .. }));
10689        assert!(index.lookup(&stale_child).is_some());
10690
10691        reconcile_subtree(&mut index, relative, &config, &mut |_| {}).expect("reconcile mount");
10692
10693        assert!(index.lookup(relative).is_some(), "the mount point itself stays visible");
10694        assert!(index.lookup(&stale_child).is_none(), "out-of-scope descendants are pruned");
10695    }
10696
10697    #[test]
10698    fn subtree_reconciliation_rejects_paths_outside_the_root() {
10699        let dir = sample_tree();
10700        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10701
10702        assert!(matches!(
10703            reconcile_subtree(
10704                &mut index,
10705                Path::new("../outside"),
10706                &ScanConfig::default(),
10707                &mut |_| {},
10708            ),
10709            Err(Error::PathEscapesRoot(_))
10710        ));
10711        assert_eq!(index.freshness(), crate::Freshness::Fresh);
10712    }
10713
10714    #[test]
10715    fn normalized_walk_errors_keep_index_and_one_shot_status_in_lockstep() {
10716        let root = Path::new("/root");
10717        let mut order: Vec<_> = (0..66).rev().collect();
10718        order.push(65);
10719        let mut errors = order
10720            .into_iter()
10721            .map(|number| {
10722                Error::io(
10723                    root.join(format!("file-{number:02}")),
10724                    std::io::Error::new(std::io::ErrorKind::PermissionDenied, "denied"),
10725                )
10726            })
10727            .collect::<Vec<_>>();
10728
10729        normalize_walk_errors(root, &mut errors);
10730        assert_eq!(errors.len(), 66, "the repeated cause is removed once");
10731
10732        let mut index = crate::Index::new(root);
10733        index.record_walk_errors(&mut errors);
10734        let status = crate::query::TreeStatus::of_walk(
10735            root,
10736            &mut ScanReport { errors, ..ScanReport::default() },
10737        );
10738
10739        assert_eq!(status.errors, index.issues());
10740        assert_eq!(status.errors_omitted, index.state().issues.omitted);
10741        assert_eq!(status.errors.len(), crate::MAX_RETAINED_ISSUES);
10742        assert_eq!(status.errors_omitted, 2);
10743    }
10744
10745    #[cfg(target_os = "linux")]
10746    #[test]
10747    fn scan_and_revalidate_keep_non_utf8_names_distinct() {
10748        use std::ffi::OsString;
10749        use std::os::unix::ffi::OsStringExt;
10750
10751        let dir = tempfile::tempdir().expect("tempdir");
10752        let first = PathBuf::from(OsString::from_vec(vec![b'n', 0x80]));
10753        let second = PathBuf::from(OsString::from_vec(vec![b'n', 0x81]));
10754        write_file(&dir.path().join(&first), b"a");
10755        write_file(&dir.path().join(&second), b"bb");
10756
10757        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10758        assert_eq!(index.total().files, 2);
10759        assert_eq!(index.total().bytes, 3);
10760
10761        fs::remove_file(dir.path().join(&first)).expect("remove first");
10762        let mut observations = Vec::new();
10763        revalidate(&index, &ScanConfig::default(), &mut |observation| {
10764            observations.push(observation);
10765        })
10766        .expect("revalidate");
10767        for observation in &observations {
10768            index.apply_ok(observation);
10769        }
10770        assert!(index.lookup(&first).is_none());
10771        assert!(index.lookup(&second).is_some());
10772        assert_eq!(index.total().bytes, 2);
10773    }
10774}