Skip to main content

fdu_core/
scan.rs

1//! The scan layer: walking a tree, producing observations, and applying reconciliation.
2//!
3//! Public scans emit upsert observations, and a revalidation sweep is the diff between
4//! what the index believes and what the filesystem says. Both speak the same
5//! [`Observation`] vocabulary as the watch layer. A detached one-shot index may consume
6//! equivalent parent-first directory groups privately because no observer can see its
7//! construction; every later mutation still crosses the shared observation boundary.
8//!
9//! # Status
10//!
11//! The serial walk is the portable `read_dir` plus non-following metadata reference.
12//! Parallel scans use the same path on most platforms; on macOS they first try a
13//! measured `getattrlistbulk` backend that returns directory entries and stat-tier
14//! metadata together. Unsupported filesystems, malformed results, mount points, and
15//! firmlinks fail closed to the portable path for the complete containing directory.
16//! Every backend produces the same [`Observation`] contract.
17
18use std::collections::{BTreeMap, BTreeSet, VecDeque};
19use std::ffi::{OsStr, OsString};
20use std::fmt::Write as _;
21use std::fs;
22use std::io::Read as _;
23use std::path::{Component, Path, PathBuf};
24
25use crate::ApplyStats;
26use crate::engine_contract::{
27    Attrs, Commit, EntryKind, Error, Observation, ObservationOp, Op, PathExpectation, PathState,
28    Result, ScanScope,
29};
30use crate::index::{
31    DetachedIndexBuilder, Index, IndexHandle, ReconcileErrors, ReconcileFinish,
32    collect_child_expectations,
33};
34use crate::query::ScopeAxis;
35use crate::stored_state::{ControlTierIdentity, EntryScope, EntryTierIdentity, SnapshotIdentity};
36
37// Keep the FFI exception at the platform boundary. The rest of the engine, including
38// every consumer of these observations, remains under the workspace's unsafe-code
39// denial.
40#[cfg(target_os = "macos")]
41#[allow(unsafe_code)]
42mod macos_bulk;
43
44#[cfg(windows)]
45#[allow(unsafe_code)]
46mod windows_metadata;
47
48/// How many ops accumulate before an observation is handed to the sink.
49///
50/// Batching matters for more than syscall economy: consumers coalesce per path within a
51/// batch and stat once per batch, and a live UI wants partial results while a large tree
52/// is still being walked rather than one delta at the end.
53const DEFAULT_BATCH_SIZE: usize = crate::platform_tuning::tuning().batch_size.get();
54
55/// Largest producer batch accepted before work must be published incrementally.
56pub const MAX_SCAN_BATCH_SIZE: usize = 64 * 1024;
57
58/// Most changed paths an exclusive parallel reconciliation may defer before applying.
59///
60/// Workers compare against one immutable index image, so mutations wait until the wave
61/// joins. Bounding that change set keeps a churned tree from turning the fast unchanged
62/// path into an unbounded allocation; overflow discards the wave and retries through
63/// the incremental serial reconciler.
64const MAX_DEFERRED_RECONCILE_OPS: usize = MAX_SCAN_BATCH_SIZE;
65
66/// Directories compared against one immutable index baseline before changes are applied.
67///
68/// The wave is large enough to amortize scoped worker creation and small enough that a
69/// changed tree publishes progress throughout a long reconciliation.
70const RECONCILE_WAVE_DIRECTORIES: usize =
71    crate::platform_tuning::tuning().reconcile_wave_directories.get();
72
73/// Identity of the fixed stat-tier reducer set.
74const REDUCERS_FINGERPRINT: u64 = 1;
75
76/// The order directories are visited in.
77///
78/// This changes *when* observations are produced, never *which* ones: both orders
79/// visit every entry exactly once and leave an identical index behind. It therefore
80/// stays out of [`ScanScope`] and cannot invalidate a cache, exactly like the worker
81/// count.
82///
83/// The choice only matters to a consumer that reads the index while the walk is still
84/// running, and there it matters a great deal.
85///
86/// # Strength of the guarantee
87///
88/// **These are scheduling preferences, not strict orders, whenever more than one worker
89/// is running** — which is the default.
90///
91/// The queue is ordered, but the *claims* are not. Workers take directories from the
92/// shared queue in the policy's order; a worker that finishes early can enqueue its
93/// children and another worker can claim them while a slower worker still holds
94/// unfinished work from a shallower level. Nothing releases a level barrier, because
95/// a barrier would idle every fast worker at each level boundary and give back most of
96/// the parallel producer's win.
97///
98/// So:
99///
100/// - With `threads: Some(1)`, [`ScanOrder::BreadthFirst`] is strict: no directory is
101///   read before one closer to the root.
102/// - With several workers it is *shallow-first*: shallow work is always preferred when
103///   a worker chooses, and deeper observations can still interleave.
104///
105/// That weaker property is what the browser use case actually needs — every top-level
106/// subtree starts filling early, so a mid-scan ranking is meaningful — and it is the
107/// property the tests pin. A caller that needs strict level order must ask for one
108/// worker and pay for it.
109#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
110pub enum ScanOrder {
111    /// Shallow directories before deep ones.
112    ///
113    /// The default, because it is the order whose partial results mean something.
114    /// Roll-ups are maintained per directory as the walk proceeds, so a consumer that
115    /// looks mid-scan sees top-level totals grow together — bars fill, rankings
116    /// converge — instead of one subtree finishing while its siblings read zero.
117    /// Interrupting early leaves a usefully complete picture of the top of the tree.
118    ///
119    /// Under several workers this is a preference rather than a guarantee; see the
120    /// type-level note above.
121    ///
122    /// Note that totals only grow *while an additive walk is running*. Monotonicity
123    /// comes from the producer being additive, not from the order — the order decides
124    /// which subtrees get to grow early.
125    #[default]
126    BreadthFirst,
127    /// One subtree toward completion before starting the next.
128    ///
129    /// Lower peak memory, since the frontier is bounded by depth rather than by the
130    /// width of a level, and better locality within a subtree. The cost is that
131    /// partial results are actively misleading: one child of the root approaches its
132    /// final total while its siblings read zero, so anything ranking by size mid-scan
133    /// ranks confidently and wrongly. Correct for a caller that only reads the
134    /// finished index and wants the smallest footprint.
135    ///
136    /// Under several workers this too is a preference: several subtrees will be in
137    /// flight at once, one per worker.
138    DepthFirst,
139}
140
141/// Knobs for a scan.
142#[derive(Clone, Debug)]
143// Four booleans, each an independent admission or observation switch with its own
144// semantic-scope consequence, not an enum in disguise: any combination is legal and
145// means what its fields say. The lint suspects flag-soup states; this is a config
146// surface whose fields are documented one by one.
147#[allow(clippy::struct_excessive_bools)]
148pub struct ScanConfig {
149    /// Maximum relative entry depth to retain. Zero keeps only the index root and `None`
150    /// means unlimited.
151    pub max_depth: Option<usize>,
152    /// Ops per emitted observation. Must be between one and [`MAX_SCAN_BATCH_SIZE`].
153    pub batch_size: usize,
154    /// Follow symlinks to directories. Off by default: following them turns a tree walk
155    /// into a graph walk with cycles, and every surveyed tool defaults to off.
156    pub follow_symlinks: bool,
157    /// Stay on the filesystem the root lives on.
158    pub one_filesystem: bool,
159    /// Hidden-component admission, or `None` to retain every component.
160    pub hidden: Option<std::sync::Arc<crate::admission::HiddenPolicy>>,
161    /// Exclude filesystem objects other than files, directories, and symlinks.
162    pub exclude_special: bool,
163    /// Directory-reading worker threads.
164    ///
165    /// A tree walk is a pile of independent, latency-bound directory reads, so it
166    /// scales with threads far better than most work does. One means the serial
167    /// walker, which stays the reference implementation and the thing every result is
168    /// checked against, with two exceptions that take the concurrent walker with one
169    /// worker instead: the detached index build, and a transient summary that reads
170    /// `.gitignore`, which must deliver each directory's control ahead of its entries
171    /// as that build consumes them. [`None`] asks for a bounded default derived from the
172    /// machine's available parallelism. The automatic pool starts conservatively and
173    /// unlocks more latency-hiding workers only when initial chunk timing identifies a
174    /// slow filesystem path.
175    ///
176    /// This is an operational knob, not a semantic one: it changes how fast the same
177    /// observations are produced, never which observations they are. That is why it
178    /// stays out of [`ScanScope`] and cannot invalidate a cache.
179    pub threads: Option<usize>,
180    /// The order directories are visited in. See [`ScanOrder`].
181    pub order: ScanOrder,
182    /// File-type rules to classify against, or `None` for the ones compiled into fdu.
183    ///
184    /// Unlike [`Self::threads`] this *is* semantic: a different taxonomy classifies the
185    /// same tree differently, which is why its fingerprint rides in [`ScanScope`] and a
186    /// change to it invalidates a snapshot. Shared rather than owned because a scan
187    /// clones its config per wave and a registry is read-only once built.
188    pub types: Option<std::sync::Arc<crate::classify::TypeRegistry>>,
189    /// Observe `.gitignore` control files and retain ignore classification.
190    ///
191    /// On by default on every surface: an [`Index`] from [`crate::open`] or a scan keeps
192    /// the exact control state it exposes and a watch maintains -- which entries are
193    /// ignored, and the ignored and unignored partitions of every roll-up -- and a one-shot
194    /// report from [`crate::prepare_report`] shows the ignored share of every row
195    /// (fdu-elnn). It costs a read of every `.gitignore` in the tree. A file past the
196    /// [`Self::control_limits`] is refused and named in [`Index::control_coverage`] rather
197    /// than ending the scan; a file that cannot be read is an error at its path, which
198    /// makes the result partial.
199    ///
200    /// Off, the scan performs no control-file I/O and retains no control table, and that is
201    /// stamped into [`ScanScope`], so an index-returning call never serves a snapshot taken
202    /// one way as the other. An [`Index`] built that way answers [`Index::is_ignored`],
203    /// [`Index::controls`], and the partition accessors with
204    /// [`crate::Error::ControlStateNotObserved`], never with "not ignored", and refuses
205    /// control input; a report's rows carry no ignored share, and a selection by ignored
206    /// state is refused. The command line spells it `--no-gitignore`.
207    ///
208    /// An opened root ([`crate::OpenedIndex`]) always observes control state, because its
209    /// ignored and unignored partitions are part of what it serves.
210    pub read_controls: bool,
211    /// Which ignored population shapes retained entries and content candidates.
212    pub population: crate::query::IgnoredEntries,
213    /// The budget and the line limit `.gitignore` files are applied under, each a size or
214    /// unbounded. See [`crate::control::ControlLimits`].
215    ///
216    /// A source that would take the table past the budget, or that has a line longer than
217    /// the line limit, is refused: its rules do not apply, the scan continues with every
218    /// size exact, and [`Index::control_coverage`] names it and the limit that fired. The
219    /// command line spells these `--gitignore-budget SIZE|all` and
220    /// `--gitignore-line-limit SIZE|all`; the Python API spells them `control_budget` and
221    /// `control_line_limit`.
222    ///
223    /// Semantic, like [`Self::read_controls`]: the limits decide which rules apply, so both
224    /// are part of [`ScanScope`] and a snapshot taken under other limits is not reused.
225    /// Ignored when control state is not observed.
226    pub control_limits: crate::control::ControlLimits,
227    /// Where to report how much of the walk has been done, or `None` to report nothing.
228    ///
229    /// An observer rather than a knob: it changes neither which observations a walk
230    /// produces nor how it produces them, so it is no part of [`ScanScope`] or of any
231    /// snapshot identity, and two configs that differ only here are the same scan.
232    /// Honoured by every walker in this module -- the cold scans, the summary fold,
233    /// [`revalidate`], and each `reconcile` entry point -- which enter
234    /// [`ProgressPhase::Scanning`](crate::ProgressPhase) or
235    /// [`ProgressPhase::Revalidating`](crate::ProgressPhase) and add their counts once
236    /// per chunk of directories, never per entry. See [`crate::Progress`] for what the
237    /// counts mean and what holds when a walk returns.
238    pub progress: Option<crate::Progress>,
239}
240
241impl Default for ScanConfig {
242    fn default() -> Self {
243        Self {
244            max_depth: None,
245            batch_size: DEFAULT_BATCH_SIZE,
246            follow_symlinks: false,
247            one_filesystem: false,
248            hidden: None,
249            exclude_special: false,
250            threads: None,
251            order: ScanOrder::default(),
252            types: None,
253            read_controls: crate::query::Request::DEFAULTS.read_controls,
254            population: crate::query::IgnoredEntries::Include,
255            control_limits: crate::query::Request::DEFAULTS.control_limits,
256            progress: None,
257        }
258    }
259}
260
261/// Why watching cannot narrow its structural scan boundary, said once for every surface.
262/// Ignored population is a supported retained-scope choice because control edits
263/// reconcile the governing directory and rebuild that population.
264///
265/// The CLI used to carry this guidance and the library carried "requires event-scope
266/// filtering", which names the implementation rather than the caller's next move -- so a
267/// library caller hitting the same wall got jargon and the CLI user got help. Two
268/// messages for one rule also drift, and the parity harness could not tell they were the
269/// same rule.
270///
271/// The knobs are named by the calling surface: `--scan-depth` on the command line,
272/// `max_depth` through the API. Everything else is identical, so the harness can verify
273/// mechanically that both surfaces state the same rule.
274pub const WATCH_SCOPE_GUIDANCE: &str = concat!(
275    "watching requires full scope and cannot be combined with max_depth or one_filesystem: ",
276    "a watcher cannot filter backend events against a narrowed boundary. Selection such as ",
277    "depth, include, and modified_since does work while watching, because it filters the ",
278    "retained index rather than narrowing the scan"
279);
280
281impl ScanConfig {
282    /// Classify this scan with `types` and include their derived identity in its scope.
283    #[must_use]
284    pub fn with_types(mut self, types: std::sync::Arc<crate::classify::TypeRegistry>) -> Self {
285        self.types = Some(types);
286        self
287    }
288
289    /// The file-type rules in effect: the supplied registry, or the compiled default.
290    pub fn types(&self) -> &crate::classify::TypeRegistry {
291        match &self.types {
292            Some(types) => types,
293            None => crate::classify::TypeRegistry::compiled(),
294        }
295    }
296
297    /// Share the file-type rules with an index that retains them.
298    pub(crate) fn types_shared(&self) -> std::sync::Arc<crate::classify::TypeRegistry> {
299        self.types
300            .as_ref()
301            .map_or_else(crate::classify::TypeRegistry::compiled_shared, std::sync::Arc::clone)
302    }
303
304    /// Hidden-component policy in effect.
305    pub fn hidden(&self) -> &crate::admission::HiddenPolicy {
306        self.hidden.as_deref().unwrap_or_else(|| crate::admission::HiddenPolicy::keep_all())
307    }
308
309    /// Semantic cache identity, excluding operational batching choices.
310    ///
311    /// Composed from [`Self::snapshot_identity`], so the scope an index records and the
312    /// tier identities a snapshot of it carries are one value in two shapes.
313    ///
314    /// No longer `const`: the type-rule fingerprint is now a property of the registry in
315    /// effect rather than a compiled-in constant, which is the whole point of letting a
316    /// caller supply one. A snapshot taken under different rules must not be reused.
317    pub fn scope(&self) -> ScanScope {
318        self.snapshot_identity().scan_scope()
319    }
320
321    /// Which entries this scan retains, the part of its scope no `.gitignore` setting
322    /// changes.
323    pub fn entry_scope(&self) -> EntryScope {
324        EntryScope {
325            max_depth: self.max_depth,
326            follow_symlinks: self.follow_symlinks,
327            one_filesystem: self.one_filesystem,
328            hidden_fingerprint: self.hidden().fingerprint(),
329            exclude_special: self.exclude_special,
330            population: self.population,
331            control_fingerprint: if self.population == crate::query::IgnoredEntries::Include {
332                0
333            } else {
334                self.control_identity().ignore_rules_fingerprint()
335            },
336        }
337    }
338
339    /// Whether this scan observes `.gitignore` control state, and under which limits.
340    ///
341    /// The limits are part of the identity only when control state is observed: a scan
342    /// that reads no control file applies none, whatever [`Self::control_limits`] says.
343    pub fn control_identity(&self) -> ControlTierIdentity {
344        if self.read_controls {
345            ControlTierIdentity::Observed { limits: self.control_limits }
346        } else {
347            ControlTierIdentity::NotObserved
348        }
349    }
350
351    /// The identity of every tier a snapshot of this scan holds.
352    pub fn snapshot_identity(&self) -> SnapshotIdentity {
353        SnapshotIdentity {
354            entries: EntryTierIdentity {
355                engine: crate::snapshot::engine_fingerprint(),
356                scope: self.entry_scope(),
357                type_rules_fingerprint: self.types().fingerprint(),
358                reducers_fingerprint: REDUCERS_FINGERPRINT,
359            },
360            controls: self.control_identity(),
361        }
362    }
363
364    /// Resolve [`Self::threads`] to the workers active when a scan begins.
365    #[cfg(any(target_os = "macos", test))]
366    fn worker_threads(&self) -> usize {
367        self.worker_pool().initial
368    }
369
370    /// Resolve the worker count for immutable-baseline reconciliation waves.
371    fn reconciliation_worker_threads(&self) -> usize {
372        match self.threads {
373            Some(threads) => threads.clamp(1, MAX_SCAN_THREADS),
374            None => std::thread::available_parallelism()
375                .map_or(1, std::num::NonZero::get)
376                .clamp(1, DEFAULT_RECONCILE_THREADS_CAP),
377        }
378    }
379
380    /// Resolve the initial and maximum worker counts for one scan.
381    #[cfg(any(target_os = "macos", test))]
382    fn worker_pool(&self) -> WorkerPool {
383        self.worker_pool_for(std::thread::available_parallelism().map_or(1, std::num::NonZero::get))
384    }
385
386    /// Resolve the worker pool from one captured operating-system parallelism value.
387    fn worker_pool_for(&self, available_parallelism: usize) -> WorkerPool {
388        match self.threads {
389            Some(threads) => WorkerPool::fixed(threads.clamp(1, MAX_SCAN_THREADS)),
390            None => automatic_worker_pool(available_parallelism),
391        }
392    }
393
394    /// The scope axis this build cannot honour, if any.
395    ///
396    /// The one statement of the capability rule, so it is asked rather than restated.
397    /// [`Request::validate`](crate::query::Request::validate) asks it before any stored
398    /// state is read, which is what makes a scope this build cannot honour refuse the same
399    /// way on every route, every cache policy, and both surfaces; [`Self::validate`] asks
400    /// it for the engine-internal callers -- a bound root, a raw scan, an observation --
401    /// that never carry a request.
402    pub(crate) const fn unsupported_axis(&self) -> Option<ScopeAxis> {
403        if self.follow_symlinks {
404            return Some(ScopeAxis::FollowSymlinks);
405        }
406        #[cfg(not(unix))]
407        if self.one_filesystem {
408            return Some(ScopeAxis::OneFilesystem);
409        }
410        None
411    }
412
413    pub(crate) fn validate(&self) -> Result<()> {
414        if !self.read_controls && self.population != crate::query::IgnoredEntries::Include {
415            return Err(Error::UnsupportedScanConfig(
416                "ignored population requires .gitignore observation",
417            ));
418        }
419        if self.batch_size == 0 || self.batch_size > MAX_SCAN_BATCH_SIZE {
420            return Err(Error::UnsupportedScanConfig(
421                "batch_size must be nonzero and no greater than MAX_SCAN_BATCH_SIZE",
422            ));
423        }
424        if let Some(axis) = self.unsupported_axis() {
425            return Err(Error::UnsupportedScanConfig(axis.reason()));
426        }
427        Ok(())
428    }
429
430    pub(crate) fn validate_for_scope(&self, indexed: ScanScope) -> Result<()> {
431        self.validate()?;
432        let requested = self.scope();
433        if indexed != requested {
434            return Err(Error::ScanScopeMismatch { indexed, requested });
435        }
436        Ok(())
437    }
438
439    /// Scope equality, plus the boundary a watcher cannot filter its backend's events
440    /// against.
441    ///
442    /// The rule belongs to the request model, which refuses a watch of a narrowed scope
443    /// before anything is opened ([`RequestError::WatchScope`](crate::query::RequestError));
444    /// this is the same rule where a watcher is bound without a request -- an opened root
445    /// that observes, and each batch the adapter applies -- and it renders the one
446    /// guidance string the model renders.
447    #[cfg(feature = "watch")]
448    pub(crate) fn validate_for_watch_scope(&self, indexed: ScanScope) -> Result<()> {
449        self.validate_for_scope(indexed)?;
450        if self.max_depth.is_some() || self.one_filesystem {
451            return Err(Error::UnsupportedScanConfig(WATCH_SCOPE_GUIDANCE));
452        }
453        Ok(())
454    }
455}
456
457impl Default for ScanScope {
458    fn default() -> Self {
459        ScanConfig::default().scope()
460    }
461}
462
463/// What a scan did, including the errors it walked past.
464///
465/// Unreadable directories are skipped rather than aborting the scan — a permission-denied
466/// subdirectory should not cost you the other 499,000 files — but they are reported
467/// rather than swallowed, so a caller can tell a complete answer from a partial one.
468#[derive(Debug, Default)]
469pub struct ScanReport {
470    /// Directories successfully listed.
471    pub dirs_read: u64,
472    /// Entries observed, directories included.
473    pub entries: u64,
474    /// Regular files whose metadata was observed.
475    pub files_walked: u64,
476    /// Apparent bytes represented by the regular files whose metadata was observed.
477    pub bytes_walked: u64,
478    /// Allocated bytes of those files: what the default size metric counts, and what a
479    /// sparse disk image or a clone makes far smaller than their apparent bytes.
480    pub allocated_walked: u64,
481    /// Paths that could not be read, with the reason.
482    pub errors: Vec<Error>,
483    /// Where the walk's time went, summed across workers.
484    pub attribution: WalkAttribution,
485}
486
487impl ScanReport {
488    /// True when every directory in scope was read successfully.
489    pub fn is_complete(&self) -> bool {
490        self.errors.is_empty()
491    }
492
493    /// Fold one worker's share of a parallel walk into the whole-walk report.
494    fn absorb(&mut self, other: Self) {
495        self.dirs_read += other.dirs_read;
496        self.entries += other.entries;
497        self.files_walked += other.files_walked;
498        self.bytes_walked += other.bytes_walked;
499        self.allocated_walked += other.allocated_walked;
500        self.errors.extend(other.errors);
501        self.attribution.absorb(other.attribution);
502    }
503
504    /// Record one successfully stated directory entry.
505    fn observe(&mut self, kind: EntryKind, attrs: Attrs) {
506        self.entries += 1;
507        if kind == EntryKind::File {
508            self.files_walked += 1;
509            self.bytes_walked += attrs.size;
510            self.allocated_walked += attrs.allocated;
511        }
512    }
513}
514
515/// One walker's running share of the progress counters.
516///
517/// Each walker keeps its own [`ScanReport`]; this remembers how much of that report it
518/// has already added to the shared [`crate::Progress`] cells, so each addition is the
519/// difference since the last. It lives on the worker's stack beside the report rather
520/// than inside it, so a report absorbed into another never carries a stale baseline.
521struct ProgressTally<'a> {
522    progress: Option<&'a crate::Progress>,
523    directories: u64,
524    files: u64,
525    bytes: u64,
526    allocated: u64,
527}
528
529impl<'a> ProgressTally<'a> {
530    const fn new(progress: Option<&'a crate::Progress>) -> Self {
531        Self { progress, directories: 0, files: 0, bytes: 0, allocated: 0 }
532    }
533
534    /// Add what `report` has counted since the last call.
535    ///
536    /// The one `Option` check is the whole cost when no handle is attached. Called once
537    /// per chunk of directories a walker hands over, never per entry.
538    fn flush(&mut self, report: &ScanReport) {
539        let Some(progress) = self.progress else { return };
540        let directories = report.dirs_read - self.directories;
541        let files = report.files_walked - self.files;
542        let bytes = report.bytes_walked - self.bytes;
543        let allocated = report.allocated_walked - self.allocated;
544        if directories != 0 || files != 0 || bytes != 0 || allocated != 0 {
545            progress.add_walked(directories, files, bytes, allocated);
546            self.directories = report.dirs_read;
547            self.files = report.files_walked;
548            self.bytes = report.bytes_walked;
549            self.allocated = report.allocated_walked;
550        }
551    }
552
553    /// Treat everything `report` holds as already added.
554    ///
555    /// For a walker that continues a report whose counts other workers added themselves.
556    fn skip_to(&mut self, report: &ScanReport) {
557        self.directories = report.dirs_read;
558        self.files = report.files_walked;
559        self.bytes = report.bytes_walked;
560        self.allocated = report.allocated_walked;
561    }
562}
563
564/// Normalize filesystem failures before one of the bounded status collectors retains them.
565///
566/// A walk may encounter the same inaccessible path from several worker paths. The report is
567/// already the full, transient set for this pass, so sorting and deduplicating it here avoids
568/// allocating or formatting a second unbounded set solely to decide which 64 details survive.
569/// I/O causes are keyed by their native root-relative path and the issue category, exactly the
570/// cause identity retained by an index. Other engine failures are left distinct: walker errors
571/// are I/O failures, and treating arbitrary engine errors as equivalent without constructing
572/// their bounded issue representation would lose information.
573pub(crate) fn normalize_walk_errors(root: &Path, errors: &mut Vec<Error>) {
574    errors.sort_by(|left, right| match (left, right) {
575        (
576            Error::Io { path: left_path, source: left_source },
577            Error::Io { path: right_path, source: right_source },
578        ) => left_path
579            .strip_prefix(root)
580            .unwrap_or(left_path)
581            .cmp(right_path.strip_prefix(root).unwrap_or(right_path))
582            .then_with(|| {
583                walk_issue_kind_rank(left_source).cmp(&walk_issue_kind_rank(right_source))
584            }),
585        (Error::Io { .. }, _) => std::cmp::Ordering::Less,
586        (_, Error::Io { .. }) => std::cmp::Ordering::Greater,
587        _ => std::cmp::Ordering::Equal,
588    });
589    errors.dedup_by(|right, left| match (left, right) {
590        (
591            Error::Io { path: left_path, source: left_source },
592            Error::Io { path: right_path, source: right_source },
593        ) => {
594            left_path.strip_prefix(root).unwrap_or(left_path)
595                == right_path.strip_prefix(root).unwrap_or(right_path)
596                && walk_issue_kind_rank(left_source) == walk_issue_kind_rank(right_source)
597        }
598        _ => false,
599    });
600}
601
602fn walk_issue_kind_rank(error: &std::io::Error) -> u8 {
603    match error.kind() {
604        std::io::ErrorKind::PermissionDenied => 0,
605        std::io::ErrorKind::NotFound => 1,
606        std::io::ErrorKind::InvalidData | std::io::ErrorKind::InvalidInput => 2,
607        _ => 5,
608    }
609}
610
611/// Schema carried by [`ScanDiagnostics`].
612///
613/// Diagnostics are an opt-in measurement contract rather than stable human output.
614/// Consumers must reject an unknown schema instead of guessing that fields retained
615/// their meaning.
616pub const SCAN_DIAGNOSTICS_SCHEMA: &str = "fdu-scan-diagnostics-v1";
617
618/// Maximum policy-window records retained by one diagnostic scan.
619///
620/// The bound is on controller evaluations, not filesystem entries. A controller that
621/// needs more history must mark the artifact truncated; claim-grade consumers reject
622/// that artifact rather than silently analyzing an incomplete policy history.
623const MAX_POLICY_TRACE_EVENTS: usize = 256;
624
625/// Opt-in, run-scoped evidence about a filesystem scan.
626///
627/// Obtain this through [`scan_with_diagnostics`] or
628/// [`scan_into_index_with_diagnostics`]. Keeping it out of [`ScanReport`] preserves the
629/// existing scan API and keeps ordinary callers off the measurement path entirely.
630#[derive(Clone, Debug, PartialEq, Eq)]
631pub struct ScanDiagnostics {
632    /// Version of this diagnostic contract.
633    pub schema: &'static str,
634    /// Automatic worker-controller history and queue state.
635    pub worker_policy: WorkerPolicyDiagnostics,
636    /// Directory-enumeration backends used by this run.
637    pub backend: ScanBackendDiagnostics,
638}
639
640/// Repository-only controller variants used by the performance evidence probe.
641///
642/// These variants are not selected by [`scan`] or [`scan_with_diagnostics`]; both keep
643/// the shipped one-shot policy. The explicit experimental APIs make candidate behavior
644/// measurable without hiding a production change behind an environment variable.
645#[doc(hidden)]
646#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
647pub enum WorkerPolicyExperiment {
648    /// The production controller: one prefix window and at most one expansion.
649    #[default]
650    ShippedOneShot,
651    /// Re-evaluate independent windows until a slow phase requests the full reserve.
652    RepeatedWindows,
653    /// Re-evaluate independent windows, gate on useful frontier/handoff backlog, and grow
654    /// the pool in stages.
655    StagedGatedWindows,
656}
657
658/// Final state of the automatic worker controller.
659#[derive(Clone, Copy, Debug, PartialEq, Eq)]
660pub enum WorkerPolicyOutcome {
661    /// The scan did no walking, for example because `max_depth` was zero.
662    NotRun,
663    /// A fixed pool had no adaptive decision to make.
664    Fixed,
665    /// The walk ended before an adaptive window became observable.
666    Undecided,
667    /// The controller measured a window and retained the initial pool.
668    Held,
669    /// The controller requested and activated the reserve workers.
670    ScaledUp,
671    /// Slow work was observed only after no useful queued or in-flight work remained.
672    HeldNoUsefulWork,
673}
674
675/// One controller evaluation over a half-open range of completed entry ordinals.
676#[derive(Clone, Debug, PartialEq, Eq)]
677pub struct WorkerPolicyWindow {
678    /// Monotonic record number within this scan.
679    pub sequence: u64,
680    /// First completed-entry ordinal represented by this window, inclusive.
681    pub start_entry_ordinal: u64,
682    /// Ordinal immediately after the last represented entry.
683    pub end_entry_ordinal: u64,
684    /// Entries contributing to the service-time signal.
685    pub observed_entries: u64,
686    /// Completed directory claims contributing to the service-time signal.
687    pub observed_chunks: u64,
688    /// Worker time contributing to the service-time signal.
689    pub observed_work_ns: u64,
690    /// Derived service time, or null when no entry made the signal observable.
691    pub work_ns_per_entry: Option<u64>,
692    /// Why `work_ns_per_entry` is null.
693    pub work_ns_per_entry_unavailable_reason: Option<&'static str>,
694    /// Directories ready to claim when the controller evaluated the window.
695    pub ready_directories: usize,
696    /// Claimed directories still being processed at that point.
697    pub in_flight_directories: usize,
698    /// Live worker threads at that point, including workers waiting for a claim.
699    pub active_workers: usize,
700    /// Observation batches sent but not yet received by the consumer.
701    pub handoff_backlog: usize,
702    /// Worker target requested by a scale decision.
703    pub requested_workers: Option<usize>,
704    /// What the controller concluded from this window.
705    pub decision: WorkerPolicyDecision,
706}
707
708/// Decision represented by a [`WorkerPolicyWindow`].
709#[derive(Clone, Copy, Debug, PartialEq, Eq)]
710pub enum WorkerPolicyDecision {
711    /// The walk ended before the window could support a decision.
712    Undecided,
713    /// The observed window retained the current pool.
714    Hold,
715    /// The observed window activated reserve workers.
716    ScaleUp,
717    /// The trigger fired after all useful work had drained.
718    HoldNoUsefulWork,
719    /// A complete window held because reserve workers had no useful frontier to claim.
720    HoldInsufficientFrontier,
721    /// A complete window held because the unbounded handoff backlog was already high.
722    HoldHandoffBacklog,
723    /// A post-decision observation window remained below the slow threshold.
724    ObserveFast,
725    /// A post-decision observation window met the slow threshold.
726    ObserveSlow,
727    /// A trailing partial window carried no new terminal decision.
728    Incomplete,
729    /// A trailing partial post-decision observation carried no policy decision.
730    ObserveIncomplete,
731}
732
733/// Worker-controller configuration, trace, and terminal queue state.
734#[derive(Clone, Debug, PartialEq, Eq)]
735pub struct WorkerPolicyDiagnostics {
736    /// Controller variant exercised by this scan.
737    pub controller: &'static str,
738    /// Parallelism reported by the operating system when the scan began.
739    pub available_parallelism: usize,
740    /// Workers in the pool before any adaptive decision.
741    pub initial_workers: usize,
742    /// Hard maximum workers this scan could activate.
743    pub maximum_workers: usize,
744    /// Entry target for an adaptive window, or null for a fixed pool.
745    pub calibration_window_entries: Option<u64>,
746    /// Slow-service trigger, or null for a fixed pool.
747    pub slow_threshold_ns_per_entry: Option<u64>,
748    /// Directory chunks folded into live controller windows.
749    pub calibration_chunks: u64,
750    /// Entries folded into live controller windows.
751    pub calibration_entries: u64,
752    /// Worker time folded into live controller windows.
753    pub calibration_work_ns: u64,
754    /// Expansion messages that caused the consumer to create more workers.
755    pub worker_expansions: u64,
756    /// Terminal policy outcome.
757    pub outcome: WorkerPolicyOutcome,
758    /// Explanation when no adaptive evaluation exists.
759    pub outcome_reason: Option<&'static str>,
760    /// Total worker threads created during the walk.
761    pub workers_spawned: usize,
762    /// Maximum simultaneously live worker threads, including workers waiting for work.
763    pub peak_active_workers: usize,
764    /// Ready directories at scan completion; a complete walk must leave zero.
765    pub ready_directories_at_finish: usize,
766    /// In-flight directories at scan completion; a complete walk must leave zero.
767    pub in_flight_directories_at_finish: usize,
768    /// Observation batches outstanding at scan completion.
769    pub handoff_backlog_at_finish: usize,
770    /// Maximum outstanding observation batches during the scan.
771    pub handoff_backlog_high_water: usize,
772    /// Bounded controller history.
773    pub windows: Vec<WorkerPolicyWindow>,
774    /// True when controller history exceeded the 256-event diagnostic bound.
775    pub events_truncated: bool,
776}
777
778/// Directory enumeration backends used by one scan.
779#[derive(Clone, Debug, PartialEq, Eq)]
780pub struct ScanBackendDiagnostics {
781    /// Portable `read_dir` calls attempted.
782    pub portable_attempts: u64,
783    /// Portable directory listings completed successfully.
784    pub portable_directory_reads: u64,
785    /// macOS bulk enumeration attempts, or null off macOS.
786    pub macos_bulk_attempts: Option<u64>,
787    /// Successful macOS bulk listings, or null off macOS.
788    pub macos_bulk_successes: Option<u64>,
789    /// Bulk attempts that fell back to portable enumeration, or null off macOS.
790    pub macos_bulk_fallbacks: Option<u64>,
791    /// Why macOS fields are null.
792    pub unavailable_reason: Option<&'static str>,
793}
794
795impl ScanDiagnostics {
796    /// Serialize this versioned diagnostic contract as compact JSON.
797    ///
798    /// This deliberately lives beside the contract instead of in a benchmark binary:
799    /// claim-grade installed-command measurements and the repository probe must emit
800    /// byte-for-byte equivalent evidence without adding a serialization dependency to
801    /// the core crate.
802    pub fn to_json(&self) -> String {
803        let policy = &self.worker_policy;
804        let backend = &self.backend;
805        let mut windows = String::from("[");
806        for (index, window) in policy.windows.iter().enumerate() {
807            if index > 0 {
808                windows.push(',');
809            }
810            let _ = write!(
811                windows,
812                concat!(
813                    "{{\"active_workers\":{},\"decision\":\"{}\",",
814                    "\"end_entry_ordinal\":{},\"handoff_backlog\":{},",
815                    "\"in_flight_directories\":{},\"observed_chunks\":{},",
816                    "\"observed_entries\":{},",
817                    "\"observed_work_ns\":{},\"ready_directories\":{},",
818                    "\"requested_workers\":{},\"sequence\":{},\"start_entry_ordinal\":{},",
819                    "\"work_ns_per_entry\":{},",
820                    "\"work_ns_per_entry_unavailable_reason\":{}}}"
821                ),
822                window.active_workers,
823                worker_policy_decision_name(window.decision),
824                window.end_entry_ordinal,
825                window.handoff_backlog,
826                window.in_flight_directories,
827                window.observed_chunks,
828                window.observed_entries,
829                window.observed_work_ns,
830                window.ready_directories,
831                json_optional_usize(window.requested_workers),
832                window.sequence,
833                window.start_entry_ordinal,
834                json_optional_u64(window.work_ns_per_entry),
835                json_optional_string(window.work_ns_per_entry_unavailable_reason),
836            );
837        }
838        windows.push(']');
839        format!(
840            concat!(
841                "{{\"backend\":{{\"macos_bulk_attempts\":{},",
842                "\"macos_bulk_fallbacks\":{},\"macos_bulk_successes\":{},",
843                "\"portable_attempts\":{},\"portable_directory_reads\":{},",
844                "\"unavailable_reason\":{}}},",
845                "\"schema\":\"{}\",\"worker_policy\":{{",
846                "\"available_parallelism\":{},\"calibration_chunks\":{},",
847                "\"calibration_entries\":{},\"calibration_window_entries\":{},",
848                "\"calibration_work_ns\":{},",
849                "\"controller\":\"{}\",",
850                "\"events_truncated\":{},\"handoff_backlog_at_finish\":{},",
851                "\"handoff_backlog_high_water\":{},\"in_flight_directories_at_finish\":{},",
852                "\"initial_workers\":{},\"maximum_workers\":{},\"outcome\":\"{}\",",
853                "\"outcome_reason\":{},\"peak_active_workers\":{},",
854                "\"ready_directories_at_finish\":{},\"slow_threshold_ns_per_entry\":{},",
855                "\"windows\":{},\"worker_expansions\":{},\"workers_spawned\":{}}}}}"
856            ),
857            json_optional_u64(backend.macos_bulk_attempts),
858            json_optional_u64(backend.macos_bulk_fallbacks),
859            json_optional_u64(backend.macos_bulk_successes),
860            backend.portable_attempts,
861            backend.portable_directory_reads,
862            json_optional_string(backend.unavailable_reason),
863            self.schema,
864            policy.available_parallelism,
865            policy.calibration_chunks,
866            policy.calibration_entries,
867            json_optional_u64(policy.calibration_window_entries),
868            policy.calibration_work_ns,
869            policy.controller,
870            policy.events_truncated,
871            policy.handoff_backlog_at_finish,
872            policy.handoff_backlog_high_water,
873            policy.in_flight_directories_at_finish,
874            policy.initial_workers,
875            policy.maximum_workers,
876            worker_policy_outcome_name(policy.outcome),
877            json_optional_string(policy.outcome_reason),
878            policy.peak_active_workers,
879            policy.ready_directories_at_finish,
880            json_optional_u64(policy.slow_threshold_ns_per_entry),
881            windows,
882            policy.worker_expansions,
883            policy.workers_spawned,
884        )
885    }
886}
887
888const fn worker_policy_outcome_name(value: WorkerPolicyOutcome) -> &'static str {
889    match value {
890        WorkerPolicyOutcome::NotRun => "not_run",
891        WorkerPolicyOutcome::Fixed => "fixed",
892        WorkerPolicyOutcome::Undecided => "undecided",
893        WorkerPolicyOutcome::Held => "held",
894        WorkerPolicyOutcome::ScaledUp => "scaled_up",
895        WorkerPolicyOutcome::HeldNoUsefulWork => "held_no_useful_work",
896    }
897}
898
899const fn worker_policy_decision_name(value: WorkerPolicyDecision) -> &'static str {
900    match value {
901        WorkerPolicyDecision::Undecided => "undecided",
902        WorkerPolicyDecision::Hold => "hold",
903        WorkerPolicyDecision::ScaleUp => "scale_up",
904        WorkerPolicyDecision::HoldNoUsefulWork => "hold_no_useful_work",
905        WorkerPolicyDecision::HoldInsufficientFrontier => "hold_insufficient_frontier",
906        WorkerPolicyDecision::HoldHandoffBacklog => "hold_handoff_backlog",
907        WorkerPolicyDecision::ObserveFast => "observe_fast",
908        WorkerPolicyDecision::ObserveSlow => "observe_slow",
909        WorkerPolicyDecision::Incomplete => "incomplete",
910        WorkerPolicyDecision::ObserveIncomplete => "observe_incomplete",
911    }
912}
913
914fn json_optional_string(value: Option<&str>) -> String {
915    value.map_or_else(|| "null".into(), |value| format!("\"{}\"", json_escape(value)))
916}
917
918fn json_optional_u64(value: Option<u64>) -> String {
919    value.map_or_else(|| "null".into(), |value| value.to_string())
920}
921
922fn json_optional_usize(value: Option<usize>) -> String {
923    value.map_or_else(|| "null".into(), |value| value.to_string())
924}
925
926fn json_escape(value: &str) -> String {
927    let mut escaped = String::new();
928    for character in value.chars() {
929        match character {
930            '"' => escaped.push_str("\\\""),
931            '\\' => escaped.push_str("\\\\"),
932            '\u{08}' => escaped.push_str("\\b"),
933            '\u{0c}' => escaped.push_str("\\f"),
934            '\n' => escaped.push_str("\\n"),
935            '\r' => escaped.push_str("\\r"),
936            '\t' => escaped.push_str("\\t"),
937            character if character <= '\u{1f}' => {
938                let _ = write!(escaped, "\\u{:04x}", u32::from(character));
939            }
940            character => escaped.push(character),
941        }
942    }
943    escaped
944}
945
946/// Where a walk's time went, so "blocked" is never one undifferentiated number.
947///
948/// The performance loop's standing question is whether a walk is bound by disk I/O,
949/// by CPU, or by coordination, and process-level counters cannot answer it: user and
950/// system time say how much CPU was burned, but a fused "blocked" number cannot say
951/// whether workers were waiting on the filesystem, on the queue lock, or on nothing
952/// at all because the queue was empty. These counters split that out at the source.
953///
954/// Everything is measured in *chunks*, never per file: one timing pair per claimed
955/// run of directories, per contended lock, per batch handoff. On the 60k-entry
956/// reference tree that is a few thousand `Instant` reads against hundreds of
957/// milliseconds of walking — the instrumentation follows the same amortization rule
958/// it exists to verify.
959///
960/// In a parallel walk the fields sum over workers, so `wall_ns` is worker-seconds
961/// (it can exceed the scan's wall clock) and every other duration is a disjoint
962/// slice of it: `work_ns + starved_ns + lock_wait_ns + send_ns <= wall_ns`, with the
963/// remainder being uninstrumented odds and ends (uncontended lock ops, loop
964/// bookkeeping). A serial walk fills only `wall_ns`, `work_ns`, and `send_ns` —
965/// there is no coordination to attribute.
966#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
967pub struct WalkAttribution {
968    /// Total time workers spent in the walk loop, summed across workers.
969    pub wall_ns: u64,
970    /// Reading directories and stating entries — the real work, syscalls plus the
971    /// compute between them. Separating disk from CPU *within* this span needs the
972    /// process-level user/system counters alongside; per-syscall timing would break
973    /// the chunk-amortization rule.
974    pub work_ns: u64,
975    /// Waiting on the queue's condvar because no work was available. Starvation:
976    /// either the frontier is momentarily narrower than the worker pool, or the walk
977    /// is ending.
978    pub starved_ns: u64,
979    /// Waiting to acquire the queue lock when another worker held it. This is the
980    /// contention the shared-queue design bets stays negligible; now it is measured
981    /// instead of argued.
982    pub lock_wait_ns: u64,
983    /// Handing observation batches to the consumer: the channel send in a parallel
984    /// walk, the inline sink call — which is the consumer actually running — in a
985    /// serial one.
986    pub send_ns: u64,
987    /// Chunks of directories claimed from the queue.
988    pub claims: u64,
989    /// Queue lock acquisitions, contended or not.
990    pub lock_ops: u64,
991    /// Lock acquisitions that found the lock already held.
992    pub lock_contended: u64,
993}
994
995impl WalkAttribution {
996    /// Fold one worker's counters into the whole-walk totals.
997    fn absorb(&mut self, other: Self) {
998        self.wall_ns += other.wall_ns;
999        self.work_ns += other.work_ns;
1000        self.starved_ns += other.starved_ns;
1001        self.lock_wait_ns += other.lock_wait_ns;
1002        self.send_ns += other.send_ns;
1003        self.claims += other.claims;
1004        self.lock_ops += other.lock_ops;
1005        self.lock_contended += other.lock_contended;
1006    }
1007
1008    /// Time attributed to a named cause, as opposed to `wall_ns`'s total.
1009    pub fn accounted_ns(&self) -> u64 {
1010        self.work_ns + self.starved_ns + self.lock_wait_ns + self.send_ns
1011    }
1012}
1013
1014/// Filesystem and index effects from an applying reconciliation pass.
1015#[derive(Debug, Default)]
1016pub struct ReconcileReport {
1017    /// Filesystem walk effects and partial errors.
1018    ///
1019    /// [`ScanReport::attribution`] remains zero for reconciliation because neither the
1020    /// serial nor parallel path has complete, comparable instrumentation yet. Zero
1021    /// means "not measured" here, not "no work".
1022    pub scan: ScanReport,
1023    /// Index arbitration and mutation effects.
1024    pub apply: ApplyStats,
1025    /// Exact producer operations considered, including no-op controls that do not
1026    /// increment an effect counter or create a commit.
1027    pub(crate) observations: u64,
1028    /// Directories this pass listed in full, with no error inside them, that the index did
1029    /// not yet hold as complete.
1030    ///
1031    /// The closing commit records each one's child set as authoritative, as discovery's
1032    /// own listing commit does, whether or not the rest of the pass completed: one transient
1033    /// child error
1034    /// elsewhere used to keep every directory the pass listed incomplete, and a directory
1035    /// first listed by such a pass stayed `Unknown { Building }` under a complete root.
1036    pub(crate) listed_incomplete: Vec<PathBuf>,
1037    /// Ownership epoch for conditional reconciliation batches.
1038    reconcile_epoch: Option<u64>,
1039    /// Retry after bounded verification evidence was superseded.
1040    retry_required: bool,
1041}
1042
1043impl ReconcileReport {
1044    /// True when the filesystem walk was complete and no conditional observation lost
1045    /// a race with another producer.
1046    pub fn is_complete(&self) -> bool {
1047        self.scan.is_complete()
1048            && self.apply.stale == 0
1049            && self.apply.resource_refused == 0
1050            && !self.retry_required
1051    }
1052
1053    /// True when a newer verification retired this pass's bounded evidence before it
1054    /// closed, so its scope is published partial and must be walked again.
1055    pub(crate) const fn retry_required(&self) -> bool {
1056        self.retry_required
1057    }
1058
1059    /// The directories whose listings this pass can vouch for, taken out of the report.
1060    ///
1061    /// None when a conditional commit lost a race or was refused: a child of any listed
1062    /// directory may then be missing from the index until the retry that race earns, and
1063    /// the retry records completeness for what it lists.
1064    pub(crate) fn take_recordable_completeness(&mut self) -> Vec<PathBuf> {
1065        let listed = std::mem::take(&mut self.listed_incomplete);
1066        if self.apply.stale > 0 || self.apply.resource_refused > 0 { Vec::new() } else { listed }
1067    }
1068}
1069
1070enum ReconcileTarget<'a> {
1071    Direct(&'a mut Index),
1072    Shared(&'a IndexHandle),
1073    Controlled { handle: &'a IndexHandle, control: &'a dyn ReconcileControl },
1074}
1075
1076/// Lifecycle checkpoints used by an owned long-running reconciliation.
1077///
1078/// The ordinary one-shot APIs use no controller. An [`crate::OpenedIndex`] supplies one
1079/// so close can stop a refresh before another write, and deterministic tests can pause
1080/// after filesystem verification but before conditional arbitration.
1081pub(crate) trait ReconcileControl {
1082    /// Fail when the owning operation may no longer publish state.
1083    fn check_active(&self) -> Result<()>;
1084
1085    /// Boundary after filesystem verification and before a conditional fact commit.
1086    fn before_conditional_commit(&self) -> Result<()>;
1087
1088    /// Atomic file-retention limit shared with every producer for this opened root.
1089    fn max_files(&self) -> Option<u64>;
1090}
1091
1092impl ReconcileTarget<'_> {
1093    fn scope(&self) -> Result<ScanScope> {
1094        match self {
1095            Self::Direct(index) => Ok(index.scope()),
1096            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.scope(),
1097        }
1098    }
1099
1100    fn root_path(&self) -> Result<PathBuf> {
1101        match self {
1102            Self::Direct(index) => Ok(index.root_path().to_path_buf()),
1103            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.root_path(),
1104        }
1105    }
1106
1107    fn expectation(&self, path: &Path) -> Result<PathExpectation> {
1108        match self {
1109            Self::Direct(index) => Ok(index.expectation(path)),
1110            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.expectation(path),
1111        }
1112    }
1113
1114    fn child_states(&self, path: &Path) -> Result<BTreeMap<OsString, PathExpectation>> {
1115        match self {
1116            Self::Direct(index) => Ok(collect_child_expectations(index, path)),
1117            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.child_states(path),
1118        }
1119    }
1120
1121    /// Child baselines for one directory listing, and whether a complete listing of it
1122    /// would be news to the index's directory completeness.
1123    ///
1124    /// A directory whose upsert has not been flushed yet is not held at all and counts as
1125    /// incomplete.
1126    fn listing_baseline(&self, path: &Path) -> Result<(BTreeMap<OsString, PathExpectation>, bool)> {
1127        match self {
1128            Self::Direct(index) => Ok((
1129                collect_child_expectations(index, path),
1130                index.directory_complete(path) != Some(true),
1131            )),
1132            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.listing_baseline(path),
1133        }
1134    }
1135
1136    fn has_control(&self, path: &Path) -> Result<bool> {
1137        match self {
1138            Self::Direct(index) => Ok(index.control_table().contains(path)),
1139            Self::Shared(handle) | Self::Controlled { handle, .. } => handle.has_control(path),
1140        }
1141    }
1142
1143    fn control_table(&self) -> Result<crate::control::ControlTable> {
1144        match self {
1145            Self::Direct(index) => Ok(index.control_table().clone()),
1146            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1147                handle.read_with(|index| index.control_table().clone())
1148            }
1149        }
1150    }
1151
1152    fn control_classification_known(&self, path: &Path) -> Result<bool> {
1153        match self {
1154            Self::Direct(index) => Ok(index.control_classification_known(path)),
1155            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1156                handle.read_with(|index| index.control_classification_known(path))
1157            }
1158        }
1159    }
1160
1161    fn apply(&mut self, started_at: u64, observation: &Observation) -> Result<crate::ApplyOutcome> {
1162        match self {
1163            Self::Direct(index) => index.apply(observation),
1164            Self::Shared(handle) => handle.apply_reconcile(started_at, observation),
1165            Self::Controlled { handle, control } => {
1166                control.before_conditional_commit()?;
1167                handle.apply_opened_reconcile(started_at, observation, control.max_files())
1168            }
1169        }
1170    }
1171
1172    fn direct_upsert_is_unchanged(
1173        &self,
1174        baseline: PathExpectation,
1175        kind: EntryKind,
1176        attrs: Attrs,
1177    ) -> bool {
1178        matches!(self, Self::Direct(_)) && baseline.state == (PathState::Present { kind, attrs })
1179    }
1180
1181    fn take_pending_invalidations(&mut self) -> Result<Vec<(PathBuf, crate::InvalidateReason)>> {
1182        match self {
1183            Self::Direct(index) => Ok(index.take_pending_invalidations()),
1184            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1185                handle.take_pending_invalidations()
1186            }
1187        }
1188    }
1189
1190    fn restore_pending_invalidations(
1191        &mut self,
1192        invalidations: Vec<(PathBuf, crate::InvalidateReason)>,
1193    ) -> Result<()> {
1194        match self {
1195            Self::Direct(index) => index.restore_pending_invalidations(invalidations),
1196            Self::Shared(handle) | Self::Controlled { handle, .. } => {
1197                handle.restore_pending_invalidations(invalidations)?;
1198            }
1199        }
1200        Ok(())
1201    }
1202
1203    /// Whether an invalidation whose reconciliation came back incomplete is queued again.
1204    ///
1205    /// A caller of the one-shot API owns its index exclusively and drains the queue when it
1206    /// chooses, so an unreadable subtree stays queued for it to retry. The shared API and an
1207    /// opened root are drained after every observed event -- by `Watcher::apply_next` and by
1208    /// the opened root's observer -- where that retry is a full walk of the same unreadable
1209    /// subtree per unrelated event, for the life of the session. There only a lost race is
1210    /// worth retrying: a stale conditional commit, or one the budget refused. A scan error
1211    /// is a settled boundary: the subtree stays partial, as it does at the observation
1212    /// handoff, and the report names the error once.
1213    fn retries_incomplete(&self, report: &ReconcileReport) -> bool {
1214        match self {
1215            Self::Direct(_) => !report.is_complete(),
1216            Self::Shared(_) | Self::Controlled { .. } => {
1217                report.apply.stale > 0 || report.apply.resource_refused > 0 || report.retry_required
1218            }
1219        }
1220    }
1221
1222    fn begin_reconcile(&mut self, path: &Path) -> Result<(u64, Option<Commit>)> {
1223        match self {
1224            Self::Direct(index) => index.begin_reconcile(path),
1225            Self::Shared(handle) => handle.begin_reconcile(path),
1226            Self::Controlled { handle, control } => {
1227                control.check_active()?;
1228                handle.begin_reconcile(path)
1229            }
1230        }
1231    }
1232
1233    fn finish_reconcile(
1234        &mut self,
1235        path: &Path,
1236        started_at: u64,
1237        complete: bool,
1238        listed_incomplete: &[PathBuf],
1239        failed_paths: &[PathBuf],
1240        errors: ReconcileErrors<'_>,
1241    ) -> Result<ReconcileFinish> {
1242        match self {
1243            Self::Direct(index) => index.finish_reconcile(
1244                path,
1245                started_at,
1246                complete,
1247                listed_incomplete,
1248                failed_paths,
1249                errors,
1250            ),
1251            Self::Shared(handle) => handle.finish_reconcile(
1252                path,
1253                started_at,
1254                complete,
1255                listed_incomplete,
1256                failed_paths,
1257                errors,
1258            ),
1259            Self::Controlled { handle, control } => {
1260                control.check_active()?;
1261                handle.finish_reconcile(
1262                    path,
1263                    started_at,
1264                    complete,
1265                    listed_incomplete,
1266                    failed_paths,
1267                    errors,
1268                )
1269            }
1270        }
1271    }
1272}
1273
1274#[cfg(unix)]
1275pub(crate) fn metadata_for_fingerprint(entry: &fs::DirEntry) -> std::io::Result<fs::Metadata> {
1276    crate::counters::bump(|c| c.stats += 1);
1277    entry.metadata()
1278}
1279
1280#[cfg(any(all(windows, test), not(any(unix, windows))))]
1281pub(crate) fn metadata_for_fingerprint(entry: &fs::DirEntry) -> std::io::Result<fs::Metadata> {
1282    crate::counters::bump(|c| c.stats += 1);
1283    // Windows serves DirEntry metadata from directory-enumeration data, which the
1284    // platform permits to be stale. Fingerprints need a fresh non-following query.
1285    fs::symlink_metadata(entry.path())
1286}
1287
1288#[cfg(test)]
1289type WalkHook = std::sync::Arc<dyn Fn(WalkHookPoint<'_>) -> Option<std::io::Error> + Send + Sync>;
1290
1291/// Where a test hook runs in a listing walk.
1292#[cfg(test)]
1293#[derive(Clone, Copy, Debug)]
1294pub(crate) enum WalkHookPoint<'a> {
1295    /// Before the metadata lookup of the listed child at this absolute path; an error
1296    /// stands in for the lookup's.
1297    ChildMetadata(&'a Path),
1298    /// After a reconciliation's listing of a directory returns its last entry; an error is
1299    /// read as one more listing item, which leaves the listing incomplete.
1300    ListingEnd,
1301    /// After a lookup of a directory's canonical control path, at this absolute path, has
1302    /// returned and before its answer is used; an error stands in for that answer.
1303    ControlLookup(&'a Path),
1304}
1305
1306/// Hooks run at each [`WalkHookPoint`], each for the paths under its root.
1307///
1308/// Process-wide, because a parallel walk looks children up on its worker threads; keyed
1309/// by root, because tests run in parallel and each walks its own temporary directory.
1310#[cfg(test)]
1311static WALK_HOOKS: std::sync::RwLock<Vec<(Vec<PathBuf>, WalkHook)>> =
1312    std::sync::RwLock::new(Vec::new());
1313
1314/// Removes its hook from [`WALK_HOOKS`] when dropped.
1315#[cfg(test)]
1316#[must_use = "the hook is removed as soon as the guard is dropped"]
1317pub(crate) struct WalkHookGuard(WalkHook);
1318
1319#[cfg(test)]
1320impl Drop for WalkHookGuard {
1321    fn drop(&mut self) {
1322        WALK_HOOKS
1323            .write()
1324            .unwrap_or_else(std::sync::PoisonError::into_inner)
1325            .retain(|(_, hook)| !std::sync::Arc::ptr_eq(hook, &self.0));
1326    }
1327}
1328
1329/// Run `hook` at every [`WalkHookPoint`] under `root`, on any thread, until the guard drops.
1330///
1331/// The hook may also change the tree before it returns. `root` matches as given and
1332/// canonical, since an opened root and a detached scan walk the canonical path.
1333#[cfg(test)]
1334pub(crate) fn install_walk_hook(
1335    root: &Path,
1336    hook: impl Fn(WalkHookPoint<'_>) -> Option<std::io::Error> + Send + Sync + 'static,
1337) -> WalkHookGuard {
1338    let mut roots = vec![root.to_path_buf()];
1339    if let Ok(canonical) = root.canonicalize() {
1340        roots.push(canonical);
1341    }
1342    let hook: WalkHook = std::sync::Arc::new(hook);
1343    WALK_HOOKS
1344        .write()
1345        .unwrap_or_else(std::sync::PoisonError::into_inner)
1346        .push((roots, std::sync::Arc::clone(&hook)));
1347    WalkHookGuard(hook)
1348}
1349
1350/// Run `hook` before every listed child's metadata lookup under `root`, with the child's
1351/// absolute path, until the guard drops.
1352#[cfg(test)]
1353pub(crate) fn install_child_metadata_hook(
1354    root: &Path,
1355    hook: impl Fn(&Path) -> Option<std::io::Error> + Send + Sync + 'static,
1356) -> WalkHookGuard {
1357    install_walk_hook(root, move |point| match point {
1358        WalkHookPoint::ChildMetadata(path) => hook(path),
1359        WalkHookPoint::ListingEnd | WalkHookPoint::ControlLookup(_) => None,
1360    })
1361}
1362
1363/// The hook installed for a root containing `path`, if any.
1364#[cfg(test)]
1365fn walk_hook(path: &Path) -> Option<WalkHook> {
1366    WALK_HOOKS
1367        .read()
1368        .unwrap_or_else(std::sync::PoisonError::into_inner)
1369        .iter()
1370        .find(|(roots, _)| roots.iter().any(|root| path.starts_with(root)))
1371        .map(|(_, hook)| std::sync::Arc::clone(hook))
1372}
1373
1374/// A reconciliation's listing of `dir`, followed by any error a test hook injects.
1375///
1376/// Callers bind the result to `listing` and iterate it as `for … in listing`, because the
1377/// admission audit (`scripts/check-admission-sites.mjs`) counts routed listing loops by that
1378/// shape. Keep the binding when editing a call site; dropping it silently removes the loop
1379/// from the audit, whose expected count would then look too high rather than wrong.
1380#[cfg(test)]
1381fn reconcile_listing(
1382    listing: fs::ReadDir,
1383    dir: &Path,
1384) -> impl Iterator<Item = std::io::Result<fs::DirEntry>> {
1385    let injected = walk_hook(dir).and_then(|hook| hook(WalkHookPoint::ListingEnd));
1386    listing.chain(injected.map(Err))
1387}
1388
1389/// A reconciliation's listing of `dir`.
1390#[cfg(not(test))]
1391fn reconcile_listing(listing: fs::ReadDir, _dir: &Path) -> fs::ReadDir {
1392    listing
1393}
1394
1395/// Metadata for one entry a directory listing returned, or `None` when it is gone.
1396///
1397/// `NotFound` for a name the listing just returned means the entry was deleted in
1398/// between, and every walk records it as it records a name the listing never returned: a
1399/// cold walk has nothing to record, and a reconciliation removes what its baseline held.
1400/// Reported as an error, it would make a walk over a tree being cleaned partial, and in a
1401/// reconciliation it would settle as a phantom entry with permanent partial freshness.
1402/// Any other error means the entry is present but unreadable.
1403#[cfg(not(windows))]
1404pub(crate) fn listed_child_metadata(entry: &fs::DirEntry) -> std::io::Result<Option<fs::Metadata>> {
1405    #[cfg(test)]
1406    {
1407        let path = entry.path();
1408        if let Some(error) =
1409            walk_hook(&path).and_then(|hook| hook(WalkHookPoint::ChildMetadata(&path)))
1410        {
1411            return missing_as_none(Err(error));
1412        }
1413    }
1414    missing_as_none(metadata_for_fingerprint(entry))
1415}
1416
1417/// Kind and attributes for one listed child.
1418///
1419/// The transient summary fold counts directories and ignores symlink attributes, so a
1420/// listing `file_type` (`d_type` on Linux) is enough for those kinds when the walk is
1421/// not bound to one filesystem. Files and specials still need a metadata lookup for
1422/// size, allocated bytes, and mtime. `one_filesystem` still stats directories because
1423/// descent compares `attrs.dev` to the root device, and `dev == 0` would otherwise
1424/// cross a mount.
1425///
1426/// Where `d_type` is `DT_UNKNOWN` (XFS without `ftype`, some FUSE/NFS mounts, older
1427/// ext3), std's `file_type` performs the non-following stat itself. The skip is a
1428/// no-op there, and the `stats` counter does not see that fallback.
1429///
1430/// Windows never takes the skip: its observation contract reads every listed entry
1431/// through a fresh non-following handle ([`observe_dir_entry`]), so the transient fold
1432/// there performs exactly the observations the retained walk performs.
1433fn listed_child_kind_and_attrs(
1434    entry: &fs::DirEntry,
1435    skip_dir_symlink_stat: bool,
1436    one_filesystem: bool,
1437) -> std::io::Result<Option<(EntryKind, Attrs)>> {
1438    #[cfg(not(windows))]
1439    {
1440        if skip_dir_symlink_stat {
1441            if let Ok(file_type) = entry.file_type() {
1442                if file_type.is_dir() && !one_filesystem {
1443                    return Ok(Some((EntryKind::Dir, Attrs::default())));
1444                }
1445                if file_type.is_symlink() {
1446                    return Ok(Some((EntryKind::Symlink, Attrs::default())));
1447                }
1448            }
1449        }
1450    }
1451    #[cfg(windows)]
1452    {
1453        let _ = (skip_dir_symlink_stat, one_filesystem);
1454    }
1455    observe_dir_entry(entry)
1456}
1457
1458fn missing_as_none<T>(lookup: std::io::Result<T>) -> std::io::Result<Option<T>> {
1459    match lookup {
1460        Ok(metadata) => Ok(Some(metadata)),
1461        Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(None),
1462        Err(error) => Err(error),
1463    }
1464}
1465
1466/// Whether a test hook observes lookups or listings under `path`, which a bulk read would
1467/// not make.
1468#[cfg(all(test, target_os = "macos"))]
1469fn walk_hook_covers(path: &Path) -> bool {
1470    walk_hook(path).is_some()
1471}
1472
1473#[cfg(all(not(test), target_os = "macos"))]
1474const fn walk_hook_covers(_path: &Path) -> bool {
1475    false
1476}
1477
1478/// Owned output from the filesystem walker before it crosses a public mutation boundary.
1479///
1480/// Only the scan and opened-discovery producers construct this type. Their admission,
1481/// depth, filesystem, and symlink checks have already selected every operation, and the
1482/// index consumes the owned paths while proving their parent identities under its write
1483/// boundary. Public scan callers receive an [`Observation`] instead and therefore keep
1484/// the full public normalization and atomic-validation contract.
1485#[derive(Debug)]
1486pub(crate) struct ScannerBatch {
1487    ops: Vec<ObservationOp>,
1488    /// When set, the consumer must return `ops` through this sender instead of dropping
1489    /// them. Workers allocate the `PathBuf`s; returning the drained vec lets glibc free
1490    /// those arenas on the producing thread. The public [`scan`] path leaves this unset.
1491    recycle: Option<std::sync::mpsc::Sender<Vec<ObservationOp>>>,
1492}
1493
1494impl ScannerBatch {
1495    pub(crate) const fn new(ops: Vec<ObservationOp>) -> Self {
1496        Self { ops, recycle: None }
1497    }
1498
1499    fn with_recycle(self, recycle: std::sync::mpsc::Sender<Vec<ObservationOp>>) -> Self {
1500        Self { recycle: Some(recycle), ..self }
1501    }
1502
1503    #[cfg(test)]
1504    pub(crate) fn from_ops(ops: Vec<Op>) -> Self {
1505        Self { ops: ops.into_iter().map(ObservationOp::unconditional).collect(), recycle: None }
1506    }
1507
1508    pub(crate) fn len(&self) -> usize {
1509        self.ops.len()
1510    }
1511
1512    pub(crate) fn ops(&self) -> &[ObservationOp] {
1513        &self.ops
1514    }
1515
1516    pub(crate) fn into_ops(self) -> Vec<ObservationOp> {
1517        self.ops
1518    }
1519
1520    fn into_observation(self) -> Observation {
1521        Observation::from_ops(self.ops)
1522    }
1523
1524    fn recycle(self) {
1525        if let Some(recycle) = self.recycle {
1526            let _ = recycle.send(self.ops);
1527        }
1528    }
1529}
1530
1531/// One direct child retained by the private detached cold-bootstrap builder.
1532///
1533/// The worker owns the component once. Unlike [`ScannerBatch`], this record does not
1534/// manufacture a full relative path or a public observation for every entry.
1535#[derive(Debug)]
1536pub(crate) struct DetachedChild {
1537    pub(crate) name: OsString,
1538    pub(crate) kind: EntryKind,
1539    pub(crate) attrs: Attrs,
1540    /// Enumeration order within the listing. An enumerator can repeat a name while its
1541    /// directory is modified, and the builder keeps the later observation, as a
1542    /// streaming re-upsert does.
1543    pub(crate) position: u32,
1544}
1545
1546/// One directory listing retained by a worker for detached bootstrap consolidation.
1547///
1548/// `path` is paid once per directory. Its children remain grouped exactly as the
1549/// filesystem enumerator produced them, so consolidation resolves the parent once and
1550/// never reconstructs a child path for nondirectories. A fixed control is retained
1551/// separately so the consumer can install the directory's complete control state
1552/// before it classifies any sibling or makes descendants visible.
1553#[derive(Debug)]
1554pub(crate) struct DetachedDirectory {
1555    pub(crate) path: PathBuf,
1556    pub(crate) children: Vec<DetachedChild>,
1557    pub(crate) control: Option<Op>,
1558}
1559
1560/// Walk `root` and emit observations describing everything found.
1561pub fn scan(
1562    root: &Path,
1563    config: &ScanConfig,
1564    sink: &mut dyn FnMut(Observation),
1565) -> Result<ScanReport> {
1566    let mut public_sink = |batch: ScannerBatch| sink(batch.into_observation());
1567    let (mut report, _diagnostics) = scan_internal(
1568        root,
1569        config,
1570        &mut public_sink,
1571        false,
1572        WorkerPolicyExperiment::ShippedOneShot,
1573        SinkMode::Retained,
1574    )?;
1575    normalize_walk_errors(root, &mut report.errors);
1576    Ok(report)
1577}
1578
1579/// Walk `root` for the transient summary tier, folding each op without retaining it.
1580///
1581/// The public [`scan`] path hands each batch to the caller as an [`Observation`], so
1582/// worker-allocated `PathBuf`s are freed on the consumer thread. This path returns
1583/// drained batches to the producing worker so each arena is allocated and freed on one
1584/// thread. Tallies must match [`scan`], and so must the normalized error set: the
1585/// summary report's status is built from these errors exactly as a retained walk's is.
1586/// A scan that reads `.gitignore` delivers each directory's control ahead of every entry
1587/// it governs, so the fold can classify each entry as the index would
1588/// ([`SinkMode::groups_directories`]).
1589pub(crate) fn scan_summary_fold(
1590    root: &Path,
1591    config: &ScanConfig,
1592    fold: &mut dyn FnMut(&ObservationOp),
1593) -> Result<ScanReport> {
1594    let mut sink = |batch: ScannerBatch| {
1595        for op in batch.ops() {
1596            fold(op);
1597        }
1598        batch.recycle();
1599    };
1600    let (mut report, _diagnostics) = scan_internal(
1601        root,
1602        config,
1603        &mut sink,
1604        false,
1605        WorkerPolicyExperiment::ShippedOneShot,
1606        SinkMode::TransientFold,
1607    )?;
1608    normalize_walk_errors(root, &mut report.errors);
1609    Ok(report)
1610}
1611
1612/// [`scan_summary_fold`] plus the diagnostic trace [`scan_with_diagnostics`] collects.
1613pub(crate) fn scan_summary_fold_with_diagnostics(
1614    root: &Path,
1615    config: &ScanConfig,
1616    fold: &mut dyn FnMut(&ObservationOp),
1617) -> Result<(ScanReport, ScanDiagnostics)> {
1618    let mut sink = |batch: ScannerBatch| {
1619        for op in batch.ops() {
1620            fold(op);
1621        }
1622        batch.recycle();
1623    };
1624    let (mut report, diagnostics) = scan_internal(
1625        root,
1626        config,
1627        &mut sink,
1628        true,
1629        WorkerPolicyExperiment::ShippedOneShot,
1630        SinkMode::TransientFold,
1631    )?;
1632    normalize_walk_errors(root, &mut report.errors);
1633    Ok((report, diagnostics.expect("diagnostic scan creates a recorder")))
1634}
1635
1636/// Walk `root`, emitting observations and a bounded run-scoped diagnostic trace.
1637///
1638/// This is the measurement counterpart to [`scan`]. It produces the same observation
1639/// stream and report while recording controller and backend evidence that ordinary
1640/// scans intentionally do not collect.
1641pub fn scan_with_diagnostics(
1642    root: &Path,
1643    config: &ScanConfig,
1644    sink: &mut dyn FnMut(Observation),
1645) -> Result<(ScanReport, ScanDiagnostics)> {
1646    scan_with_policy_diagnostics(root, config, sink, WorkerPolicyExperiment::ShippedOneShot)
1647}
1648
1649/// Exercise a repository-only worker-controller candidate and retain its trace.
1650#[doc(hidden)]
1651pub fn scan_with_policy_diagnostics(
1652    root: &Path,
1653    config: &ScanConfig,
1654    sink: &mut dyn FnMut(Observation),
1655    policy: WorkerPolicyExperiment,
1656) -> Result<(ScanReport, ScanDiagnostics)> {
1657    let mut public_sink = |batch: ScannerBatch| sink(batch.into_observation());
1658    let (mut report, diagnostics) =
1659        scan_internal(root, config, &mut public_sink, true, policy, SinkMode::Retained)?;
1660    normalize_walk_errors(root, &mut report.errors);
1661    Ok((report, diagnostics.expect("diagnostic scan creates a recorder")))
1662}
1663
1664/// What the caller does with each batch of observations.
1665///
1666/// Two measured keeps hang off this one concept, and both were measured on the
1667/// transient fold alone: returning drained batches to the producing worker (H147,
1668/// exp-151) and taking directory and symlink kind from the listing without a stat
1669/// (H72, exp-153). They are named here as properties of the mode rather than passed as
1670/// one flag under one of their names, so a measurement on another platform can move
1671/// one without silently moving the other.
1672///
1673/// A third property is semantic rather than measured: a transient fold that observes
1674/// `.gitignore` classifies on its consumer, and that needs each directory's control
1675/// before its entries ([`Self::groups_directories`]).
1676#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1677enum SinkMode {
1678    /// The consumer keeps the observations: the public [`scan`] and the index.
1679    Retained,
1680    /// The consumer folds each batch and drops it: the transient summary tier.
1681    TransientFold,
1682}
1683
1684impl SinkMode {
1685    /// Drained batches go back to the worker that allocated them (H147).
1686    fn recycles_batches(self) -> bool {
1687        self == Self::TransientFold
1688    }
1689
1690    /// Directory and symlink kind come from the listing without a stat (H72).
1691    fn skips_dir_symlink_stat(self) -> bool {
1692        self == Self::TransientFold
1693    }
1694
1695    /// Each directory's control reaches the consumer ahead of every entry it governs
1696    /// (fdu-1ovb).
1697    ///
1698    /// The transient summary classifies every entry against `.gitignore` on its
1699    /// consumer, as the detached index builder does, and the builder applies a
1700    /// directory's control before it classifies any child because it receives each
1701    /// listing whole. A streaming batch keeps listing order, where `.gitignore` can come
1702    /// last, so a worker holds a listing's observations and moves the control ahead of
1703    /// them when the listing ends, or, if the batch fills first, reads the directory's
1704    /// control directly and lets that read stand for the listing
1705    /// (`StreamingEmission::send_if_full`). The retained stream keeps listing order: the
1706    /// index reclassifies the subtree a control governs when that control arrives.
1707    fn groups_directories(self, config: &ScanConfig) -> bool {
1708        self == Self::TransientFold && config.read_controls
1709    }
1710}
1711
1712fn scan_internal(
1713    root: &Path,
1714    config: &ScanConfig,
1715    sink: &mut dyn FnMut(ScannerBatch),
1716    collect_diagnostics: bool,
1717    policy: WorkerPolicyExperiment,
1718    sink_mode: SinkMode,
1719) -> Result<(ScanReport, Option<ScanDiagnostics>)> {
1720    config.validate()?;
1721    if let Some(progress) = &config.progress {
1722        progress.enter(crate::ProgressPhase::Scanning);
1723    }
1724    let root_meta = {
1725        crate::counters::bump(|c| c.stats += 1);
1726        fs::symlink_metadata(root)
1727    }
1728    .map_err(|e| Error::io(root, e))?;
1729    if !root_meta.is_dir() {
1730        return Err(Error::io(
1731            root,
1732            std::io::Error::new(std::io::ErrorKind::NotADirectory, "scan root is not a directory"),
1733        ));
1734    }
1735    let root_dev = root_device(root, &root_meta).map_err(|error| Error::io(root, error))?;
1736    let available_parallelism =
1737        std::thread::available_parallelism().map_or(1, std::num::NonZero::get);
1738    // Narrow populations need control admission before child enumeration. Keep one ordered
1739    // producer and control table so a bounded budget makes the same decisions as the
1740    // index that consumes the observations.
1741    let pool = if config.population == crate::query::IgnoredEntries::Include {
1742        config.worker_pool_for(available_parallelism)
1743    } else {
1744        WorkerPool::fixed(1)
1745    };
1746    let diagnostics = collect_diagnostics
1747        .then(|| ScanDiagnosticsRecorder::new(pool, available_parallelism, policy));
1748
1749    // The serial walk below emits in listing order and never groups a directory, so a
1750    // transient fold that classifies takes the concurrent walk even with one worker. That
1751    // is the walk the detached index takes at every worker count, and one worker visits
1752    // directories in the order that builder consumes them, so a control budget admits
1753    // the same files on both routes. A narrowed population keeps the serial walk, which
1754    // reads each directory's control before listing it.
1755    let groups = sink_mode.groups_directories(config)
1756        && config.population == crate::query::IgnoredEntries::Include;
1757    if config.max_depth != Some(0) && (pool.initial > 1 || groups) {
1758        let report = scan_concurrent(
1759            root,
1760            config,
1761            root_dev,
1762            sink,
1763            pool,
1764            diagnostics.as_ref(),
1765            policy,
1766            sink_mode,
1767        );
1768        return Ok((report, diagnostics.as_ref().map(|value| value.finish())));
1769    }
1770
1771    let mut report = ScanReport::default();
1772    if config.max_depth == Some(0) {
1773        if let Some(diagnostics) = &diagnostics {
1774            diagnostics.mark_not_run();
1775            diagnostics.record_queue_finish(0, 0);
1776        }
1777        return Ok((report, diagnostics.as_ref().map(|value| value.finish())));
1778    }
1779    let worker_guard = diagnostics.as_ref().map(ScanDiagnosticsRecorder::worker_guard);
1780    let walk_started = std::time::Instant::now();
1781    let mut batch: Vec<ObservationOp> = Vec::with_capacity(config.batch_size);
1782    let mut queue: VecDeque<(PathBuf, usize)> = VecDeque::from(vec![(PathBuf::new(), 0)]);
1783    let mut controls = (config.population != crate::query::IgnoredEntries::Include)
1784        .then(|| crate::control::ControlTable::with_limits(config.control_limits));
1785    let mut unreadable_controls = std::collections::BTreeSet::new();
1786    let mut tally = ProgressTally::new(config.progress.as_ref());
1787    // Every batch leaves through here, so the batch is where the serial walk reports
1788    // its progress: the handoff the consumer already pays for, never the entry.
1789    let mut emit = |ops: Vec<ObservationOp>, report: &mut ScanReport| {
1790        let send_started = std::time::Instant::now();
1791        sink(ScannerBatch::new(ops));
1792        report.attribution.send_ns += elapsed_ns(send_started);
1793        tally.flush(report);
1794    };
1795
1796    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
1797        let abs_dir = root.join(&rel_dir);
1798        if let Some(controls) = controls.as_mut() {
1799            let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
1800            match read_directory_control(config, root, &control_path) {
1801                Ok(Some(op)) => {
1802                    apply_discovery_control(controls, &op)?;
1803                    batch.push(ObservationOp::unconditional(op));
1804                }
1805                Ok(None) => {}
1806                Err(error) => {
1807                    unreadable_controls.insert(rel_dir.clone());
1808                    report.errors.push(error);
1809                }
1810            }
1811        }
1812        crate::counters::bump(|c| c.dir_opens += 1);
1813        if let Some(diagnostics) = &diagnostics {
1814            diagnostics.portable_attempted();
1815        }
1816        let listing = match fs::read_dir(&abs_dir) {
1817            Ok(listing) => {
1818                if let Some(diagnostics) = &diagnostics {
1819                    diagnostics.portable_succeeded();
1820                }
1821                listing
1822            }
1823            Err(e) => {
1824                report.errors.push(Error::io(abs_dir, e));
1825                continue;
1826            }
1827        };
1828        report.dirs_read += 1;
1829
1830        for item in listing {
1831            let item = match item {
1832                Ok(item) => item,
1833                Err(e) => {
1834                    report.errors.push(Error::io(&abs_dir, e));
1835                    continue;
1836                }
1837            };
1838            crate::counters::bump(|c| c.dir_entries += 1);
1839            let name = item.file_name();
1840            let rel_path = rel_dir.join(&name);
1841            let (kind, attrs) = match listed_child_kind_and_attrs(
1842                &item,
1843                sink_mode.skips_dir_symlink_stat(),
1844                config.one_filesystem,
1845            ) {
1846                Ok(Some(observed)) => observed,
1847                Ok(None) => continue,
1848                Err(error) => {
1849                    report.errors.push(Error::io(item.path(), error));
1850                    continue;
1851                }
1852            };
1853            let disposition =
1854                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
1855            if disposition == crate::admission::Disposition::Reject {
1856                continue;
1857            }
1858            if population_prunes(
1859                config.population,
1860                &rel_path,
1861                kind,
1862                disposition,
1863                controls.as_ref(),
1864                &unreadable_controls,
1865            ) {
1866                continue;
1867            }
1868            // A narrowed walk looked the directory's control up before listing it, and that
1869            // lookup stands for every spelling the listing shows.
1870            let control =
1871                match if controls.is_some() && crate::control::control_spelling(&name).is_some() {
1872                    Ok(None)
1873                } else {
1874                    read_control_op(config, root, &rel_path, kind)
1875                } {
1876                    Ok(control) => control,
1877                    Err(error) => {
1878                        report.errors.push(error);
1879                        None
1880                    }
1881                };
1882            if disposition == crate::admission::Disposition::ControlOnly {
1883                if let Some(control) = control {
1884                    batch.push(ObservationOp::unconditional(control));
1885                    if batch.len() >= config.batch_size {
1886                        emit(std::mem::take(&mut batch), &mut report);
1887                        batch.reserve(config.batch_size);
1888                    }
1889                }
1890                continue;
1891            }
1892            report.observe(kind, attrs);
1893            batch.push(ObservationOp::unconditional(Op::Upsert {
1894                path: rel_path.clone(),
1895                kind,
1896                attrs,
1897            }));
1898            if batch.len() >= config.batch_size {
1899                emit(std::mem::take(&mut batch), &mut report);
1900                batch.reserve(config.batch_size);
1901            }
1902            if let Some(control) = control {
1903                batch.push(ObservationOp::unconditional(control));
1904                if batch.len() >= config.batch_size {
1905                    emit(std::mem::take(&mut batch), &mut report);
1906                    batch.reserve(config.batch_size);
1907                }
1908            }
1909
1910            if should_descend(kind, attrs, depth, root_dev, config) {
1911                queue.push_back((rel_path, depth + 1));
1912            }
1913        }
1914    }
1915
1916    if !batch.is_empty() {
1917        emit(batch, &mut report);
1918    }
1919    // A walk whose last directories filled no batch has counted them and sent nothing.
1920    tally.flush(&report);
1921    // A serial walk has no coordination to attribute: wall is the loop, "send" is the
1922    // inline sink — which is the consumer actually running — and work is the rest.
1923    report.attribution.wall_ns = elapsed_ns(walk_started);
1924    report.attribution.work_ns =
1925        report.attribution.wall_ns.saturating_sub(report.attribution.send_ns);
1926    drop(worker_guard);
1927    if let Some(diagnostics) = &diagnostics {
1928        diagnostics.record_queue_finish(0, 0);
1929    }
1930    Ok((report, diagnostics.as_ref().map(|value| value.finish())))
1931}
1932
1933/// Take the next directory in the configured order.
1934///
1935/// Both orders push to the back; only the end they are taken from differs, which is
1936/// what keeps this a one-line policy rather than two walkers.
1937fn take_next(queue: &mut VecDeque<(PathBuf, usize)>, order: ScanOrder) -> Option<(PathBuf, usize)> {
1938    match order {
1939        ScanOrder::BreadthFirst => queue.pop_front(),
1940        ScanOrder::DepthFirst => queue.pop_back(),
1941    }
1942}
1943
1944/// Largest worker pool a caller may ask for explicitly.
1945///
1946/// Well past anything measured to help. It exists so a caller that computes a thread
1947/// count from something silly cannot spawn thousands of threads.
1948const MAX_SCAN_THREADS: usize = 32;
1949
1950/// Ceiling on the workers active at the start of an automatic scan.
1951///
1952/// Measured, not guessed. On a 10-core machine walking a 60k-entry `node_modules`
1953/// tree, wall time fell 37% at two workers and 50% at four, then stopped improving:
1954/// six matched four within noise and eight was 4% worse than four. The walk becomes
1955/// bound by the single index consumer, so past this point extra workers buy queue
1956/// contention and efficiency-core scheduling rather than throughput. See
1957/// `docs/project/reports/report-2026-08-10-fdu-performance-experiments.md`.
1958///
1959/// That measurement was taken on macOS, and every constant in this group now reads its
1960/// value from [`crate::platform_tuning`], which records per platform whether the number
1961/// was measured there or inherited. On Linux these are inherited.
1962const DEFAULT_SCAN_THREADS_CAP: usize = crate::platform_tuning::tuning().scan_threads_cap.get();
1963
1964/// Ceiling on automatic workers for an immutable-baseline reconciliation wave.
1965///
1966/// Reconciliation reads both filesystem and index state. Its measured knee arrives
1967/// before the cold producer's because additional metadata calls amplify kernel work
1968/// after the index comparisons already saturate the performance cores.
1969const DEFAULT_RECONCILE_THREADS_CAP: usize =
1970    crate::platform_tuning::tuning().reconcile_threads_cap.get();
1971
1972/// Ceiling an automatic scan may unlock after it establishes that the tree is large.
1973///
1974/// Sixteen was the knee on the 720k-entry cache-pressure corpus in exp-015. Thirty-two
1975/// did not improve on it and spent substantially more worker time waiting at the end.
1976const ADAPTIVE_SCAN_THREADS_CAP: usize =
1977    crate::platform_tuning::tuning().adaptive_scan_threads_cap.get();
1978
1979/// Maximum reserve depth relative to the host's reported parallelism.
1980const ADAPTIVE_SCAN_PARALLELISM_MULTIPLIER: usize =
1981    crate::platform_tuning::tuning().adaptive_scan_parallelism_multiplier.get();
1982
1983/// Entries used to calibrate the initial workers' filesystem service time.
1984const ADAPTIVE_SCAN_CALIBRATION_ENTRIES: u64 =
1985    crate::platform_tuning::tuning().adaptive_scan_calibration_entries.get();
1986
1987/// Average worker time per observed entry that identifies a latency-bound scan.
1988///
1989/// Whole-run attribution separated the measured regimes: roughly 18 microseconds on
1990/// the 60k tree, 22 on the 120k boundary, and 42 or more on the 720k cache-pressure
1991/// tree. Thirty leaves margin between them. The calibration uses the same chunk timing
1992/// already collected for attribution, so it adds no per-entry clock reads.
1993///
1994/// Those are APFS regimes. The Linux warm floor is about 1.5 µs per entry, twenty times
1995/// below this threshold, so the trigger may never fire there — which is exactly the kind
1996/// of inherited constant [`crate::platform_tuning`] exists to make visible (H84).
1997const ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY: u64 =
1998    crate::platform_tuning::tuning().adaptive_scan_slow_work_ns_per_entry.get();
1999
2000#[derive(Clone, Copy, Debug, PartialEq, Eq)]
2001struct WorkerPool {
2002    initial: usize,
2003    maximum: usize,
2004    calibration: Option<WorkerCalibration>,
2005}
2006
2007#[derive(Clone, Copy, Debug, PartialEq, Eq)]
2008struct WorkerCalibration {
2009    minimum_entries: u64,
2010    slow_work_ns_per_entry: u64,
2011    entries: u64,
2012    work_ns: u64,
2013    chunks: u64,
2014}
2015
2016#[derive(Clone, Copy, Debug)]
2017struct PolicyWindowSnapshot {
2018    sequence: u64,
2019    start_entry_ordinal: u64,
2020    end_entry_ordinal: u64,
2021    observed_entries: u64,
2022    observed_chunks: u64,
2023    observed_work_ns: u64,
2024    ready_directories: usize,
2025    in_flight_directories: usize,
2026    active_workers: usize,
2027    handoff_backlog: usize,
2028    requested_workers: Option<usize>,
2029    decision: WorkerPolicyDecision,
2030}
2031
2032struct PolicyTraceState {
2033    outcome: WorkerPolicyOutcome,
2034    outcome_reason: Option<&'static str>,
2035    outcome_sequence: Option<u64>,
2036    windows: Vec<WorkerPolicyWindow>,
2037    events_truncated: bool,
2038    ready_directories_at_finish: usize,
2039    in_flight_directories_at_finish: usize,
2040}
2041
2042/// Shared state used only by the opt-in diagnostic scan APIs.
2043///
2044/// Normal scans pass no recorder and therefore never touch these atomics or locks. The
2045/// trace mutex is deliberately separate from the directory queue: recording a policy
2046/// window may add diagnostic cost, but it cannot alter the queue's synchronization or
2047/// the controller's decision.
2048struct ScanDiagnosticsRecorder {
2049    available_parallelism: usize,
2050    pool: WorkerPool,
2051    policy: WorkerPolicyExperiment,
2052    trace: std::sync::Mutex<PolicyTraceState>,
2053    workers_spawned: std::sync::atomic::AtomicUsize,
2054    active_workers: std::sync::atomic::AtomicUsize,
2055    peak_active_workers: std::sync::atomic::AtomicUsize,
2056    handoff_backlog: std::sync::atomic::AtomicUsize,
2057    handoff_backlog_high_water: std::sync::atomic::AtomicUsize,
2058    calibration_chunks: std::sync::atomic::AtomicU64,
2059    calibration_entries: std::sync::atomic::AtomicU64,
2060    calibration_work_ns: std::sync::atomic::AtomicU64,
2061    worker_expansions: std::sync::atomic::AtomicU64,
2062    portable_attempts: std::sync::atomic::AtomicU64,
2063    portable_successes: std::sync::atomic::AtomicU64,
2064    #[cfg(target_os = "macos")]
2065    macos_bulk_attempts: std::sync::atomic::AtomicU64,
2066    #[cfg(target_os = "macos")]
2067    macos_bulk_successes: std::sync::atomic::AtomicU64,
2068    #[cfg(target_os = "macos")]
2069    macos_bulk_fallbacks: std::sync::atomic::AtomicU64,
2070}
2071
2072impl ScanDiagnosticsRecorder {
2073    fn new(
2074        pool: WorkerPool,
2075        available_parallelism: usize,
2076        policy: WorkerPolicyExperiment,
2077    ) -> std::sync::Arc<Self> {
2078        let (outcome, outcome_reason) = if pool.calibration.is_some() {
2079            (
2080                WorkerPolicyOutcome::Undecided,
2081                Some("the adaptive calibration window has not completed"),
2082            )
2083        } else {
2084            (WorkerPolicyOutcome::Fixed, Some("this worker pool has no adaptive reserve"))
2085        };
2086        std::sync::Arc::new(Self {
2087            available_parallelism,
2088            pool,
2089            policy,
2090            trace: std::sync::Mutex::new(PolicyTraceState {
2091                outcome,
2092                outcome_reason,
2093                outcome_sequence: None,
2094                windows: Vec::new(),
2095                events_truncated: false,
2096                ready_directories_at_finish: 0,
2097                in_flight_directories_at_finish: 0,
2098            }),
2099            workers_spawned: std::sync::atomic::AtomicUsize::new(0),
2100            active_workers: std::sync::atomic::AtomicUsize::new(0),
2101            peak_active_workers: std::sync::atomic::AtomicUsize::new(0),
2102            handoff_backlog: std::sync::atomic::AtomicUsize::new(0),
2103            handoff_backlog_high_water: std::sync::atomic::AtomicUsize::new(0),
2104            calibration_chunks: std::sync::atomic::AtomicU64::new(0),
2105            calibration_entries: std::sync::atomic::AtomicU64::new(0),
2106            calibration_work_ns: std::sync::atomic::AtomicU64::new(0),
2107            worker_expansions: std::sync::atomic::AtomicU64::new(0),
2108            portable_attempts: std::sync::atomic::AtomicU64::new(0),
2109            portable_successes: std::sync::atomic::AtomicU64::new(0),
2110            #[cfg(target_os = "macos")]
2111            macos_bulk_attempts: std::sync::atomic::AtomicU64::new(0),
2112            #[cfg(target_os = "macos")]
2113            macos_bulk_successes: std::sync::atomic::AtomicU64::new(0),
2114            #[cfg(target_os = "macos")]
2115            macos_bulk_fallbacks: std::sync::atomic::AtomicU64::new(0),
2116        })
2117    }
2118
2119    fn worker_guard(self: &std::sync::Arc<Self>) -> ScanWorkerGuard {
2120        let active = self
2121            .active_workers
2122            .fetch_add(1, std::sync::atomic::Ordering::Relaxed)
2123            .saturating_add(1);
2124        self.workers_spawned.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2125        atomic_update_max(&self.peak_active_workers, active);
2126        ScanWorkerGuard { recorder: self.clone() }
2127    }
2128
2129    fn record_policy_window(&self, snapshot: PolicyWindowSnapshot) {
2130        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2131        let sequence = snapshot.sequence;
2132        let supersedes = trace.outcome_sequence.is_none_or(|current| sequence >= current);
2133        match snapshot.decision {
2134            WorkerPolicyDecision::Undecided if supersedes => {
2135                trace.outcome = WorkerPolicyOutcome::Undecided;
2136                trace.outcome_reason =
2137                    Some("the walk ended before the adaptive calibration window completed");
2138                trace.outcome_sequence = Some(sequence);
2139            }
2140            WorkerPolicyDecision::Hold
2141                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2142            {
2143                trace.outcome = WorkerPolicyOutcome::Held;
2144                trace.outcome_reason = None;
2145                trace.outcome_sequence = Some(sequence);
2146            }
2147            WorkerPolicyDecision::ScaleUp => {
2148                trace.outcome = WorkerPolicyOutcome::ScaledUp;
2149                trace.outcome_reason = None;
2150                trace.outcome_sequence = Some(sequence);
2151            }
2152            WorkerPolicyDecision::HoldNoUsefulWork
2153                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2154            {
2155                trace.outcome = WorkerPolicyOutcome::HeldNoUsefulWork;
2156                trace.outcome_reason =
2157                    Some("the slow trigger fired only after ready and in-flight work had drained");
2158                trace.outcome_sequence = Some(sequence);
2159            }
2160            WorkerPolicyDecision::HoldInsufficientFrontier
2161                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2162            {
2163                trace.outcome = WorkerPolicyOutcome::Held;
2164                trace.outcome_reason =
2165                    Some("the observed frontier could not use additional workers");
2166                trace.outcome_sequence = Some(sequence);
2167            }
2168            WorkerPolicyDecision::HoldHandoffBacklog
2169                if trace.outcome != WorkerPolicyOutcome::ScaledUp && supersedes =>
2170            {
2171                trace.outcome = WorkerPolicyOutcome::Held;
2172                trace.outcome_reason =
2173                    Some("the observation handoff backlog was already at the controller limit");
2174                trace.outcome_sequence = Some(sequence);
2175            }
2176            WorkerPolicyDecision::Incomplete
2177                if supersedes && trace.outcome == WorkerPolicyOutcome::Undecided =>
2178            {
2179                trace.outcome_reason =
2180                    Some("the walk ended before any adaptive calibration window completed");
2181                trace.outcome_sequence = Some(sequence);
2182            }
2183            _ => {}
2184        }
2185        if sequence >= MAX_POLICY_TRACE_EVENTS as u64 {
2186            trace.events_truncated = true;
2187            return;
2188        }
2189        let work_ns_per_entry = (snapshot.observed_entries > 0)
2190            .then(|| snapshot.observed_work_ns / snapshot.observed_entries);
2191        trace.windows.push(WorkerPolicyWindow {
2192            sequence,
2193            start_entry_ordinal: snapshot.start_entry_ordinal,
2194            end_entry_ordinal: snapshot.end_entry_ordinal,
2195            observed_entries: snapshot.observed_entries,
2196            observed_chunks: snapshot.observed_chunks,
2197            observed_work_ns: snapshot.observed_work_ns,
2198            work_ns_per_entry,
2199            work_ns_per_entry_unavailable_reason: work_ns_per_entry
2200                .is_none()
2201                .then_some("the window observed no entries"),
2202            ready_directories: snapshot.ready_directories,
2203            in_flight_directories: snapshot.in_flight_directories,
2204            active_workers: snapshot.active_workers,
2205            handoff_backlog: snapshot.handoff_backlog,
2206            requested_workers: snapshot.requested_workers,
2207            decision: snapshot.decision,
2208        });
2209    }
2210
2211    fn mark_not_run(&self) {
2212        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2213        trace.outcome = WorkerPolicyOutcome::NotRun;
2214        trace.outcome_reason = Some("max_depth zero requested no directory walk");
2215    }
2216
2217    fn record_queue_finish(&self, ready_directories: usize, in_flight_directories: usize) {
2218        let mut trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2219        trace.ready_directories_at_finish = ready_directories;
2220        trace.in_flight_directories_at_finish = in_flight_directories;
2221    }
2222
2223    fn handoff_sent(&self) {
2224        let backlog = self
2225            .handoff_backlog
2226            .fetch_add(1, std::sync::atomic::Ordering::Relaxed)
2227            .saturating_add(1);
2228        atomic_update_max(&self.handoff_backlog_high_water, backlog);
2229    }
2230
2231    fn handoff_received(&self) {
2232        let previous = self.handoff_backlog.fetch_sub(1, std::sync::atomic::Ordering::Relaxed);
2233        debug_assert!(previous > 0, "received handoff must have been sent");
2234    }
2235
2236    fn calibration_chunk(&self, entries: u64, work_ns: u64) {
2237        self.calibration_chunks.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2238        self.calibration_entries.fetch_add(entries, std::sync::atomic::Ordering::Relaxed);
2239        self.calibration_work_ns.fetch_add(work_ns, std::sync::atomic::Ordering::Relaxed);
2240    }
2241
2242    fn worker_expanded(&self) {
2243        self.worker_expansions.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2244    }
2245
2246    fn portable_attempted(&self) {
2247        self.portable_attempts.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2248    }
2249
2250    fn portable_succeeded(&self) {
2251        self.portable_successes.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2252    }
2253
2254    #[cfg(target_os = "macos")]
2255    fn macos_bulk_attempted(&self) {
2256        self.macos_bulk_attempts.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2257    }
2258
2259    #[cfg(target_os = "macos")]
2260    fn macos_bulk_succeeded(&self) {
2261        self.macos_bulk_successes.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2262    }
2263
2264    #[cfg(target_os = "macos")]
2265    fn macos_bulk_fell_back(&self) {
2266        self.macos_bulk_fallbacks.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
2267    }
2268
2269    fn finish(&self) -> ScanDiagnostics {
2270        let trace = self.trace.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
2271        let calibration = self.pool.calibration;
2272        let backend = ScanBackendDiagnostics {
2273            portable_attempts: self.portable_attempts.load(std::sync::atomic::Ordering::Relaxed),
2274            portable_directory_reads: self
2275                .portable_successes
2276                .load(std::sync::atomic::Ordering::Relaxed),
2277            #[cfg(target_os = "macos")]
2278            macos_bulk_attempts: Some(
2279                self.macos_bulk_attempts.load(std::sync::atomic::Ordering::Relaxed),
2280            ),
2281            #[cfg(not(target_os = "macos"))]
2282            macos_bulk_attempts: None,
2283            #[cfg(target_os = "macos")]
2284            macos_bulk_successes: Some(
2285                self.macos_bulk_successes.load(std::sync::atomic::Ordering::Relaxed),
2286            ),
2287            #[cfg(not(target_os = "macos"))]
2288            macos_bulk_successes: None,
2289            #[cfg(target_os = "macos")]
2290            macos_bulk_fallbacks: Some(
2291                self.macos_bulk_fallbacks.load(std::sync::atomic::Ordering::Relaxed),
2292            ),
2293            #[cfg(not(target_os = "macos"))]
2294            macos_bulk_fallbacks: None,
2295            #[cfg(target_os = "macos")]
2296            unavailable_reason: None,
2297            #[cfg(not(target_os = "macos"))]
2298            unavailable_reason: Some(
2299                "macOS bulk directory enumeration is unavailable on this platform",
2300            ),
2301        };
2302        ScanDiagnostics {
2303            schema: SCAN_DIAGNOSTICS_SCHEMA,
2304            worker_policy: WorkerPolicyDiagnostics {
2305                controller: worker_policy_experiment_name(self.policy),
2306                available_parallelism: self.available_parallelism,
2307                initial_workers: self.pool.initial,
2308                maximum_workers: self.pool.maximum,
2309                calibration_window_entries: calibration.map(|value| value.minimum_entries),
2310                slow_threshold_ns_per_entry: calibration.map(|value| value.slow_work_ns_per_entry),
2311                calibration_chunks: self
2312                    .calibration_chunks
2313                    .load(std::sync::atomic::Ordering::Relaxed),
2314                calibration_entries: self
2315                    .calibration_entries
2316                    .load(std::sync::atomic::Ordering::Relaxed),
2317                calibration_work_ns: self
2318                    .calibration_work_ns
2319                    .load(std::sync::atomic::Ordering::Relaxed),
2320                worker_expansions: self
2321                    .worker_expansions
2322                    .load(std::sync::atomic::Ordering::Relaxed),
2323                outcome: trace.outcome,
2324                outcome_reason: trace.outcome_reason,
2325                workers_spawned: self.workers_spawned.load(std::sync::atomic::Ordering::Relaxed),
2326                peak_active_workers: self
2327                    .peak_active_workers
2328                    .load(std::sync::atomic::Ordering::Relaxed),
2329                ready_directories_at_finish: trace.ready_directories_at_finish,
2330                in_flight_directories_at_finish: trace.in_flight_directories_at_finish,
2331                handoff_backlog_at_finish: self
2332                    .handoff_backlog
2333                    .load(std::sync::atomic::Ordering::Relaxed),
2334                handoff_backlog_high_water: self
2335                    .handoff_backlog_high_water
2336                    .load(std::sync::atomic::Ordering::Relaxed),
2337                windows: {
2338                    let mut windows = trace.windows.clone();
2339                    windows.sort_by_key(|window| window.sequence);
2340                    windows
2341                },
2342                events_truncated: trace.events_truncated,
2343            },
2344            backend,
2345        }
2346    }
2347}
2348
2349const fn worker_policy_experiment_name(value: WorkerPolicyExperiment) -> &'static str {
2350    match value {
2351        WorkerPolicyExperiment::ShippedOneShot => "shipped_one_shot",
2352        WorkerPolicyExperiment::RepeatedWindows => "repeated_windows",
2353        WorkerPolicyExperiment::StagedGatedWindows => "staged_gated_windows",
2354    }
2355}
2356
2357struct ScanWorkerGuard {
2358    recorder: std::sync::Arc<ScanDiagnosticsRecorder>,
2359}
2360
2361impl Drop for ScanWorkerGuard {
2362    fn drop(&mut self) {
2363        let previous =
2364            self.recorder.active_workers.fetch_sub(1, std::sync::atomic::Ordering::Relaxed);
2365        debug_assert!(previous > 0, "worker guard must balance worker start");
2366    }
2367}
2368
2369fn atomic_update_max(target: &std::sync::atomic::AtomicUsize, value: usize) {
2370    let mut observed = target.load(std::sync::atomic::Ordering::Relaxed);
2371    while value > observed {
2372        match target.compare_exchange_weak(
2373            observed,
2374            value,
2375            std::sync::atomic::Ordering::Relaxed,
2376            std::sync::atomic::Ordering::Relaxed,
2377        ) {
2378            Ok(_) => break,
2379            Err(actual) => observed = actual,
2380        }
2381    }
2382}
2383
2384enum WalkMessage {
2385    Batch(ScannerBatch),
2386    DetachedDirectories {
2387        directories: Vec<DetachedDirectory>,
2388        /// The worker that allocated `directories`. The consumer drains each listing
2389        /// into the index and sends the emptied listings back here, so their path and
2390        /// child buffers are freed or reused on that worker's thread (H159).
2391        recycle: std::sync::mpsc::Sender<Vec<DetachedDirectory>>,
2392    },
2393    ScaleUp {
2394        sender: std::sync::mpsc::Sender<Self>,
2395        target_workers: usize,
2396    },
2397}
2398
2399impl WorkerPool {
2400    const fn fixed(workers: usize) -> Self {
2401        Self { initial: workers, maximum: workers, calibration: None }
2402    }
2403}
2404
2405impl WorkerCalibration {
2406    const fn new(minimum_entries: u64, slow_work_ns_per_entry: u64) -> Self {
2407        Self { minimum_entries, slow_work_ns_per_entry, entries: 0, work_ns: 0, chunks: 0 }
2408    }
2409
2410    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<bool> {
2411        self.chunks = self.chunks.saturating_add(1);
2412        self.entries = self.entries.saturating_add(entries);
2413        self.work_ns = self.work_ns.saturating_add(work_ns);
2414        (self.entries >= self.minimum_entries)
2415            .then(|| self.work_ns / self.entries >= self.slow_work_ns_per_entry)
2416    }
2417}
2418
2419#[derive(Clone, Copy, Debug)]
2420struct CalibrationWindow {
2421    start_entry_ordinal: u64,
2422    end_entry_ordinal: u64,
2423    entries: u64,
2424    chunks: u64,
2425    work_ns: u64,
2426    slow: bool,
2427}
2428
2429#[derive(Debug)]
2430struct RepeatedCalibration {
2431    minimum_entries: u64,
2432    slow_work_ns_per_entry: u64,
2433    window_start: u64,
2434    entries: u64,
2435    chunks: u64,
2436    work_ns: u64,
2437    completed_windows: u64,
2438}
2439
2440impl RepeatedCalibration {
2441    const fn new(calibration: WorkerCalibration) -> Self {
2442        Self {
2443            minimum_entries: calibration.minimum_entries,
2444            slow_work_ns_per_entry: calibration.slow_work_ns_per_entry,
2445            window_start: 0,
2446            entries: 0,
2447            chunks: 0,
2448            work_ns: 0,
2449            completed_windows: 0,
2450        }
2451    }
2452
2453    const fn starting_at(calibration: WorkerCalibration, window_start: u64) -> Self {
2454        let mut repeated = Self::new(calibration);
2455        repeated.window_start = window_start;
2456        repeated
2457    }
2458
2459    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<CalibrationWindow> {
2460        self.chunks = self.chunks.saturating_add(1);
2461        self.entries = self.entries.saturating_add(entries);
2462        self.work_ns = self.work_ns.saturating_add(work_ns);
2463        if self.entries < self.minimum_entries {
2464            return None;
2465        }
2466        let end_entry_ordinal = self.window_start.saturating_add(self.entries);
2467        let window = CalibrationWindow {
2468            start_entry_ordinal: self.window_start,
2469            end_entry_ordinal,
2470            entries: self.entries,
2471            chunks: self.chunks,
2472            work_ns: self.work_ns,
2473            slow: self.work_ns / self.entries >= self.slow_work_ns_per_entry,
2474        };
2475        self.window_start = end_entry_ordinal;
2476        self.entries = 0;
2477        self.chunks = 0;
2478        self.work_ns = 0;
2479        self.completed_windows = self.completed_windows.saturating_add(1);
2480        Some(window)
2481    }
2482}
2483
2484#[derive(Debug)]
2485enum WorkerController {
2486    OneShot(WorkerCalibration),
2487    Repeated { calibration: RepeatedCalibration, staged_gated: bool },
2488}
2489
2490impl WorkerController {
2491    fn new(calibration: WorkerCalibration, policy: WorkerPolicyExperiment) -> Self {
2492        match policy {
2493            WorkerPolicyExperiment::ShippedOneShot => Self::OneShot(calibration),
2494            WorkerPolicyExperiment::RepeatedWindows => Self::Repeated {
2495                calibration: RepeatedCalibration::new(calibration),
2496                staged_gated: false,
2497            },
2498            WorkerPolicyExperiment::StagedGatedWindows => Self::Repeated {
2499                calibration: RepeatedCalibration::new(calibration),
2500                staged_gated: true,
2501            },
2502        }
2503    }
2504
2505    fn observe(&mut self, entries: u64, work_ns: u64) -> Option<CalibrationWindow> {
2506        match self {
2507            Self::OneShot(calibration) => {
2508                let slow = calibration.observe(entries, work_ns)?;
2509                Some(CalibrationWindow {
2510                    start_entry_ordinal: 0,
2511                    end_entry_ordinal: calibration.entries,
2512                    entries: calibration.entries,
2513                    chunks: calibration.chunks,
2514                    work_ns: calibration.work_ns,
2515                    slow,
2516                })
2517            }
2518            Self::Repeated { calibration, .. } => calibration.observe(entries, work_ns),
2519        }
2520    }
2521
2522    fn partial_window(&self) -> (CalibrationWindow, WorkerPolicyDecision) {
2523        match self {
2524            Self::OneShot(calibration) => (
2525                CalibrationWindow {
2526                    start_entry_ordinal: 0,
2527                    end_entry_ordinal: calibration.entries,
2528                    entries: calibration.entries,
2529                    chunks: calibration.chunks,
2530                    work_ns: calibration.work_ns,
2531                    slow: false,
2532                },
2533                WorkerPolicyDecision::Undecided,
2534            ),
2535            Self::Repeated { calibration, .. } => (
2536                CalibrationWindow {
2537                    start_entry_ordinal: calibration.window_start,
2538                    end_entry_ordinal: calibration.window_start.saturating_add(calibration.entries),
2539                    entries: calibration.entries,
2540                    chunks: calibration.chunks,
2541                    work_ns: calibration.work_ns,
2542                    slow: false,
2543                },
2544                if calibration.completed_windows == 0 {
2545                    WorkerPolicyDecision::Undecided
2546                } else {
2547                    WorkerPolicyDecision::Incomplete
2548                },
2549            ),
2550        }
2551    }
2552
2553    const fn is_staged_gated(&self) -> bool {
2554        matches!(self, Self::Repeated { staged_gated: true, .. })
2555    }
2556
2557    const fn is_one_shot(&self) -> bool {
2558        matches!(self, Self::OneShot(_))
2559    }
2560
2561    const fn calibration_spec(&self) -> WorkerCalibration {
2562        match self {
2563            Self::OneShot(calibration) => WorkerCalibration::new(
2564                calibration.minimum_entries,
2565                calibration.slow_work_ns_per_entry,
2566            ),
2567            Self::Repeated { calibration, .. } => WorkerCalibration::new(
2568                calibration.minimum_entries,
2569                calibration.slow_work_ns_per_entry,
2570            ),
2571        }
2572    }
2573}
2574
2575fn automatic_worker_pool(available: usize) -> WorkerPool {
2576    let initial = available.clamp(1, DEFAULT_SCAN_THREADS_CAP);
2577    // Preserve the serial fallback when the platform cannot report more than one
2578    // available processor. There is no measured basis for inventing parallelism there.
2579    if initial == 1 {
2580        return WorkerPool::fixed(1);
2581    }
2582    let maximum = available
2583        .saturating_mul(ADAPTIVE_SCAN_PARALLELISM_MULTIPLIER)
2584        .clamp(initial, ADAPTIVE_SCAN_THREADS_CAP);
2585    WorkerPool {
2586        initial,
2587        maximum,
2588        calibration: (maximum > initial).then_some(WorkerCalibration::new(
2589            ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
2590            ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
2591        )),
2592    }
2593}
2594
2595/// Directories handed to a worker in one go.
2596///
2597/// Popping one directory at a time makes the queue lock the bottleneck on a wide,
2598/// shallow tree; taking a small run amortizes the lock without letting one worker
2599/// starve the others by hoarding the queue.
2600const DIR_CLAIM: usize = 4;
2601
2602/// A parallel directory walk that produces exactly the observations the serial walk does.
2603///
2604/// The shape is deliberate. Workers read directories and *produce* observations; they
2605/// never touch an index. A single consumer — the caller's sink, on this thread —
2606/// applies them. That keeps the crate's one mutation contract intact: parallelism is a
2607/// property of the producer, and the index still sees one ordered stream of observations.
2608///
2609/// Ordering across independent subtrees is not fixed, but a directory observation is
2610/// published before that directory becomes claimable. The index therefore sees a
2611/// parent-first causal stream without imposing a global level barrier or serializing
2612/// filesystem work. The resulting index is byte-identical to the serial walker's,
2613/// which the benchmark harness re-proves on every trial by comparing engine digests
2614/// against an independent oracle.
2615#[allow(clippy::too_many_arguments)]
2616fn scan_concurrent(
2617    root: &Path,
2618    config: &ScanConfig,
2619    root_dev: u64,
2620    sink: &mut dyn FnMut(ScannerBatch),
2621    pool: WorkerPool,
2622    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2623    policy: WorkerPolicyExperiment,
2624    sink_mode: SinkMode,
2625) -> ScanReport {
2626    let mut consume = |message| match message {
2627        WalkMessage::Batch(batch) => {
2628            if let Some(diagnostics) = diagnostics {
2629                diagnostics.handoff_received();
2630            }
2631            sink(batch);
2632        }
2633        WalkMessage::DetachedDirectories { .. } => {
2634            unreachable!("the streaming walker never publishes detached directories")
2635        }
2636        WalkMessage::ScaleUp { .. } => {
2637            unreachable!("the shared runner consumes scale-up messages")
2638        }
2639    };
2640    run_concurrent_walk(
2641        root,
2642        config,
2643        root_dev,
2644        pool,
2645        diagnostics,
2646        policy,
2647        match sink_mode {
2648            SinkMode::Retained => walk_worker,
2649            SinkMode::TransientFold => walk_worker_transient_fold,
2650        },
2651        &mut consume,
2652    )
2653}
2654
2655/// Parallel cold walk for a detached index that has no streaming consumer.
2656///
2657/// Workers publish directory-shaped facts before making their children claimable. The
2658/// caller consumes those groups into a private builder while filesystem work continues,
2659/// preserving parent-first causality and pipeline overlap without sending one full path
2660/// or public observation per entry.
2661fn scan_concurrent_detached(
2662    root: &Path,
2663    config: &ScanConfig,
2664    root_dev: u64,
2665    pool: WorkerPool,
2666    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2667    policy: WorkerPolicyExperiment,
2668) -> Result<(ScanReport, DetachedIndexBuilder)> {
2669    let mut builder = DetachedIndexBuilder::new(root, config.scope(), config.types_shared())
2670        .with_control_limits(config.control_limits);
2671    let mut build_error = None;
2672    let output = {
2673        let mut consume = |message| match message {
2674            WalkMessage::Batch(_) => {
2675                unreachable!("the detached walker never publishes scanner batches")
2676            }
2677            WalkMessage::DetachedDirectories { mut directories, recycle } => {
2678                if let Some(diagnostics) = diagnostics {
2679                    diagnostics.handoff_received();
2680                }
2681                if build_error.is_none() {
2682                    for directory in &mut directories {
2683                        if let Err(error) = builder.push_directory(directory) {
2684                            build_error = Some(error);
2685                            break;
2686                        }
2687                    }
2688                }
2689                // A worker that has already left has dropped its receiver, and then the
2690                // listings are freed here, as every listing was before H159.
2691                let _ = recycle.send(directories);
2692            }
2693            WalkMessage::ScaleUp { .. } => {
2694                unreachable!("the shared runner consumes scale-up messages")
2695            }
2696        };
2697        run_concurrent_walk(
2698            root,
2699            config,
2700            root_dev,
2701            pool,
2702            diagnostics,
2703            policy,
2704            walk_detached_worker,
2705            &mut consume,
2706        )
2707    };
2708    if let Some(error) = build_error {
2709        return Err(error);
2710    }
2711    Ok((output, builder))
2712}
2713
2714type WalkWorker = fn(
2715    &Path,
2716    &ScanConfig,
2717    u64,
2718    &DirectoryQueue,
2719    &std::sync::mpsc::Sender<WalkMessage>,
2720    Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2721) -> ScanReport;
2722
2723/// Run the shared pool, scaling controller, diagnostics, and report reduction.
2724///
2725/// Streaming and detached scans differ only in their worker emission and main-thread
2726/// consumer. Keeping orchestration here prevents fixes to termination, diagnostics, or
2727/// panic handling from diverging between the two cold paths.
2728#[allow(clippy::too_many_arguments)]
2729fn run_concurrent_walk<C>(
2730    root: &Path,
2731    config: &ScanConfig,
2732    root_dev: u64,
2733    pool: WorkerPool,
2734    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
2735    policy: WorkerPolicyExperiment,
2736    worker: WalkWorker,
2737    consume: &mut C,
2738) -> ScanReport
2739where
2740    C: FnMut(WalkMessage),
2741{
2742    let diagnostics = diagnostics.cloned();
2743    let queue = DirectoryQueue::new_with_policy(
2744        (PathBuf::new(), 0),
2745        config.order,
2746        pool.calibration,
2747        diagnostics.clone(),
2748        pool.initial,
2749        pool.maximum,
2750        policy,
2751    );
2752    let (sender, receiver) = std::sync::mpsc::channel::<WalkMessage>();
2753
2754    let mut report = std::thread::scope(|scope| {
2755        let mut handles: Vec<_> = (0..pool.initial)
2756            .map(|_| {
2757                let sender = sender.clone();
2758                let queue = &queue;
2759                let diagnostics = diagnostics.clone();
2760                scope.spawn(move || {
2761                    worker(root, config, root_dev, queue, &sender, diagnostics.as_ref())
2762                })
2763            })
2764            .collect();
2765        // The loop below ends when every sender is gone, so this one must go first.
2766        drop(sender);
2767
2768        let mut spawned_workers = pool.initial;
2769        for message in receiver {
2770            match message {
2771                WalkMessage::ScaleUp { sender, target_workers }
2772                    if target_workers > spawned_workers =>
2773                {
2774                    let target_workers = target_workers.min(pool.maximum);
2775                    record_adaptive_worker_expansion(diagnostics.as_ref());
2776                    for _ in spawned_workers..target_workers {
2777                        let sender = sender.clone();
2778                        let queue = &queue;
2779                        let diagnostics = diagnostics.clone();
2780                        handles.push(scope.spawn(move || {
2781                            worker(root, config, root_dev, queue, &sender, diagnostics.as_ref())
2782                        }));
2783                    }
2784                    spawned_workers = target_workers;
2785                }
2786                WalkMessage::ScaleUp { .. } => {}
2787                output => consume(output),
2788            }
2789        }
2790
2791        // A walk that ends before its calibration window fills never observed enough to
2792        // decide anything. That is an *unobservable* policy, not a decision to hold the
2793        // initial pool, and an artifact that conflated the two would report a held pool
2794        // as if the walk had measured one and chosen it.
2795        let queue_finish = {
2796            let mut state = queue.lock();
2797            let mut trailing_window = None;
2798            if let Some(controller) = &state.controller {
2799                let (window, decision) = controller.partial_window();
2800                if decision == WorkerPolicyDecision::Undecided {
2801                    crate::counters::bump(|counts| {
2802                        counts.adaptive_policy_undecided =
2803                            counts.adaptive_policy_undecided.saturating_add(1);
2804                    });
2805                }
2806                if diagnostics.is_some() {
2807                    let sequence = state.allocate_policy_sequence();
2808                    trailing_window = Some(PolicyWindowSnapshot {
2809                        sequence,
2810                        start_entry_ordinal: window.start_entry_ordinal,
2811                        end_entry_ordinal: window.end_entry_ordinal,
2812                        observed_entries: window.entries,
2813                        observed_chunks: window.chunks,
2814                        observed_work_ns: window.work_ns,
2815                        ready_directories: state.ready_directories,
2816                        in_flight_directories: state.in_flight_directories,
2817                        active_workers: diagnostics.as_ref().map_or(0, |diagnostics| {
2818                            diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
2819                        }),
2820                        handoff_backlog: diagnostics.as_ref().map_or(0, |diagnostics| {
2821                            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
2822                        }),
2823                        requested_workers: None,
2824                        decision,
2825                    });
2826                }
2827            } else if diagnostics.is_some() {
2828                let shadow = state.shadow_calibration.as_ref().and_then(|shadow| {
2829                    (shadow.entries > 0).then_some((
2830                        shadow.window_start,
2831                        shadow.entries,
2832                        shadow.chunks,
2833                        shadow.work_ns,
2834                    ))
2835                });
2836                if let Some((window_start, entries, chunks, work_ns)) = shadow {
2837                    let sequence = state.allocate_policy_sequence();
2838                    trailing_window = Some(PolicyWindowSnapshot {
2839                        sequence,
2840                        start_entry_ordinal: window_start,
2841                        end_entry_ordinal: window_start.saturating_add(entries),
2842                        observed_entries: entries,
2843                        observed_chunks: chunks,
2844                        observed_work_ns: work_ns,
2845                        ready_directories: state.ready_directories,
2846                        in_flight_directories: state.in_flight_directories,
2847                        active_workers: diagnostics.as_ref().map_or(0, |diagnostics| {
2848                            diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
2849                        }),
2850                        handoff_backlog: diagnostics.as_ref().map_or(0, |diagnostics| {
2851                            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
2852                        }),
2853                        requested_workers: None,
2854                        decision: WorkerPolicyDecision::ObserveIncomplete,
2855                    });
2856                }
2857            }
2858            (state.ready_directories, state.in_flight_directories, trailing_window)
2859        };
2860        if let Some(diagnostics) = &diagnostics {
2861            diagnostics.record_queue_finish(queue_finish.0, queue_finish.1);
2862            if let Some(window) = queue_finish.2 {
2863                diagnostics.record_policy_window(window);
2864            }
2865        }
2866
2867        let mut report = ScanReport::default();
2868        for handle in handles {
2869            match handle.join() {
2870                Ok(worker) => report.absorb(worker),
2871                Err(_) => {
2872                    // A worker panic leaves directories unaccounted for. Preserve that
2873                    // as a partial scan instead of reporting a short tree as complete.
2874                    report.errors.push(Error::io(
2875                        root,
2876                        std::io::Error::other("a scan worker thread panicked"),
2877                    ));
2878                }
2879            }
2880        }
2881        report
2882    });
2883
2884    // Workers finish in filesystem order, so normalize errors before they escape.
2885    report.errors.sort_by_cached_key(ToString::to_string);
2886    report
2887}
2888
2889/// Compile-time adapter for the one directory walker.
2890///
2891/// The filesystem, queue, admission, and diagnostics logic stays singular. Generic
2892/// emission keeps the public streaming path and private detached path branch-free in
2893/// their per-entry loops after monomorphization.
2894trait WalkEmission {
2895    type Directory;
2896
2897    fn begin_directory(&mut self, path: &Path) -> Self::Directory;
2898
2899    #[allow(clippy::too_many_arguments)]
2900    fn record_entry(
2901        &mut self,
2902        root: &Path,
2903        rel_dir: &Path,
2904        depth: usize,
2905        region: RegionId,
2906        name: &OsStr,
2907        kind: EntryKind,
2908        attrs: Attrs,
2909        root_dev: u64,
2910        config: &ScanConfig,
2911        directory: &mut Self::Directory,
2912        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
2913        report: &mut ScanReport,
2914        sender: &std::sync::mpsc::Sender<WalkMessage>,
2915        chunk_send_ns: &mut u64,
2916        diagnostics: Option<&ScanDiagnosticsRecorder>,
2917    ) -> bool;
2918
2919    fn finish_directory(&mut self, directory: Self::Directory);
2920
2921    fn publish_before_discovery(
2922        &mut self,
2923        has_discovered: bool,
2924        sender: &std::sync::mpsc::Sender<WalkMessage>,
2925        chunk_send_ns: &mut u64,
2926        diagnostics: Option<&ScanDiagnosticsRecorder>,
2927    ) -> bool;
2928
2929    fn finish(
2930        &mut self,
2931        sender: &std::sync::mpsc::Sender<WalkMessage>,
2932        report: &mut ScanReport,
2933        diagnostics: Option<&ScanDiagnosticsRecorder>,
2934    );
2935
2936    /// Transient summary can take directory and symlink kind from the listing.
2937    fn skip_dir_symlink_stat(&self) -> bool {
2938        false
2939    }
2940}
2941
2942struct StreamingEmission {
2943    batch: Vec<ObservationOp>,
2944    batch_size: usize,
2945    recycle_tx: Option<std::sync::mpsc::Sender<Vec<ObservationOp>>>,
2946    recycle_rx: Option<std::sync::mpsc::Receiver<Vec<ObservationOp>>>,
2947    skip_dir_symlink_stat: bool,
2948    /// Whether each listing's control goes ahead of its entries
2949    /// ([`SinkMode::groups_directories`]).
2950    group_directories: bool,
2951    /// Where the listing being recorded begins in `batch`.
2952    directory_start: usize,
2953    /// Whether that listing's control already leads its observations, so the batch may be
2954    /// sent before the listing ends.
2955    listing_settled: bool,
2956    /// Where that listing stands on its directory's control.
2957    listing_control: ListingControl,
2958}
2959
2960/// Where the listing a grouping worker is recording stands on its directory's control.
2961#[derive(Clone, Copy, PartialEq, Eq, Debug)]
2962enum ListingControl {
2963    /// No listed spelling of `.gitignore` has read the directory's control yet.
2964    Unread,
2965    /// A listed spelling read the directory's control, or failed to; its observation, if
2966    /// any, is among the held ones.
2967    Listed,
2968    /// The held observations filled the batch before any listed spelling read the
2969    /// control, so the directory's control was looked up directly (`read_directory_control`)
2970    /// and stands for the whole listing, as in the narrowed-population walk.
2971    ///
2972    /// A spelling listed afterwards is recorded as a row and not read again, so the
2973    /// directory's control file is read once. On a tree nothing modifies during the walk
2974    /// the probed file is the listed one; if the file changes between the two, the listing
2975    /// keeps what the probe read, and an error the second read would have met is never
2976    /// met.
2977    Probed,
2978}
2979
2980impl StreamingEmission {
2981    /// An emission with the properties [`SinkMode`] names for `mode` under `config`.
2982    ///
2983    /// The recycle channel returns drained `PathBuf` arenas to this worker so glibc
2984    /// frees them on the thread that allocated them.
2985    fn for_sink(config: &ScanConfig, mode: SinkMode) -> Self {
2986        let batch_size = config.batch_size;
2987        let (recycle_tx, recycle_rx) = if mode.recycles_batches() {
2988            let (tx, rx) = std::sync::mpsc::channel();
2989            (Some(tx), Some(rx))
2990        } else {
2991            (None, None)
2992        };
2993        Self {
2994            batch: Vec::with_capacity(batch_size),
2995            batch_size,
2996            recycle_tx,
2997            recycle_rx,
2998            skip_dir_symlink_stat: mode.skips_dir_symlink_stat(),
2999            group_directories: mode.groups_directories(config),
3000            directory_start: 0,
3001            listing_settled: false,
3002            listing_control: ListingControl::Unread,
3003        }
3004    }
3005
3006    /// Send the batch once it is full, settling a grouped listing's control first.
3007    ///
3008    /// A grouped listing's observations are held until its control leads them: at the
3009    /// listing's end ([`WalkEmission::finish_directory`]), or here, when the batch fills
3010    /// first. So a batch never holds more than `batch_size` observations and one control,
3011    /// however long the listing, and a fill that finds no spelling of `.gitignore` listed
3012    /// yet probes for it once per listing: one metadata lookup, and a read on a hit
3013    /// (`read_directory_control`). Returns whether the consumer is still there.
3014    #[allow(clippy::too_many_arguments)]
3015    fn send_if_full(
3016        &mut self,
3017        root: &Path,
3018        rel_dir: &Path,
3019        config: &ScanConfig,
3020        report: &mut ScanReport,
3021        sender: &std::sync::mpsc::Sender<WalkMessage>,
3022        chunk_send_ns: &mut u64,
3023        diagnostics: Option<&ScanDiagnosticsRecorder>,
3024    ) -> bool {
3025        if self.batch.len() < self.batch_size {
3026            return true;
3027        }
3028        if self.group_directories && !self.listing_settled {
3029            self.settle_listing(root, rel_dir, config, report);
3030        }
3031        let send_started = std::time::Instant::now();
3032        let sent = self.send_full(sender, diagnostics);
3033        *chunk_send_ns += elapsed_ns(send_started);
3034        sent
3035    }
3036
3037    /// Put the current listing's control ahead of its held observations.
3038    fn settle_listing(
3039        &mut self,
3040        root: &Path,
3041        rel_dir: &Path,
3042        config: &ScanConfig,
3043        report: &mut ScanReport,
3044    ) {
3045        self.listing_settled = true;
3046        if self.listing_control != ListingControl::Unread {
3047            controls_first(&mut self.batch[self.directory_start..]);
3048            return;
3049        }
3050        self.listing_control = ListingControl::Probed;
3051        let control = rel_dir.join(crate::control::CONTROL_FILE_NAME);
3052        // The probe is the lookup the rule names, so a hit stands for the directory
3053        // whichever spelling it resolved to, exactly as a listed spelling's read would.
3054        match read_directory_control(config, root, &control) {
3055            Ok(Some(op)) => {
3056                self.batch.insert(self.directory_start, ObservationOp::unconditional(op));
3057            }
3058            Ok(None) => {}
3059            Err(error) => report.errors.push(error),
3060        }
3061    }
3062
3063    /// Whether the entries this listing still lists must not read their control: its
3064    /// directory's control was already probed, and one read stands for the listing, as in
3065    /// the narrowed-population walk (`read_listed_control_op`).
3066    fn control_probed(&self) -> bool {
3067        self.group_directories && self.listing_control == ListingControl::Probed
3068    }
3069
3070    /// Note that a listed entry read this listing's control, or failed to.
3071    fn note_listed_control(&mut self, control: Option<&Op>, control_error: Option<&Error>) {
3072        if control.is_some() || control_error.is_some() {
3073            self.listing_control = ListingControl::Listed;
3074        }
3075    }
3076
3077    fn wrap(&self, ops: Vec<ObservationOp>) -> ScannerBatch {
3078        match &self.recycle_tx {
3079            Some(recycle) => ScannerBatch::new(ops).with_recycle(recycle.clone()),
3080            None => ScannerBatch::new(ops),
3081        }
3082    }
3083
3084    fn next_vec(&self) -> Vec<ObservationOp> {
3085        // The retained path keeps its pre-H147 shape: an empty vec that grows by
3086        // doubling. Pre-sizing every batch there was never measured, and the public
3087        // `scan` is what a library caller pays for.
3088        let Some(recycle_rx) = &self.recycle_rx else {
3089            return Vec::new();
3090        };
3091        let mut kept = None;
3092        while let Ok(mut recycled) = recycle_rx.try_recv() {
3093            recycled.clear();
3094            kept = Some(recycled);
3095        }
3096        kept.unwrap_or_else(|| Vec::with_capacity(self.batch_size))
3097    }
3098
3099    fn send_full(
3100        &mut self,
3101        sender: &std::sync::mpsc::Sender<WalkMessage>,
3102        diagnostics: Option<&ScanDiagnosticsRecorder>,
3103    ) -> bool {
3104        let ops = std::mem::take(&mut self.batch);
3105        let sent = send_scanner_batch(sender, self.wrap(ops), diagnostics);
3106        self.batch = self.next_vec();
3107        // A listing sent partway continues at the start of the new batch.
3108        self.directory_start = 0;
3109        sent
3110    }
3111}
3112
3113impl WalkEmission for StreamingEmission {
3114    type Directory = ();
3115
3116    fn begin_directory(&mut self, _path: &Path) {
3117        self.directory_start = self.batch.len();
3118        self.listing_settled = false;
3119        self.listing_control = ListingControl::Unread;
3120    }
3121
3122    #[allow(clippy::too_many_arguments)]
3123    fn record_entry(
3124        &mut self,
3125        root: &Path,
3126        rel_dir: &Path,
3127        depth: usize,
3128        region: RegionId,
3129        name: &OsStr,
3130        kind: EntryKind,
3131        attrs: Attrs,
3132        root_dev: u64,
3133        config: &ScanConfig,
3134        _directory: &mut Self::Directory,
3135        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3136        report: &mut ScanReport,
3137        sender: &std::sync::mpsc::Sender<WalkMessage>,
3138        chunk_send_ns: &mut u64,
3139        diagnostics: Option<&ScanDiagnosticsRecorder>,
3140    ) -> bool {
3141        record_walk_entry(
3142            root,
3143            rel_dir,
3144            depth,
3145            region,
3146            name,
3147            kind,
3148            attrs,
3149            root_dev,
3150            config,
3151            self,
3152            discovered,
3153            report,
3154            sender,
3155            chunk_send_ns,
3156            diagnostics,
3157        )
3158    }
3159
3160    fn finish_directory(&mut self, _directory: Self::Directory) {
3161        if self.group_directories && !self.listing_settled {
3162            controls_first(&mut self.batch[self.directory_start..]);
3163        }
3164    }
3165
3166    fn publish_before_discovery(
3167        &mut self,
3168        has_discovered: bool,
3169        sender: &std::sync::mpsc::Sender<WalkMessage>,
3170        chunk_send_ns: &mut u64,
3171        diagnostics: Option<&ScanDiagnosticsRecorder>,
3172    ) -> bool {
3173        if self.batch.is_empty() || !has_discovered {
3174            return true;
3175        }
3176        let send_started = std::time::Instant::now();
3177        let sent = self.send_full(sender, diagnostics);
3178        *chunk_send_ns += elapsed_ns(send_started);
3179        sent
3180    }
3181
3182    fn finish(
3183        &mut self,
3184        sender: &std::sync::mpsc::Sender<WalkMessage>,
3185        report: &mut ScanReport,
3186        diagnostics: Option<&ScanDiagnosticsRecorder>,
3187    ) {
3188        if self.batch.is_empty() {
3189            return;
3190        }
3191        let send_started = std::time::Instant::now();
3192        // The walk is over: do not ask `send_full` for a replacement vec that no
3193        // later `record_entry` would use.
3194        let ops = std::mem::take(&mut self.batch);
3195        let _ = send_scanner_batch(sender, self.wrap(ops), diagnostics);
3196        self.batch = Vec::new();
3197        report.attribution.send_ns += elapsed_ns(send_started);
3198    }
3199
3200    fn skip_dir_symlink_stat(&self) -> bool {
3201        self.skip_dir_symlink_stat
3202    }
3203}
3204
3205/// Emptied listings one detached worker keeps for reuse: one chunk's worth.
3206///
3207/// A bound on retention, not a tuned speed value. A chunk claims at most [`DIR_CLAIM`]
3208/// directories, and the worker takes returned listings back once per chunk, so this
3209/// covers the next chunk; a listing returned past it is freed at once, still on its own
3210/// thread, which is the property H159 needs. Retained listings are memory the consumer
3211/// can no longer reuse for the index: four chunks' worth of listings up to 256 children
3212/// each cost 1.0 MiB (+1.4%) of peak RSS on a 158,705-entry macOS subject and 2.3 MiB
3213/// (+5.0%) on a 77,159-entry one.
3214const DETACHED_SPARE_LISTINGS: usize = DIR_CLAIM;
3215
3216/// The largest child buffer, in children, a spare listing may keep.
3217///
3218/// Also a bound on retention rather than a tuned value: most directories are small, and
3219/// a listing whose buffer grew past this, at about 80 bytes per child, is freed when it
3220/// comes back, on its own thread, instead of being pinned for the rest of the walk.
3221const DETACHED_SPARE_CHILD_CAPACITY: usize = 64;
3222
3223/// One worker's detached emission: listings built here and published to the consumer.
3224///
3225/// Every path and child buffer in a listing is allocated on this worker's thread. The
3226/// consumer drains each listing into the index and sends the emptied listings back
3227/// (H159), and the worker reuses them or frees them itself. Under glibc a chunk freed on
3228/// another thread goes back to the arena that allocated it, under that arena's lock,
3229/// while this worker is allocating from it; the 2026-09-27 Linux comparison's allocator
3230/// screen and context-switch profile point at that contention for the index tier's gap
3231/// to its peers. A reused listing carries exactly the facts a fresh one would: the same
3232/// path bytes, and children and control only from this directory's listing.
3233struct DetachedEmission {
3234    directories: Vec<DetachedDirectory>,
3235    /// Emptied listings ready to be reused by [`WalkEmission::begin_directory`].
3236    spare: Vec<DetachedDirectory>,
3237    /// An emptied list to publish the next chunk's listings in.
3238    spare_list: Vec<DetachedDirectory>,
3239    recycle_tx: std::sync::mpsc::Sender<Vec<DetachedDirectory>>,
3240    recycle_rx: std::sync::mpsc::Receiver<Vec<DetachedDirectory>>,
3241}
3242
3243impl DetachedEmission {
3244    fn new() -> Self {
3245        let (recycle_tx, recycle_rx) = std::sync::mpsc::channel();
3246        Self {
3247            directories: Vec::new(),
3248            spare: Vec::new(),
3249            spare_list: Vec::new(),
3250            recycle_tx,
3251            recycle_rx,
3252        }
3253    }
3254
3255    /// Take back every list the consumer has returned since the last chunk.
3256    ///
3257    /// A listing the consumer skipped, after a build error or for a repeated directory,
3258    /// comes back with its children and control still in it; they are dropped here, on
3259    /// the thread that allocated them, before the listing can be reused.
3260    fn collect_returned(&mut self) {
3261        while let Ok(mut returned) = self.recycle_rx.try_recv() {
3262            for mut directory in returned.drain(..) {
3263                directory.children.clear();
3264                directory.control = None;
3265                if self.spare.len() < DETACHED_SPARE_LISTINGS
3266                    && directory.children.capacity() <= DETACHED_SPARE_CHILD_CAPACITY
3267                {
3268                    self.spare.push(directory);
3269                }
3270            }
3271            if self.spare_list.capacity() == 0 {
3272                self.spare_list = returned;
3273            }
3274        }
3275    }
3276}
3277
3278impl WalkEmission for DetachedEmission {
3279    type Directory = DetachedDirectory;
3280
3281    fn begin_directory(&mut self, path: &Path) -> Self::Directory {
3282        let Some(mut directory) = self.spare.pop() else {
3283            return DetachedDirectory {
3284                path: path.to_path_buf(),
3285                children: Vec::new(),
3286                control: None,
3287            };
3288        };
3289        // The same bytes `to_path_buf` would copy, into a buffer this thread already owns.
3290        let buffer = directory.path.as_mut_os_string();
3291        buffer.clear();
3292        buffer.push(path);
3293        debug_assert!(directory.children.is_empty() && directory.control.is_none());
3294        directory
3295    }
3296
3297    #[allow(clippy::too_many_arguments)]
3298    fn record_entry(
3299        &mut self,
3300        root: &Path,
3301        rel_dir: &Path,
3302        depth: usize,
3303        region: RegionId,
3304        name: &OsStr,
3305        kind: EntryKind,
3306        attrs: Attrs,
3307        root_dev: u64,
3308        config: &ScanConfig,
3309        directory: &mut Self::Directory,
3310        discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3311        report: &mut ScanReport,
3312        _sender: &std::sync::mpsc::Sender<WalkMessage>,
3313        _chunk_send_ns: &mut u64,
3314        _diagnostics: Option<&ScanDiagnosticsRecorder>,
3315    ) -> bool {
3316        record_detached_entry(
3317            root,
3318            rel_dir,
3319            depth,
3320            region,
3321            name,
3322            kind,
3323            attrs,
3324            root_dev,
3325            config,
3326            &mut directory.children,
3327            &mut directory.control,
3328            discovered,
3329            report,
3330        );
3331        true
3332    }
3333
3334    fn finish_directory(&mut self, directory: Self::Directory) {
3335        self.directories.push(directory);
3336    }
3337
3338    fn publish_before_discovery(
3339        &mut self,
3340        _has_discovered: bool,
3341        sender: &std::sync::mpsc::Sender<WalkMessage>,
3342        chunk_send_ns: &mut u64,
3343        diagnostics: Option<&ScanDiagnosticsRecorder>,
3344    ) -> bool {
3345        if self.directories.is_empty() {
3346            return true;
3347        }
3348        // Taking back returned listings is handoff work, timed with the send so the
3349        // chunk's work time, which calibrates the worker pool, stays the walk's own.
3350        let send_started = std::time::Instant::now();
3351        self.collect_returned();
3352        let next = std::mem::take(&mut self.spare_list);
3353        let directories = std::mem::replace(&mut self.directories, next);
3354        let sent =
3355            send_detached_directories(sender, directories, self.recycle_tx.clone(), diagnostics);
3356        *chunk_send_ns += elapsed_ns(send_started);
3357        sent
3358    }
3359
3360    fn finish(
3361        &mut self,
3362        _sender: &std::sync::mpsc::Sender<WalkMessage>,
3363        _report: &mut ScanReport,
3364        _diagnostics: Option<&ScanDiagnosticsRecorder>,
3365    ) {
3366    }
3367}
3368
3369fn walk_detached_worker(
3370    root: &Path,
3371    config: &ScanConfig,
3372    root_dev: u64,
3373    queue: &DirectoryQueue,
3374    sender: &std::sync::mpsc::Sender<WalkMessage>,
3375    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3376) -> ScanReport {
3377    let report = walk_worker_with(
3378        root,
3379        config,
3380        root_dev,
3381        queue,
3382        sender,
3383        diagnostics,
3384        DetachedEmission::new(),
3385    );
3386    // A walker leaves only when the queue is empty with nothing in flight, or when its
3387    // consumer is gone, so the walk is over. The index may still be assembling the
3388    // listings already sent; the counters have stopped, and the phase says why.
3389    if let Some(progress) = &config.progress {
3390        progress.enter(crate::ProgressPhase::Indexing);
3391    }
3392    report
3393}
3394
3395/// One worker's share of the public observation walk.
3396fn walk_worker(
3397    root: &Path,
3398    config: &ScanConfig,
3399    root_dev: u64,
3400    queue: &DirectoryQueue,
3401    sender: &std::sync::mpsc::Sender<WalkMessage>,
3402    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3403) -> ScanReport {
3404    walk_worker_with(
3405        root,
3406        config,
3407        root_dev,
3408        queue,
3409        sender,
3410        diagnostics,
3411        StreamingEmission::for_sink(config, SinkMode::Retained),
3412    )
3413}
3414
3415/// One worker's share of the transient summary walk.
3416fn walk_worker_transient_fold(
3417    root: &Path,
3418    config: &ScanConfig,
3419    root_dev: u64,
3420    queue: &DirectoryQueue,
3421    sender: &std::sync::mpsc::Sender<WalkMessage>,
3422    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3423) -> ScanReport {
3424    walk_worker_with(
3425        root,
3426        config,
3427        root_dev,
3428        queue,
3429        sender,
3430        diagnostics,
3431        StreamingEmission::for_sink(config, SinkMode::TransientFold),
3432    )
3433}
3434
3435fn walk_worker_with<E: WalkEmission>(
3436    root: &Path,
3437    config: &ScanConfig,
3438    root_dev: u64,
3439    queue: &DirectoryQueue,
3440    sender: &std::sync::mpsc::Sender<WalkMessage>,
3441    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
3442    mut emission: E,
3443) -> ScanReport {
3444    let _counter_guard = crate::counters::thread_flush_guard();
3445    let _worker_guard = diagnostics.map(ScanDiagnosticsRecorder::worker_guard);
3446    let worker_started = std::time::Instant::now();
3447    let mut report = ScanReport::default();
3448    let mut tally = ProgressTally::new(config.progress.as_ref());
3449    let mut claimed: Vec<(PathBuf, usize, RegionId)> = Vec::with_capacity(DIR_CLAIM);
3450    let mut discovered: Vec<(PathBuf, usize, RegionId)> = Vec::new();
3451    let mut consumer_gone = false;
3452    #[cfg(target_os = "macos")]
3453    let mut bulk_reader = macos_bulk::Reader::new();
3454
3455    'walk: while let Some(claim) = queue.claim(&mut claimed, &mut report.attribution) {
3456        // One timing pair per claimed chunk, never per entry: the chunk is the unit
3457        // the amortization argument is made in, so it is the unit the evidence is
3458        // collected in.
3459        let chunk_started = std::time::Instant::now();
3460        let mut chunk_send_ns: u64 = 0;
3461        let entries_before = report.entries;
3462        for (rel_dir, depth, region) in claimed.drain(..) {
3463            let abs_dir = root.join(&rel_dir);
3464            let mut directory = emission.begin_directory(&rel_dir);
3465            #[cfg(target_os = "macos")]
3466            {
3467                if let Some(diagnostics) = diagnostics {
3468                    diagnostics.macos_bulk_attempted();
3469                }
3470                if let Some(entries) =
3471                    (!walk_hook_covers(&abs_dir)).then(|| bulk_reader.read(&abs_dir)).flatten()
3472                {
3473                    if let Some(diagnostics) = diagnostics {
3474                        diagnostics.macos_bulk_succeeded();
3475                    }
3476                    report.dirs_read += 1;
3477                    for entry in entries {
3478                        if !emission.record_entry(
3479                            root,
3480                            &rel_dir,
3481                            depth,
3482                            region,
3483                            &entry.name,
3484                            entry.kind,
3485                            entry.attrs,
3486                            root_dev,
3487                            config,
3488                            &mut directory,
3489                            &mut discovered,
3490                            &mut report,
3491                            sender,
3492                            &mut chunk_send_ns,
3493                            diagnostics.map(AsRef::as_ref),
3494                        ) {
3495                            consumer_gone = true;
3496                            break 'walk;
3497                        }
3498                    }
3499                    emission.finish_directory(directory);
3500                    continue;
3501                }
3502                if let Some(diagnostics) = diagnostics {
3503                    diagnostics.macos_bulk_fell_back();
3504                }
3505            }
3506
3507            crate::counters::bump(|c| c.dir_opens += 1);
3508            if let Some(diagnostics) = diagnostics {
3509                diagnostics.portable_attempted();
3510            }
3511
3512            let listing = match fs::read_dir(&abs_dir) {
3513                Ok(listing) => {
3514                    if let Some(diagnostics) = diagnostics {
3515                        diagnostics.portable_succeeded();
3516                    }
3517                    listing
3518                }
3519                Err(e) => {
3520                    report.errors.push(Error::io(abs_dir, e));
3521                    continue;
3522                }
3523            };
3524            report.dirs_read += 1;
3525
3526            for item in listing {
3527                let item = match item {
3528                    Ok(item) => item,
3529                    Err(e) => {
3530                        report.errors.push(Error::io(&abs_dir, e));
3531                        continue;
3532                    }
3533                };
3534                crate::counters::bump(|c| c.dir_entries += 1);
3535                let name = item.file_name();
3536                let (kind, attrs) = match listed_child_kind_and_attrs(
3537                    &item,
3538                    emission.skip_dir_symlink_stat(),
3539                    config.one_filesystem,
3540                ) {
3541                    Ok(Some(observed)) => observed,
3542                    Ok(None) => continue,
3543                    Err(error) => {
3544                        report.errors.push(Error::io(item.path(), error));
3545                        continue;
3546                    }
3547                };
3548                if !emission.record_entry(
3549                    root,
3550                    &rel_dir,
3551                    depth,
3552                    region,
3553                    &name,
3554                    kind,
3555                    attrs,
3556                    root_dev,
3557                    config,
3558                    &mut directory,
3559                    &mut discovered,
3560                    &mut report,
3561                    sender,
3562                    &mut chunk_send_ns,
3563                    diagnostics.map(AsRef::as_ref),
3564                ) {
3565                    // The consumer is gone; nothing further will be read.
3566                    consumer_gone = true;
3567                    break 'walk;
3568                }
3569            }
3570            emission.finish_directory(directory);
3571        }
3572        // Publish facts that authorize newly discovered directories before making
3573        // those directories claimable. Both emission modes preserve this boundary.
3574        if !emission.publish_before_discovery(
3575            !discovered.is_empty(),
3576            sender,
3577            &mut chunk_send_ns,
3578            diagnostics.map(AsRef::as_ref),
3579        ) {
3580            report.attribution.send_ns += chunk_send_ns;
3581            report.attribution.work_ns += elapsed_ns(chunk_started).saturating_sub(chunk_send_ns);
3582            consumer_gone = true;
3583            break 'walk;
3584        }
3585        report.attribution.send_ns += chunk_send_ns;
3586        let chunk_work_ns = elapsed_ns(chunk_started).saturating_sub(chunk_send_ns);
3587        report.attribution.work_ns += chunk_work_ns;
3588        // Progress is reported per chunk for the same reason timing is: the chunk is
3589        // the unit of handoff, so it is the unit the shared counters are touched in.
3590        tally.flush(&report);
3591
3592        // Publish new work before releasing the claim so a worker that finds nothing
3593        // new does not hold work that others could be doing.
3594        if !discovered.is_empty() {
3595            queue.extend(discovered.drain(..), &mut report.attribution);
3596        }
3597        if let Some(target_workers) = claim.release(
3598            report.entries.saturating_sub(entries_before),
3599            chunk_work_ns,
3600            &mut report.attribution,
3601        ) {
3602            // Carry a sender in-band so the consumer can create the reserve workers
3603            // without retaining a channel endpoint that would keep a small scan alive.
3604            // Only the release that completes a slow calibration returns true, so one
3605            // message expands the pool exactly once.
3606            let _ = sender.send(WalkMessage::ScaleUp { sender: sender.clone(), target_workers });
3607        }
3608    }
3609
3610    if !consumer_gone {
3611        emission.finish(sender, &mut report, diagnostics.map(AsRef::as_ref));
3612    }
3613    // A worker that left mid-chunk because its consumer was gone still read what it
3614    // read, and the report it returns says so.
3615    tally.flush(&report);
3616    report.attribution.wall_ns = elapsed_ns(worker_started);
3617    report
3618}
3619
3620#[allow(clippy::too_many_arguments)]
3621fn record_detached_entry(
3622    root: &Path,
3623    rel_dir: &Path,
3624    depth: usize,
3625    region: RegionId,
3626    name: &OsStr,
3627    kind: EntryKind,
3628    attrs: Attrs,
3629    root_dev: u64,
3630    config: &ScanConfig,
3631    children: &mut Vec<DetachedChild>,
3632    control: &mut Option<Op>,
3633    discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3634    report: &mut ScanReport,
3635) {
3636    let disposition = crate::admission::decide(name, kind, config.hidden(), config.exclude_special);
3637    if disposition == crate::admission::Disposition::Reject {
3638        return;
3639    }
3640    // Construct a full path only for a spelling of the control name. The scanner's public
3641    // preparation builds one for every retained entry because that path escapes in an
3642    // observation; this private builder keeps ordinary children component-only.
3643    if config.read_controls && crate::control::control_spelling(name).is_some() {
3644        let path = rel_dir.join(name);
3645        match read_control_op(config, root, &path, kind) {
3646            // A listing can repeat the control name while the directory changes, and a
3647            // case-sensitive one can list `.gitignore` beside `.GITIGNORE`, whose lookup
3648            // reads the same file. The later read wins, as the builder keeps the later
3649            // observation of the entry.
3650            Ok(observed) => *control = observed,
3651            Err(error) => report.errors.push(error),
3652        }
3653    }
3654    if disposition != crate::admission::Disposition::Retain {
3655        return;
3656    }
3657    report.observe(kind, attrs);
3658    // Positions only order repeated names, and no real listing reaches `u32::MAX` entries.
3659    let position = u32::try_from(children.len()).unwrap_or(u32::MAX);
3660    children.push(DetachedChild { name: name.to_os_string(), kind, attrs, position });
3661    if should_descend(kind, attrs, depth, root_dev, config) {
3662        let child_region = if depth == 0 { RegionId::UNASSIGNED } else { region };
3663        discovered.push((rel_dir.join(name), depth + 1, child_region));
3664    }
3665}
3666
3667/// One filesystem entry after the scan's shared admission, control, and descent rules.
3668pub(crate) struct PreparedWalkEntry {
3669    pub(crate) path: PathBuf,
3670    pub(crate) kind: EntryKind,
3671    pub(crate) attrs: Attrs,
3672    pub(crate) retained: bool,
3673    pub(crate) control: Option<Op>,
3674    pub(crate) descend: bool,
3675    pub(crate) control_error: Option<Error>,
3676}
3677
3678/// Apply the producer-independent part of a directory walk to one verified entry.
3679///
3680/// Both blocking and opened-root scans call this after obtaining non-following metadata,
3681/// which keeps admission, fixed controls, and traversal boundaries from drifting.
3682#[allow(clippy::too_many_arguments)]
3683pub(crate) fn prepare_walk_entry(
3684    root: &Path,
3685    rel_dir: &Path,
3686    depth: usize,
3687    name: &OsStr,
3688    kind: EntryKind,
3689    attrs: Attrs,
3690    root_dev: u64,
3691    config: &ScanConfig,
3692) -> Option<PreparedWalkEntry> {
3693    prepare_walk_entry_reading(root, rel_dir, depth, name, kind, attrs, root_dev, config, true)
3694}
3695
3696/// [`prepare_walk_entry`], reading the entry's control only when `read_control` allows.
3697///
3698/// A grouping emission whose listing already probed its directory's control passes
3699/// `false`, so a `.gitignore` listed afterwards is not read a second time
3700/// (`StreamingEmission::control_probed`).
3701#[allow(clippy::too_many_arguments)]
3702fn prepare_walk_entry_reading(
3703    root: &Path,
3704    rel_dir: &Path,
3705    depth: usize,
3706    name: &OsStr,
3707    kind: EntryKind,
3708    attrs: Attrs,
3709    root_dev: u64,
3710    config: &ScanConfig,
3711    read_control: bool,
3712) -> Option<PreparedWalkEntry> {
3713    let disposition = crate::admission::decide(name, kind, config.hidden(), config.exclude_special);
3714    if disposition == crate::admission::Disposition::Reject {
3715        return None;
3716    }
3717    let path = rel_dir.join(name);
3718    let (control, control_error) = if read_control {
3719        match read_control_op(config, root, &path, kind) {
3720            Ok(control) => (control, None),
3721            Err(error) => (None, Some(error)),
3722        }
3723    } else {
3724        (None, None)
3725    };
3726    Some(PreparedWalkEntry {
3727        path,
3728        kind,
3729        attrs,
3730        retained: disposition == crate::admission::Disposition::Retain,
3731        control,
3732        descend: should_descend(kind, attrs, depth, root_dev, config),
3733        control_error,
3734    })
3735}
3736
3737#[allow(clippy::too_many_arguments)]
3738fn record_walk_entry(
3739    root: &Path,
3740    rel_dir: &Path,
3741    depth: usize,
3742    region: RegionId,
3743    name: &OsStr,
3744    kind: EntryKind,
3745    attrs: Attrs,
3746    root_dev: u64,
3747    config: &ScanConfig,
3748    emission: &mut StreamingEmission,
3749    discovered: &mut Vec<(PathBuf, usize, RegionId)>,
3750    report: &mut ScanReport,
3751    sender: &std::sync::mpsc::Sender<WalkMessage>,
3752    chunk_send_ns: &mut u64,
3753    diagnostics: Option<&ScanDiagnosticsRecorder>,
3754) -> bool {
3755    let read_control = !emission.control_probed();
3756    let Some(prepared) = prepare_walk_entry_reading(
3757        root,
3758        rel_dir,
3759        depth,
3760        name,
3761        kind,
3762        attrs,
3763        root_dev,
3764        config,
3765        read_control,
3766    ) else {
3767        return true;
3768    };
3769    emission.note_listed_control(prepared.control.as_ref(), prepared.control_error.as_ref());
3770    let (mut control, control_error) = (prepared.control, prepared.control_error);
3771    if let Some(error) = control_error {
3772        report.errors.push(error);
3773    }
3774    // A grouping emission holds an entry's control ahead of the entry itself, so no send
3775    // can take the entry, `.gitignore` included, before the control that governs it.
3776    if !prepared.retained || emission.group_directories {
3777        if let Some(control) = control.take() {
3778            emission.batch.push(ObservationOp::unconditional(control));
3779            if !emission.send_if_full(
3780                root,
3781                rel_dir,
3782                config,
3783                report,
3784                sender,
3785                chunk_send_ns,
3786                diagnostics,
3787            ) {
3788                return false;
3789            }
3790        }
3791        if !prepared.retained {
3792            return true;
3793        }
3794    }
3795    report.observe(kind, attrs);
3796    emission.batch.push(ObservationOp::unconditional(Op::Upsert {
3797        path: prepared.path.clone(),
3798        kind,
3799        attrs,
3800    }));
3801    if !emission.send_if_full(root, rel_dir, config, report, sender, chunk_send_ns, diagnostics) {
3802        return false;
3803    }
3804    if let Some(control) = control {
3805        emission.batch.push(ObservationOp::unconditional(control));
3806        if !emission.send_if_full(root, rel_dir, config, report, sender, chunk_send_ns, diagnostics)
3807        {
3808            return false;
3809        }
3810    }
3811    if prepared.descend {
3812        // A child of the root seeds a new region; everything deeper inherits its
3813        // parent's. Region membership therefore costs one integer copy and never
3814        // inspects a path.
3815        let child_region = if depth == 0 { RegionId::UNASSIGNED } else { region };
3816        discovered.push((prepared.path, depth + 1, child_region));
3817    }
3818    true
3819}
3820
3821/// Move one listing's control observations ahead of its entries, each kind in its order.
3822///
3823/// A listing that repeats `.gitignore` while the directory changes keeps its reads in
3824/// order, so the later one still wins, as it does in the detached builder. Only the rare
3825/// listing that holds a control moves; every other listing costs one pass over the tags
3826/// it has just written.
3827fn controls_first(listing: &mut [ObservationOp]) {
3828    let mut placed = 0;
3829    for position in 0..listing.len() {
3830        if matches!(listing[position].op, Op::ControlUpsert { .. } | Op::ControlRemove { .. }) {
3831            listing[placed..=position].rotate_right(1);
3832            placed += 1;
3833        }
3834    }
3835}
3836
3837/// Observe the control an entry at `path` stands for, if the scan's policy asks for
3838/// control state at all.
3839///
3840/// Every control observation goes through here -- each walk and reconcile site, and the
3841/// watch layer's verification -- so the policy cannot be forgotten at one of them. A
3842/// watch must honor it like a scan does: its scope has to equal the index's, the scope
3843/// carries this bit, and a verifier that read control files regardless would grow a
3844/// partial rule set, from whichever sources events touched, under a scope that says
3845/// there is none.
3846///
3847/// It is also where a listed name becomes a control read, by the rule the module
3848/// documentation of [`crate::control`] states: the directory's control is what a lookup
3849/// of `<dir>/.gitignore` resolves to. An entry named exactly `.gitignore` is that
3850/// lookup's target on every filesystem, so it is read through its own path with the kind
3851/// its listing observed, and costs nothing more than it did. A case variant such as
3852/// `.GITIGNORE` may or may not be the target, so the canonical path is looked up
3853/// ([`read_directory_control`]): the entry decides only whether to look. The observation
3854/// names the canonical path either way, and its bytes are the ones git reads. A name
3855/// that spells no control is answered without a system call.
3856pub(crate) fn read_control_op(
3857    config: &ScanConfig,
3858    root: &Path,
3859    path: &Path,
3860    kind: EntryKind,
3861) -> Result<Option<Op>> {
3862    if !config.read_controls {
3863        return Ok(None);
3864    }
3865    match crate::control::path_control_spelling(path) {
3866        Some(crate::control::ControlSpelling::Exact) => {
3867            read_control_op_unconditional(root, path, kind, config.control_limits.budget)
3868        }
3869        Some(crate::control::ControlSpelling::Variant) => {
3870            read_directory_control(config, root, &crate::control::sibling_control_path(path))
3871        }
3872        None => Ok(None),
3873    }
3874}
3875
3876/// Read a directory's control by looking up its canonical path, `<dir>/.gitignore`.
3877///
3878/// This is the rule itself: on a case-insensitive directory the lookup resolves to
3879/// whichever spelling the directory stores, as git's open does, and on a case-sensitive
3880/// one only to the exact name. `Ok(None)` means nothing resolves. It costs one metadata
3881/// lookup, and a read on a hit, so it runs only where a listing cannot stand in for it: a
3882/// narrowed population reads each directory's control before listing it, a classifying
3883/// transient fold whose batch fills first probes for it (`StreamingEmission::send_if_full`),
3884/// and a listed case variant resolves through it ([`read_control_op`]).
3885fn read_directory_control(
3886    config: &ScanConfig,
3887    root: &Path,
3888    control_path: &Path,
3889) -> Result<Option<Op>> {
3890    if !config.read_controls {
3891        return Ok(None);
3892    }
3893    let absolute = control_lookup_path(root, control_path);
3894    let found = look_up_control(&absolute, control_path, config.control_limits.budget);
3895    #[cfg(test)]
3896    {
3897        if let Some(error) =
3898            walk_hook(&absolute).and_then(|hook| hook(WalkHookPoint::ControlLookup(&absolute)))
3899        {
3900            return Err(Error::io(&absolute, error));
3901        }
3902    }
3903    found
3904}
3905
3906/// The lookup [`read_directory_control`] makes, at `absolute`, of the control it names
3907/// `control_path`.
3908fn look_up_control(
3909    absolute: &Path,
3910    control_path: &Path,
3911    budget: Option<usize>,
3912) -> Result<Option<Op>> {
3913    crate::counters::bump(|counts| counts.stats = counts.stats.saturating_add(1));
3914    let kind = match fs::symlink_metadata(absolute) {
3915        Ok(metadata) if metadata.file_type().is_file() => EntryKind::File,
3916        Ok(_) => EntryKind::Other,
3917        Err(error) if error.kind() == std::io::ErrorKind::NotFound => return Ok(None),
3918        Err(error) => return Err(Error::io(absolute, error)),
3919    };
3920    read_control_source(absolute, control_path, kind, budget)
3921}
3922
3923/// [`read_directory_control`], answering a miss with the removal of the directory's rules.
3924///
3925/// For a producer that saw one spelling of the control name vanish or fail: whether the
3926/// directory still has a control is the lookup's question, and a removal of rules the
3927/// table does not hold is inert.
3928pub(crate) fn read_directory_control_or_removal(
3929    config: &ScanConfig,
3930    root: &Path,
3931    control_path: &Path,
3932) -> Result<Option<Op>> {
3933    if !config.read_controls {
3934        return Ok(None);
3935    }
3936    Ok(Some(
3937        read_directory_control(config, root, control_path)?
3938            .unwrap_or_else(|| Op::ControlRemove { path: control_path.to_path_buf() }),
3939    ))
3940}
3941
3942/// Where a lookup of the control path `control_path` under `root` goes.
3943#[cfg(not(test))]
3944fn control_lookup_path(root: &Path, control_path: &Path) -> PathBuf {
3945    root.join(control_path)
3946}
3947
3948/// Where a lookup of the control path `control_path` under `root` goes, which a test may
3949/// resolve as a case-insensitive directory would ([`install_case_folding_control_lookup`]).
3950#[cfg(test)]
3951fn control_lookup_path(root: &Path, control_path: &Path) -> PathBuf {
3952    let absolute = root.join(control_path);
3953    let folds = CASE_FOLDING_LOOKUPS
3954        .read()
3955        .unwrap_or_else(std::sync::PoisonError::into_inner)
3956        .iter()
3957        .any(|folded| absolute.starts_with(folded));
3958    let Some(directory) = absolute.parent().filter(|_| folds) else {
3959        return absolute;
3960    };
3961    // A case-insensitive directory holds at most one spelling, and the lookup returns it.
3962    // A test tree may hold two only where the host is case-sensitive, and then the exact
3963    // name is what the host itself resolves.
3964    if fs::symlink_metadata(&absolute).is_ok() {
3965        return absolute;
3966    }
3967    let Ok(mut names) = fs::read_dir(directory) else {
3968        return absolute;
3969    };
3970    names
3971        .find_map(|item| {
3972            let name = item.ok()?.file_name();
3973            crate::control::control_spelling(&name).map(|_| directory.join(name))
3974        })
3975        .unwrap_or(absolute)
3976}
3977
3978/// Roots whose control lookups [`control_lookup_path`] resolves case-insensitively.
3979#[cfg(test)]
3980static CASE_FOLDING_LOOKUPS: std::sync::RwLock<Vec<PathBuf>> = std::sync::RwLock::new(Vec::new());
3981
3982/// Removes its root from [`CASE_FOLDING_LOOKUPS`] when dropped.
3983#[cfg(test)]
3984#[must_use = "the lookups fold only until the guard is dropped"]
3985pub(crate) struct CaseFoldingGuard(Vec<PathBuf>);
3986
3987#[cfg(test)]
3988impl Drop for CaseFoldingGuard {
3989    fn drop(&mut self) {
3990        CASE_FOLDING_LOOKUPS
3991            .write()
3992            .unwrap_or_else(std::sync::PoisonError::into_inner)
3993            .retain(|root| !self.0.contains(root));
3994    }
3995}
3996
3997/// Resolve every control lookup under `root` as a case-insensitive directory would, until
3998/// the guard drops: `<dir>/.gitignore` opens whichever spelling `<dir>` stores.
3999///
4000/// A case-sensitive test host cannot make a case-insensitive directory (an ext4 casefold
4001/// directory needs a kernel built with Unicode support and an empty directory to flag), so
4002/// this is how the rule's case-insensitive branch runs on every CI runner. It changes only
4003/// the lookup the rule depends on: listings still show stored names, and every other path
4004/// resolves as the host resolves it. A real case-insensitive volume is tested where the
4005/// temporary directory is one. `root` matches as given and canonical.
4006#[cfg(test)]
4007pub(crate) fn install_case_folding_control_lookup(root: &Path) -> CaseFoldingGuard {
4008    let mut roots = vec![root.to_path_buf()];
4009    if let Ok(canonical) = root.canonicalize() {
4010        if canonical != root {
4011            roots.push(canonical);
4012        }
4013    }
4014    CASE_FOLDING_LOOKUPS
4015        .write()
4016        .unwrap_or_else(std::sync::PoisonError::into_inner)
4017        .extend(roots.iter().cloned());
4018    CaseFoldingGuard(roots)
4019}
4020
4021fn read_listed_control_op(
4022    config: &ScanConfig,
4023    root: &Path,
4024    path: &Path,
4025    kind: EntryKind,
4026) -> Result<Option<Op>> {
4027    if config.population != crate::query::IgnoredEntries::Include
4028        && crate::control::path_control_spelling(path).is_some()
4029    {
4030        // The exclusion walk already looked the directory's control up before any
4031        // sibling, whichever spelling holds it.
4032        Ok(None)
4033    } else {
4034        read_control_op(config, root, path, kind)
4035    }
4036}
4037
4038fn population_prunes(
4039    population: crate::query::IgnoredEntries,
4040    path: &Path,
4041    kind: EntryKind,
4042    disposition: crate::admission::Disposition,
4043    controls: Option<&crate::control::ControlTable>,
4044    unreadable_controls: &std::collections::BTreeSet<PathBuf>,
4045) -> bool {
4046    // The fixed control source stays in the entry tier even for `only`: removing its
4047    // entry would also remove the retained rule source during a watch reconciliation.
4048    // Every spelling is kept, because the listing cannot tell which one a case-insensitive
4049    // directory resolves `.gitignore` to without the lookup the walk made before it.
4050    if crate::control::path_control_spelling(path).is_some() {
4051        return false;
4052    }
4053    let Some(table) = controls else { return false };
4054    if !table.classification_known(path)
4055        || path.ancestors().skip(1).any(|ancestor| unreadable_controls.contains(ancestor))
4056    {
4057        return false;
4058    }
4059    let ignored = table.is_ignored(path, kind.is_dir());
4060    match population {
4061        crate::query::IgnoredEntries::Include => false,
4062        crate::query::IgnoredEntries::Exclude => ignored,
4063        crate::query::IgnoredEntries::Only => {
4064            !ignored && !kind.is_dir() && disposition != crate::admission::Disposition::ControlOnly
4065        }
4066    }
4067}
4068
4069fn apply_discovery_control(table: &mut crate::control::ControlTable, op: &Op) -> Result<()> {
4070    match op {
4071        Op::ControlUpsert { path, source } => {
4072            table.upsert(path, source.clone())?;
4073        }
4074        Op::ControlRemove { path } => {
4075            table.remove(path)?;
4076        }
4077        _ => unreachable!("directory control probe emits only control operations"),
4078    }
4079    Ok(())
4080}
4081
4082/// Read one fixed control source without allowing a raced or hostile file to allocate
4083/// beyond the index-wide control budget.
4084///
4085/// A file longer than the budget is read only to one byte past it. No table under that
4086/// budget can admit a source that long, since every retained byte is charged at least
4087/// once, so the truncated source it sends is refused for the budget rather than parsed.
4088///
4089/// Private to this module, so no caller elsewhere can step around the policy gate in
4090/// `read_control_op`.
4091fn read_control_op_unconditional(
4092    root: &Path,
4093    path: &Path,
4094    kind: EntryKind,
4095    budget: Option<usize>,
4096) -> Result<Option<Op>> {
4097    if !crate::control::is_control_file(path) {
4098        return Ok(None);
4099    }
4100    read_control_source(&root.join(path), path, kind, budget)
4101}
4102
4103/// Read the control at `absolute`, whose kind is `kind`, as an observation of `path`.
4104///
4105/// `absolute` is where the lookup went and `path` the canonical control path the
4106/// observation names; on a real filesystem the first is `root.join(path)`.
4107fn read_control_source(
4108    absolute: &Path,
4109    path: &Path,
4110    kind: EntryKind,
4111    budget: Option<usize>,
4112) -> Result<Option<Op>> {
4113    if kind != EntryKind::File {
4114        return Ok(Some(Op::ControlRemove { path: path.to_path_buf() }));
4115    }
4116    let file = open_control_file(absolute).map_err(|error| Error::io(absolute, error))?;
4117    if !file.metadata().map_err(|error| Error::io(absolute, error))?.file_type().is_file() {
4118        return Ok(Some(Op::ControlRemove { path: path.to_path_buf() }));
4119    }
4120    let read_limit = budget
4121        .map_or(u64::MAX, |budget| u64::try_from(budget).unwrap_or(u64::MAX).saturating_add(1));
4122    let mut source = Vec::new();
4123    file.take(read_limit).read_to_end(&mut source).map_err(|error| Error::io(absolute, error))?;
4124    crate::counters::bump(|counts| counts.control_reads = counts.control_reads.saturating_add(1));
4125    Ok(Some(Op::ControlUpsert { path: path.to_path_buf(), source }))
4126}
4127
4128#[cfg(unix)]
4129fn open_control_file(path: &Path) -> std::io::Result<fs::File> {
4130    use std::os::unix::fs::OpenOptionsExt as _;
4131
4132    fs::OpenOptions::new().read(true).custom_flags(libc::O_NONBLOCK | libc::O_NOFOLLOW).open(path)
4133}
4134
4135#[cfg(not(unix))]
4136fn open_control_file(path: &Path) -> std::io::Result<fs::File> {
4137    fs::File::open(path)
4138}
4139
4140fn send_scanner_batch(
4141    sender: &std::sync::mpsc::Sender<WalkMessage>,
4142    batch: ScannerBatch,
4143    diagnostics: Option<&ScanDiagnosticsRecorder>,
4144) -> bool {
4145    if let Some(diagnostics) = diagnostics {
4146        diagnostics.handoff_sent();
4147    }
4148    let sent = sender.send(WalkMessage::Batch(batch)).is_ok();
4149    if !sent {
4150        // Balance the reservation when the receiver disappeared before accepting it.
4151        if let Some(diagnostics) = diagnostics {
4152            diagnostics.handoff_received();
4153        }
4154    }
4155    sent
4156}
4157
4158fn send_detached_directories(
4159    sender: &std::sync::mpsc::Sender<WalkMessage>,
4160    directories: Vec<DetachedDirectory>,
4161    recycle: std::sync::mpsc::Sender<Vec<DetachedDirectory>>,
4162    diagnostics: Option<&ScanDiagnosticsRecorder>,
4163) -> bool {
4164    if let Some(diagnostics) = diagnostics {
4165        diagnostics.handoff_sent();
4166    }
4167    let sent = sender.send(WalkMessage::DetachedDirectories { directories, recycle }).is_ok();
4168    if !sent {
4169        if let Some(diagnostics) = diagnostics {
4170            diagnostics.handoff_received();
4171        }
4172    }
4173    sent
4174}
4175
4176/// Nanoseconds since `started`, saturating rather than panicking on the absurd.
4177fn elapsed_ns(started: std::time::Instant) -> u64 {
4178    u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)
4179}
4180
4181/// A top-level subtree, used to spread workers across the breadth of the tree.
4182///
4183/// Every directory below the root belongs to the region seeded by its depth-1
4184/// ancestor, inherited from its parent rather than recomputed from its path. The root
4185/// itself is [`RegionId::ROOT`], which exists only to bootstrap.
4186#[derive(Clone, Copy, PartialEq, Eq, Debug)]
4187struct RegionId(usize);
4188
4189impl RegionId {
4190    /// The root's own region, which exists only to bootstrap the walk.
4191    const ROOT: Self = Self(0);
4192    /// "Allocate a fresh region for this directory." Resolved by
4193    /// [`DirectoryQueueState::push`], which is the only place holding the lock that
4194    /// owns the region table.
4195    const UNASSIGNED: Self = Self(usize::MAX);
4196}
4197
4198/// Directories still to read, plus enough state to know when the walk is finished.
4199///
4200/// The termination condition is the only subtle part: the queue being empty does not
4201/// mean the walk is done, because a worker that is mid-directory may be about to push
4202/// its children. So a worker holds a claim from the moment it takes work until the
4203/// moment it has published everything that work produced, and the walk ends only when
4204/// the queue is empty *and* no claim is outstanding.
4205///
4206/// # Why breadth-first is region-scheduled rather than a global FIFO
4207///
4208/// A global FIFO orders the *queue*, but claims are unordered: workers take whatever
4209/// is at the front, which on a real tree means several workers grinding through the
4210/// same top-level subtree while others sit untouched. Measured on the branching
4211/// fixture, that left a global-FIFO walk starting the same 7–8 of 12 subtrees at the
4212/// halfway mark as depth-first did — the ordering bought nothing a consumer could see.
4213/// It also made the pending set hold a whole level of the tree, which is where the
4214/// +1.5–3.7% peak RSS in exp-012 came from.
4215///
4216/// So breadth-first keeps work in per-region buckets and hands each free worker a
4217/// *different* region, round-robin. Within a region the bucket is LIFO, which restores
4218/// depth-first's locality and spine-bounded memory. Nothing waits on a level boundary:
4219/// if only one region has work, every worker takes it. The result is a scheduler whose
4220/// shallow preference is expressed in *which subtree a worker picks up*, not in the
4221/// order a single queue drains — which is the property progressive consumers actually
4222/// need.
4223struct DirectoryQueue {
4224    state: std::sync::Mutex<DirectoryQueueState>,
4225    ready: std::sync::Condvar,
4226    order: ScanOrder,
4227    diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
4228}
4229
4230/// One outstanding claim, held for exactly as long as the worker owes the queue the
4231/// work it took.
4232///
4233/// Giving the claim back is the queue's liveness condition, not a courtesy: [`claim`]
4234/// parks every other worker on the condvar while `outstanding` is nonzero, so a single
4235/// claim that is never returned stops the whole walk and the scoped join that waits on
4236/// it. That makes `Drop` the only safe place to put the release, because the paths that
4237/// skip a hand-written call are exactly the ones that matter — the `break` taken when
4238/// the consumer disconnects, and an unwinding panic inside a directory read.
4239///
4240/// [`claim`]: DirectoryQueue::claim
4241struct DirectoryClaim<'a> {
4242    queue: &'a DirectoryQueue,
4243    /// Directories represented by the claim, for exact in-flight accounting.
4244    directories: usize,
4245    /// Whether the worker already returned this claim through [`Self::release`].
4246    released: bool,
4247}
4248
4249impl DirectoryClaim<'_> {
4250    /// Return the claim at the end of a completed chunk, feeding the chunk's own
4251    /// measurements to the shared calibration.
4252    ///
4253    /// Returns the queue's scale-up decision, which is why the normal path cannot be
4254    /// `Drop`: a destructor has neither the chunk's timing nor anywhere to put an
4255    /// answer.
4256    fn release(
4257        mut self,
4258        entries: u64,
4259        work_ns: u64,
4260        timing: &mut WalkAttribution,
4261    ) -> Option<usize> {
4262        self.released = true;
4263        self.queue.release(self.directories, entries, work_ns, timing)
4264    }
4265}
4266
4267impl Drop for DirectoryClaim<'_> {
4268    fn drop(&mut self) {
4269        if self.released {
4270            return;
4271        }
4272        // An abandoned chunk: the consumer went away mid-directory, or a read panicked.
4273        // Either way the partial timing describes an aborted chunk rather than the cost
4274        // of reading directories, so it must not reach the calibration that sizes the
4275        // worker pool. Returning the claim is the whole job.
4276        self.queue.abandon(self.directories);
4277    }
4278}
4279
4280struct DirectoryQueueState {
4281    /// Depth-first's single stack. Unused under breadth-first.
4282    pending: VecDeque<(PathBuf, usize, RegionId)>,
4283    /// Breadth-first's per-region work, indexed by [`RegionId`]. Each is a LIFO stack.
4284    regions: Vec<Vec<(PathBuf, usize, RegionId)>>,
4285    /// Regions with work, in round-robin order. A region appears at most once; the
4286    /// flag array is what keeps that true without scanning the ring.
4287    ready_ring: VecDeque<RegionId>,
4288    /// Whether each region is currently in `ready_ring`.
4289    enqueued: Vec<bool>,
4290    /// Directories currently available for a future claim.
4291    ready_directories: usize,
4292    /// Directories held by outstanding claims.
4293    in_flight_directories: usize,
4294    outstanding: usize,
4295    finished: bool,
4296    controller: Option<WorkerController>,
4297    /// Observation-only windows retained after the shipped one-shot decision.
4298    shadow_calibration: Option<RepeatedCalibration>,
4299    /// Completion-order sequence assigned under the queue lock.
4300    next_policy_sequence: u64,
4301    worker_target: usize,
4302    maximum_workers: usize,
4303}
4304
4305impl DirectoryQueueState {
4306    fn seeded(
4307        root: (PathBuf, usize),
4308        order: ScanOrder,
4309        calibration: Option<WorkerCalibration>,
4310        initial_workers: usize,
4311        maximum_workers: usize,
4312        policy: WorkerPolicyExperiment,
4313    ) -> Self {
4314        let mut state = Self {
4315            pending: VecDeque::new(),
4316            regions: Vec::new(),
4317            ready_ring: VecDeque::new(),
4318            enqueued: Vec::new(),
4319            ready_directories: 0,
4320            in_flight_directories: 0,
4321            outstanding: 0,
4322            finished: false,
4323            controller: calibration.map(|value| WorkerController::new(value, policy)),
4324            shadow_calibration: None,
4325            next_policy_sequence: 0,
4326            worker_target: initial_workers,
4327            maximum_workers,
4328        };
4329        state.push((root.0, root.1, RegionId::ROOT), order);
4330        state
4331    }
4332
4333    /// Push one directory into the structure the order uses.
4334    fn push(&mut self, item: (PathBuf, usize, RegionId), order: ScanOrder) {
4335        self.ready_directories = self.ready_directories.saturating_add(1);
4336        match order {
4337            ScanOrder::DepthFirst => self.pending.push_back(item),
4338            ScanOrder::BreadthFirst => {
4339                let region = if item.2 == RegionId::UNASSIGNED {
4340                    // One region per top-level subtree, numbered as they are found.
4341                    self.regions.len().max(1)
4342                } else {
4343                    item.2.0
4344                };
4345                if region >= self.regions.len() {
4346                    self.regions.resize_with(region + 1, Vec::new);
4347                    self.enqueued.resize(region + 1, false);
4348                }
4349                // Resolve the id *into* the item, so every directory discovered beneath
4350                // this one inherits a concrete region instead of the sentinel. Without
4351                // this the sentinel propagates and each directory allocates a region of
4352                // its own, degenerating the scheduler into round-robin over the whole
4353                // frontier.
4354                let mut item = item;
4355                item.2 = RegionId(region);
4356                self.regions[region].push(item);
4357                if !self.enqueued[region] {
4358                    self.enqueued[region] = true;
4359                    self.ready_ring.push_back(RegionId(region));
4360                }
4361            }
4362        }
4363    }
4364
4365    /// Whether any work is available.
4366    fn is_empty(&self, order: ScanOrder) -> bool {
4367        match order {
4368            ScanOrder::DepthFirst => self.pending.is_empty(),
4369            ScanOrder::BreadthFirst => self.ready_ring.is_empty(),
4370        }
4371    }
4372
4373    /// Take up to `limit` directories from the next region in the round-robin ring.
4374    ///
4375    /// Every region holding work is in the ring exactly once, so popping it always
4376    /// finds work and always moves to a *different* subtree than the previous claim.
4377    /// An earlier version preferred the caller's previous region for locality, which
4378    /// pinned each worker to one subtree: with twelve deep chains and six workers only
4379    /// six subtrees ever advanced, and depth-first — whose four-directory claims
4380    /// happen to fan across the root's children — spread wider than breadth-first did.
4381    /// Locality still comes from the claim being a run of directories out of one
4382    /// region; it must not come from a worker refusing to leave.
4383    fn take(
4384        &mut self,
4385        limit: usize,
4386        order: ScanOrder,
4387        into: &mut Vec<(PathBuf, usize, RegionId)>,
4388    ) -> usize {
4389        let before = into.len();
4390        match order {
4391            ScanOrder::DepthFirst => {
4392                let take = self.pending.len().min(limit);
4393                let start = self.pending.len() - take;
4394                into.extend(self.pending.drain(start..));
4395            }
4396            ScanOrder::BreadthFirst => {
4397                let Some(region) = self.ready_ring.pop_front() else { return 0 };
4398                self.enqueued[region.0] = false;
4399                let bucket = &mut self.regions[region.0];
4400                let take = bucket.len().min(limit);
4401                let start = bucket.len() - take;
4402                into.extend(bucket.drain(start..));
4403                // Re-arm the region only if work remains and it is not already queued,
4404                // so a busy region cannot appear twice and starve the others.
4405                if !bucket.is_empty() && !self.enqueued[region.0] {
4406                    self.enqueued[region.0] = true;
4407                    self.ready_ring.push_back(region);
4408                }
4409            }
4410        }
4411        into.len().saturating_sub(before)
4412    }
4413
4414    fn allocate_policy_sequence(&mut self) -> u64 {
4415        let sequence = self.next_policy_sequence;
4416        self.next_policy_sequence = self.next_policy_sequence.saturating_add(1);
4417        sequence
4418    }
4419}
4420
4421impl DirectoryQueue {
4422    /// Seed the queue with the root, which is region zero until its children fan out.
4423    #[cfg(test)]
4424    fn new(
4425        root: (PathBuf, usize),
4426        order: ScanOrder,
4427        calibration: Option<WorkerCalibration>,
4428        diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
4429    ) -> Self {
4430        Self::new_with_policy(
4431            root,
4432            order,
4433            calibration,
4434            diagnostics,
4435            1,
4436            2,
4437            WorkerPolicyExperiment::ShippedOneShot,
4438        )
4439    }
4440
4441    #[allow(clippy::too_many_arguments)]
4442    fn new_with_policy(
4443        root: (PathBuf, usize),
4444        order: ScanOrder,
4445        calibration: Option<WorkerCalibration>,
4446        diagnostics: Option<std::sync::Arc<ScanDiagnosticsRecorder>>,
4447        initial_workers: usize,
4448        maximum_workers: usize,
4449        policy: WorkerPolicyExperiment,
4450    ) -> Self {
4451        let state = DirectoryQueueState::seeded(
4452            root,
4453            order,
4454            calibration,
4455            initial_workers,
4456            maximum_workers,
4457            policy,
4458        );
4459        Self {
4460            state: std::sync::Mutex::new(state),
4461            ready: std::sync::Condvar::new(),
4462            order,
4463            diagnostics,
4464        }
4465    }
4466
4467    /// Take up to [`DIR_CLAIM`] directories, blocking until there is work or the walk
4468    /// is over. Returns `None` once no more work will ever arrive.
4469    ///
4470    /// Time spent waiting is charged to `timing`: lock acquisition to `lock_wait_ns`
4471    /// when contended, condvar waits to `starved_ns`. The condvar span includes the
4472    /// lock re-acquisition on wake, which slightly overstates starvation rather than
4473    /// understating contention — the fail-honest direction for the number that is
4474    /// supposed to stay near zero.
4475    fn claim<'a>(
4476        &'a self,
4477        into: &mut Vec<(PathBuf, usize, RegionId)>,
4478        timing: &mut WalkAttribution,
4479    ) -> Option<DirectoryClaim<'a>> {
4480        let mut state = self.lock_timed(timing);
4481        loop {
4482            if !state.is_empty(self.order) {
4483                let directories = state.take(DIR_CLAIM, self.order, into);
4484                state.ready_directories = state.ready_directories.saturating_sub(directories);
4485                state.in_flight_directories =
4486                    state.in_flight_directories.saturating_add(directories);
4487                state.outstanding += 1;
4488                timing.claims += 1;
4489                return Some(DirectoryClaim { queue: self, directories, released: false });
4490            }
4491            if state.finished {
4492                return None;
4493            }
4494            if state.outstanding == 0 {
4495                state.finished = true;
4496                self.ready.notify_all();
4497                return None;
4498            }
4499            let started = std::time::Instant::now();
4500            state = self.ready.wait(state).unwrap_or_else(std::sync::PoisonError::into_inner);
4501            timing.starved_ns += elapsed_ns(started);
4502        }
4503    }
4504
4505    fn extend(
4506        &self,
4507        directories: impl Iterator<Item = (PathBuf, usize, RegionId)>,
4508        timing: &mut WalkAttribution,
4509    ) {
4510        let mut state = self.lock_timed(timing);
4511        for item in directories {
4512            state.push(item, self.order);
4513        }
4514        drop(state);
4515        self.ready.notify_all();
4516    }
4517
4518    /// Give up a claim whose chunk never finished. Wakes everyone if it was the last.
4519    ///
4520    /// Reached only from [`DirectoryClaim::drop`], where there is no `WalkAttribution`
4521    /// to charge and nothing worth charging: an abandoned chunk read some unknown
4522    /// fraction of its directories, so its lock wait says nothing about contention
4523    /// during the walk.
4524    fn abandon(&self, directories: usize) {
4525        let mut state = self.lock();
4526        state.outstanding -= 1;
4527        state.in_flight_directories = state.in_flight_directories.saturating_sub(directories);
4528        if state.outstanding == 0 && state.is_empty(self.order) {
4529            state.finished = true;
4530            drop(state);
4531            self.ready.notify_all();
4532        }
4533    }
4534
4535    /// Give up a claim taken by [`claim`]. Wakes everyone if this was the last one.
4536    ///
4537    /// The chunk's own entry count and work time feed the shared service-time
4538    /// calibration, so the decision uses the timing the walk already collects for
4539    /// attribution rather than a second clock. Returns the new worker target when a
4540    /// controller requests expansion. A release that ends the walk returns no target,
4541    /// because there is no longer useful work for a reserve worker to take.
4542    fn release(
4543        &self,
4544        directories: usize,
4545        observed_entries: u64,
4546        observed_work_ns: u64,
4547        timing: &mut WalkAttribution,
4548    ) -> Option<usize> {
4549        let mut state = self.lock_timed(timing);
4550        // Whether this chunk reached the calibration at all. Only chunks released while
4551        // it is still live are policy history; later ones are ordinary walk work.
4552        let calibrating = state.controller.is_some();
4553        let one_shot = state.controller.as_ref().is_some_and(WorkerController::is_one_shot);
4554        let controller_spec = state.controller.as_ref().map(WorkerController::calibration_spec);
4555        let staged_gated = state.controller.as_ref().is_some_and(WorkerController::is_staged_gated);
4556        let completed_window = state
4557            .controller
4558            .as_mut()
4559            .and_then(|value| value.observe(observed_entries, observed_work_ns));
4560        let observed_window = state
4561            .shadow_calibration
4562            .as_mut()
4563            .and_then(|value| value.observe(observed_entries, observed_work_ns));
4564        let window_completed = completed_window.is_some();
4565        state.outstanding -= 1;
4566        state.in_flight_directories = state.in_flight_directories.saturating_sub(directories);
4567        let finished = state.outstanding == 0 && state.is_empty(self.order);
4568        if finished {
4569            state.finished = true;
4570        }
4571        let handoff_backlog = self.diagnostics.as_ref().map_or(0, |diagnostics| {
4572            diagnostics.handoff_backlog.load(std::sync::atomic::Ordering::Relaxed)
4573        });
4574        let mut requested_workers = None;
4575        let policy_window = completed_window.map(|window| {
4576            let useful_frontier =
4577                state.ready_directories.saturating_add(state.in_flight_directories);
4578            let decision = if !window.slow {
4579                WorkerPolicyDecision::Hold
4580            } else if finished {
4581                WorkerPolicyDecision::HoldNoUsefulWork
4582            } else if staged_gated && useful_frontier <= state.worker_target {
4583                WorkerPolicyDecision::HoldInsufficientFrontier
4584            } else if staged_gated && handoff_backlog >= state.worker_target {
4585                WorkerPolicyDecision::HoldHandoffBacklog
4586            } else {
4587                let target = if staged_gated {
4588                    state.worker_target.saturating_mul(2).min(state.maximum_workers)
4589                } else {
4590                    state.maximum_workers
4591                };
4592                if target > state.worker_target {
4593                    state.worker_target = target;
4594                    requested_workers = Some(target);
4595                    WorkerPolicyDecision::ScaleUp
4596                } else {
4597                    WorkerPolicyDecision::HoldInsufficientFrontier
4598                }
4599            };
4600            let sequence = state.allocate_policy_sequence();
4601            PolicyWindowSnapshot {
4602                sequence,
4603                start_entry_ordinal: window.start_entry_ordinal,
4604                end_entry_ordinal: window.end_entry_ordinal,
4605                observed_entries: window.entries,
4606                observed_chunks: window.chunks,
4607                observed_work_ns: window.work_ns,
4608                ready_directories: state.ready_directories,
4609                in_flight_directories: state.in_flight_directories,
4610                active_workers: self.diagnostics.as_ref().map_or(0, |diagnostics| {
4611                    diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
4612                }),
4613                handoff_backlog,
4614                requested_workers,
4615                decision,
4616            }
4617        });
4618        let shadow_window = observed_window.map(|window| {
4619            let sequence = state.allocate_policy_sequence();
4620            PolicyWindowSnapshot {
4621                sequence,
4622                start_entry_ordinal: window.start_entry_ordinal,
4623                end_entry_ordinal: window.end_entry_ordinal,
4624                observed_entries: window.entries,
4625                observed_chunks: window.chunks,
4626                observed_work_ns: window.work_ns,
4627                ready_directories: state.ready_directories,
4628                in_flight_directories: state.in_flight_directories,
4629                active_workers: self.diagnostics.as_ref().map_or(0, |diagnostics| {
4630                    diagnostics.active_workers.load(std::sync::atomic::Ordering::Relaxed)
4631                }),
4632                handoff_backlog,
4633                requested_workers: None,
4634                decision: if window.slow {
4635                    WorkerPolicyDecision::ObserveSlow
4636                } else {
4637                    WorkerPolicyDecision::ObserveFast
4638                },
4639            }
4640        });
4641        let controller_terminates = (one_shot || finished) && window_completed
4642            || requested_workers.is_some_and(|target| target == state.maximum_workers);
4643        if controller_terminates {
4644            // Continue observing after every terminal decision, including a candidate
4645            // that reached the maximum pool. Without this shadow history a slow-prefix
4646            // expansion makes a later fast phase unobservable, precisely the
4647            // irreversible over-expansion case the evidence matrix must detect.
4648            if let (Some(spec), Some(window)) = (controller_spec, completed_window) {
4649                if self.diagnostics.is_some() && !finished {
4650                    state.shadow_calibration =
4651                        Some(RepeatedCalibration::starting_at(spec, window.end_entry_ordinal));
4652                }
4653            }
4654            state.controller = None;
4655        }
4656        drop(state);
4657
4658        // The sequence was assigned while the queue was locked, but the trace lock and
4659        // bounded-vector update stay outside that critical section. Recorder arrival
4660        // may differ from completion order; `finish` sorts the retained prefix.
4661        if let (Some(diagnostics), Some(window)) = (&self.diagnostics, policy_window) {
4662            diagnostics.record_policy_window(window);
4663        }
4664        if let (Some(diagnostics), Some(window)) = (&self.diagnostics, shadow_window) {
4665            diagnostics.record_policy_window(window);
4666        }
4667
4668        // Recorded outside the lock: the sampling is off by default, and a disabled
4669        // counter must not lengthen the critical section it observes.
4670        if calibrating {
4671            record_adaptive_calibration_chunk(
4672                self.diagnostics.as_ref(),
4673                observed_entries,
4674                observed_work_ns,
4675            );
4676        }
4677
4678        if finished {
4679            self.ready.notify_all();
4680        }
4681        requested_workers
4682    }
4683
4684    /// Acquire the state lock, charging any contention to `timing`.
4685    ///
4686    /// The fast path is a `try_lock` that succeeds and costs one counter increment;
4687    /// only the contended path pays for reading the clock. Poisoning is tolerated for
4688    /// the same reason as [`Self::lock`].
4689    fn lock_timed(
4690        &self,
4691        timing: &mut WalkAttribution,
4692    ) -> std::sync::MutexGuard<'_, DirectoryQueueState> {
4693        timing.lock_ops += 1;
4694        match self.state.try_lock() {
4695            Ok(guard) => guard,
4696            Err(std::sync::TryLockError::Poisoned(poisoned)) => poisoned.into_inner(),
4697            Err(std::sync::TryLockError::WouldBlock) => {
4698                timing.lock_contended += 1;
4699                let started = std::time::Instant::now();
4700                let guard = self.lock();
4701                timing.lock_wait_ns += elapsed_ns(started);
4702                guard
4703            }
4704        }
4705    }
4706
4707    /// A poisoned queue means a worker panicked mid-walk. The data behind the lock is
4708    /// a plain work list with no invariant that a panic could have broken, and the
4709    /// caller already reports the panic as a scan error, so recovering the list is
4710    /// strictly better than propagating a second panic into every other worker.
4711    fn lock(&self) -> std::sync::MutexGuard<'_, DirectoryQueueState> {
4712        self.state.lock().unwrap_or_else(std::sync::PoisonError::into_inner)
4713    }
4714}
4715
4716fn record_adaptive_calibration_chunk(
4717    diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>,
4718    entries: u64,
4719    work_ns: u64,
4720) {
4721    if let Some(diagnostics) = diagnostics {
4722        diagnostics.calibration_chunk(entries, work_ns);
4723    }
4724    crate::counters::bump(|counts| {
4725        counts.adaptive_calibration_chunks = counts.adaptive_calibration_chunks.saturating_add(1);
4726        counts.adaptive_calibration_entries =
4727            counts.adaptive_calibration_entries.saturating_add(entries);
4728        counts.adaptive_calibration_work_us =
4729            counts.adaptive_calibration_work_us.saturating_add(work_ns / 1_000);
4730    });
4731}
4732
4733fn record_adaptive_worker_expansion(diagnostics: Option<&std::sync::Arc<ScanDiagnosticsRecorder>>) {
4734    if let Some(diagnostics) = diagnostics {
4735        diagnostics.worker_expanded();
4736    }
4737    crate::counters::bump(|counts| {
4738        counts.adaptive_scale_ups = counts.adaptive_scale_ups.saturating_add(1);
4739    });
4740}
4741
4742fn scan_detached_directories(
4743    root: &Path,
4744    config: &ScanConfig,
4745    collect_diagnostics: bool,
4746    policy: WorkerPolicyExperiment,
4747) -> Result<(ScanReport, DetachedIndexBuilder, Option<ScanDiagnostics>)> {
4748    if let Some(progress) = &config.progress {
4749        progress.enter(crate::ProgressPhase::Scanning);
4750    }
4751    let root_metadata = {
4752        crate::counters::bump(|counts| counts.stats += 1);
4753        fs::symlink_metadata(root)
4754    }
4755    .map_err(|error| Error::io(root, error))?;
4756    if !root_metadata.is_dir() {
4757        return Err(Error::io(
4758            root,
4759            std::io::Error::new(std::io::ErrorKind::NotADirectory, "scan root is not a directory"),
4760        ));
4761    }
4762    let root_dev = root_device(root, &root_metadata).map_err(|error| Error::io(root, error))?;
4763    let available_parallelism =
4764        std::thread::available_parallelism().map_or(1, std::num::NonZero::get);
4765    let pool = config.worker_pool_for(available_parallelism);
4766    let diagnostics = collect_diagnostics
4767        .then(|| ScanDiagnosticsRecorder::new(pool, available_parallelism, policy));
4768
4769    if config.max_depth == Some(0) {
4770        if let Some(diagnostics) = &diagnostics {
4771            diagnostics.mark_not_run();
4772            diagnostics.record_queue_finish(0, 0);
4773        }
4774        return Ok((
4775            ScanReport::default(),
4776            DetachedIndexBuilder::new(root, config.scope(), config.types_shared())
4777                .with_control_limits(config.control_limits),
4778            diagnostics.as_ref().map(|value| value.finish()),
4779        ));
4780    }
4781
4782    let walk_started = crate::counters::enabled().then(std::time::Instant::now);
4783    let (output, builder) =
4784        scan_concurrent_detached(root, config, root_dev, pool, diagnostics.as_ref(), policy)?;
4785    // Also reached by a walk no worker left, such as a single-threaded one, so every
4786    // cold index ends its walk in the same phase.
4787    if let Some(progress) = &config.progress {
4788        progress.enter(crate::ProgressPhase::Indexing);
4789    }
4790    if let Some(started) = walk_started {
4791        let elapsed = elapsed_ns(started) / 1_000;
4792        crate::counters::bump(|counts| {
4793            counts.detached_walk_us = counts.detached_walk_us.saturating_add(elapsed);
4794        });
4795    }
4796    Ok((output, builder, diagnostics.as_ref().map(|value| value.finish())))
4797}
4798
4799fn consolidate_detached_index(
4800    mut output: ScanReport,
4801    builder: DetachedIndexBuilder,
4802) -> (Index, ScanReport) {
4803    let entries = output.entries;
4804    let consolidate_started = crate::counters::enabled().then(std::time::Instant::now);
4805    let mut index = builder.finish();
4806    if let Some(started) = consolidate_started {
4807        let elapsed = elapsed_ns(started) / 1_000;
4808        crate::counters::bump(|counts| {
4809            counts.detached_builds = counts.detached_builds.saturating_add(1);
4810            counts.detached_entries = counts.detached_entries.saturating_add(entries);
4811            counts.detached_finish_us = counts.detached_finish_us.saturating_add(elapsed);
4812        });
4813    }
4814    index.record_walk_errors(&mut output.errors);
4815    index.set_initial_scan_freshness(&output.errors);
4816    (index, output)
4817}
4818
4819/// Walk `root` and return a fully populated index.
4820pub fn scan_into_index(root: &Path, config: &ScanConfig) -> Result<(Index, ScanReport)> {
4821    config.validate()?;
4822    let root = root.canonicalize().map_err(|error| Error::io(root, error))?;
4823    if config.population != crate::query::IgnoredEntries::Include {
4824        let (index, report, _) = scan_into_index_with_scanner(
4825            &root,
4826            config,
4827            false,
4828            WorkerPolicyExperiment::ShippedOneShot,
4829        )?;
4830        return Ok((index, report));
4831    }
4832    let (output, builder, _diagnostics) =
4833        scan_detached_directories(&root, config, false, WorkerPolicyExperiment::ShippedOneShot)?;
4834    Ok(consolidate_detached_index(output, builder))
4835}
4836
4837fn scan_into_index_with_scanner(
4838    root: &Path,
4839    config: &ScanConfig,
4840    collect_diagnostics: bool,
4841    policy: WorkerPolicyExperiment,
4842) -> Result<(Index, ScanReport, Option<ScanDiagnostics>)> {
4843    let mut index = Index::new_with_scope_and_types(root, config.scope(), config.types_shared());
4844    index.set_control_limits(config.control_limits);
4845    let mut apply_error: Option<Error> = None;
4846    let (mut report, diagnostics) = scan_internal(
4847        root,
4848        config,
4849        &mut |batch| {
4850            if apply_error.is_none() {
4851                if let Err(error) = index.apply_scanner_baseline(batch) {
4852                    apply_error = Some(error);
4853                }
4854            }
4855        },
4856        collect_diagnostics,
4857        policy,
4858        SinkMode::Retained,
4859    )?;
4860    if let Some(error) = apply_error {
4861        return Err(error);
4862    }
4863    index.record_walk_errors(&mut report.errors);
4864    index.set_initial_scan_freshness(&report.errors);
4865    Ok((index, report, diagnostics))
4866}
4867
4868#[cfg(test)]
4869fn scan_into_index_via_scanner(root: &Path, config: &ScanConfig) -> Result<(Index, ScanReport)> {
4870    let (index, report, _) =
4871        scan_into_index_with_scanner(root, config, false, WorkerPolicyExperiment::ShippedOneShot)?;
4872    Ok((index, report))
4873}
4874
4875/// Walk `root` into an index and retain the opt-in diagnostic trace for that run.
4876///
4877/// The index and [`ScanReport`] have the same semantics as [`scan_into_index`]. The
4878/// additional trace is versioned independently so evidence tooling can fail closed on
4879/// changes without coupling cache or engine behavior to measurement details.
4880pub fn scan_into_index_with_diagnostics(
4881    root: &Path,
4882    config: &ScanConfig,
4883) -> Result<(Index, ScanReport, ScanDiagnostics)> {
4884    scan_into_index_with_policy_diagnostics(root, config, WorkerPolicyExperiment::ShippedOneShot)
4885}
4886
4887/// Exercise a repository-only worker-controller candidate while building an index.
4888#[doc(hidden)]
4889pub fn scan_into_index_with_policy_diagnostics(
4890    root: &Path,
4891    config: &ScanConfig,
4892    policy: WorkerPolicyExperiment,
4893) -> Result<(Index, ScanReport, ScanDiagnostics)> {
4894    config.validate()?;
4895    let root = root.canonicalize().map_err(|error| Error::io(root, error))?;
4896    if config.population != crate::query::IgnoredEntries::Include {
4897        let (index, report, diagnostics) =
4898            scan_into_index_with_scanner(&root, config, true, policy)?;
4899        return Ok((index, report, diagnostics.expect("diagnostic scanner creates a recorder")));
4900    }
4901    let (output, builder, diagnostics) = scan_detached_directories(&root, config, true, policy)?;
4902    let (index, report) = consolidate_detached_index(output, builder);
4903    Ok((index, report, diagnostics.expect("diagnostic detached scan creates a recorder")))
4904}
4905
4906/// Diff the filesystem against an existing index and emit conditional observations.
4907///
4908/// This is cache tier 2: after a snapshot is loaded, a sweep like this is what makes the
4909/// answer trustworthy rather than merely fast. Unchanged entries produce upserts whose
4910/// fingerprints already match, which the index discards as no-ops, so the caller can
4911/// apply the whole stream without filtering it first.
4912///
4913/// Entries the index holds but the filesystem no longer has become [`Op::Remove`],
4914/// detected per directory rather than by accumulating every visited path in memory.
4915///
4916/// This observation-only reference API assumes its emitted stream is applied to the same
4917/// unchanged baseline after the borrow ends. Use [`reconcile`] or [`reconcile_handle`]
4918/// when other producers can write concurrently; those paths capture the stronger
4919/// generation/revision/absence expectations returned by [`Index::expectation`].
4920pub fn revalidate(
4921    index: &Index,
4922    config: &ScanConfig,
4923    sink: &mut dyn FnMut(Observation),
4924) -> Result<ScanReport> {
4925    config.validate_for_scope(index.scope())?;
4926    let root = index.root_path().to_path_buf();
4927    let root_meta = {
4928        crate::counters::bump(|c| c.stats += 1);
4929        fs::symlink_metadata(&root)
4930    }
4931    .map_err(|error| Error::io(&root, error))?;
4932    if !root_meta.is_dir() {
4933        return Err(Error::io(
4934            &root,
4935            std::io::Error::new(
4936                std::io::ErrorKind::NotADirectory,
4937                "revalidation root is not a directory",
4938            ),
4939        ));
4940    }
4941    let root_dev = root_device(&root, &root_meta).map_err(|error| Error::io(&root, error))?;
4942    if let Some(progress) = &config.progress {
4943        progress.enter(crate::ProgressPhase::Revalidating);
4944    }
4945    let mut report = ScanReport::default();
4946    let mut tally = ProgressTally::new(config.progress.as_ref());
4947    let batch_limit = config.batch_size.max(1);
4948    let mut batch: Vec<ObservationOp> = Vec::with_capacity(batch_limit);
4949    if config.max_depth == Some(0) {
4950        if let Some(children) = index.children(Path::new("")) {
4951            for (name, _) in children {
4952                let path = PathBuf::from(name);
4953                batch.push(ObservationOp::if_state(
4954                    Op::Remove { path: path.clone() },
4955                    index.relaxed_expectation(&path),
4956                ));
4957                if batch.len() >= batch_limit {
4958                    sink(Observation::from_ops(std::mem::take(&mut batch)));
4959                    batch.reserve(batch_limit);
4960                }
4961            }
4962        }
4963        if !batch.is_empty() {
4964            sink(Observation::from_ops(batch));
4965        }
4966        return Ok(report);
4967    }
4968    let mut queue: VecDeque<(PathBuf, usize)> = VecDeque::from(vec![(PathBuf::new(), 0)]);
4969    let mut controls = (config.population != crate::query::IgnoredEntries::Include)
4970        .then(|| index.control_table().clone());
4971    let mut unreadable_controls = std::collections::BTreeSet::new();
4972
4973    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
4974        let abs_dir = root.join(&rel_dir);
4975        let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
4976        let mut had_control = index.control_table().contains(&control_path);
4977        let mut listed_control = ListedControl::default();
4978        if let Some(table) = controls.as_mut() {
4979            let baseline = index.relaxed_expectation(&control_path);
4980            let lookup = read_directory_control(config, &root, &control_path);
4981            listed_control.lookup(&lookup);
4982            match lookup {
4983                Ok(Some(op)) => {
4984                    apply_discovery_control(table, &op)?;
4985                    batch.push(ObservationOp::if_state(op, baseline));
4986                }
4987                Ok(None) => {
4988                    table.remove(&control_path)?;
4989                    if had_control {
4990                        batch.push(ObservationOp::if_state(
4991                            Op::ControlRemove { path: control_path.clone() },
4992                            baseline,
4993                        ));
4994                        had_control = false;
4995                    }
4996                }
4997                Err(error) => {
4998                    unreadable_controls.insert(rel_dir.clone());
4999                    report.errors.push(error);
5000                }
5001            }
5002        }
5003        crate::counters::bump(|c| c.dir_opens += 1);
5004        let listing = match fs::read_dir(&abs_dir) {
5005            Ok(listing) => listing,
5006            Err(e) => {
5007                report.errors.push(Error::io(abs_dir, e));
5008                continue;
5009            }
5010        };
5011        report.dirs_read += 1;
5012
5013        let mut seen: BTreeSet<OsString> = BTreeSet::new();
5014        let mut listing_complete = true;
5015        let listing = reconcile_listing(listing, &abs_dir);
5016        for item in listing {
5017            let item = match item {
5018                Ok(item) => item,
5019                Err(e) => {
5020                    listing_complete = false;
5021                    report.errors.push(Error::io(&abs_dir, e));
5022                    continue;
5023                }
5024            };
5025            let name = item.file_name();
5026            // Seeing the name proves it is not absent even when a following metadata
5027            // lookup fails. Record it before any fallible per-entry work so an
5028            // operational error cannot become a false removal in the missing sweep.
5029            seen.insert(name.clone());
5030            let rel_path = rel_dir.join(&name);
5031            let baseline = index.relaxed_expectation(&rel_path);
5032            let (kind, attrs) = match observe_dir_entry(&item) {
5033                Ok(Some(observed)) => observed,
5034                Ok(None) => {
5035                    let entry_held = baseline.state != PathState::Absent;
5036                    for removal in
5037                        vanished_child_removals(&rel_dir, &name, entry_held, &mut had_control)
5038                            .into_iter()
5039                            .flatten()
5040                    {
5041                        batch.push(ObservationOp::if_state(removal, baseline));
5042                    }
5043                    if batch.len() >= batch_limit {
5044                        sink(Observation::from_ops(std::mem::take(&mut batch)));
5045                        batch.reserve(batch_limit);
5046                    }
5047                    continue;
5048                }
5049                Err(e) => {
5050                    listed_control.listed(&name);
5051                    report.errors.push(Error::io(item.path(), e));
5052                    continue;
5053                }
5054            };
5055            listed_control.listed(&name);
5056            let disposition =
5057                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
5058            if population_prunes(
5059                config.population,
5060                &rel_path,
5061                kind,
5062                disposition,
5063                controls.as_ref(),
5064                &unreadable_controls,
5065            ) {
5066                if baseline.state != PathState::Absent {
5067                    batch.push(ObservationOp::if_state(Op::Remove { path: rel_path }, baseline));
5068                }
5069                continue;
5070            }
5071            let read = read_listed_control_op(config, &root, &rel_path, kind);
5072            listed_control.read(&name, &read);
5073            let control = match read {
5074                Ok(control) => control,
5075                Err(error) => {
5076                    report.errors.push(error);
5077                    None
5078                }
5079            };
5080            if disposition != crate::admission::Disposition::Retain {
5081                if baseline.state != PathState::Absent {
5082                    batch.push(ObservationOp::if_state(
5083                        Op::Remove { path: rel_path.clone() },
5084                        baseline,
5085                    ));
5086                }
5087                if disposition == crate::admission::Disposition::ControlOnly {
5088                    if let Some(control) = control {
5089                        let guard = control_guard(&rel_path, baseline, |path| {
5090                            index.relaxed_expectation(path)
5091                        });
5092                        batch.push(ObservationOp::if_state(control, guard));
5093                    }
5094                }
5095                if batch.len() >= batch_limit {
5096                    sink(Observation::from_ops(std::mem::take(&mut batch)));
5097                    batch.reserve(batch_limit);
5098                }
5099                continue;
5100            }
5101            report.observe(kind, attrs);
5102            batch.push(ObservationOp::if_state(
5103                Op::Upsert { path: rel_path.clone(), kind, attrs },
5104                baseline,
5105            ));
5106            if batch.len() >= batch_limit {
5107                sink(Observation::from_ops(std::mem::take(&mut batch)));
5108                batch.reserve(batch_limit);
5109            }
5110            if let Some(control) = control {
5111                let guard =
5112                    control_guard(&rel_path, baseline, |path| index.relaxed_expectation(path));
5113                batch.push(ObservationOp::if_state(control, guard));
5114                if batch.len() >= batch_limit {
5115                    sink(Observation::from_ops(std::mem::take(&mut batch)));
5116                    batch.reserve(batch_limit);
5117                }
5118            }
5119
5120            if should_descend(kind, attrs, depth, root_dev, config) {
5121                queue.push_back((rel_path, depth + 1));
5122            } else if kind.is_dir() {
5123                if let Some(children) = index.children(&rel_path) {
5124                    for (child_name, _) in children {
5125                        let child_path = rel_path.join(child_name);
5126                        batch.push(ObservationOp::if_state(
5127                            Op::Remove { path: child_path.clone() },
5128                            index.relaxed_expectation(&child_path),
5129                        ));
5130                        if batch.len() >= batch_limit {
5131                            sink(Observation::from_ops(std::mem::take(&mut batch)));
5132                            batch.reserve(batch_limit);
5133                        }
5134                    }
5135                }
5136            }
5137        }
5138
5139        // Anything the index still lists here but the filesystem did not return is gone.
5140        if listing_complete {
5141            if let Some(known) = index.children(&rel_dir) {
5142                for (name, _) in known {
5143                    if !seen.contains(name) {
5144                        let path = rel_dir.join(name);
5145                        batch.push(ObservationOp::if_state(
5146                            Op::Remove { path: path.clone() },
5147                            index.relaxed_expectation(&path),
5148                        ));
5149                        // In the same batch, so the rules the removal drops are back
5150                        // before any commit shows the directory without them.
5151                        if let Some(control) = listed_control.restatement_after_removing(name) {
5152                            batch.push(ObservationOp::if_state(
5153                                control,
5154                                index.relaxed_expectation(&control_path),
5155                            ));
5156                        }
5157                    }
5158                }
5159            }
5160            if had_control && !listed_control.seen {
5161                batch.push(ObservationOp::if_state(
5162                    Op::ControlRemove { path: control_path.clone() },
5163                    index.relaxed_expectation(&control_path),
5164                ));
5165            }
5166        }
5167        // Per directory: an unchanged tree fills no batch, so the batch cannot be the
5168        // unit here without the counters standing still for the whole walk.
5169        tally.flush(&report);
5170    }
5171
5172    if !batch.is_empty() {
5173        sink(Observation::from_ops(batch));
5174    }
5175    tally.flush(&report);
5176    Ok(report)
5177}
5178
5179/// Reconcile the full index and publish each exact commit as it lands.
5180pub fn reconcile(
5181    index: &mut Index,
5182    config: &ScanConfig,
5183    sink: &mut dyn FnMut(&Commit),
5184) -> Result<ReconcileReport> {
5185    reconcile_subtree(index, Path::new(""), config, sink)
5186}
5187
5188/// Reconcile one relative subtree, applying effective changes during the walk.
5189///
5190/// If an ancestor vanished or became a non-directory, reconciliation widens to that
5191/// ancestor so a child invalidation can converge instead of retrying `ENOTDIR` forever.
5192pub fn reconcile_subtree(
5193    index: &mut Index,
5194    subtree: &Path,
5195    config: &ScanConfig,
5196    sink: &mut dyn FnMut(&Commit),
5197) -> Result<ReconcileReport> {
5198    reconcile_target(&mut ReconcileTarget::Direct(index), subtree, config, sink)
5199}
5200
5201/// Reconcile a shared index while allowing readers between applied batches.
5202pub fn reconcile_handle(
5203    handle: &IndexHandle,
5204    config: &ScanConfig,
5205    sink: &mut dyn FnMut(&Commit),
5206) -> Result<ReconcileReport> {
5207    reconcile_subtree_handle(handle, Path::new(""), config, sink)
5208}
5209
5210/// Reconcile one subtree of a shared index, widening to a missing/non-directory ancestor
5211/// when necessary.
5212pub fn reconcile_subtree_handle(
5213    handle: &IndexHandle,
5214    subtree: &Path,
5215    config: &ScanConfig,
5216    sink: &mut dyn FnMut(&Commit),
5217) -> Result<ReconcileReport> {
5218    reconcile_target(&mut ReconcileTarget::Shared(handle), subtree, config, sink)
5219}
5220
5221/// Internal effects of one opened-root multi-path reconciliation.
5222#[derive(Debug, Default)]
5223pub(crate) struct ReconcilePathsReport {
5224    pub(crate) reconciliation: ReconcileReport,
5225    pub(crate) accepted: Vec<PathBuf>,
5226    pub(crate) rejected: Vec<crate::RejectedRefreshPath>,
5227}
5228
5229/// Reconcile one bounded path set under an opened-root lifecycle controller.
5230///
5231/// Classification precedes I/O, overlapping descendants fold into one walk, and all
5232/// surviving scopes enter `Reconciling` before the first is read. `forbid_expansion`
5233/// is the conservative resource-stop rule: removals and same-file verification remain
5234/// legal, while work that could retain another file or discover children is refused.
5235pub(crate) fn reconcile_paths_handle_controlled(
5236    handle: &IndexHandle,
5237    paths: &[PathBuf],
5238    config: &ScanConfig,
5239    forbid_expansion: bool,
5240    control: &dyn ReconcileControl,
5241    sink: &mut dyn FnMut(&Commit),
5242) -> Result<ReconcilePathsReport> {
5243    let mut target = ReconcileTarget::Controlled { handle, control };
5244    reconcile_paths_target(&mut target, paths, config, forbid_expansion, sink)
5245}
5246
5247fn reconcile_paths_target(
5248    target: &mut ReconcileTarget<'_>,
5249    paths: &[PathBuf],
5250    config: &ScanConfig,
5251    forbid_expansion: bool,
5252    sink: &mut dyn FnMut(&Commit),
5253) -> Result<ReconcilePathsReport> {
5254    config.validate_for_scope(target.scope()?)?;
5255    let mut report = ReconcilePathsReport::default();
5256    let mut accepted = BTreeSet::new();
5257
5258    for requested in paths {
5259        let reject = |reason| crate::RejectedRefreshPath { path: requested.clone(), reason };
5260        let Ok(path) = normalize_subtree(requested) else {
5261            report.rejected.push(reject(crate::RefreshRejection::OutsideRoot));
5262            continue;
5263        };
5264        if config.max_depth.is_some_and(|maximum| path.components().count() > maximum) {
5265            report.rejected.push(reject(crate::RefreshRejection::BeyondDepth));
5266            continue;
5267        }
5268        // This is lexical admission before the final kind is observed. Treating the
5269        // boundary as a file preserves the fixed hidden `.gitignore` control exception;
5270        // the verified walk still applies the real kind and special-object policy.
5271        if crate::admission::decide_path(&path, EntryKind::File, config.hidden(), false)
5272            == crate::admission::Disposition::Reject
5273        {
5274            report.rejected.push(reject(crate::RefreshRejection::NotAdmitted));
5275            continue;
5276        }
5277        if forbid_expansion && refresh_may_expand(target, &path, &mut report.reconciliation.scan)? {
5278            report.rejected.push(reject(crate::RefreshRejection::ResourceBudget));
5279            continue;
5280        }
5281        accepted.insert(path);
5282    }
5283
5284    report.accepted = accepted.into_iter().collect();
5285    let mut resolved = Vec::new();
5286    let mut unsafe_roots = Vec::new();
5287    for requested_root in covering_roots(report.accepted.clone()) {
5288        match resolve_subtree_root(target, &requested_root, config) {
5289            Ok(root) => resolved.push(root),
5290            Err(Error::SubtreeOutsideScanScope { .. }) => unsafe_roots.push(requested_root),
5291            Err(error) => return Err(error),
5292        }
5293    }
5294    if !unsafe_roots.is_empty() {
5295        let mut retained = Vec::with_capacity(report.accepted.len());
5296        for path in std::mem::take(&mut report.accepted) {
5297            if unsafe_roots.iter().any(|root| path.starts_with(root)) {
5298                report.rejected.push(crate::RejectedRefreshPath {
5299                    path,
5300                    reason: crate::RefreshRejection::UnsafeAncestry,
5301                });
5302            } else {
5303                retained.push(path);
5304            }
5305        }
5306        report.accepted = retained;
5307    }
5308    let walked = covering_roots(resolved);
5309    if walked.is_empty() {
5310        return Ok(report);
5311    }
5312
5313    let mut opened = Vec::with_capacity(walked.len());
5314    for subtree in walked {
5315        let (started_at, commit) = target.begin_reconcile(&subtree)?;
5316        if let Some(commit) = commit.as_ref() {
5317            sink(commit);
5318        }
5319        opened.push((subtree, started_at));
5320    }
5321
5322    // Each subtree closes on its own walk's outcome, so a subtree that could not be read
5323    // neither marks a verified sibling partial nor withholds the completeness its listing
5324    // earned.
5325    let mut failure = None;
5326    let mut outcomes = Vec::with_capacity(opened.len());
5327    for (subtree, started_at) in &opened {
5328        if failure.is_some() {
5329            outcomes.push((false, false));
5330            continue;
5331        }
5332        match reconcile_target_inner(
5333            target,
5334            subtree,
5335            *started_at,
5336            config,
5337            MAX_DEFERRED_RECONCILE_OPS,
5338            sink,
5339        ) {
5340            Ok(mut reconciliation) => {
5341                outcomes.push((
5342                    reconciliation.is_complete(),
5343                    reconciliation.apply.stale == 0 && reconciliation.apply.resource_refused == 0,
5344                ));
5345                reconciliation.listed_incomplete = reconciliation.take_recordable_completeness();
5346                merge_reconcile_report(&mut report.reconciliation, reconciliation);
5347            }
5348            Err(error) => {
5349                outcomes.push((false, false));
5350                failure = Some(error);
5351            }
5352        }
5353    }
5354
5355    let listed_incomplete = std::mem::take(&mut report.reconciliation.listed_incomplete);
5356    let root = target.root_path()?;
5357    normalize_walk_errors(&root, &mut report.reconciliation.scan.errors);
5358    let failed_paths = failure_paths(target, &report.reconciliation.scan.errors)?;
5359    for ((subtree, started_at), (complete, disproves_old)) in opened.into_iter().zip(outcomes) {
5360        let commit = target.finish_reconcile(
5361            &subtree,
5362            started_at,
5363            complete,
5364            &listed_incomplete,
5365            &failed_paths,
5366            ReconcileErrors {
5367                errors: &report.reconciliation.scan.errors,
5368                terminal: failure.as_ref(),
5369                disproves_old,
5370            },
5371        )?;
5372        if let Some(commit) = commit.commit.as_ref() {
5373            sink(commit);
5374        }
5375        report.reconciliation.retry_required |= commit.retry;
5376    }
5377
5378    match failure {
5379        Some(error) => Err(error),
5380        None => Ok(report),
5381    }
5382}
5383
5384/// Root-relative paths whose filesystem facts a failed reconciliation could not verify.
5385///
5386/// An unscoped error returns an empty set, which makes the closer conservatively mark the
5387/// whole requested subtree partial. Precise I/O paths let verified siblings remain fresh.
5388fn failure_paths(target: &ReconcileTarget<'_>, errors: &[Error]) -> Result<Vec<PathBuf>> {
5389    if errors.is_empty() {
5390        return Ok(Vec::new());
5391    }
5392    let root = target.root_path()?;
5393    let mut paths = Vec::with_capacity(errors.len());
5394    for error in errors {
5395        let Some(path) = crate::Issue::from_error_under(&root, error).path else {
5396            return Ok(Vec::new());
5397        };
5398        if path.is_absolute() {
5399            return Ok(Vec::new());
5400        }
5401        paths.push(path);
5402    }
5403    paths.sort();
5404    paths.dedup();
5405    Ok(paths)
5406}
5407
5408/// Whether verification could increase the retained-file set.
5409///
5410/// This deliberately recognizes only cases that prove non-expansion. At a resource
5411/// boundary, uncertainty is a refusal rather than permission to exceed the bound.
5412fn refresh_may_expand(
5413    target: &ReconcileTarget<'_>,
5414    path: &Path,
5415    work: &mut ScanReport,
5416) -> Result<bool> {
5417    let current = target.expectation(path)?.state;
5418    let absolute = target.root_path()?.join(path);
5419    let observed = match fs::symlink_metadata(&absolute) {
5420        Ok(metadata) => {
5421            let Ok((kind, attrs)) = observe(&absolute, &metadata) else {
5422                return Ok(true);
5423            };
5424            work.observe(kind, attrs);
5425            Some(kind)
5426        }
5427        Err(error)
5428            if matches!(
5429                error.kind(),
5430                std::io::ErrorKind::NotFound | std::io::ErrorKind::NotADirectory
5431            ) =>
5432        {
5433            None
5434        }
5435        Err(_) => return Ok(true),
5436    };
5437    Ok(!matches!(
5438        (current, observed),
5439        (PathState::Present { kind: EntryKind::File, .. }, Some(EntryKind::File)) | (_, None)
5440    ))
5441}
5442
5443/// Drop every path covered by a shallower member of the same sorted set.
5444fn covering_roots(mut paths: Vec<PathBuf>) -> Vec<PathBuf> {
5445    paths.sort();
5446    paths.dedup();
5447    if paths.first().is_some_and(|first| first.as_os_str().is_empty()) {
5448        return vec![PathBuf::new()];
5449    }
5450    let mut roots: Vec<PathBuf> = Vec::with_capacity(paths.len());
5451    for path in paths {
5452        if roots.last().is_some_and(|kept| path.starts_with(kept)) {
5453            continue;
5454        }
5455        roots.push(path);
5456    }
5457    roots
5458}
5459
5460fn reconcile_target(
5461    target: &mut ReconcileTarget<'_>,
5462    subtree: &Path,
5463    config: &ScanConfig,
5464    sink: &mut dyn FnMut(&Commit),
5465) -> Result<ReconcileReport> {
5466    config.validate_for_scope(target.scope()?)?;
5467    if let Some(progress) = &config.progress {
5468        progress.enter(crate::ProgressPhase::Revalidating);
5469    }
5470    let subtree = normalize_subtree(subtree)?;
5471    if config.max_depth.is_some_and(|maximum| subtree.components().count() > maximum) {
5472        return Err(Error::SubtreeOutsideScanScope { path: subtree, scope: config.scope() });
5473    }
5474    let subtree = if config.population != crate::query::IgnoredEntries::Include
5475        && !subtree.as_os_str().is_empty()
5476    {
5477        // A narrowed tier may have pruned the requested entry or an ancestor. Start
5478        // from its nearest retained parent so the governing control is read before the
5479        // directory listing decides whether the boundary itself belongs in the tier.
5480        // A control-file edit also needs this parent listing to discover siblings that
5481        // were absent under the previous rule.
5482        let mut parent = subtree.parent().map_or_else(PathBuf::new, Path::to_path_buf);
5483        while !parent.as_os_str().is_empty()
5484            && (target.expectation(&parent)?.state == PathState::Absent
5485                || !target.control_classification_known(&parent)?)
5486        {
5487            parent = parent.parent().map_or_else(PathBuf::new, Path::to_path_buf);
5488        }
5489        parent
5490    } else {
5491        subtree
5492    };
5493    let subtree = resolve_subtree_root(target, &subtree, config)?;
5494    let (started_at, started) = target.begin_reconcile(&subtree)?;
5495    if let Some(commit) = started.as_ref() {
5496        sink(commit);
5497    }
5498    match reconcile_target_inner(
5499        target,
5500        &subtree,
5501        started_at,
5502        config,
5503        MAX_DEFERRED_RECONCILE_OPS,
5504        sink,
5505    ) {
5506        Ok(mut report) => {
5507            let root = target.root_path()?;
5508            normalize_walk_errors(&root, &mut report.scan.errors);
5509            let listed_incomplete = report.take_recordable_completeness();
5510            let failed_paths = failure_paths(target, &report.scan.errors)?;
5511            let finished = target.finish_reconcile(
5512                &subtree,
5513                started_at,
5514                report.is_complete(),
5515                &listed_incomplete,
5516                &failed_paths,
5517                ReconcileErrors {
5518                    errors: &report.scan.errors,
5519                    terminal: None,
5520                    disproves_old: report.apply.stale == 0 && report.apply.resource_refused == 0,
5521                },
5522            )?;
5523            if let Some(commit) = finished.commit.as_ref() {
5524                sink(commit);
5525            }
5526            report.retry_required |= finished.retry;
5527            Ok(report)
5528        }
5529        Err(error) => {
5530            let finished = target.finish_reconcile(
5531                &subtree,
5532                started_at,
5533                false,
5534                &[],
5535                &[],
5536                ReconcileErrors { errors: &[], terminal: Some(&error), disproves_old: false },
5537            )?;
5538            if let Some(commit) = finished.commit.as_ref() {
5539                sink(commit);
5540            }
5541            Err(error)
5542        }
5543    }
5544}
5545
5546fn reconcile_target_inner(
5547    target: &mut ReconcileTarget<'_>,
5548    subtree: &Path,
5549    started_at: u64,
5550    config: &ScanConfig,
5551    max_deferred_ops: usize,
5552    sink: &mut dyn FnMut(&Commit),
5553) -> Result<ReconcileReport> {
5554    let root = target.root_path()?;
5555    let root_meta = {
5556        crate::counters::bump(|c| c.stats += 1);
5557        fs::symlink_metadata(&root)
5558    }
5559    .map_err(|error| Error::io(&root, error))?;
5560    if !root_meta.is_dir() {
5561        return Err(Error::io(
5562            &root,
5563            std::io::Error::new(
5564                std::io::ErrorKind::NotADirectory,
5565                "reconciliation root is not a directory",
5566            ),
5567        ));
5568    }
5569    let root_dev = root_device(&root, &root_meta).map_err(|error| Error::io(&root, error))?;
5570    let start_depth = subtree.components().count();
5571    let mut report =
5572        ReconcileReport { reconcile_epoch: Some(started_at), ..ReconcileReport::default() };
5573    let mut tally = ProgressTally::new(config.progress.as_ref());
5574    let mut retry_frontier = None;
5575    let mut batch: Vec<ObservationOp> = Vec::with_capacity(config.batch_size.max(1));
5576
5577    if config.max_depth == Some(0) {
5578        remove_known_children(target, Path::new(""), config, &mut batch, sink, &mut report)?;
5579        return Ok(report);
5580    }
5581
5582    if !subtree.as_os_str().is_empty() {
5583        let baseline = target.expectation(subtree)?;
5584        let absolute = root.join(subtree);
5585        let meta = match fs::symlink_metadata(&absolute) {
5586            Ok(meta) => meta,
5587            Err(error) if error.kind() == std::io::ErrorKind::NotFound => {
5588                batch.push(ObservationOp::if_state(
5589                    Op::Remove { path: subtree.to_path_buf() },
5590                    baseline,
5591                ));
5592                push_lost_spelling_control(
5593                    target,
5594                    &root,
5595                    config,
5596                    subtree,
5597                    baseline,
5598                    &mut batch,
5599                    &mut report,
5600                )?;
5601                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5602                return Ok(report);
5603            }
5604            Err(error) => {
5605                report.scan.errors.push(Error::io(&absolute, error));
5606                if baseline.state != PathState::Absent {
5607                    batch.push(ObservationOp::if_state(
5608                        Op::Remove { path: subtree.to_path_buf() },
5609                        baseline,
5610                    ));
5611                }
5612                push_lost_spelling_control(
5613                    target,
5614                    &root,
5615                    config,
5616                    subtree,
5617                    baseline,
5618                    &mut batch,
5619                    &mut report,
5620                )?;
5621                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5622                return Ok(report);
5623            }
5624        };
5625        let (kind, attrs) = match observe(&absolute, &meta) {
5626            Ok(observed) => observed,
5627            Err(error) => {
5628                report.scan.errors.push(Error::io(absolute, error));
5629                return Ok(report);
5630            }
5631        };
5632        let disposition =
5633            crate::admission::decide_path(subtree, kind, config.hidden(), config.exclude_special);
5634        if disposition != crate::admission::Disposition::Retain {
5635            if baseline.state != PathState::Absent {
5636                batch.push(ObservationOp::if_state(
5637                    Op::Remove { path: subtree.to_path_buf() },
5638                    baseline,
5639                ));
5640            }
5641            if disposition == crate::admission::Disposition::ControlOnly {
5642                let read = read_control_op(config, &root, subtree, kind);
5643                push_read_control(target, subtree, baseline, read, &mut batch, &mut report)?;
5644            }
5645            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5646            return Ok(report);
5647        }
5648        report.scan.observe(kind, attrs);
5649        push_reconcile_upsert(target, subtree, kind, attrs, baseline, &mut batch, &mut report);
5650        // A retained control file at the root of the walk reads its rules here, as the
5651        // listing walk does for every retained entry it lists: a file does not descend,
5652        // so nothing below would read them, and the table kept the old source while the
5653        // pass reported complete and marked the path fresh. In the same batch as the
5654        // upsert, so both are arbitrated against one baseline.
5655        let read = read_control_op(config, &root, subtree, kind);
5656        push_read_control(target, subtree, baseline, read, &mut batch, &mut report)?;
5657        flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5658        if !should_descend(kind, attrs, start_depth.saturating_sub(1), root_dev, config) {
5659            if kind.is_dir() {
5660                remove_known_children(target, subtree, config, &mut batch, sink, &mut report)?;
5661            }
5662            tally.flush(&report.scan);
5663            return Ok(report);
5664        }
5665    }
5666
5667    if subtree.as_os_str().is_empty()
5668        && config.population == crate::query::IgnoredEntries::Include
5669        && config.reconciliation_worker_threads() > 1
5670    {
5671        if let ReconcileTarget::Direct(index) = target {
5672            match reconcile_direct_parallel(index, &root, root_dev, config, max_deferred_ops, sink)?
5673            {
5674                DirectParallelOutcome::Complete(parallel) => return Ok(parallel),
5675                DirectParallelOutcome::RetrySerial { prefix, remaining } => {
5676                    report = prefix;
5677                    report.reconcile_epoch = Some(started_at);
5678                    retry_frontier = Some(remaining);
5679                    // The wave workers reported the prefix themselves, and the wave
5680                    // that overflowed as well: progress counts that wave's reads twice,
5681                    // once there and once as the serial retry rereads it, while this
5682                    // report counts each directory once. Work done, not the answer.
5683                    tally.skip_to(&report.scan);
5684                }
5685            }
5686        }
5687    }
5688
5689    let mut queue: VecDeque<(PathBuf, usize)> = retry_frontier
5690        .unwrap_or_else(|| VecDeque::from(vec![(subtree.to_path_buf(), start_depth)]));
5691    let mut controls = if config.population == crate::query::IgnoredEntries::Include {
5692        None
5693    } else {
5694        Some(target.control_table()?)
5695    };
5696    let mut unreadable_controls = std::collections::BTreeSet::new();
5697    #[cfg(target_os = "macos")]
5698    let mut bulk_reader = (config.worker_threads() > 1).then(macos_bulk::Reader::new);
5699    while let Some((rel_dir, depth)) = take_next(&mut queue, config.order) {
5700        let errors_before = report.scan.errors.len();
5701        let abs_dir = root.join(&rel_dir);
5702        let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
5703        let mut had_control = target.has_control(&control_path)?;
5704        let mut listed_control = ListedControl::default();
5705        if let Some(table) = controls.as_mut() {
5706            let lookup = read_directory_control(config, &root, &control_path);
5707            listed_control.lookup(&lookup);
5708            match lookup {
5709                Ok(Some(op)) => {
5710                    let baseline = target.expectation(&control_path)?;
5711                    apply_discovery_control(table, &op)?;
5712                    batch.push(ObservationOp::if_state(op, baseline));
5713                }
5714                Ok(None) => {
5715                    table.remove(&control_path)?;
5716                    if had_control {
5717                        let baseline = target.expectation(&control_path)?;
5718                        batch.push(ObservationOp::if_state(
5719                            Op::ControlRemove { path: control_path.clone() },
5720                            baseline,
5721                        ));
5722                        had_control = false;
5723                    }
5724                }
5725                Err(error) => {
5726                    unreadable_controls.insert(rel_dir.clone());
5727                    report.scan.errors.push(error);
5728                }
5729            }
5730            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5731            if report.apply.stale > 0 || report.apply.resource_refused > 0 {
5732                // This directory's population decisions use the source just read.
5733                // If its conditional control commit lost ownership, walking siblings
5734                // against the local copy could prune under rules the index rejected.
5735                report.retry_required = true;
5736                return Ok(report);
5737            }
5738        }
5739        let (mut known, records_completeness) = target.listing_baseline(&rel_dir)?;
5740        let mut listing_complete = true;
5741        let process_entry = |name: OsString,
5742                             kind: EntryKind,
5743                             attrs: Attrs,
5744                             baseline: PathExpectation,
5745                             listed_control: &mut ListedControl,
5746                             target: &mut ReconcileTarget<'_>,
5747                             queue: &mut VecDeque<(PathBuf, usize)>,
5748                             batch: &mut Vec<ObservationOp>,
5749                             sink: &mut dyn FnMut(&Commit),
5750                             report: &mut ReconcileReport|
5751         -> Result<()> {
5752            let rel_path = rel_dir.join(&name);
5753            listed_control.listed(&name);
5754            let disposition =
5755                crate::admission::decide(&name, kind, config.hidden(), config.exclude_special);
5756            if population_prunes(
5757                config.population,
5758                &rel_path,
5759                kind,
5760                disposition,
5761                controls.as_ref(),
5762                &unreadable_controls,
5763            ) {
5764                if baseline.state != PathState::Absent {
5765                    batch.push(ObservationOp::if_state(Op::Remove { path: rel_path }, baseline));
5766                }
5767                return Ok(());
5768            }
5769            if disposition != crate::admission::Disposition::Retain {
5770                if baseline.state != PathState::Absent {
5771                    batch.push(ObservationOp::if_state(
5772                        Op::Remove { path: rel_path.clone() },
5773                        baseline,
5774                    ));
5775                }
5776                if disposition == crate::admission::Disposition::ControlOnly {
5777                    let read = read_listed_control_op(config, &root, &rel_path, kind);
5778                    listed_control.read(&name, &read);
5779                    push_read_control(target, &rel_path, baseline, read, batch, report)?;
5780                }
5781                if batch.len() >= config.batch_size.max(1) {
5782                    flush_reconcile_batch(target, batch, sink, report)?;
5783                }
5784                return Ok(());
5785            }
5786            report.scan.observe(kind, attrs);
5787            push_reconcile_upsert(target, &rel_path, kind, attrs, baseline, batch, report);
5788            if batch.len() >= config.batch_size.max(1) {
5789                flush_reconcile_batch(target, batch, sink, report)?;
5790            }
5791            let read = read_listed_control_op(config, &root, &rel_path, kind);
5792            listed_control.read(&name, &read);
5793            push_read_control(target, &rel_path, baseline, read, batch, report)?;
5794            if batch.len() >= config.batch_size.max(1) {
5795                flush_reconcile_batch(target, batch, sink, report)?;
5796            }
5797
5798            if should_descend(kind, attrs, depth, root_dev, config) {
5799                queue.push_back((rel_path, depth + 1));
5800            } else if kind.is_dir() {
5801                remove_known_children(target, &rel_path, config, batch, sink, report)?;
5802            }
5803            Ok(())
5804        };
5805
5806        #[cfg(target_os = "macos")]
5807        let used_bulk = if walk_hook_covers(&abs_dir) {
5808            false
5809        } else if let Some(entries) = bulk_reader.as_mut().and_then(|reader| reader.read(&abs_dir))
5810        {
5811            report.scan.dirs_read += 1;
5812            for entry in entries {
5813                let baseline = match known.remove(&entry.name) {
5814                    Some(baseline) => baseline,
5815                    None => target.expectation(&rel_dir.join(&entry.name))?,
5816                };
5817                process_entry(
5818                    entry.name,
5819                    entry.kind,
5820                    entry.attrs,
5821                    baseline,
5822                    &mut listed_control,
5823                    target,
5824                    &mut queue,
5825                    &mut batch,
5826                    sink,
5827                    &mut report,
5828                )?;
5829            }
5830            true
5831        } else {
5832            false
5833        };
5834        #[cfg(not(target_os = "macos"))]
5835        let used_bulk = false;
5836
5837        if !used_bulk {
5838            crate::counters::bump(|c| c.dir_opens += 1);
5839            let listing = match fs::read_dir(&abs_dir) {
5840                Ok(listing) => listing,
5841                Err(error) => {
5842                    report.scan.errors.push(Error::io(&abs_dir, error));
5843                    remove_known_children(target, &rel_dir, config, &mut batch, sink, &mut report)?;
5844                    if had_control {
5845                        let baseline = target.expectation(&control_path)?;
5846                        batch.push(ObservationOp::if_state(
5847                            Op::ControlRemove { path: control_path },
5848                            baseline,
5849                        ));
5850                        flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5851                    }
5852                    continue;
5853                }
5854            };
5855            report.scan.dirs_read += 1;
5856            let listing = reconcile_listing(listing, &abs_dir);
5857            for item in listing {
5858                let item = match item {
5859                    Ok(item) => item,
5860                    Err(error) => {
5861                        listing_complete = false;
5862                        report.scan.errors.push(Error::io(&abs_dir, error));
5863                        continue;
5864                    }
5865                };
5866                let name = item.file_name();
5867                // Seeing the name proves it is not absent even if the following
5868                // metadata lookup fails. Remove it from the missing set before that
5869                // fallible lookup so an operational error cannot turn an existing
5870                // entry into a deletion.
5871                let baseline = match known.remove(&name) {
5872                    Some(baseline) => baseline,
5873                    None => target.expectation(&rel_dir.join(&name))?,
5874                };
5875                let (kind, attrs) = match observe_dir_entry(&item) {
5876                    Ok(Some(observed)) => observed,
5877                    Ok(None) => {
5878                        let entry_held = baseline.state != PathState::Absent;
5879                        for removal in
5880                            vanished_child_removals(&rel_dir, &name, entry_held, &mut had_control)
5881                                .into_iter()
5882                                .flatten()
5883                        {
5884                            batch.push(ObservationOp::if_state(removal, baseline));
5885                        }
5886                        if batch.len() >= config.batch_size.max(1) {
5887                            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5888                        }
5889                        continue;
5890                    }
5891                    Err(error) => {
5892                        listed_control.listed(&name);
5893                        report.scan.errors.push(Error::io(item.path(), error));
5894                        if baseline.state != PathState::Absent {
5895                            batch.push(ObservationOp::if_state(
5896                                Op::Remove { path: rel_dir.join(&name) },
5897                                baseline,
5898                            ));
5899                        }
5900                        if name == crate::control::CONTROL_FILE_NAME && had_control {
5901                            batch.push(ObservationOp::if_state(
5902                                Op::ControlRemove { path: rel_dir.join(&name) },
5903                                baseline,
5904                            ));
5905                            had_control = false;
5906                        }
5907                        if batch.len() >= config.batch_size.max(1) {
5908                            flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5909                        }
5910                        continue;
5911                    }
5912                };
5913                process_entry(
5914                    name,
5915                    kind,
5916                    attrs,
5917                    baseline,
5918                    &mut listed_control,
5919                    target,
5920                    &mut queue,
5921                    &mut batch,
5922                    sink,
5923                    &mut report,
5924                )?;
5925            }
5926        }
5927
5928        for (name, baseline) in known {
5929            let restatement = listed_control.restatement_after_removing(&name);
5930            batch.push(ObservationOp::if_state(Op::Remove { path: rel_dir.join(name) }, baseline));
5931            // Before the batch can be flushed, so the rules the removal drops are back in the
5932            // same commit.
5933            if let Some(control) = restatement {
5934                let guard = target.expectation(&control_path)?;
5935                batch.push(ObservationOp::if_state(control, guard));
5936            }
5937            if batch.len() >= config.batch_size.max(1) {
5938                flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5939            }
5940        }
5941        if had_control && !listed_control.seen {
5942            let baseline = target.expectation(&control_path)?;
5943            batch.push(ObservationOp::if_state(Op::ControlRemove { path: control_path }, baseline));
5944        }
5945        if listing_complete {
5946            // Only a directory with no error inside its own processing vouches for its
5947            // child set; an error under a sibling or a descendant is that directory's
5948            // to answer for, as discovery decides completeness per directory.
5949            if records_completeness && report.scan.errors.len() == errors_before {
5950                report.listed_incomplete.push(rel_dir);
5951            }
5952        }
5953        // Per directory, as in `revalidate`: an unchanged tree hands the sink nothing.
5954        tally.flush(&report.scan);
5955    }
5956
5957    flush_reconcile_batch(target, &mut batch, sink, &mut report)?;
5958    report.scan.errors.sort_by_cached_key(ToString::to_string);
5959    tally.flush(&report.scan);
5960    Ok(report)
5961}
5962
5963#[derive(Debug, Default)]
5964struct DeferredReconcile {
5965    scan: ScanReport,
5966    unchanged: u64,
5967    operations: Vec<Op>,
5968    /// Each removal of a stale `.gitignore` entry whose rules a case variant now holds,
5969    /// followed by the restatement of those rules (`ListedControl`). Applied after the
5970    /// wave's sorted operations, in one batch, because the sort moves every removal after
5971    /// every other operation and a batch boundary could fall between the two.
5972    restated: Vec<Op>,
5973    discovered: Vec<(PathBuf, usize, RegionId)>,
5974    listed_incomplete: Vec<PathBuf>,
5975}
5976
5977enum DirectParallelOutcome {
5978    Complete(ReconcileReport),
5979    RetrySerial { prefix: ReconcileReport, remaining: VecDeque<(PathBuf, usize)> },
5980}
5981
5982/// Reconcile an exclusive full tree in bounded immutable-baseline waves.
5983///
5984/// No index write occurs while a wave's workers hold shared baseline references. That
5985/// lets each worker discard exact no-ops where they are observed instead of funnelling
5986/// every entry through one consumer. Effective changes still enter through ordinary
5987/// observations between waves, preserving both the index's sole mutation contract and
5988/// progressive delta delivery.
5989///
5990/// Unlike every other reconciliation path, the operations a wave defers are
5991/// unconditional: they carry no [`ObservationOp::if_state`] guard. Three properties
5992/// have to hold together for that to be safe, and a change to any one of them puts the
5993/// guards back. The target is an exclusive `&mut Index`, so no other producer can
5994/// commit between a worker's read and the wave's write. Nothing is applied while
5995/// workers run, so no baseline a worker compared against can go stale beneath it. And
5996/// a directory is only reconciled in a wave after the wave that discovered it has
5997/// committed, so a parent is never absent when its children arrive.
5998fn reconcile_direct_parallel(
5999    index: &mut Index,
6000    root: &Path,
6001    root_dev: u64,
6002    config: &ScanConfig,
6003    max_deferred_ops: usize,
6004    sink: &mut dyn FnMut(&Commit),
6005) -> Result<DirectParallelOutcome> {
6006    let mut frontier = DirectoryQueueState::seeded(
6007        (PathBuf::new(), 0),
6008        config.order,
6009        None,
6010        1,
6011        1,
6012        WorkerPolicyExperiment::ShippedOneShot,
6013    );
6014    let mut report = ReconcileReport::default();
6015    while !frontier.is_empty(config.order) {
6016        let mut wave = Vec::with_capacity(RECONCILE_WAVE_DIRECTORIES);
6017        while wave.len() < RECONCILE_WAVE_DIRECTORIES && !frontier.is_empty(config.order) {
6018            let remaining = RECONCILE_WAVE_DIRECTORIES - wave.len();
6019            frontier.take(remaining.min(DIR_CLAIM), config.order, &mut wave);
6020        }
6021
6022        let next = std::sync::atomic::AtomicUsize::new(0);
6023        let deferred_count = std::sync::atomic::AtomicUsize::new(0);
6024        let overflowed = std::sync::atomic::AtomicBool::new(false);
6025        let workers = config.reconciliation_worker_threads().min(wave.len());
6026        let baseline: &Index = index;
6027        let results: Vec<DeferredReconcile> = std::thread::scope(|scope| {
6028            let handles: Vec<_> = (0..workers)
6029                .map(|_| {
6030                    scope.spawn(|| {
6031                        reconcile_wave_worker(
6032                            baseline,
6033                            root,
6034                            root_dev,
6035                            config,
6036                            &wave,
6037                            &next,
6038                            &deferred_count,
6039                            &overflowed,
6040                            max_deferred_ops,
6041                        )
6042                    })
6043                })
6044                .collect();
6045
6046            handles
6047                .into_iter()
6048                .map(|handle| {
6049                    if let Ok(worker) = handle.join() {
6050                        return worker;
6051                    }
6052                    let mut worker = DeferredReconcile::default();
6053                    worker.scan.errors.push(Error::io(
6054                        root,
6055                        std::io::Error::other("a reconciliation worker thread panicked"),
6056                    ));
6057                    worker
6058                })
6059                .collect()
6060        });
6061
6062        // Nothing from an overflowing wave was applied, so the ordinary incremental
6063        // reconciler can resume at that wave. Completed waves and their statistics are
6064        // retained exactly once; restarting from the root would count their unchanged
6065        // entries again and misreport the logical reconciliation pass.
6066        if overflowed.load(std::sync::atomic::Ordering::Relaxed) {
6067            let mut remaining: VecDeque<_> =
6068                wave.into_iter().map(|(path, depth, _region)| (path, depth)).collect();
6069            let mut deferred = Vec::with_capacity(DIR_CLAIM);
6070            while !frontier.is_empty(config.order) {
6071                frontier.take(DIR_CLAIM, config.order, &mut deferred);
6072                remaining.extend(deferred.drain(..).map(|(path, depth, _region)| (path, depth)));
6073            }
6074            return Ok(DirectParallelOutcome::RetrySerial { prefix: report, remaining });
6075        }
6076
6077        let operation_count = deferred_count.load(std::sync::atomic::Ordering::Relaxed);
6078        let mut operations = Vec::with_capacity(operation_count);
6079        let mut restated = Vec::new();
6080        for worker in results {
6081            report.listed_incomplete.extend(worker.listed_incomplete);
6082            report.scan.absorb(worker.scan);
6083            report.apply.unchanged += worker.unchanged;
6084            report.observations = report.observations.saturating_add(worker.unchanged);
6085            operations.extend(worker.operations);
6086            restated.extend(worker.restated);
6087            for directory in worker.discovered {
6088                frontier.push(directory, config.order);
6089            }
6090        }
6091        apply_deferred_reconcile(
6092            index,
6093            &mut operations,
6094            restated,
6095            config,
6096            sink,
6097            &mut report.apply,
6098            &mut report.observations,
6099        )?;
6100    }
6101    report.scan.errors.sort_by_cached_key(ToString::to_string);
6102    Ok(DirectParallelOutcome::Complete(report))
6103}
6104
6105fn apply_deferred_reconcile(
6106    index: &mut Index,
6107    operations: &mut Vec<Op>,
6108    restated: Vec<Op>,
6109    config: &ScanConfig,
6110    sink: &mut dyn FnMut(&Commit),
6111    stats: &mut ApplyStats,
6112    observations: &mut u64,
6113) -> Result<()> {
6114    // Parent upserts establish real directory attributes before children arrive.
6115    // Removals run deepest first so a parent removal never precedes an independently
6116    // observed descendant operation. Deterministic causal order also makes emitted
6117    // commits stable for callers.
6118    operations.sort_by(|left, right| {
6119        let left_remove = matches!(left, Op::Remove { .. });
6120        let right_remove = matches!(right, Op::Remove { .. });
6121        left_remove.cmp(&right_remove).then_with(|| {
6122            let left_depth = left.path().components().count();
6123            let right_depth = right.path().components().count();
6124            if left_remove {
6125                right_depth.cmp(&left_depth).then_with(|| left.path().cmp(right.path()))
6126            } else {
6127                left_depth.cmp(&right_depth).then_with(|| left.path().cmp(right.path()))
6128            }
6129        })
6130    });
6131
6132    let batch_limit = config.batch_size.max(1);
6133    let mut batch = Vec::with_capacity(batch_limit.min(operations.len()));
6134    for operation in operations.drain(..) {
6135        batch.push(operation);
6136        if batch.len() >= batch_limit {
6137            flush_direct_reconcile_batch(index, &mut batch, sink, stats, observations)?;
6138        }
6139    }
6140    flush_direct_reconcile_batch(index, &mut batch, sink, stats, observations)?;
6141    // Whole, so no commit shows a directory without the rules a restatement puts back.
6142    let mut restated = restated;
6143    flush_direct_reconcile_batch(index, &mut restated, sink, stats, observations)
6144}
6145
6146#[allow(clippy::too_many_arguments)]
6147fn reconcile_wave_worker(
6148    index: &Index,
6149    root: &Path,
6150    root_dev: u64,
6151    config: &ScanConfig,
6152    wave: &[(PathBuf, usize, RegionId)],
6153    next: &std::sync::atomic::AtomicUsize,
6154    deferred_count: &std::sync::atomic::AtomicUsize,
6155    overflowed: &std::sync::atomic::AtomicBool,
6156    max_deferred_ops: usize,
6157) -> DeferredReconcile {
6158    let _counter_guard = crate::counters::thread_flush_guard();
6159    let mut result = DeferredReconcile::default();
6160    let mut tally = ProgressTally::new(config.progress.as_ref());
6161    #[cfg(target_os = "macos")]
6162    let mut bulk_reader = macos_bulk::Reader::new();
6163
6164    loop {
6165        let start = next.fetch_add(DIR_CLAIM, std::sync::atomic::Ordering::Relaxed);
6166        if start >= wave.len() {
6167            break;
6168        }
6169        let end = start.saturating_add(DIR_CLAIM).min(wave.len());
6170        for (rel_dir, depth, region) in &wave[start..end] {
6171            let errors_before = result.scan.errors.len();
6172            let mut known = collect_child_expectations(index, rel_dir);
6173            let abs_dir = root.join(rel_dir);
6174            let control_path = rel_dir.join(crate::control::CONTROL_FILE_NAME);
6175            let mut had_control = index.control_table().contains(&control_path);
6176            let mut listed_control = ListedControl::default();
6177            let mut control_errors = Vec::new();
6178            let mut vanished = Vec::new();
6179            let mut unverified = Vec::new();
6180            let mut control_read_failed = false;
6181            let mut listing_open_failed = false;
6182
6183            {
6184                let mut process_entry =
6185                    |name: OsString,
6186                     kind: EntryKind,
6187                     attrs: Attrs,
6188                     baseline: PathExpectation,
6189                     listed_control: &mut ListedControl| {
6190                        let rel_path = rel_dir.join(&name);
6191                        listed_control.listed(&name);
6192                        let disposition = crate::admission::decide(
6193                            &name,
6194                            kind,
6195                            config.hidden(),
6196                            config.exclude_special,
6197                        );
6198                        if disposition == crate::admission::Disposition::Reject {
6199                            if baseline.state != PathState::Absent {
6200                                defer_reconcile_op(
6201                                    Op::Remove { path: rel_path },
6202                                    &mut result.operations,
6203                                    deferred_count,
6204                                    overflowed,
6205                                    max_deferred_ops,
6206                                );
6207                            }
6208                            return;
6209                        }
6210                        if disposition == crate::admission::Disposition::ControlOnly {
6211                            let read = read_control_op(config, root, &rel_path, kind);
6212                            listed_control.read(&name, &read);
6213                            match read {
6214                                Ok(Some(Op::ControlUpsert { path, source })) => {
6215                                    if !index.control_table().source_is(&path, &source) {
6216                                        defer_reconcile_op(
6217                                            Op::ControlUpsert { path, source },
6218                                            &mut result.operations,
6219                                            deferred_count,
6220                                            overflowed,
6221                                            max_deferred_ops,
6222                                        );
6223                                    }
6224                                }
6225                                Ok(Some(Op::ControlRemove { path })) => {
6226                                    if index.control_table().contains(&path) {
6227                                        defer_reconcile_op(
6228                                            Op::ControlRemove { path },
6229                                            &mut result.operations,
6230                                            deferred_count,
6231                                            overflowed,
6232                                            max_deferred_ops,
6233                                        );
6234                                    }
6235                                }
6236                                Ok(Some(_) | None) => {}
6237                                Err(error) => {
6238                                    control_errors.push(error);
6239                                    control_read_failed = true;
6240                                }
6241                            }
6242                            return;
6243                        }
6244                        result.scan.entries += 1;
6245                        if kind == EntryKind::File {
6246                            result.scan.files_walked += 1;
6247                            result.scan.bytes_walked += attrs.size;
6248                            result.scan.allocated_walked += attrs.allocated;
6249                        }
6250                        if baseline.state == (PathState::Present { kind, attrs }) {
6251                            result.unchanged += 1;
6252                        } else {
6253                            defer_reconcile_op(
6254                                Op::Upsert { path: rel_path.clone(), kind, attrs },
6255                                &mut result.operations,
6256                                deferred_count,
6257                                overflowed,
6258                                max_deferred_ops,
6259                            );
6260                        }
6261                        let read = read_control_op(config, root, &rel_path, kind);
6262                        listed_control.read(&name, &read);
6263                        match read {
6264                            Ok(Some(Op::ControlUpsert { path, source })) => {
6265                                if !index.control_table().source_is(&path, &source) {
6266                                    defer_reconcile_op(
6267                                        Op::ControlUpsert { path, source },
6268                                        &mut result.operations,
6269                                        deferred_count,
6270                                        overflowed,
6271                                        max_deferred_ops,
6272                                    );
6273                                }
6274                            }
6275                            // The exact name's upsert as a non-file drops its rules by
6276                            // itself; a case variant's does not, so its lookup's removal
6277                            // is sent.
6278                            Ok(Some(Op::ControlRemove { path }))
6279                                if crate::control::control_spelling(&name)
6280                                    == Some(crate::control::ControlSpelling::Variant)
6281                                    && index.control_table().contains(&path) =>
6282                            {
6283                                defer_reconcile_op(
6284                                    Op::ControlRemove { path },
6285                                    &mut result.operations,
6286                                    deferred_count,
6287                                    overflowed,
6288                                    max_deferred_ops,
6289                                );
6290                            }
6291                            Ok(Some(_) | None) => {}
6292                            Err(error) => {
6293                                control_errors.push(error);
6294                                // The failed read was of the directory's control, through
6295                                // the entry for the exact name and a lookup for a variant.
6296                                control_read_failed = true;
6297                            }
6298                        }
6299
6300                        if should_descend(kind, attrs, *depth, root_dev, config) {
6301                            let child_region =
6302                                if *depth == 0 { RegionId::UNASSIGNED } else { *region };
6303                            result.discovered.push((rel_path, depth + 1, child_region));
6304                        } else if kind.is_dir() {
6305                            for name in collect_child_expectations(index, &rel_path).into_keys() {
6306                                defer_reconcile_op(
6307                                    Op::Remove { path: rel_path.join(name) },
6308                                    &mut result.operations,
6309                                    deferred_count,
6310                                    overflowed,
6311                                    max_deferred_ops,
6312                                );
6313                            }
6314                        }
6315                    };
6316
6317                #[cfg(target_os = "macos")]
6318                let used_bulk = if let Some(entries) =
6319                    (!walk_hook_covers(&abs_dir)).then(|| bulk_reader.read(&abs_dir)).flatten()
6320                {
6321                    result.scan.dirs_read += 1;
6322                    for entry in entries {
6323                        let baseline = known
6324                            .remove(&entry.name)
6325                            .unwrap_or_else(|| index.expectation(&rel_dir.join(&entry.name)));
6326                        process_entry(
6327                            entry.name,
6328                            entry.kind,
6329                            entry.attrs,
6330                            baseline,
6331                            &mut listed_control,
6332                        );
6333                    }
6334                    true
6335                } else {
6336                    false
6337                };
6338                #[cfg(not(target_os = "macos"))]
6339                let used_bulk = false;
6340
6341                if !used_bulk {
6342                    crate::counters::bump(|c| c.dir_opens += 1);
6343                    let listing = match fs::read_dir(&abs_dir) {
6344                        Ok(listing) => Some(listing),
6345                        Err(error) => {
6346                            result.scan.errors.push(Error::io(&abs_dir, error));
6347                            listing_open_failed = true;
6348                            None
6349                        }
6350                    };
6351                    if let Some(listing) = listing {
6352                        result.scan.dirs_read += 1;
6353                        let listing = reconcile_listing(listing, &abs_dir);
6354                        for item in listing {
6355                            let item = match item {
6356                                Ok(item) => item,
6357                                Err(error) => {
6358                                    result.scan.errors.push(Error::io(&abs_dir, error));
6359                                    continue;
6360                                }
6361                            };
6362                            let name = item.file_name();
6363                            // Match the serial path: an entry whose name was enumerated is
6364                            // not missing merely because its metadata could not be read.
6365                            let baseline = known
6366                                .remove(&name)
6367                                .unwrap_or_else(|| index.expectation(&rel_dir.join(&name)));
6368                            let (kind, attrs) = match observe_dir_entry(&item) {
6369                                Ok(Some(observed)) => observed,
6370                                Ok(None) => {
6371                                    // Removed once this directory's listing is done.
6372                                    vanished.push((name, baseline.state != PathState::Absent));
6373                                    continue;
6374                                }
6375                                Err(error) => {
6376                                    listed_control.listed(&name);
6377                                    result.scan.errors.push(Error::io(item.path(), error));
6378                                    unverified.push((name, baseline.state != PathState::Absent));
6379                                    continue;
6380                                }
6381                            };
6382                            process_entry(name, kind, attrs, baseline, &mut listed_control);
6383                        }
6384                    }
6385                }
6386            }
6387            control_read_failed |= listing_open_failed && had_control;
6388            result.scan.errors.append(&mut control_errors);
6389            if index.directory_complete(rel_dir) != Some(true)
6390                && result.scan.errors.len() == errors_before
6391            {
6392                result.listed_incomplete.push(rel_dir.clone());
6393            }
6394            for (name, entry_held) in vanished {
6395                for removal in vanished_child_removals(rel_dir, &name, entry_held, &mut had_control)
6396                    .into_iter()
6397                    .flatten()
6398                {
6399                    defer_reconcile_op(
6400                        removal,
6401                        &mut result.operations,
6402                        deferred_count,
6403                        overflowed,
6404                        max_deferred_ops,
6405                    );
6406                }
6407            }
6408            for (name, entry_held) in unverified {
6409                if entry_held {
6410                    defer_reconcile_op(
6411                        Op::Remove { path: rel_dir.join(&name) },
6412                        &mut result.operations,
6413                        deferred_count,
6414                        overflowed,
6415                        max_deferred_ops,
6416                    );
6417                }
6418                if name == crate::control::CONTROL_FILE_NAME {
6419                    control_read_failed = true;
6420                }
6421            }
6422            for (name, _) in known {
6423                let restatement = listed_control.restatement_after_removing(&name);
6424                let removal = Op::Remove { path: rel_dir.join(name) };
6425                match restatement {
6426                    Some(control) => {
6427                        for operation in [removal, control] {
6428                            defer_reconcile_op(
6429                                operation,
6430                                &mut result.restated,
6431                                deferred_count,
6432                                overflowed,
6433                                max_deferred_ops,
6434                            );
6435                        }
6436                    }
6437                    None => defer_reconcile_op(
6438                        removal,
6439                        &mut result.operations,
6440                        deferred_count,
6441                        overflowed,
6442                        max_deferred_ops,
6443                    ),
6444                }
6445            }
6446            if had_control && (!listed_control.seen || control_read_failed) {
6447                defer_reconcile_op(
6448                    Op::ControlRemove { path: control_path },
6449                    &mut result.operations,
6450                    deferred_count,
6451                    overflowed,
6452                    max_deferred_ops,
6453                );
6454            }
6455        }
6456        // Once per claimed chunk, as the cold walker reports, so a long wave on a slow
6457        // filesystem moves the counters while it runs rather than when it lands.
6458        tally.flush(&result.scan);
6459    }
6460    result
6461}
6462
6463/// What one reconciliation listing has shown about its directory's control file.
6464///
6465/// A listing names the control by the spelling the directory stores. The exact name
6466/// `.gitignore` is the control wherever it is listed, so listing it proves a control
6467/// present even when its read then fails. A case variant is the control only where a
6468/// lookup of `.gitignore` resolves to it (`read_control_op`), so what listing one proves is
6469/// what that lookup returned: a control when it found one, nothing when it missed, and
6470/// nothing when the variant's own metadata could not be read and no lookup was made.
6471/// That last case leaves the rules unknown, which the walk error records
6472/// (`crate::control::unreadable_control`); the sweep removes rules nothing showed, so the
6473/// table ends as a cold walk's does, having read nothing there either.
6474#[derive(Default)]
6475struct ListedControl {
6476    /// The listing showed the directory's control, or failed to read it, so the closing
6477    /// sweep must not remove it.
6478    seen: bool,
6479    /// The listing showed a case variant of the control name, such as `.GITIGNORE`.
6480    variant_listed: bool,
6481    /// The last control a lookup of the canonical path found during this listing: a
6482    /// listed case variant's, or a narrowed walk's before the listing.
6483    ///
6484    /// A case-only rename on a case-insensitive volume (`.gitignore` to `.GITIGNORE`) leaves
6485    /// the old spelling's entry in the index, and the closing sweep removes it. Removing
6486    /// the entry at the canonical path drops the rules it governs
6487    /// (`Index::projected_controls`), yet those rules are now the variant's, so when this
6488    /// listing showed the variant the sweep restates this observation right after that
6489    /// removal ([`Self::restatement_after_removing`]).
6490    looked_up: Option<Op>,
6491}
6492
6493impl ListedControl {
6494    /// Note that `name` was listed, before its metadata or its control is read.
6495    fn listed(&mut self, name: &OsStr) {
6496        match crate::control::control_spelling(name) {
6497            Some(crate::control::ControlSpelling::Exact) => self.seen = true,
6498            Some(crate::control::ControlSpelling::Variant) => self.variant_listed = true,
6499            None => {}
6500        }
6501    }
6502
6503    /// Note what reading the control through the listed `name` returned. Only a case
6504    /// variant's read is a lookup; the exact name already counted when it was listed.
6505    fn read(&mut self, name: &OsStr, read: &Result<Option<Op>>) {
6506        if crate::control::control_spelling(name) == Some(crate::control::ControlSpelling::Variant)
6507        {
6508            self.lookup(read);
6509        }
6510    }
6511
6512    /// Note what a lookup of the directory's canonical control path returned.
6513    fn lookup(&mut self, lookup: &Result<Option<Op>>) {
6514        match lookup {
6515            Ok(Some(observed)) => {
6516                self.seen = true;
6517                self.looked_up = Some(observed.clone());
6518            }
6519            Ok(None) => {}
6520            Err(_) => self.seen = true,
6521        }
6522    }
6523
6524    /// The observation to push right after the sweep removes the entry `name`: what the
6525    /// lookup found, when `name` is the stale exact spelling and this listing showed a case
6526    /// variant in its place.
6527    ///
6528    /// Only a listed variant can hold rules the removal would drop. Without one, the exact
6529    /// file is simply gone: a narrowed walk's lookup found it before the listing and it was
6530    /// deleted in between, and the removal leaves the table as bare as the directory, as a
6531    /// cold walk would. Restating that lookup would keep rules for a file that no longer
6532    /// exists until the next pass. That window remains only where a case-sensitive
6533    /// directory stores a variant beside the exact file deleted in it, a variant that never
6534    /// held the rules there.
6535    fn restatement_after_removing(&self, name: &OsStr) -> Option<Op> {
6536        if name == crate::control::CONTROL_FILE_NAME && self.variant_listed {
6537            self.looked_up.clone()
6538        } else {
6539            None
6540        }
6541    }
6542}
6543
6544/// The expectation that guards the control observation a listed entry at `path` produced,
6545/// given the entry's own, `entry`.
6546///
6547/// A guard is the expectation of the path its operation names. That is the entry's path
6548/// for the exact name, but a case variant's observation names the canonical control path,
6549/// so only a variant pays for `expect` to read that path's expectation. Asked only once an
6550/// observation exists, so an ordinary entry never pays for it.
6551fn control_guard<T>(path: &Path, entry: T, expect: impl FnOnce(&Path) -> T) -> T {
6552    match crate::control::path_control_spelling(path) {
6553        Some(crate::control::ControlSpelling::Variant) => {
6554            expect(&crate::control::sibling_control_path(path))
6555        }
6556        _ => entry,
6557    }
6558}
6559
6560/// Push what reading the control through the entry at `path` returned: the observation,
6561/// guarded by the expectation of the path it names ([`control_guard`]), or, when the read
6562/// failed, the removal of rules the table holds there, since nothing verifies them now.
6563fn push_read_control(
6564    target: &ReconcileTarget<'_>,
6565    path: &Path,
6566    baseline: PathExpectation,
6567    read: Result<Option<Op>>,
6568    batch: &mut Vec<ObservationOp>,
6569    report: &mut ReconcileReport,
6570) -> Result<()> {
6571    match read {
6572        Ok(Some(control)) => {
6573            let guard = control_guard(path, Ok(baseline), |named| target.expectation(named))?;
6574            batch.push(ObservationOp::if_state(control, guard));
6575        }
6576        Ok(None) => {}
6577        Err(error) => {
6578            // The failed read was of the entry itself for the exact name, and of the
6579            // canonical path a case variant looked up: the canonical path either way.
6580            let control_path = crate::control::sibling_control_path(path);
6581            if target.has_control(&control_path)? {
6582                let guard = control_guard(path, Ok(baseline), |named| target.expectation(named))?;
6583                batch
6584                    .push(ObservationOp::if_state(Op::ControlRemove { path: control_path }, guard));
6585            }
6586            report.scan.errors.push(error);
6587        }
6588    }
6589    Ok(())
6590}
6591
6592/// Push what a reconcile root at `path` that vanished, or could not be observed, leaves of
6593/// its directory's control, when `path` spells the control name.
6594///
6595/// The exact name was the lookup's target, so its loss removes the rules the table holds,
6596/// as it always has. A case variant's loss says nothing by itself: another spelling may
6597/// still resolve, or the variant was never the control, so the canonical path is looked
6598/// up and its answer stands, a miss removing the rules.
6599fn push_lost_spelling_control(
6600    target: &ReconcileTarget<'_>,
6601    root: &Path,
6602    config: &ScanConfig,
6603    path: &Path,
6604    baseline: PathExpectation,
6605    batch: &mut Vec<ObservationOp>,
6606    report: &mut ReconcileReport,
6607) -> Result<()> {
6608    match crate::control::path_control_spelling(path) {
6609        Some(crate::control::ControlSpelling::Exact) => {
6610            if target.has_control(path)? {
6611                batch.push(ObservationOp::if_state(
6612                    Op::ControlRemove { path: path.to_path_buf() },
6613                    baseline,
6614                ));
6615            }
6616        }
6617        Some(crate::control::ControlSpelling::Variant) => {
6618            let control_path = crate::control::sibling_control_path(path);
6619            let read = read_directory_control_or_removal(config, root, &control_path);
6620            push_read_control(target, path, baseline, read, batch, report)?;
6621        }
6622        None => {}
6623    }
6624    Ok(())
6625}
6626
6627/// What a listed child gone at its stat removes: its entry, if the baseline holds one, and
6628/// its rules, if it is the directory's control file and the table holds them.
6629///
6630/// The stat's `NotFound` is positive evidence that both are gone, so neither waits for a
6631/// complete listing; only a name the listing never returned has to, because it may merely
6632/// be unread. Removing a retained control file's entry drops its rules too, but a
6633/// hidden-pruned one has no entry, and without its own removal one unreadable sibling would
6634/// leave its rules applied with no file behind them. Clears `had_control` once the rules
6635/// are removed, so the listing's closing removals do not repeat it.
6636fn vanished_child_removals(
6637    dir: &Path,
6638    name: &OsStr,
6639    entry_held: bool,
6640    had_control: &mut bool,
6641) -> [Option<Op>; 2] {
6642    let path = dir.join(name);
6643    let rules = (*had_control && name == crate::control::CONTROL_FILE_NAME).then(|| {
6644        *had_control = false;
6645        Op::ControlRemove { path: path.clone() }
6646    });
6647    [entry_held.then_some(Op::Remove { path }), rules]
6648}
6649
6650fn defer_reconcile_op(
6651    operation: Op,
6652    operations: &mut Vec<Op>,
6653    deferred_count: &std::sync::atomic::AtomicUsize,
6654    overflowed: &std::sync::atomic::AtomicBool,
6655    max_deferred_ops: usize,
6656) {
6657    let position = deferred_count.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
6658    if position < max_deferred_ops {
6659        operations.push(operation);
6660    } else {
6661        overflowed.store(true, std::sync::atomic::Ordering::Relaxed);
6662    }
6663}
6664
6665fn flush_direct_reconcile_batch(
6666    index: &mut Index,
6667    batch: &mut Vec<Op>,
6668    sink: &mut dyn FnMut(&Commit),
6669    stats: &mut ApplyStats,
6670    observations: &mut u64,
6671) -> Result<()> {
6672    if batch.is_empty() {
6673        return Ok(());
6674    }
6675    *observations = observations.saturating_add(u64::try_from(batch.len()).unwrap_or(u64::MAX));
6676    let outcome = index.apply(&Observation::new(std::mem::take(batch)))?;
6677    merge_apply_stats(stats, outcome.stats);
6678    if let Some(commit) = outcome.commit.as_ref() {
6679        sink(commit);
6680    }
6681    Ok(())
6682}
6683
6684/// Drain and reconcile every pending invalidation, collapsing nested requests.
6685///
6686/// An invalidation whose reconciliation comes back incomplete -- a subtree that could not
6687/// be read, or a conditional commit that lost a race -- is queued again, so the next call
6688/// retries it. That suits a caller that drains when it chooses. A caller that drains after
6689/// every event would re-walk an unreadable subtree each time, and belongs on
6690/// [`reconcile_pending_handle`], which settles it instead.
6691pub fn reconcile_pending(
6692    index: &mut Index,
6693    config: &ScanConfig,
6694    sink: &mut dyn FnMut(&Commit),
6695) -> Result<ReconcileReport> {
6696    let mut target = ReconcileTarget::Direct(index);
6697    reconcile_pending_target(&mut target, config, sink)
6698}
6699
6700/// Drain and reconcile invalidations on a shared index.
6701///
6702/// Unlike [`reconcile_pending`], a subtree that could not be read is not queued again.
6703/// `Watcher::apply_next` drains after every event, and a retry there re-walks the same
6704/// unreadable subtree on each unrelated one, for the life of the watch. The error is a
6705/// settled boundary instead: the subtree stays [`crate::Freshness::Partial`], the returned
6706/// report names the error once, and the watcher retains its cause as an issue. Only a lost
6707/// race -- a stale conditional commit -- is queued for the next call. Invalidate the subtree
6708/// again to retry it deliberately.
6709pub fn reconcile_pending_handle(
6710    handle: &IndexHandle,
6711    config: &ScanConfig,
6712    sink: &mut dyn FnMut(&Commit),
6713) -> Result<ReconcileReport> {
6714    let mut target = ReconcileTarget::Shared(handle);
6715    reconcile_pending_target(&mut target, config, sink)
6716}
6717
6718/// Drain and reconcile invalidations under an opened-root lifecycle and resource bound.
6719#[cfg(feature = "watch")]
6720pub(crate) fn reconcile_pending_handle_controlled(
6721    handle: &IndexHandle,
6722    config: &ScanConfig,
6723    control: &dyn ReconcileControl,
6724    sink: &mut dyn FnMut(&Commit),
6725) -> Result<ReconcileReport> {
6726    let mut target = ReconcileTarget::Controlled { handle, control };
6727    reconcile_pending_target(&mut target, config, sink)
6728}
6729
6730fn reconcile_pending_target(
6731    target: &mut ReconcileTarget<'_>,
6732    config: &ScanConfig,
6733    sink: &mut dyn FnMut(&Commit),
6734) -> Result<ReconcileReport> {
6735    config.validate_for_scope(target.scope()?)?;
6736    let roots = take_invalidation_roots(target)?;
6737    let mut combined = ReconcileReport::default();
6738    for (position, (root, reason)) in roots.iter().enumerate() {
6739        match reconcile_target(target, root, config, sink) {
6740            Ok(report) => {
6741                if target.retries_incomplete(&report) {
6742                    target.restore_pending_invalidations(vec![(root.clone(), *reason)])?;
6743                }
6744                merge_reconcile_report(&mut combined, report);
6745            }
6746            Err(error) => {
6747                target.restore_pending_invalidations(roots[position..].to_vec())?;
6748                return Err(error);
6749            }
6750        }
6751    }
6752    Ok(combined)
6753}
6754
6755fn take_invalidation_roots(
6756    target: &mut ReconcileTarget<'_>,
6757) -> Result<Vec<(PathBuf, crate::InvalidateReason)>> {
6758    let mut pending = target.take_pending_invalidations()?;
6759    pending.sort_by(|(left, _), (right, _)| {
6760        left.components().count().cmp(&right.components().count()).then_with(|| left.cmp(right))
6761    });
6762    let mut roots: Vec<(PathBuf, crate::InvalidateReason)> = Vec::new();
6763    for (path, reason) in pending {
6764        if roots.iter().any(|(root, _)| path.starts_with(root)) {
6765            continue;
6766        }
6767        roots.push((path, reason));
6768    }
6769
6770    Ok(roots)
6771}
6772
6773fn remove_known_children(
6774    target: &mut ReconcileTarget<'_>,
6775    path: &Path,
6776    config: &ScanConfig,
6777    batch: &mut Vec<ObservationOp>,
6778    sink: &mut dyn FnMut(&Commit),
6779    report: &mut ReconcileReport,
6780) -> Result<()> {
6781    for (name, baseline) in target.child_states(path)? {
6782        batch.push(ObservationOp::if_state(Op::Remove { path: path.join(name) }, baseline));
6783        if batch.len() >= config.batch_size.max(1) {
6784            flush_reconcile_batch(target, batch, sink, report)?;
6785        }
6786    }
6787    flush_reconcile_batch(target, batch, sink, report)
6788}
6789
6790fn push_reconcile_upsert(
6791    target: &ReconcileTarget<'_>,
6792    path: &Path,
6793    kind: EntryKind,
6794    attrs: Attrs,
6795    baseline: PathExpectation,
6796    batch: &mut Vec<ObservationOp>,
6797    report: &mut ReconcileReport,
6798) {
6799    // An exclusive Index borrow cannot race another index producer. If filesystem
6800    // metadata exactly matches the captured state, applying this upsert can only be a
6801    // no-op, so avoid allocating an owned op and walking the index again. Shared
6802    // reconciliation keeps the conditional observation so ABA arbitration remains
6803    // authoritative between its read and write lock boundaries.
6804    if target.direct_upsert_is_unchanged(baseline, kind, attrs) {
6805        report.observations = report.observations.saturating_add(1);
6806        report.apply.unchanged = report.apply.unchanged.saturating_add(1);
6807        return;
6808    }
6809    batch.push(ObservationOp::if_state(
6810        Op::Upsert { path: path.to_path_buf(), kind, attrs },
6811        baseline,
6812    ));
6813}
6814
6815fn flush_reconcile_batch(
6816    target: &mut ReconcileTarget<'_>,
6817    batch: &mut Vec<ObservationOp>,
6818    sink: &mut dyn FnMut(&Commit),
6819    report: &mut ReconcileReport,
6820) -> Result<()> {
6821    if batch.is_empty() {
6822        return Ok(());
6823    }
6824    report.observations =
6825        report.observations.saturating_add(u64::try_from(batch.len()).unwrap_or(u64::MAX));
6826    let started_at = report.reconcile_epoch.expect("reconciliation report has an owner");
6827    let outcome = target.apply(started_at, &Observation::from_ops(std::mem::take(batch)))?;
6828    merge_apply_stats(&mut report.apply, outcome.stats);
6829    if let Some(commit) = outcome.commit.as_ref() {
6830        sink(commit);
6831    }
6832    Ok(())
6833}
6834
6835fn merge_apply_stats(total: &mut ApplyStats, addition: ApplyStats) {
6836    total.inserted += addition.inserted;
6837    total.updated += addition.updated;
6838    total.removed += addition.removed;
6839    total.unchanged += addition.unchanged;
6840    total.invalidated += addition.invalidated;
6841    total.controls += addition.controls;
6842    total.reclassified += addition.reclassified;
6843    total.stale += addition.stale;
6844    total.resource_refused += addition.resource_refused;
6845}
6846
6847fn merge_reconcile_report(total: &mut ReconcileReport, addition: ReconcileReport) {
6848    total.retry_required |= addition.retry_required;
6849    total.scan.dirs_read += addition.scan.dirs_read;
6850    total.scan.entries += addition.scan.entries;
6851    total.scan.files_walked += addition.scan.files_walked;
6852    total.scan.bytes_walked += addition.scan.bytes_walked;
6853    total.scan.allocated_walked += addition.scan.allocated_walked;
6854    total.scan.errors.extend(addition.scan.errors);
6855    total.observations = total.observations.saturating_add(addition.observations);
6856    merge_apply_stats(&mut total.apply, addition.apply);
6857    total.listed_incomplete.extend(addition.listed_incomplete);
6858}
6859
6860fn should_descend(
6861    kind: EntryKind,
6862    attrs: Attrs,
6863    parent_depth: usize,
6864    root_dev: u64,
6865    config: &ScanConfig,
6866) -> bool {
6867    crate::admission::should_descend(
6868        kind,
6869        attrs,
6870        parent_depth,
6871        root_dev,
6872        config.max_depth,
6873        config.one_filesystem,
6874    )
6875}
6876
6877pub(crate) fn normalize_subtree(path: &Path) -> Result<PathBuf> {
6878    let mut normalized = PathBuf::new();
6879    for component in path.components() {
6880        match component {
6881            Component::Normal(part) => normalized.push(part),
6882            Component::CurDir => {}
6883            Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
6884                return Err(Error::PathEscapesRoot(path.to_path_buf()));
6885            }
6886        }
6887    }
6888    Ok(normalized)
6889}
6890
6891fn resolve_subtree_root(
6892    target: &ReconcileTarget<'_>,
6893    subtree: &Path,
6894    config: &ScanConfig,
6895) -> Result<PathBuf> {
6896    if subtree.as_os_str().is_empty() {
6897        return Ok(PathBuf::new());
6898    }
6899    let root = target.root_path()?;
6900    crate::counters::bump(|c| c.stats += 1);
6901    let Ok(root_metadata) = fs::symlink_metadata(&root) else {
6902        // The applying pass reports operational root failures as partial.
6903        return Ok(subtree.to_path_buf());
6904    };
6905    if !root_metadata.is_dir() {
6906        return Ok(subtree.to_path_buf());
6907    }
6908    let Ok(root_dev) = root_device(&root, &root_metadata) else {
6909        return Ok(subtree.to_path_buf());
6910    };
6911    let mut prefix = PathBuf::new();
6912    let mut components = subtree.components().peekable();
6913    while let Some(component) = components.next() {
6914        if components.peek().is_none() {
6915            break; // The boundary entry itself remains visible even when descent stops.
6916        }
6917        prefix.push(component.as_os_str());
6918        let metadata = match fs::symlink_metadata(root.join(&prefix)) {
6919            Ok(metadata) => metadata,
6920            Err(error)
6921                if matches!(
6922                    error.kind(),
6923                    std::io::ErrorKind::NotFound | std::io::ErrorKind::NotADirectory
6924                ) =>
6925            {
6926                return Ok(prefix);
6927            }
6928            Err(_) => break, // The applying pass records operational failures as partial.
6929        };
6930        if metadata.file_type().is_symlink() {
6931            return Err(Error::SubtreeOutsideScanScope {
6932                path: subtree.to_path_buf(),
6933                scope: config.scope(),
6934            });
6935        }
6936        if !metadata.is_dir() {
6937            return Ok(prefix);
6938        }
6939        let Ok(attrs) = attrs_from(&root.join(&prefix), &metadata) else {
6940            break;
6941        };
6942        if config.one_filesystem && root_dev != 0 && attrs.dev != 0 && attrs.dev != root_dev {
6943            return Err(Error::SubtreeOutsideScanScope {
6944                path: subtree.to_path_buf(),
6945                scope: config.scope(),
6946            });
6947        }
6948    }
6949    Ok(subtree.to_path_buf())
6950}
6951
6952/// Read an entry's kind and roll-up attributes out of its metadata.
6953///
6954/// Exposed so the watch layer verifies entries exactly the way the walker records them —
6955/// two stat interpretations that could drift would show up as an index that disagrees
6956/// with itself depending on which producer last touched a path.
6957///
6958/// On Windows the observation comes from a fresh non-following handle, and `meta` is
6959/// what answers for an entry whose handle cannot be opened because it is locked or
6960/// access is denied — the same fallback std's `metadata` makes, with identity and change
6961/// time unavailable for that entry.
6962pub fn observe(path: &Path, meta: &fs::Metadata) -> std::io::Result<(EntryKind, Attrs)> {
6963    #[cfg(windows)]
6964    {
6965        windows_metadata::observe(path, || Ok(meta.clone()))
6966    }
6967    #[cfg(not(windows))]
6968    {
6969        Ok((kind_from(meta), attrs_from(path, meta)?))
6970    }
6971}
6972
6973pub(crate) fn observe_dir_entry(
6974    entry: &fs::DirEntry,
6975) -> std::io::Result<Option<(EntryKind, Attrs)>> {
6976    #[cfg(windows)]
6977    {
6978        crate::counters::bump(|c| c.stats += 1);
6979        #[cfg(test)]
6980        {
6981            let path = entry.path();
6982            if let Some(error) =
6983                walk_hook(&path).and_then(|hook| hook(WalkHookPoint::ChildMetadata(&path)))
6984            {
6985                return missing_as_none(Err(error));
6986            }
6987        }
6988        // The listing already holds the entry's enumeration data; it is read only when
6989        // the handle cannot be opened, so the ordinary path allocates nothing more.
6990        missing_as_none(windows_metadata::observe(&entry.path(), || entry.metadata()))
6991    }
6992    #[cfg(not(windows))]
6993    {
6994        let Some(meta) = listed_child_metadata(entry)? else {
6995            return Ok(None);
6996        };
6997        Ok(Some((kind_from(&meta), attrs_from(Path::new(""), &meta)?)))
6998    }
6999}
7000
7001#[cfg(not(windows))]
7002fn kind_from(meta: &fs::Metadata) -> EntryKind {
7003    let file_type = meta.file_type();
7004    if file_type.is_symlink() {
7005        EntryKind::Symlink
7006    } else if file_type.is_dir() {
7007        EntryKind::Dir
7008    } else if file_type.is_file() {
7009        EntryKind::File
7010    } else {
7011        EntryKind::Other
7012    }
7013}
7014
7015#[cfg(unix)]
7016#[allow(clippy::unnecessary_wraps)] // Windows observation is fallible; keep one call contract.
7017pub(crate) fn attrs_from(_path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
7018    use std::os::unix::fs::MetadataExt;
7019    Ok(Attrs {
7020        size: meta.size(),
7021        // st_blocks is in 512-byte units by POSIX convention regardless of the
7022        // filesystem's own block size.
7023        allocated: meta.blocks().saturating_mul(512),
7024        mtime_ns: compose_ns(meta.mtime(), meta.mtime_nsec()),
7025        ctime_ns: compose_ns(meta.ctime(), meta.ctime_nsec()),
7026        inode: meta.ino(),
7027        dev: meta.dev(),
7028    })
7029}
7030
7031#[cfg(unix)]
7032fn compose_ns(secs: i64, nanos: i64) -> i64 {
7033    secs.saturating_mul(1_000_000_000).saturating_add(nanos)
7034}
7035
7036#[cfg(windows)]
7037pub(crate) fn attrs_from(path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
7038    windows_metadata::observe(path, || Ok(meta.clone())).map(|(_, attrs)| attrs)
7039}
7040
7041#[cfg(not(any(unix, windows)))]
7042#[allow(clippy::unnecessary_wraps)] // Windows observation is fallible; keep one call contract.
7043pub(crate) fn attrs_from(_path: &Path, meta: &fs::Metadata) -> std::io::Result<Attrs> {
7044    let mtime_ns = meta.modified().map_or(0, system_time_ns);
7045    Ok(Attrs {
7046        size: meta.len(),
7047        // No allocated size without platform-specific calls; apparent size is the
7048        // honest fallback rather than a guess at block rounding.
7049        allocated: meta.len(),
7050        mtime_ns,
7051        // Windows has no ctime in the Unix sense. Leaving it zero means the fingerprint
7052        // degrades to size + mtime there, which is what every portable tool does.
7053        ctime_ns: 0,
7054        inode: 0,
7055        dev: 0,
7056    })
7057}
7058
7059/// The device a walk's root is on, which bounds a one-filesystem walk.
7060///
7061/// Only the device is needed, and on Windows it is read without demanding a consistent
7062/// observation of the root's times, which change whenever a child is created or removed.
7063pub(crate) fn root_device(root: &Path, meta: &fs::Metadata) -> std::io::Result<u64> {
7064    #[cfg(windows)]
7065    {
7066        let _ = meta;
7067        windows_metadata::volume_serial(root)
7068    }
7069    #[cfg(not(windows))]
7070    {
7071        attrs_from(root, meta).map(|attrs| attrs.dev)
7072    }
7073}
7074
7075pub(crate) fn attrs_from_file(file: &fs::File, meta: &fs::Metadata) -> std::io::Result<Attrs> {
7076    #[cfg(windows)]
7077    {
7078        let _ = meta;
7079        windows_metadata::attrs_from_file(file)
7080    }
7081    #[cfg(not(windows))]
7082    {
7083        let _ = file;
7084        attrs_from(Path::new(""), meta)
7085    }
7086}
7087
7088#[cfg(any(not(any(unix, windows)), test))]
7089fn system_time_ns(time: std::time::SystemTime) -> i64 {
7090    match time.duration_since(std::time::UNIX_EPOCH) {
7091        Ok(duration) => i64::try_from(duration.as_nanos()).unwrap_or(i64::MAX),
7092        Err(error) => {
7093            i64::try_from(error.duration().as_nanos()).map_or(i64::MIN, i64::saturating_neg)
7094        }
7095    }
7096}
7097
7098#[cfg(test)]
7099mod tests {
7100    use super::*;
7101    use std::fs::File;
7102    use std::io::Write;
7103
7104    fn write_file(path: &Path, contents: &[u8]) {
7105        if let Some(parent) = path.parent() {
7106            fs::create_dir_all(parent).expect("create parent");
7107        }
7108        let mut f = File::create(path).expect("create file");
7109        f.write_all(contents).expect("write");
7110    }
7111
7112    fn sample_tree() -> tempfile::TempDir {
7113        let dir = tempfile::tempdir().expect("tempdir");
7114        write_file(&dir.path().join("a.txt"), b"hello");
7115        write_file(&dir.path().join("src/main.rs"), b"fn main() {}");
7116        write_file(&dir.path().join("src/deep/nested.rs"), b"// nested");
7117        dir
7118    }
7119
7120    /// A counter that silently reads zero is worse than a missing one, because a report
7121    /// full of zeroes invites the conclusion that the work did not happen.
7122    ///
7123    /// This has already gone wrong twice: once when the per-entry counter was added to
7124    /// the serial walk while the parallel walk went uninstrumented, and once when a
7125    /// clippy fix hoisted a `read_dir` out of a match scrutinee and took the counter
7126    /// with it. Both builds compiled, passed every other test, and reported zero. This
7127    /// asserts the relationships a real walk must satisfy, so the next such edit fails
7128    /// here instead of in a report someone believes.
7129    #[test]
7130    fn a_walk_moves_every_counter_it_should() {
7131        // Both walkers, because they are separate loops with separate call sites. The
7132        // first version of this test only exercised the parallel one, and deleting the
7133        // serial walker's counter still passed — a guard covering one path gives false
7134        // confidence about the other.
7135        let _serial = crate::counters::test_serial();
7136        crate::counters::enable(true);
7137        for threads in [Some(1), Some(4)] {
7138            let dir = sample_tree();
7139            let config = ScanConfig { threads, ..ScanConfig::default() };
7140
7141            // Deltas around the scan, not absolute totals. The counters are
7142            // process-global, so a test running beside this one can add to them — and
7143            // `test_serial` cannot prevent that, since it only serializes tests that
7144            // take it, not every test that happens to walk a tree.
7145            let before = crate::counters::snapshot();
7146            let report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
7147            crate::counters::flush_thread();
7148            let after = crate::counters::snapshot();
7149            let observed_entries = after.dir_entries - before.dir_entries;
7150            let observed_opens = after.dir_opens - before.dir_opens;
7151            let observed_stats = after.stats - before.stats;
7152
7153            // `>=` rather than `==`, and the direction is the whole point: concurrent
7154            // work can only inflate these, never deflate them. So a counter that is too
7155            // low means a path ran uninstrumented, which is the failure worth catching
7156            // and the one that has actually happened — the macOS bulk reader reported
7157            // zero opens against three real ones. Equality would catch double-counting
7158            // too, and would be flaky for it.
7159            assert!(
7160                observed_entries >= report.entries,
7161                "every enumerated entry is counted at {threads:?}: {observed_entries} < {}",
7162                report.entries
7163            );
7164            assert!(
7165                observed_opens >= report.dirs_read,
7166                "every directory open is counted at {threads:?}: {observed_opens} < {}",
7167                report.dirs_read
7168            );
7169            assert!(
7170                observed_stats >= report.entries,
7171                "every entry is stated at {threads:?}: {observed_stats} < {}",
7172                report.entries
7173            );
7174            #[cfg(target_os = "macos")]
7175            {
7176                let observed_enum = after.dir_enumeration_calls - before.dir_enumeration_calls;
7177                // The serial walker is the portable `read_dir` path, which cannot see
7178                // getdents multiplicity. Enumeration calls are a bulk-backend fact.
7179                if threads != Some(1) {
7180                    assert!(
7181                        observed_enum >= report.dirs_read,
7182                        "every successful bulk directory issues at least one enumeration \
7183                         call at {threads:?}: {observed_enum} < {}",
7184                        report.dirs_read
7185                    );
7186                }
7187            }
7188
7189            // Deliberately not asserted: `allocs` stays zero in a library test, because
7190            // allocation counting needs a binary to install `CountingAlloc` as its
7191            // global allocator and a test harness installs its own. The probe covers
7192            // that half; this covers the counters the library itself drives.
7193        }
7194        crate::counters::enable(false);
7195    }
7196
7197    #[test]
7198    fn summary_fold_skips_stat_on_directories_and_symlinks() {
7199        let _serial = crate::counters::test_serial();
7200        crate::counters::enable(true);
7201        let dir = tempfile::tempdir().expect("tempdir");
7202        fs::create_dir(dir.path().join("src")).expect("directory");
7203        write_file(&dir.path().join("a.txt"), b"hi");
7204        #[cfg(unix)]
7205        std::os::unix::fs::symlink("a.txt", dir.path().join("link")).expect("symlink");
7206        let config = ScanConfig { threads: Some(1), read_controls: false, ..ScanConfig::default() };
7207
7208        crate::counters::test_thread_reset();
7209        let scan_report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
7210        let scan_stats = crate::counters::test_thread_snapshot().stats;
7211
7212        crate::counters::test_thread_reset();
7213        let fold_report = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
7214        let fold_stats = crate::counters::test_thread_snapshot().stats;
7215        crate::counters::enable(false);
7216
7217        assert_eq!(fold_report.entries, scan_report.entries);
7218        assert_eq!(fold_report.files_walked, scan_report.files_walked);
7219        assert_eq!(fold_report.bytes_walked, scan_report.bytes_walked);
7220        // Windows observes every listed entry through a fresh handle on both paths, so the
7221        // fold performs exactly the retained walk's observations there; the skip is a
7222        // non-Windows saving.
7223        #[cfg(not(windows))]
7224        assert!(
7225            fold_stats < scan_stats,
7226            "fold {fold_stats} should skip directory/symlink stats versus scan {scan_stats}"
7227        );
7228        #[cfg(unix)]
7229        assert_eq!(scan_stats.saturating_sub(fold_stats), 2);
7230        #[cfg(not(any(unix, windows)))]
7231        assert_eq!(scan_stats.saturating_sub(fold_stats), 1);
7232        #[cfg(windows)]
7233        assert_eq!(fold_stats, scan_stats);
7234    }
7235
7236    #[cfg(unix)]
7237    #[test]
7238    fn summary_fold_still_stats_directories_when_bound_to_one_filesystem() {
7239        let _serial = crate::counters::test_serial();
7240        crate::counters::enable(true);
7241        let dir = tempfile::tempdir().expect("tempdir");
7242        fs::create_dir(dir.path().join("src")).expect("directory");
7243        write_file(&dir.path().join("a.txt"), b"hi");
7244        std::os::unix::fs::symlink("a.txt", dir.path().join("link")).expect("symlink");
7245        let config = ScanConfig {
7246            threads: Some(1),
7247            read_controls: false,
7248            one_filesystem: true,
7249            ..ScanConfig::default()
7250        };
7251
7252        crate::counters::test_thread_reset();
7253        let scan_report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
7254        let scan_stats = crate::counters::test_thread_snapshot().stats;
7255
7256        crate::counters::test_thread_reset();
7257        let fold_report = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
7258        let fold_stats = crate::counters::test_thread_snapshot().stats;
7259        crate::counters::enable(false);
7260
7261        assert_eq!(fold_report.entries, scan_report.entries);
7262        assert_eq!(scan_stats.saturating_sub(fold_stats), 1);
7263    }
7264
7265    #[test]
7266    fn summary_fold_reuses_cleared_recycled_batches() {
7267        // Four workers and a batch of three force StreamingEmission to send more than
7268        // once per worker on this tree. Without `recycled.clear()`, the next send
7269        // re-folds the previous ops and files/bytes/dirs double-count.
7270        const DIRS: usize = 16;
7271        const FILES_PER_DIR: usize = 40;
7272        let dir = tempfile::tempdir().expect("tempdir");
7273        let mut expected_bytes = 0u64;
7274        for directory in 0..DIRS {
7275            let child = dir.path().join(format!("d{directory:02}"));
7276            fs::create_dir(&child).expect("directory");
7277            for file in 0..FILES_PER_DIR {
7278                let size = directory * FILES_PER_DIR + file + 1;
7279                expected_bytes += size as u64;
7280                write_file(&child.join(format!("f{file:02}.dat")), &vec![b'x'; size]);
7281            }
7282        }
7283        let expected_files = (DIRS * FILES_PER_DIR) as u64;
7284        let expected_dirs = DIRS as u64;
7285        let expected_entries = expected_files + expected_dirs;
7286        let config = ScanConfig {
7287            threads: Some(4),
7288            batch_size: 3,
7289            read_controls: false,
7290            ..ScanConfig::default()
7291        };
7292        let mut files = 0u64;
7293        let mut bytes = 0u64;
7294        let mut dirs = 0u64;
7295        let mut ops = 0u64;
7296        let report = scan_summary_fold(dir.path(), &config, &mut |observed| {
7297            ops += 1;
7298            let Op::Upsert { kind, attrs, .. } = &observed.op else {
7299                return;
7300            };
7301            match kind {
7302                EntryKind::File => {
7303                    files += 1;
7304                    bytes += attrs.size;
7305                }
7306                EntryKind::Dir => dirs += 1,
7307                EntryKind::Symlink | EntryKind::Other => {}
7308            }
7309        })
7310        .expect("fold");
7311        assert_eq!(files, expected_files);
7312        assert_eq!(bytes, expected_bytes);
7313        assert_eq!(dirs, expected_dirs);
7314        assert_eq!(ops, report.entries);
7315        assert_eq!(report.entries, expected_entries);
7316        assert_eq!(report.files_walked, expected_files);
7317        assert_eq!(report.bytes_walked, expected_bytes);
7318    }
7319
7320    /// Publish one listing through `emission` and receive what the consumer would.
7321    fn publish_detached(
7322        emission: &mut DetachedEmission,
7323        directory: DetachedDirectory,
7324        sender: &std::sync::mpsc::Sender<WalkMessage>,
7325        receiver: &std::sync::mpsc::Receiver<WalkMessage>,
7326    ) -> (Vec<DetachedDirectory>, std::sync::mpsc::Sender<Vec<DetachedDirectory>>) {
7327        emission.finish_directory(directory);
7328        let mut send_ns = 0;
7329        assert!(emission.publish_before_discovery(true, sender, &mut send_ns, None));
7330        match receiver.try_recv().expect("a published chunk") {
7331            WalkMessage::DetachedDirectories { directories, recycle } => (directories, recycle),
7332            _ => panic!("the detached walker publishes only listings"),
7333        }
7334    }
7335
7336    #[test]
7337    fn detached_emission_reuses_returned_listings_as_fresh_ones() {
7338        // H159: the consumer hands drained listings back to the worker that allocated
7339        // them. A reused listing must be indistinguishable from a fresh one, including
7340        // after the consumer returned it unapplied, as it does after a build error.
7341        let (sender, receiver) = std::sync::mpsc::channel();
7342        let mut emission = DetachedEmission::new();
7343
7344        let mut skipped = emission.begin_directory(Path::new("a/much/longer/relative/path"));
7345        skipped.children.push(DetachedChild {
7346            name: OsString::from("stale.txt"),
7347            kind: EntryKind::File,
7348            attrs: Attrs::default(),
7349            position: 0,
7350        });
7351        skipped.control = Some(Op::ControlRemove { path: PathBuf::from("a/.gitignore") });
7352        let (directories, recycle) = publish_detached(&mut emission, skipped, &sender, &receiver);
7353        let returned_list = directories.as_ptr();
7354        let returned_children = directories[0].children.as_ptr();
7355        recycle.send(directories).expect("the worker still listens");
7356
7357        // Returned listings are taken back at the next publish, so this one is fresh.
7358        let mut large = emission.begin_directory(Path::new("large"));
7359        assert_eq!(large.children.capacity(), 0);
7360        large.children.reserve_exact(DETACHED_SPARE_CHILD_CAPACITY + 1);
7361        let (directories, recycle) = publish_detached(&mut emission, large, &sender, &receiver);
7362        recycle.send(directories).expect("the worker still listens");
7363
7364        let reused = emission.begin_directory(Path::new("b"));
7365        assert_eq!(reused.path.as_os_str(), OsStr::new("b"));
7366        assert!(reused.children.is_empty(), "a returned listing's children are dropped");
7367        assert!(reused.control.is_none(), "a returned listing's control is dropped");
7368        assert_eq!(reused.children.as_ptr(), returned_children, "the child buffer is reused");
7369        let (directories, _recycle) = publish_detached(&mut emission, reused, &sender, &receiver);
7370        assert_eq!(directories.as_ptr(), returned_list, "the published list is reused");
7371
7372        // The large listing came back past the retention bound and was freed instead.
7373        let fresh = emission.begin_directory(Path::new("c"));
7374        assert_eq!(fresh.path.as_os_str(), OsStr::new("c"));
7375        assert_eq!(fresh.children.capacity(), 0);
7376    }
7377
7378    /// A transient fold that reads `.gitignore` receives each directory's control, once,
7379    /// ahead of every entry in that directory, whatever the batch size, the worker count,
7380    /// or where `.gitignore` falls in the listing (fdu-1ovb). Its consumer classifies each
7381    /// entry as it arrives, so an entry folded before its directory's control would miss
7382    /// that control's rules. Batches smaller than a listing force the probe path; the
7383    /// default batch holds whole listings and moves the listed control.
7384    #[test]
7385    fn a_classifying_summary_fold_receives_each_control_ahead_of_its_entries() {
7386        const DIRS: usize = 12;
7387        const FILES_PER_DIR: usize = 30;
7388        let dir = tempfile::tempdir().expect("tempdir");
7389        write_file(&dir.path().join(".gitignore"), b"*.log\n");
7390        for directory in 0..DIRS {
7391            let child = dir.path().join(format!("d{directory:02}"));
7392            write_file(&child.join(".gitignore"), b"*.tmp\n");
7393            for file in 0..FILES_PER_DIR {
7394                write_file(&child.join(format!("f{file:02}.dat")), b"x");
7395            }
7396            write_file(&child.join("nested/n.dat"), b"n");
7397        }
7398        let default_batch = ScanConfig::default().batch_size;
7399        for (threads, batch_size) in [
7400            (Some(1), 1),
7401            (Some(4), 1),
7402            (Some(4), 3),
7403            (None, 7),
7404            (Some(4), 16),
7405            (None, default_batch),
7406        ] {
7407            let config = ScanConfig { batch_size, threads, ..ScanConfig::default() };
7408            assert!(config.read_controls, "observation is the default");
7409            // Per directory: the positions of its control observations and of its entries.
7410            let mut seen: std::collections::BTreeMap<PathBuf, (Vec<usize>, Vec<usize>)> =
7411                std::collections::BTreeMap::new();
7412            let mut position = 0;
7413            scan_summary_fold(dir.path(), &config, &mut |observed| {
7414                let (path, control) = match &observed.op {
7415                    Op::Upsert { path, .. } => (path, false),
7416                    Op::ControlUpsert { path, .. } | Op::ControlRemove { path } => (path, true),
7417                    other => panic!("a cold walk emitted {other:?}"),
7418                };
7419                let directory = path.parent().unwrap_or_else(|| Path::new("")).to_path_buf();
7420                let (controls, entries) = seen.entry(directory).or_default();
7421                if control { controls } else { entries }.push(position);
7422                position += 1;
7423            })
7424            .expect("fold");
7425
7426            assert_eq!(seen.len(), 1 + 2 * DIRS, "the root, each child, and each nested");
7427            for (directory, (controls, entries)) in &seen {
7428                let label = format!("{directory:?} with {threads:?} workers, batch {batch_size}");
7429                let governed = directory.as_os_str().is_empty()
7430                    || directory.file_name().is_some_and(|name| name != "nested");
7431                assert_eq!(controls.len(), usize::from(governed), "{label}: one control read");
7432                if let (Some(last_control), Some(first_entry)) =
7433                    (controls.iter().max(), entries.iter().min())
7434                {
7435                    assert!(last_control < first_entry, "{label}: its control comes first");
7436                }
7437            }
7438        }
7439    }
7440
7441    /// A directory's control is what a lookup of `<dir>/.gitignore` resolves to, as git
7442    /// opens it (fdu-0w1b). A listed `.GITIGNORE` reads exactly what that lookup finds,
7443    /// named by the canonical path: its rules where the directory is case-insensitive, and
7444    /// nothing where it is not. The probe a transient fold makes is that same lookup.
7445    #[test]
7446    fn a_directory_control_is_what_a_lookup_of_its_canonical_path_resolves_to() {
7447        use crate::test_support::CaseLookups;
7448
7449        let dir = tempfile::tempdir().expect("tempdir");
7450        write_file(&dir.path().join("exact/.gitignore"), b"*.log\n");
7451        write_file(&dir.path().join("variant/.GITIGNORE"), b"*.tmp\n");
7452        fs::create_dir_all(dir.path().join("shape/.GitIgnore")).expect("a directory so named");
7453        write_file(&dir.path().join("absent/README"), b"no rules");
7454        let config = ScanConfig::default();
7455        let upsert = |path: &str, source: &[u8]| {
7456            Some(Op::ControlUpsert { path: PathBuf::from(path), source: source.to_vec() })
7457        };
7458
7459        for (lookups, insensitive) in CaseLookups::on_this_host(dir.path()) {
7460            let _lookups = lookups.install(dir.path());
7461            let label = format!("{lookups:?} lookups");
7462            let lookup = |directory: &str| {
7463                read_directory_control(
7464                    &config,
7465                    dir.path(),
7466                    &Path::new(directory).join(".gitignore"),
7467                )
7468                .expect("lookup")
7469            };
7470            let listed = |path: &str, kind| {
7471                read_control_op(&config, dir.path(), Path::new(path), kind).expect("listed read")
7472            };
7473
7474            assert_eq!(lookup("exact"), upsert("exact/.gitignore", b"*.log\n"), "{label}");
7475            assert_eq!(listed("exact/.gitignore", EntryKind::File), lookup("exact"), "{label}");
7476            assert_eq!(
7477                lookup("variant"),
7478                if insensitive { upsert("variant/.gitignore", b"*.tmp\n") } else { None },
7479                "{label}"
7480            );
7481            assert_eq!(
7482                listed("variant/.GITIGNORE", EntryKind::File),
7483                lookup("variant"),
7484                "{label}: a listed variant reads what the lookup finds"
7485            );
7486            assert_eq!(
7487                listed("shape/.GitIgnore", EntryKind::Dir),
7488                insensitive.then(|| Op::ControlRemove { path: PathBuf::from("shape/.gitignore") }),
7489                "{label}: a directory is resolved to, and holds no rules"
7490            );
7491            assert_eq!(lookup("absent"), None, "{label}");
7492            assert_eq!(listed("absent/README", EntryKind::File), None, "{label}");
7493
7494            let blind = ScanConfig { read_controls: false, ..ScanConfig::default() };
7495            assert_eq!(
7496                read_control_op(
7497                    &blind,
7498                    dir.path(),
7499                    Path::new("variant/.GITIGNORE"),
7500                    EntryKind::File
7501                )
7502                .expect("read"),
7503                None,
7504                "{label}: a scan that reads no rules looks nothing up"
7505            );
7506        }
7507    }
7508
7509    /// Every entry's kind and ignored classification, and every retained control source:
7510    /// what a route's index says about a tree's `.gitignore` rules.
7511    type Classification = (BTreeMap<PathBuf, (EntryKind, Option<bool>)>, Vec<(PathBuf, Vec<u8>)>);
7512
7513    fn classification(index: &Index) -> Classification {
7514        let mut entries = BTreeMap::new();
7515        let mut pending = vec![PathBuf::new()];
7516        while let Some(directory) = pending.pop() {
7517            let children: Vec<(PathBuf, crate::EntryId)> = index
7518                .children(&directory)
7519                .into_iter()
7520                .flatten()
7521                .map(|(name, id)| (directory.join(name), id))
7522                .collect();
7523            for (path, id) in children {
7524                let kind = index.kind_of(id).expect("a live child");
7525                entries.insert(path.clone(), (kind, index.is_ignored(&path).expect("observed")));
7526                if kind.is_dir() {
7527                    pending.push(path);
7528                }
7529            }
7530        }
7531        let sources = index
7532            .controls()
7533            .expect("observed")
7534            .sources()
7535            .map(|(path, source)| (path, source.to_vec()))
7536            .collect();
7537        (entries, sources)
7538    }
7539
7540    /// A tree whose `up` directory holds its rules only in a case variant of the control
7541    /// name, beside a root `.gitignore` those rules partly override.
7542    fn case_variant_tree() -> tempfile::TempDir {
7543        let dir = tempfile::tempdir().expect("tempdir");
7544        write_file(&dir.path().join(".gitignore"), b"*.log\n");
7545        write_file(&dir.path().join("a.log"), b"root rule");
7546        write_file(&dir.path().join("keep.txt"), b"kept");
7547        write_file(&dir.path().join("up/.GITIGNORE"), b"*.tmp\n!keep.log\nbuild/\n");
7548        write_file(&dir.path().join("up/x.tmp"), b"variant rule");
7549        write_file(&dir.path().join("up/keep.log"), b"negated by the variant");
7550        write_file(&dir.path().join("up/y.log"), b"root rule");
7551        write_file(&dir.path().join("up/build/out.bin"), b"variant rule");
7552        write_file(&dir.path().join("up/deep/z.tmp"), b"variant rule");
7553        for file in 0..12 {
7554            write_file(&dir.path().join(format!("up/f{file:02}.dat")), b"unignored");
7555        }
7556        dir
7557    }
7558
7559    /// What [`case_variant_tree`] classifies where its variant does and does not govern.
7560    fn assert_case_variant_classification(classified: &Classification, governs: bool, label: &str) {
7561        let (entries, sources) = classified;
7562        let ignored = |path: &str| entries.get(Path::new(path)).map(|(_, ignored)| *ignored);
7563        assert_eq!(ignored("a.log"), Some(Some(true)), "{label}");
7564        assert_eq!(ignored("up/y.log"), Some(Some(true)), "{label}");
7565        assert_eq!(ignored("up/x.tmp"), Some(Some(governs)), "{label}");
7566        assert_eq!(ignored("up/deep/z.tmp"), Some(Some(governs)), "{label}");
7567        assert_eq!(ignored("up/build"), Some(Some(governs)), "{label}");
7568        assert_eq!(ignored("up/build/out.bin"), Some(Some(governs)), "{label}");
7569        assert_eq!(ignored("up/keep.log"), Some(Some(!governs)), "{label}: negation");
7570        assert_eq!(ignored("up/f00.dat"), Some(Some(false)), "{label}");
7571        let governing: Vec<&Path> = sources.iter().map(|(path, _)| path.as_path()).collect();
7572        let expected: Vec<&Path> = if governs {
7573            vec![Path::new(".gitignore"), Path::new("up/.gitignore")]
7574        } else {
7575            vec![Path::new(".gitignore")]
7576        };
7577        assert_eq!(governing, expected, "{label}: rules are recorded by the canonical path");
7578    }
7579
7580    /// The detached builder, the retained scanner stream (serial and concurrent), and the
7581    /// narrowed-population walks classify a tree whose rules sit in `.GITIGNORE` alike, and
7582    /// apply those rules exactly where a lookup of `.gitignore` resolves to it (fdu-0w1b).
7583    /// Hidden pruning keeps the variant a control signal, as it keeps `.gitignore` one.
7584    #[test]
7585    fn every_cold_route_classifies_a_case_variant_control_alike() {
7586        use crate::test_support::CaseLookups;
7587
7588        let dir = case_variant_tree();
7589        let root = dir.path();
7590        let pruned = Some(std::sync::Arc::new(crate::HiddenPolicy::prune_hidden(Vec::<
7591            std::ffi::OsString,
7592        >::new())));
7593        for (lookups, governs) in CaseLookups::on_this_host(root) {
7594            let _lookups = lookups.install(root);
7595            let reference = ScanConfig { threads: Some(1), ..ScanConfig::default() };
7596            let (index, report) = scan_into_index(root, &reference).expect("detached scan");
7597            assert!(report.is_complete(), "{:?}", report.errors);
7598            let expected = classification(&index);
7599            assert_case_variant_classification(&expected, governs, &format!("{lookups:?}"));
7600
7601            for threads in [Some(1), Some(4)] {
7602                for batch_size in [ScanConfig::default().batch_size, 1, 3] {
7603                    let config = ScanConfig { batch_size, threads, ..ScanConfig::default() };
7604                    let label = format!("{lookups:?}, {threads:?} workers, batch {batch_size}");
7605                    let (detached, _) = scan_into_index(root, &config).expect("detached");
7606                    assert_eq!(classification(&detached), expected, "{label}: detached");
7607                    let (streamed, _) = scan_into_index_via_scanner(root, &config).expect("stream");
7608                    assert_eq!(classification(&streamed), expected, "{label}: scanner stream");
7609                }
7610            }
7611
7612            // Pruning hidden entries drops the variant's row, never its rules.
7613            let hidden = ScanConfig { hidden: pruned.clone(), ..ScanConfig::default() };
7614            let (without_hidden, _) = scan_into_index(root, &hidden).expect("hidden pruned");
7615            let (entries, sources) = classification(&without_hidden);
7616            assert_eq!(sources, expected.1, "{lookups:?}: hidden pruning keeps the rules");
7617            for (path, fact) in &entries {
7618                assert_eq!(expected.0.get(path), Some(fact), "{lookups:?}: {path:?}");
7619            }
7620            assert!(!entries.contains_key(Path::new("up/.GITIGNORE")), "{lookups:?}");
7621
7622            // A narrowed population reads each control before listing its directory, which
7623            // is the lookup itself; it keeps every spelling's row, as it keeps `.gitignore`.
7624            let spelled = |path: &Path| crate::control::path_control_spelling(path).is_some();
7625            for population in
7626                [crate::query::IgnoredEntries::Exclude, crate::query::IgnoredEntries::Only]
7627            {
7628                let config = ScanConfig { population, ..ScanConfig::default() };
7629                let (narrowed, report) = scan_into_index(root, &config).expect("narrowed");
7630                assert!(report.is_complete(), "{:?}", report.errors);
7631                let (entries, sources) = classification(&narrowed);
7632                let label = format!("{lookups:?}, {population:?}");
7633                assert_eq!(sources, expected.1, "{label}");
7634                let files =
7635                    |entries: &BTreeMap<PathBuf, (EntryKind, Option<bool>)>| -> Vec<PathBuf> {
7636                        entries
7637                            .iter()
7638                            .filter(|(_, (kind, _))| *kind == EntryKind::File)
7639                            .map(|(path, _)| path.clone())
7640                            .collect()
7641                    };
7642                let wanted: BTreeMap<PathBuf, (EntryKind, Option<bool>)> = expected
7643                    .0
7644                    .iter()
7645                    .filter(|(path, (_, ignored))| {
7646                        spelled(path)
7647                            || match population {
7648                                crate::query::IgnoredEntries::Exclude => *ignored == Some(false),
7649                                _ => *ignored == Some(true),
7650                            }
7651                    })
7652                    .map(|(path, fact)| (path.clone(), *fact))
7653                    .collect();
7654                assert_eq!(files(&entries), files(&wanted), "{label}: retained files");
7655            }
7656        }
7657    }
7658
7659    /// Reconciliation keeps a case-variant control exactly as a cold walk finds it through
7660    /// every change: a case-only rename of `.gitignore` to `.GITIGNORE`, an edit, a
7661    /// removal, and a refresh of the variant's own path after it is created and after it
7662    /// is gone. Serial and parallel reconciliation, and revalidation, agree with a cold
7663    /// walk after each (fdu-0w1b).
7664    ///
7665    /// The rename is the case a canonical path cannot follow on its own: the index still
7666    /// holds the old `.gitignore` entry, and removing an entry at the canonical path drops
7667    /// the rules it governs, which on a case-insensitive directory the variant still holds.
7668    #[test]
7669    fn reconciliation_follows_a_case_variant_control_through_every_change() {
7670        use crate::test_support::CaseLookups;
7671
7672        let probe = tempfile::tempdir().expect("tempdir");
7673        for (lookups, governs) in CaseLookups::on_this_host(probe.path()) {
7674            for threads in [Some(1), Some(4)] {
7675                let dir = tempfile::tempdir().expect("tempdir");
7676                let root = dir.path().canonicalize().expect("canonical root");
7677                let _lookups = lookups.install(&root);
7678                let config = ScanConfig { threads, batch_size: 3, ..ScanConfig::default() };
7679                write_file(&root.join(".gitignore"), b"*.log\n");
7680                write_file(&root.join("up/.gitignore"), b"*.tmp\n");
7681                write_file(&root.join("up/x.tmp"), b"governed");
7682                write_file(&root.join("up/y.bin"), b"governed after the edit");
7683                for file in 0..8 {
7684                    write_file(&root.join(format!("up/f{file}.dat")), b"unignored");
7685                }
7686                let (mut index, _) = scan_into_index(&root, &config).expect("cold scan");
7687                let cold = |label: &str| {
7688                    let (cold, report) = scan_into_index(&root, &config).expect("cold");
7689                    assert!(report.is_complete(), "{label}: {:?}", report.errors);
7690                    classification(&cold)
7691                };
7692                let reconciled = |index: &mut Index, label: &str| {
7693                    let report = reconcile(index, &config, &mut |_| {}).expect("reconcile");
7694                    assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
7695                    assert_eq!(report.apply.stale, 0, "{label}: no observation lost a race");
7696                };
7697                let revalidated = |snapshot: &Index, label: &str| {
7698                    let mut revalidated = snapshot.clone();
7699                    let mut observations = Vec::new();
7700                    revalidate(&revalidated, &config, &mut |observation| {
7701                        observations.push(observation);
7702                    })
7703                    .expect("revalidate");
7704                    for observation in &observations {
7705                        let outcome = revalidated.apply(observation).expect("apply");
7706                        assert_eq!(outcome.stats.stale, 0, "{label}: revalidation lost a race");
7707                    }
7708                    classification(&revalidated)
7709                };
7710                let governed = |label: &str| {
7711                    let (entries, _) = cold(label);
7712                    entries.get(Path::new("up/x.tmp")).map(|(_, ignored)| *ignored)
7713                };
7714
7715                let before_rename = index.clone();
7716                fs::rename(root.join("up/.gitignore"), root.join("up/.GITIGNORE")).expect("recase");
7717                let label = format!("{lookups:?}, {threads:?} workers: case-only rename");
7718                assert_eq!(governed(&label), Some(Some(governs)), "{label}");
7719                assert_eq!(
7720                    revalidated(&before_rename, &label),
7721                    cold(&label),
7722                    "{label}: revalidate"
7723                );
7724                reconciled(&mut index, &label);
7725                assert_eq!(classification(&index), cold(&label), "{label}");
7726
7727                write_file(&root.join("up/.GITIGNORE"), b"*.bin\n");
7728                let label = format!("{lookups:?}, {threads:?} workers: edit");
7729                reconciled(&mut index, &label);
7730                assert_eq!(classification(&index), cold(&label), "{label}");
7731
7732                fs::remove_file(root.join("up/.GITIGNORE")).expect("remove the variant");
7733                let label = format!("{lookups:?}, {threads:?} workers: removal");
7734                reconciled(&mut index, &label);
7735                assert_eq!(classification(&index), cold(&label), "{label}");
7736
7737                // A refresh of the variant's own path looks the directory's control up
7738                // whether the variant is there or gone.
7739                write_file(&root.join("up/.GITIGNORE"), b"*.tmp\n");
7740                for (step, label) in [
7741                    (None, format!("{lookups:?}, {threads:?} workers: refresh of a new variant")),
7742                    (
7743                        Some(()),
7744                        format!("{lookups:?}, {threads:?} workers: refresh of a removed variant"),
7745                    ),
7746                ] {
7747                    if step.is_some() {
7748                        fs::remove_file(root.join("up/.GITIGNORE")).expect("remove");
7749                    }
7750                    let report = reconcile_subtree(
7751                        &mut index,
7752                        Path::new("up/.GITIGNORE"),
7753                        &config,
7754                        &mut |_| {},
7755                    )
7756                    .expect("refresh");
7757                    assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
7758                    assert_eq!(classification(&index), cold(&label), "{label}");
7759                }
7760            }
7761        }
7762    }
7763
7764    /// Revalidate a copy of `index` and apply every observation it sends, none of which may
7765    /// lose a race.
7766    fn revalidated_copy(index: &Index, config: &ScanConfig, label: &str) -> Index {
7767        let mut revalidated = index.clone();
7768        let mut observations = Vec::new();
7769        let report = revalidate(&revalidated, config, &mut |observation| {
7770            observations.push(observation);
7771        })
7772        .expect("revalidate");
7773        assert!(report.is_complete(), "{label}: {:?}", report.errors);
7774        for observation in &observations {
7775            let outcome = revalidated.apply(observation).expect("apply");
7776            assert_eq!(outcome.stats.stale, 0, "{label}: revalidation lost a race");
7777        }
7778        revalidated
7779    }
7780
7781    /// A narrowed population's reconciliation keeps a case-variant control through a
7782    /// case-only rename, as a cold walk finds it. Its lookup before the listing is what
7783    /// finds the variant's rules, and the listing shows the variant, so the sweep restates
7784    /// the rules its removal of the stale `.gitignore` entry drops (fdu-0w1b).
7785    #[test]
7786    fn a_narrowed_reconciliation_follows_a_case_only_rename() {
7787        use crate::test_support::CaseLookups;
7788
7789        let probe = tempfile::tempdir().expect("tempdir");
7790        for (lookups, governs) in CaseLookups::on_this_host(probe.path()) {
7791            for population in
7792                [crate::query::IgnoredEntries::Exclude, crate::query::IgnoredEntries::Only]
7793            {
7794                let label = format!("{lookups:?}, {population:?}");
7795                let dir = tempfile::tempdir().expect("tempdir");
7796                let root = dir.path().canonicalize().expect("canonical root");
7797                let _lookups = lookups.install(&root);
7798                let config = ScanConfig { population, batch_size: 3, ..ScanConfig::default() };
7799                write_file(&root.join("up/.gitignore"), b"*.tmp\n");
7800                write_file(&root.join("up/x.tmp"), b"governed");
7801                for file in 0..4 {
7802                    write_file(&root.join(format!("up/f{file}.dat")), b"unignored");
7803                }
7804                let (mut index, _) = scan_into_index(&root, &config).expect("cold scan");
7805                fs::rename(root.join("up/.gitignore"), root.join("up/.GITIGNORE")).expect("recase");
7806                let (cold, report) = scan_into_index(&root, &config).expect("cold");
7807                assert!(report.is_complete(), "{label}: {:?}", report.errors);
7808                let cold = classification(&cold);
7809                assert_eq!(
7810                    cold.1.iter().any(|(path, _)| path == Path::new("up/.gitignore")),
7811                    governs,
7812                    "{label}: the variant governs exactly where the lookup resolves to it"
7813                );
7814
7815                let revalidated = revalidated_copy(&index, &config, &label);
7816                assert_eq!(classification(&revalidated), cold, "{label}: revalidate");
7817                let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
7818                assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
7819                assert_eq!(report.apply.stale, 0, "{label}: no observation lost a race");
7820                assert_eq!(classification(&index), cold, "{label}: reconcile");
7821            }
7822        }
7823    }
7824
7825    /// A narrowed population's reconciliation looks a directory's control up before
7826    /// listing it. When the `.gitignore` that lookup found is deleted before the listing,
7827    /// the sweep removes the stale entry, and the rules go with it: no case variant was
7828    /// listed to hold them, so nothing is restated, and the table is as bare as the
7829    /// directory. The next pass agrees with a cold walk.
7830    #[test]
7831    fn a_control_deleted_between_its_lookup_and_the_listing_leaves_no_rules() {
7832        for population in
7833            [crate::query::IgnoredEntries::Exclude, crate::query::IgnoredEntries::Only]
7834        {
7835            for revalidating in [true, false] {
7836                let label = format!("{population:?}, revalidating: {revalidating}");
7837                let dir = tempfile::tempdir().expect("tempdir");
7838                let root = dir.path().canonicalize().expect("canonical root");
7839                let config = ScanConfig { population, ..ScanConfig::default() };
7840                write_file(&root.join("up/.gitignore"), b"*.tmp\n");
7841                write_file(&root.join("up/x.tmp"), b"governed");
7842                write_file(&root.join("up/y.dat"), b"unignored");
7843                let control_path = Path::new("up/.gitignore");
7844                let (mut index, _) = scan_into_index(&root, &config).expect("cold scan");
7845                assert!(index.control_table().contains(control_path), "{label}");
7846
7847                let control = root.join(control_path);
7848                let deleted = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
7849                let hook_deleted = std::sync::Arc::clone(&deleted);
7850                let race = install_walk_hook(&root, move |point| {
7851                    if let WalkHookPoint::ControlLookup(looked_up) = point {
7852                        if looked_up == control && fs::remove_file(looked_up).is_ok() {
7853                            hook_deleted.store(true, std::sync::atomic::Ordering::SeqCst);
7854                        }
7855                    }
7856                    None
7857                });
7858                if revalidating {
7859                    index = revalidated_copy(&index, &config, &label);
7860                } else {
7861                    let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
7862                    assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
7863                }
7864                drop(race);
7865                assert!(
7866                    deleted.load(std::sync::atomic::Ordering::SeqCst),
7867                    "{label}: the lookup found the control before it was deleted"
7868                );
7869                assert_eq!(index.path_state(control_path), PathState::Absent, "{label}");
7870                assert!(
7871                    !index.control_table().contains(control_path),
7872                    "{label}: no rules remain for a file that is gone"
7873                );
7874
7875                let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
7876                assert!(report.is_complete(), "{label}: {:?}", report.scan.errors);
7877                let (cold, _) = scan_into_index(&root, &config).expect("cold");
7878                assert_eq!(classification(&index), classification(&cold), "{label}");
7879            }
7880        }
7881    }
7882
7883    /// The sweep restates what a lookup found only after removing the stale `.gitignore`
7884    /// entry of a listing that showed a case variant, the one spelling that can still hold
7885    /// the rules that removal drops. Without a listed variant the exact file is gone, and
7886    /// so are its rules.
7887    #[test]
7888    fn a_listing_restates_looked_up_rules_only_after_showing_a_case_variant() {
7889        let found =
7890            Op::ControlUpsert { path: PathBuf::from("up/.gitignore"), source: b"*.tmp\n".to_vec() };
7891        let exact = OsStr::new(".gitignore");
7892
7893        let mut deleted = ListedControl::default();
7894        deleted.lookup(&Ok(Some(found.clone())));
7895        deleted.listed(OsStr::new("x.tmp"));
7896        assert_eq!(deleted.restatement_after_removing(exact), None, "the exact file is gone");
7897
7898        let mut recased = ListedControl::default();
7899        recased.lookup(&Ok(Some(found.clone())));
7900        recased.listed(OsStr::new(".GITIGNORE"));
7901        assert_eq!(recased.restatement_after_removing(exact), Some(found.clone()));
7902        assert_eq!(
7903            recased.restatement_after_removing(OsStr::new(".GitIgnore")),
7904            None,
7905            "only removing the canonical entry drops rules"
7906        );
7907        assert_eq!(recased.restatement_after_removing(OsStr::new("x.tmp")), None);
7908
7909        // An included population looks the control up through the listed variant itself.
7910        let mut read = ListedControl::default();
7911        read.listed(OsStr::new(".GitIgnore"));
7912        read.read(OsStr::new(".GitIgnore"), &Ok(Some(found.clone())));
7913        assert_eq!(read.restatement_after_removing(exact), Some(found));
7914
7915        let mut missed = ListedControl::default();
7916        missed.lookup(&Ok(None));
7917        missed.listed(OsStr::new(".GITIGNORE"));
7918        assert_eq!(missed.restatement_after_removing(exact), None, "nothing was found");
7919    }
7920
7921    /// Where a directory is case-sensitive it can list `.gitignore` beside `.GITIGNORE`,
7922    /// and only the exact name governs, on every route and through every change: the
7923    /// variant's lookup resolves to the exact file, and removing the variant leaves the
7924    /// rules while removing the exact name takes them (fdu-0w1b).
7925    #[test]
7926    fn a_case_sensitive_directory_listing_both_spellings_takes_only_the_exact_name() {
7927        let dir = tempfile::tempdir().expect("tempdir");
7928        let root = dir.path().canonicalize().expect("canonical root");
7929        write_file(&root.join(".gitignore"), b"*.log\n");
7930        write_file(&root.join(".GITIGNORE"), b"*.tmp\n");
7931        let listed = fs::read_dir(&root).expect("list").count();
7932        if listed != 2 {
7933            eprintln!(
7934                "skipped: the temporary directory is case-insensitive, so it cannot hold both \
7935                 spellings"
7936            );
7937            return;
7938        }
7939        write_file(&root.join("x.log"), b"exact rule");
7940        write_file(&root.join("y.tmp"), b"variant, not a rule here");
7941        let ignored =
7942            |index: &Index, path: &str| index.is_ignored(Path::new(path)).expect("observed");
7943
7944        for threads in [Some(1), Some(4)] {
7945            let config = ScanConfig { threads, batch_size: 1, ..ScanConfig::default() };
7946            let (detached, _) = scan_into_index(&root, &config).expect("detached");
7947            let (streamed, _) = scan_into_index_via_scanner(&root, &config).expect("stream");
7948            for index in [&detached, &streamed] {
7949                assert_eq!(ignored(index, "x.log"), Some(true), "{threads:?}");
7950                assert_eq!(ignored(index, "y.tmp"), Some(false), "{threads:?}");
7951                assert_eq!(classification(index), classification(&detached), "{threads:?}");
7952            }
7953        }
7954
7955        let config = ScanConfig::default();
7956        let (mut index, _) = scan_into_index(&root, &config).expect("cold");
7957        fs::remove_file(root.join(".GITIGNORE")).expect("remove the variant");
7958        reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
7959        assert_eq!(ignored(&index, "x.log"), Some(true), "the exact name still governs");
7960        write_file(&root.join(".GITIGNORE"), b"*.tmp\n");
7961        fs::remove_file(root.join(".gitignore")).expect("remove the exact name");
7962        reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
7963        assert_eq!(ignored(&index, "x.log"), Some(false), "no rules remain");
7964        assert_eq!(ignored(&index, "y.tmp"), Some(false), "the variant never governs here");
7965        let (cold, _) = scan_into_index(&root, &config).expect("cold");
7966        assert_eq!(classification(&index), classification(&cold));
7967    }
7968
7969    /// A listing whose control was already probed does not read its `.gitignore` again:
7970    /// the entry is prepared with no control and, here, no error from a file that cannot
7971    /// be read, where reading it would have produced one.
7972    #[test]
7973    #[cfg(unix)]
7974    fn a_probed_listing_prepares_its_control_entry_without_reading_it() {
7975        use std::os::unix::fs::PermissionsExt;
7976        if !crate::test_support::require_permission_bits() {
7977            return;
7978        }
7979        let dir = tempfile::tempdir().expect("tempdir");
7980        let control = dir.path().join(".gitignore");
7981        write_file(&control, b"*.log\n");
7982        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("deny");
7983        let config = ScanConfig::default();
7984        let prepare = |read_control| {
7985            prepare_walk_entry_reading(
7986                dir.path(),
7987                Path::new(""),
7988                0,
7989                OsStr::new(".gitignore"),
7990                EntryKind::File,
7991                Attrs::default(),
7992                0,
7993                &config,
7994                read_control,
7995            )
7996            .expect("admitted")
7997        };
7998        let read = prepare(true);
7999        let skipped = prepare(false);
8000        fs::set_permissions(&control, fs::Permissions::from_mode(0o600)).expect("restore");
8001
8002        assert!(read.control.is_none() && read.control_error.is_some(), "the read was made");
8003        assert!(skipped.control.is_none() && skipped.control_error.is_none(), "no read");
8004        assert!(skipped.retained, "the entry is still a row");
8005    }
8006
8007    /// An automatic walk too short to fill its calibration window must say so.
8008    ///
8009    /// The failure this guards is quiet: such a walk runs on its initial pool, which is
8010    /// indistinguishable in the artifacts from a walk that measured the filesystem and
8011    /// chose to hold — unless the undecided case is recorded separately. Reading the
8012    /// first as the second is how a policy with no evidence behind it comes to look
8013    /// like a policy with evidence behind it.
8014    #[test]
8015    fn a_short_automatic_walk_records_an_undecided_policy() {
8016        let _serial = crate::counters::test_serial();
8017        let available = std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get);
8018        if automatic_worker_pool(available).calibration.is_none() {
8019            // A host reporting one processor has no reserve to unlock, so there is no
8020            // policy here to leave undecided.
8021            return;
8022        }
8023
8024        crate::counters::enable(true);
8025        let dir = sample_tree();
8026        let config = ScanConfig { threads: None, ..ScanConfig::default() };
8027        let before = crate::counters::snapshot();
8028        scan(dir.path(), &config, &mut |_| {}).expect("scan");
8029        crate::counters::flush_thread();
8030        let after = crate::counters::snapshot();
8031        crate::counters::enable(false);
8032
8033        // A strict increase, so a counter inflated by a test running beside this one
8034        // cannot turn the assertion into a false pass.
8035        assert!(
8036            after.adaptive_policy_undecided > before.adaptive_policy_undecided,
8037            "a three-file tree cannot fill a {ADAPTIVE_SCAN_CALIBRATION_ENTRIES}-entry window"
8038        );
8039    }
8040
8041    #[test]
8042    fn diagnostics_make_a_fixed_pool_and_backend_choice_explicit() {
8043        let dir = sample_tree();
8044        let config = ScanConfig { threads: Some(1), ..ScanConfig::default() };
8045
8046        let (report, diagnostics) =
8047            scan_with_diagnostics(dir.path(), &config, &mut |_| {}).expect("diagnostic scan");
8048
8049        assert_eq!(diagnostics.schema, SCAN_DIAGNOSTICS_SCHEMA);
8050        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Fixed);
8051        assert_eq!(diagnostics.worker_policy.initial_workers, 1);
8052        assert_eq!(diagnostics.worker_policy.maximum_workers, 1);
8053        assert_eq!(diagnostics.worker_policy.peak_active_workers, 1);
8054        assert!(diagnostics.worker_policy.windows.is_empty());
8055        assert!(!diagnostics.worker_policy.events_truncated);
8056        assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
8057        assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
8058        assert_eq!(diagnostics.backend.portable_directory_reads, report.dirs_read);
8059
8060        #[cfg(target_os = "macos")]
8061        {
8062            assert_eq!(diagnostics.backend.macos_bulk_attempts, Some(0));
8063            assert_eq!(diagnostics.backend.macos_bulk_successes, Some(0));
8064            assert_eq!(diagnostics.backend.macos_bulk_fallbacks, Some(0));
8065            assert!(diagnostics.backend.unavailable_reason.is_none());
8066        }
8067        #[cfg(not(target_os = "macos"))]
8068        {
8069            assert_eq!(diagnostics.backend.macos_bulk_attempts, None);
8070            assert_eq!(diagnostics.backend.macos_bulk_successes, None);
8071            assert_eq!(diagnostics.backend.macos_bulk_fallbacks, None);
8072            assert_eq!(
8073                diagnostics.backend.unavailable_reason,
8074                Some("macOS bulk directory enumeration is unavailable on this platform")
8075            );
8076        }
8077    }
8078
8079    #[test]
8080    fn diagnostics_fail_closed_when_an_automatic_window_is_incomplete() {
8081        let available = std::thread::available_parallelism().map_or(1, std::num::NonZeroUsize::get);
8082        let pool = automatic_worker_pool(available);
8083        if pool.calibration.is_none() {
8084            return;
8085        }
8086        let dir = sample_tree();
8087        let config = ScanConfig { threads: None, ..ScanConfig::default() };
8088
8089        let (report, diagnostics) =
8090            scan_with_diagnostics(dir.path(), &config, &mut |_| {}).expect("diagnostic scan");
8091
8092        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Undecided);
8093        assert_eq!(diagnostics.worker_policy.available_parallelism, available);
8094        assert_eq!(diagnostics.worker_policy.initial_workers, pool.initial);
8095        assert_eq!(diagnostics.worker_policy.maximum_workers, pool.maximum);
8096        assert_eq!(
8097            diagnostics.worker_policy.calibration_window_entries,
8098            Some(ADAPTIVE_SCAN_CALIBRATION_ENTRIES)
8099        );
8100        assert_eq!(
8101            diagnostics.worker_policy.slow_threshold_ns_per_entry,
8102            Some(ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY)
8103        );
8104        assert_eq!(diagnostics.worker_policy.windows.len(), 1);
8105        let window = &diagnostics.worker_policy.windows[0];
8106        assert_eq!(window.sequence, 0);
8107        assert_eq!(window.start_entry_ordinal, 0);
8108        assert_eq!(window.end_entry_ordinal, report.entries);
8109        assert_eq!(window.observed_entries, report.entries);
8110        assert_eq!(window.decision, WorkerPolicyDecision::Undecided);
8111        assert!(window.end_entry_ordinal < ADAPTIVE_SCAN_CALIBRATION_ENTRIES);
8112        assert!(window.active_workers <= diagnostics.worker_policy.peak_active_workers);
8113        assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
8114        assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
8115        assert!(diagnostics.worker_policy.handoff_backlog_high_water >= 1);
8116    }
8117
8118    #[test]
8119    fn diagnostic_trace_is_bounded_and_marks_truncation() {
8120        let recorder = ScanDiagnosticsRecorder::new(
8121            WorkerPool::fixed(2),
8122            2,
8123            WorkerPolicyExperiment::ShippedOneShot,
8124        );
8125        for sequence in 0..=MAX_POLICY_TRACE_EVENTS {
8126            recorder.record_policy_window(PolicyWindowSnapshot {
8127                sequence: sequence as u64,
8128                start_entry_ordinal: sequence as u64,
8129                end_entry_ordinal: sequence as u64 + 1,
8130                observed_entries: 1,
8131                observed_chunks: 1,
8132                observed_work_ns: 10,
8133                ready_directories: 1,
8134                in_flight_directories: 1,
8135                active_workers: 1,
8136                handoff_backlog: 0,
8137                requested_workers: None,
8138                decision: WorkerPolicyDecision::Hold,
8139            });
8140        }
8141
8142        let diagnostics = recorder.finish();
8143        assert_eq!(diagnostics.worker_policy.windows.len(), MAX_POLICY_TRACE_EVENTS);
8144        assert!(diagnostics.worker_policy.events_truncated);
8145    }
8146
8147    #[test]
8148    fn diagnostic_trace_preserves_queue_order_when_recorders_arrive_out_of_order() {
8149        let pool =
8150            WorkerPool { initial: 2, maximum: 4, calibration: Some(WorkerCalibration::new(1, 1)) };
8151        let recorder =
8152            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::RepeatedWindows);
8153        let snapshot = |sequence, decision| PolicyWindowSnapshot {
8154            sequence,
8155            start_entry_ordinal: sequence,
8156            end_entry_ordinal: sequence + 1,
8157            observed_entries: 1,
8158            observed_chunks: 1,
8159            observed_work_ns: 1,
8160            ready_directories: 0,
8161            in_flight_directories: 0,
8162            active_workers: 1,
8163            handoff_backlog: 0,
8164            requested_workers: None,
8165            decision,
8166        };
8167
8168        recorder.record_policy_window(snapshot(1, WorkerPolicyDecision::HoldNoUsefulWork));
8169        recorder.record_policy_window(snapshot(0, WorkerPolicyDecision::Hold));
8170
8171        let diagnostics = recorder.finish();
8172        assert_eq!(
8173            diagnostics
8174                .worker_policy
8175                .windows
8176                .iter()
8177                .map(|window| window.sequence)
8178                .collect::<Vec<_>>(),
8179            vec![0, 1]
8180        );
8181        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::HeldNoUsefulWork);
8182    }
8183
8184    #[test]
8185    fn diagnostic_policy_aggregates_cross_check_runtime_counters() {
8186        let _serial = crate::counters::test_serial();
8187        let pool = WorkerPool {
8188            initial: 2,
8189            maximum: 4,
8190            calibration: Some(WorkerCalibration::new(17, 100)),
8191        };
8192        let recorder =
8193            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::ShippedOneShot);
8194
8195        crate::counters::enable(true);
8196        let before = crate::counters::snapshot();
8197        record_adaptive_calibration_chunk(Some(&recorder), 17, 2_100);
8198        record_adaptive_worker_expansion(Some(&recorder));
8199        crate::counters::flush_thread();
8200        let after = crate::counters::snapshot();
8201        crate::counters::enable(false);
8202
8203        recorder.record_policy_window(PolicyWindowSnapshot {
8204            sequence: 0,
8205            start_entry_ordinal: 0,
8206            end_entry_ordinal: 17,
8207            observed_entries: 17,
8208            observed_chunks: 1,
8209            observed_work_ns: 2_100,
8210            ready_directories: 2,
8211            in_flight_directories: 2,
8212            active_workers: 2,
8213            handoff_backlog: 0,
8214            requested_workers: Some(4),
8215            decision: WorkerPolicyDecision::ScaleUp,
8216        });
8217        let diagnostics = recorder.finish();
8218        let policy = diagnostics.worker_policy;
8219        assert_eq!(policy.calibration_chunks, 1);
8220        assert_eq!(policy.calibration_entries, 17);
8221        assert_eq!(policy.calibration_work_ns, 2_100);
8222        assert_eq!(policy.worker_expansions, 1);
8223        assert_eq!(policy.windows[0].observed_chunks, policy.calibration_chunks);
8224        assert_eq!(policy.windows[0].observed_entries, policy.calibration_entries);
8225        assert_eq!(policy.windows[0].observed_work_ns, policy.calibration_work_ns);
8226
8227        // Other tests can record while this process-global interval is enabled, so the
8228        // counter delta may be larger but must never be smaller than this run-scoped
8229        // trace. The shared helpers above make the two observations one event.
8230        assert!(
8231            after.adaptive_calibration_chunks - before.adaptive_calibration_chunks
8232                >= policy.calibration_chunks
8233        );
8234        assert!(
8235            after.adaptive_calibration_entries - before.adaptive_calibration_entries
8236                >= policy.calibration_entries
8237        );
8238        assert!(
8239            after.adaptive_calibration_work_us - before.adaptive_calibration_work_us
8240                >= policy.calibration_work_ns / 1_000
8241        );
8242        assert!(after.adaptive_scale_ups - before.adaptive_scale_ups >= policy.worker_expansions);
8243    }
8244
8245    #[test]
8246    fn diagnostic_index_scan_preserves_the_regular_result() {
8247        let dir = branching_tree();
8248        let config = ScanConfig { threads: Some(4), ..ScanConfig::default() };
8249        let (plain, plain_report) = scan_into_index(dir.path(), &config).expect("plain scan");
8250        let (diagnostic, diagnostic_report, diagnostics) =
8251            scan_into_index_with_diagnostics(dir.path(), &config).expect("diagnostic scan");
8252
8253        assert_eq!(index_fingerprint(&plain), index_fingerprint(&diagnostic));
8254        assert_eq!(plain_report.entries, diagnostic_report.entries);
8255        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Fixed);
8256    }
8257
8258    #[test]
8259    fn detached_bootstrap_matches_the_streaming_reducer_for_each_worker_count() {
8260        let dir = branching_tree();
8261        for threads in 1..=4 {
8262            let config = ScanConfig {
8263                read_controls: false,
8264                threads: Some(threads),
8265                ..ScanConfig::default()
8266            };
8267            let _ = detached_and_streaming_indexes(dir.path(), &config);
8268        }
8269    }
8270
8271    #[test]
8272    fn detached_control_bootstrap_matches_the_streaming_reducer_for_each_worker_count() {
8273        let dir = controlled_branching_tree();
8274        for threads in 1..=4 {
8275            let config =
8276                ScanConfig { read_controls: true, threads: Some(threads), ..ScanConfig::default() };
8277            let _ = detached_and_streaming_indexes(dir.path(), &config);
8278        }
8279    }
8280
8281    #[test]
8282    fn detached_bootstrap_preserves_the_exact_first_mutation() {
8283        let dir = branching_tree();
8284        let config = ScanConfig { read_controls: false, threads: Some(4), ..ScanConfig::default() };
8285        let (mut detached, mut streaming) = detached_and_streaming_indexes(dir.path(), &config);
8286        let created = dir.path().join("t3/m2/after-bootstrap.rs");
8287        write_file(&created, b"new fact");
8288        let attrs =
8289            attrs_from(&created, &fs::symlink_metadata(&created).expect("new file metadata"))
8290                .expect("observe new file");
8291        let observation = Observation::new(vec![Op::Upsert {
8292            path: PathBuf::from("t3/m2/after-bootstrap.rs"),
8293            kind: EntryKind::File,
8294            attrs,
8295        }]);
8296
8297        let detached_outcome = detached.apply(&observation).expect("detached mutation");
8298        let streaming_outcome = streaming.apply(&observation).expect("streaming mutation");
8299        assert_eq!(detached_outcome, streaming_outcome);
8300        assert_indexes_equal(&detached, &streaming);
8301    }
8302
8303    #[test]
8304    fn detached_control_bootstrap_preserves_the_exact_first_mutation() {
8305        let dir = controlled_branching_tree();
8306        let config = ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
8307        let (mut detached, mut streaming) = detached_and_streaming_indexes(dir.path(), &config);
8308        let observation = Observation::new(vec![Op::ControlUpsert {
8309            path: PathBuf::from(".gitignore"),
8310            source: b"leaf-2.dat\n".to_vec(),
8311        }]);
8312
8313        let detached_outcome = detached.apply(&observation).expect("detached control mutation");
8314        let streaming_outcome = streaming.apply(&observation).expect("streaming control mutation");
8315        assert_eq!(detached_outcome, streaming_outcome);
8316        assert_indexes_equal(&detached, &streaming);
8317    }
8318
8319    fn observed_coverage(index: &Index) -> crate::control::ControlObservation {
8320        match index.control_coverage() {
8321            crate::control::ControlCoverage::Observed(observation) => observation,
8322            crate::control::ControlCoverage::NotObserved => panic!("controls were observed"),
8323        }
8324    }
8325
8326    /// Both bootstrap lanes refuse a line over the limit and a file over the budget, and
8327    /// neither ends the scan or makes it partial. Both refusals are order-independent, so
8328    /// the lanes agree on exactly which files they refused.
8329    #[test]
8330    fn both_bootstrap_lanes_refuse_over_bound_controls_without_ending_the_scan() {
8331        let dir = tempfile::tempdir().expect("tempdir");
8332        let mut long_line = b"*.log\n".to_vec();
8333        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
8334        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
8335        write_file(&dir.path().join("guarded/kept.log"), b"guarded");
8336        write_file(
8337            &dir.path().join("huge/.gitignore"),
8338            &b"x\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
8339        );
8340        write_file(&dir.path().join("applied/.gitignore"), b"*.log\n");
8341        write_file(&dir.path().join("applied/dropped.log"), b"applied");
8342        let config = ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
8343
8344        let (detached, _) = detached_and_streaming_indexes(dir.path(), &config);
8345        let (_, report) = scan_into_index(dir.path(), &config).expect("scan");
8346
8347        assert!(report.is_complete(), "{:?}", report.errors);
8348        let coverage = observed_coverage(&detached);
8349        assert_eq!((coverage.applied, coverage.refused), (1, 2));
8350        assert_eq!(
8351            coverage.refusals,
8352            vec![
8353                crate::control::RefusedControl {
8354                    path: PathBuf::from("guarded/.gitignore"),
8355                    reason: crate::control::ControlRefusalReason::LineLimit,
8356                },
8357                crate::control::RefusedControl {
8358                    path: PathBuf::from("huge/.gitignore"),
8359                    reason: crate::control::ControlRefusalReason::Budget,
8360                },
8361            ]
8362        );
8363        assert_eq!(detached.is_ignored(Path::new("guarded/kept.log")).expect("observed"), None);
8364        assert_eq!(
8365            detached.is_ignored(Path::new("applied/dropped.log")).expect("observed"),
8366            Some(true)
8367        );
8368    }
8369
8370    /// The control counters attribute what a scan's control state cost: files read, sources
8371    /// refused, and sources that shared a retained content instead of parsing their own.
8372    ///
8373    /// Off by default and compiled in, like every counter, so the numbers a speed check
8374    /// reads come from the shipped path rather than an instrumented build.
8375    #[test]
8376    fn control_counters_attribute_reads_refusals_and_sharing() {
8377        let _serial = crate::counters::test_serial();
8378        let dir = tempfile::tempdir().expect("tempdir");
8379        let shared = b"*.log\n".to_vec();
8380        write_file(&dir.path().join(".gitignore"), &shared);
8381        write_file(&dir.path().join("twin/.gitignore"), &shared);
8382        let mut long_line = b"*.tmp\n".to_vec();
8383        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
8384        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
8385        let config = ScanConfig { read_controls: true, threads: Some(1), ..ScanConfig::default() };
8386
8387        crate::counters::enable(true);
8388        // Deltas around the scan rather than absolute totals, for the reason
8389        // `a_walk_moves_every_counter_it_should` gives: the counters are process-global,
8390        // `test_serial` only serializes the tests that take it, and every report in this
8391        // binary now reads `.gitignore` by default, so a test running beside this one can
8392        // add control reads of its own.
8393        let before = crate::counters::snapshot();
8394        let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
8395        crate::counters::flush_thread();
8396        let after = crate::counters::snapshot();
8397        crate::counters::enable(false);
8398
8399        assert!(report.is_complete(), "{:?}", report.errors);
8400        assert_eq!(observed_coverage(&index).refused, 1);
8401        // `>=` in the one direction concurrency can move them. A count that is too low
8402        // means a path ran uninstrumented, which is the defect worth catching; too high
8403        // is another test's tree, which is not.
8404        for (label, observed, expected) in [
8405            ("one read per .gitignore", after.control_reads - before.control_reads, 3),
8406            ("the line over the limit", after.control_refused - before.control_refused, 1),
8407            (
8408                "the twin shares one parsed content",
8409                after.control_sources_shared - before.control_sources_shared,
8410                1,
8411            ),
8412        ] {
8413            assert!(observed >= expected, "{label}: counted {observed}, expected {expected}");
8414        }
8415    }
8416
8417    /// Both limits are part of the scope, and each lifts only its own refusals: no budget
8418    /// still refuses a long line, and no line limit still refuses a file past the budget.
8419    #[test]
8420    fn each_control_limit_is_scope_and_lifts_only_its_own_refusals() {
8421        use crate::control::{ControlLimits, ControlRefusalReason, RefusedControl};
8422
8423        let with = |limits| ScanConfig { control_limits: limits, ..ScanConfig::default() };
8424        let defaults = ControlLimits::default();
8425        let default = ScanConfig::default();
8426        let no_budget = with(ControlLimits { budget: None, ..defaults });
8427        let no_line_limit = with(ControlLimits { line_limit: None, ..defaults });
8428        let configs = [
8429            default.clone(),
8430            with(ControlLimits { budget: Some(16 * 1024 * 1024), ..defaults }),
8431            no_budget.clone(),
8432            with(ControlLimits { line_limit: Some(64 * 1024), ..defaults }),
8433            no_line_limit.clone(),
8434            with(ControlLimits { budget: None, line_limit: None }),
8435            // The same values in each other's places are a different scope.
8436            with(ControlLimits { budget: defaults.line_limit, line_limit: defaults.budget }),
8437        ];
8438        let scopes: Vec<ScanScope> = configs.iter().map(ScanConfig::scope).collect();
8439        for (index, scope) in scopes.iter().enumerate() {
8440            assert!(scope.observes_controls());
8441            assert!(scopes[index + 1..].iter().all(|other| other != scope), "{scopes:?}");
8442        }
8443        for config in &configs {
8444            let blind = ScanConfig { read_controls: false, ..config.clone() };
8445            assert_eq!(blind.scope().ignore_rules_fingerprint, 0, "unobserved has one scope");
8446        }
8447
8448        let dir = tempfile::tempdir().expect("tempdir");
8449        let mut long_line = b"*.log\n".to_vec();
8450        long_line.extend(std::iter::repeat_n(b'x', crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1));
8451        write_file(&dir.path().join("guarded/.gitignore"), &long_line);
8452        write_file(&dir.path().join("guarded/dropped.log"), b"log");
8453        write_file(
8454            &dir.path().join("huge/.gitignore"),
8455            &b"x\n".repeat(crate::control::DEFAULT_CONTROL_BUDGET / 2),
8456        );
8457        let refused = |path: &str, reason| RefusedControl { path: PathBuf::from(path), reason };
8458        let (bounded, _) = scan_into_index(dir.path(), &default).expect("default scan");
8459        assert_eq!(observed_coverage(&bounded).refused, 2);
8460
8461        let (budget_lifted, _) = detached_and_streaming_indexes(dir.path(), &no_budget);
8462        let coverage = observed_coverage(&budget_lifted);
8463        assert_eq!(coverage.limits, no_budget.control_limits);
8464        assert_eq!(
8465            coverage.refusals,
8466            [refused("guarded/.gitignore", ControlRefusalReason::LineLimit)]
8467        );
8468        assert_eq!(
8469            budget_lifted.is_ignored(Path::new("guarded/dropped.log")).expect("observed"),
8470            None
8471        );
8472
8473        let (line_limit_lifted, _) = detached_and_streaming_indexes(dir.path(), &no_line_limit);
8474        let coverage = observed_coverage(&line_limit_lifted);
8475        assert_eq!(coverage.refusals, [refused("huge/.gitignore", ControlRefusalReason::Budget)]);
8476        assert_eq!(
8477            line_limit_lifted.is_ignored(Path::new("guarded/dropped.log")).expect("observed"),
8478            Some(true)
8479        );
8480        assert_eq!(line_limit_lifted.scope(), no_line_limit.scope());
8481    }
8482
8483    /// The synthetic tree that ended a cold scan (fdu-1onj): 1,105 directories, each with
8484    /// a distinct 510-byte `.gitignore` of short rules, plus one line over the limit. The
8485    /// scan completes with every size exact and names what it refused, on both lanes.
8486    #[test]
8487    fn a_tree_past_both_control_bounds_completes_with_exact_sizes() {
8488        const DIRECTORIES: usize = 1_105;
8489        let dir = tempfile::tempdir().expect("tempdir");
8490        for directory in 0..DIRECTORIES {
8491            let mut source = Vec::new();
8492            for line in 0..63 {
8493                source.extend(format!("p{directory:04}{line:02}\n").bytes());
8494            }
8495            source.extend(format!("q{directory:04}\n").bytes());
8496            assert_eq!(source.len(), 510);
8497            let root = dir.path().join(format!("d{directory:04}"));
8498            write_file(&root.join(".gitignore"), &source);
8499            write_file(&root.join("file.txt"), b"contents");
8500        }
8501        write_file(
8502            &dir.path().join("a-guard/.gitignore"),
8503            &vec![b'x'; crate::control::DEFAULT_CONTROL_LINE_LIMIT + 1],
8504        );
8505        let observing =
8506            ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() };
8507        let blind = ScanConfig { read_controls: false, ..observing.clone() };
8508
8509        let (unobserved, _) = scan_into_index(dir.path(), &blind).expect("controls-off scan");
8510        let canonical = dir.path().canonicalize().expect("canonical root");
8511        let lanes = [
8512            scan_into_index(dir.path(), &observing).expect("detached scan"),
8513            scan_into_index_via_scanner(&canonical, &observing).expect("streaming scan"),
8514        ];
8515        for (index, report) in &lanes {
8516            assert!(report.is_complete(), "{:?}", report.errors);
8517            assert_eq!(index.total(), unobserved.total(), "sizes do not depend on controls");
8518            let coverage = observed_coverage(index);
8519            assert!(coverage.refused > 1, "the budget refused sources: {coverage:?}");
8520            assert_eq!(
8521                coverage.applied + coverage.refused,
8522                u64::try_from(DIRECTORIES + 1).expect("small")
8523            );
8524            assert_eq!(coverage.refusals.len(), crate::MAX_RETAINED_ISSUES);
8525            assert!(!coverage.lists_every_refusal());
8526            assert_eq!(
8527                coverage.refusals[0],
8528                crate::control::RefusedControl {
8529                    path: PathBuf::from("a-guard/.gitignore"),
8530                    reason: crate::control::ControlRefusalReason::LineLimit,
8531                }
8532            );
8533            assert!(
8534                index.control_table().retained_cost() <= crate::control::DEFAULT_CONTROL_BUDGET
8535            );
8536        }
8537    }
8538
8539    #[test]
8540    fn fingerprint_metadata_observes_mutation_after_directory_enumeration() {
8541        let dir = tempfile::tempdir().expect("tempdir");
8542        let path = dir.path().join("changing.bin");
8543        write_file(&path, b"before");
8544        let entry = fs::read_dir(dir.path())
8545            .expect("read directory")
8546            .next()
8547            .expect("one entry")
8548            .expect("read entry");
8549
8550        write_file(&path, b"after mutation");
8551
8552        let metadata = metadata_for_fingerprint(&entry).expect("fresh metadata");
8553        assert_eq!(metadata.len(), b"after mutation".len() as u64);
8554    }
8555
8556    /// A tree wide and deep enough that workers genuinely interleave.
8557    ///
8558    /// A three-file fixture would pass every one of these tests with a broken queue,
8559    /// because one worker would finish before another started.
8560    fn branching_tree() -> tempfile::TempDir {
8561        let dir = tempfile::tempdir().expect("tempdir");
8562        for top in 0..12 {
8563            for middle in 0..6 {
8564                for leaf in 0..7 {
8565                    write_file(
8566                        &dir.path().join(format!("t{top}/m{middle}/leaf-{leaf}.dat")),
8567                        &vec![b'x'; leaf * 13],
8568                    );
8569                }
8570            }
8571            // A deep chain alongside the wide fan-out, so depth and width are both
8572            // exercised by the same walk.
8573            write_file(&dir.path().join(format!("t{top}/a/b/c/d/e/deep.txt")), b"deep");
8574        }
8575        dir
8576    }
8577
8578    fn controlled_branching_tree() -> tempfile::TempDir {
8579        let dir = branching_tree();
8580        write_file(&dir.path().join(".gitignore"), b"leaf-1.dat\nt7/\n");
8581        write_file(&dir.path().join("t3/.gitignore"), b"!m2/leaf-1.dat\n*.tmp\n");
8582        write_file(&dir.path().join("t3/m2/generated.tmp"), b"ignored by nested control");
8583        write_file(&dir.path().join("t7/.gitignore"), b"!m0/leaf-1.dat\n");
8584        fs::create_dir_all(dir.path().join("t5/.gitignore")).expect("non-file control directory");
8585        write_file(&dir.path().join("t5/.gitignore/ordinary.txt"), b"ordinary child");
8586        dir
8587    }
8588
8589    fn index_fingerprint(index: &Index) -> Vec<(PathBuf, EntryKind, Attrs)> {
8590        let mut entries: Vec<(PathBuf, EntryKind, Attrs)> = Vec::new();
8591        let mut queue = vec![PathBuf::new()];
8592        while let Some(path) = queue.pop() {
8593            let Some(children) = index.children(&path) else {
8594                continue;
8595            };
8596            let names: Vec<PathBuf> = children.map(|(name, _id)| path.join(name)).collect();
8597            for child_path in names {
8598                let kind = index.kind(&child_path).expect("child has a kind");
8599                let attrs = *index.attrs(&child_path).expect("child has attrs");
8600                entries.push((child_path.clone(), kind, attrs));
8601                if kind.is_dir() {
8602                    queue.push(child_path);
8603                }
8604            }
8605        }
8606        entries.sort_by(|left, right| left.0.cmp(&right.0));
8607        entries
8608    }
8609
8610    fn detached_and_streaming_indexes(root: &Path, config: &ScanConfig) -> (Index, Index) {
8611        let canonical = root.canonicalize().expect("canonical test root");
8612        let (streaming, streaming_report) =
8613            scan_into_index_via_scanner(&canonical, config).expect("streaming oracle");
8614        let (detached, detached_report) = scan_into_index(root, config).expect("detached scan");
8615
8616        assert_eq!(detached_report.dirs_read, streaming_report.dirs_read);
8617        assert_eq!(detached_report.entries, streaming_report.entries);
8618        assert_eq!(detached_report.files_walked, streaming_report.files_walked);
8619        assert_eq!(detached_report.bytes_walked, streaming_report.bytes_walked);
8620        assert_eq!(
8621            detached_report.errors.iter().map(ToString::to_string).collect::<Vec<_>>(),
8622            streaming_report.errors.iter().map(ToString::to_string).collect::<Vec<_>>()
8623        );
8624        assert_indexes_equal(&detached, &streaming);
8625        (detached, streaming)
8626    }
8627
8628    fn assert_indexes_equal(left: &Index, right: &Index) {
8629        assert_eq!(index_fingerprint(left), index_fingerprint(right));
8630        assert_eq!(left.total(), right.total());
8631        assert_eq!(left.partition_total().ok(), right.partition_total().ok());
8632        assert_eq!(left.scope(), right.scope());
8633        assert_eq!(left.freshness(), right.freshness());
8634        assert_eq!(left.state(), right.state());
8635        assert_eq!(left.clock(), right.clock());
8636        assert_eq!(left.len(), right.len());
8637        assert_eq!(left.issues(), right.issues());
8638        assert_eq!(left.observes_controls(), right.observes_controls());
8639        assert_eq!(left.control_coverage(), right.control_coverage());
8640        assert_eq!(
8641            left.control_table()
8642                .sources()
8643                .map(|(path, source)| (path, source.to_vec()))
8644                .collect::<Vec<_>>(),
8645            right
8646                .control_table()
8647                .sources()
8648                .map(|(path, source)| (path, source.to_vec()))
8649                .collect::<Vec<_>>()
8650        );
8651        for (path, _, _) in index_fingerprint(left) {
8652            assert_eq!(left.is_ignored(&path).ok(), right.is_ignored(&path).ok(), "{path:?}");
8653        }
8654    }
8655
8656    /// A small tree whose mutation crosses every structural reconciliation boundary.
8657    fn reconciliation_transition_tree() -> tempfile::TempDir {
8658        let dir = tempfile::tempdir().expect("tempdir");
8659        write_file(&dir.path().join("changed.txt"), b"before");
8660        write_file(&dir.path().join("removed.txt"), b"remove me");
8661        write_file(&dir.path().join("directory-to-file/old.rs"), b"old child");
8662        write_file(&dir.path().join("file-to-directory"), b"old file");
8663        write_file(&dir.path().join("removed-tree/nested/gone.md"), b"gone");
8664        write_file(&dir.path().join("stable/deep/kept.rs"), b"kept");
8665        dir
8666    }
8667
8668    fn mutate_reconciliation_transition_tree(root: &Path) {
8669        write_file(&root.join("changed.txt"), b"after, with a distinct size");
8670        fs::remove_file(root.join("removed.txt")).expect("remove root file");
8671
8672        fs::remove_dir_all(root.join("directory-to-file")).expect("remove old directory");
8673        write_file(&root.join("directory-to-file"), b"replacement file");
8674
8675        fs::remove_file(root.join("file-to-directory")).expect("remove old file");
8676        write_file(&root.join("file-to-directory/new.txt"), b"replacement child");
8677
8678        fs::remove_dir_all(root.join("removed-tree")).expect("remove nested tree");
8679        write_file(&root.join("added-tree/nested/new.md"), b"new nested file");
8680    }
8681
8682    fn effective_ops(commits: &[Commit]) -> Vec<Op> {
8683        let mut operations: Vec<_> = commits
8684            .iter()
8685            .flat_map(|commit| commit.changes.iter())
8686            .filter_map(|change| match change {
8687                crate::EffectiveChange::Inserted { path, kind, attrs } => {
8688                    Some(Op::Upsert { path: path.clone(), kind: *kind, attrs: *attrs })
8689                }
8690                crate::EffectiveChange::Updated { path, kind, current, .. } => {
8691                    Some(Op::Upsert { path: path.clone(), kind: *kind, attrs: *current })
8692                }
8693                crate::EffectiveChange::Removed { path, .. } => {
8694                    Some(Op::Remove { path: path.clone() })
8695                }
8696                crate::EffectiveChange::Invalidated { path, reason } => {
8697                    Some(Op::InvalidateSubtree { path: path.clone(), reason: *reason })
8698                }
8699                crate::EffectiveChange::ControlUpdated { .. }
8700                | crate::EffectiveChange::ControlRefusalUpdated { .. }
8701                | crate::EffectiveChange::Reclassified { .. } => None,
8702            })
8703            .collect();
8704        operations.sort_by(|left, right| left.path().cmp(right.path()));
8705        operations
8706    }
8707
8708    fn commit_touches(commit: &Commit, path: &Path) -> bool {
8709        commit.changes.iter().any(|change| change.path() == path)
8710    }
8711
8712    #[test]
8713    fn parallel_and_serial_walks_produce_the_same_index() {
8714        let dir = branching_tree();
8715        let serial_config = ScanConfig { threads: Some(1), ..ScanConfig::default() };
8716        let (serial, serial_report) =
8717            scan_into_index(dir.path(), &serial_config).expect("serial scan");
8718        assert!(serial_report.is_complete());
8719
8720        for threads in [2_usize, 3, 8] {
8721            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
8722            let (parallel, report) = scan_into_index(dir.path(), &config).expect("parallel scan");
8723            assert!(report.is_complete(), "{threads} threads reported errors");
8724            assert_eq!(report.entries, serial_report.entries, "{threads} threads");
8725            assert_eq!(report.dirs_read, serial_report.dirs_read, "{threads} threads");
8726            assert_eq!(report.files_walked, serial_report.files_walked, "{threads} threads");
8727            assert_eq!(report.bytes_walked, serial_report.bytes_walked, "{threads} threads");
8728            // Public roll-ups carry extension names even though the internal merge path
8729            // uses ids whose assignment order differs between serial and parallel walks.
8730            let (serial_total, parallel_total) = (serial.total(), parallel.total());
8731            assert_eq!(
8732                (
8733                    parallel_total.files,
8734                    parallel_total.dirs,
8735                    parallel_total.bytes,
8736                    parallel_total.allocated,
8737                    parallel_total.newest_mtime_ns,
8738                ),
8739                (
8740                    serial_total.files,
8741                    serial_total.dirs,
8742                    serial_total.bytes,
8743                    serial_total.allocated,
8744                    serial_total.newest_mtime_ns,
8745                ),
8746                "{threads} threads roll-up"
8747            );
8748            assert_eq!(
8749                parallel_total.by_ext, serial_total.by_ext,
8750                "{threads} threads per-extension roll-up"
8751            );
8752            assert_eq!(
8753                index_fingerprint(&parallel),
8754                index_fingerprint(&serial),
8755                "{threads} threads produced a different index"
8756            );
8757        }
8758    }
8759
8760    #[test]
8761    fn parallel_walk_emits_every_entry_exactly_once() {
8762        let dir = branching_tree();
8763        let config = ScanConfig { threads: Some(4), batch_size: 16, ..ScanConfig::default() };
8764        let mut seen: BTreeMap<PathBuf, usize> = BTreeMap::new();
8765        let report = scan(dir.path(), &config, &mut |observation| {
8766            for op in &observation.ops {
8767                if let Op::Upsert { path, .. } = &op.op {
8768                    *seen.entry(path.clone()).or_default() += 1;
8769                }
8770            }
8771        })
8772        .expect("parallel scan");
8773
8774        assert!(report.is_complete());
8775        assert_eq!(seen.len() as u64, report.entries, "entry count disagrees with the report");
8776        let duplicated: Vec<_> =
8777            seen.iter().filter(|(_path, count)| **count != 1).map(|(path, _)| path).collect();
8778        assert!(duplicated.is_empty(), "paths emitted more than once: {duplicated:?}");
8779    }
8780
8781    #[test]
8782    fn parallel_walk_honours_max_depth() {
8783        let dir = branching_tree();
8784        for threads in [1_usize, 4] {
8785            let config =
8786                ScanConfig { threads: Some(threads), max_depth: Some(2), ..ScanConfig::default() };
8787            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
8788            assert!(report.is_complete());
8789            for (path, _kind, _attrs) in index_fingerprint(&index) {
8790                assert!(
8791                    path.components().count() <= 2,
8792                    "{threads} threads kept {path:?} past the depth limit"
8793                );
8794            }
8795        }
8796    }
8797
8798    #[test]
8799    fn scan_order_never_changes_the_resulting_index() {
8800        let dir = branching_tree();
8801        let depth_first =
8802            ScanConfig { order: ScanOrder::DepthFirst, threads: Some(1), ..ScanConfig::default() };
8803        let (expected, expected_report) =
8804            scan_into_index(dir.path(), &depth_first).expect("depth-first scan");
8805
8806        for (order, threads) in
8807            [(ScanOrder::BreadthFirst, 1), (ScanOrder::BreadthFirst, 4), (ScanOrder::DepthFirst, 4)]
8808        {
8809            let config = ScanConfig { order, threads: Some(threads), ..ScanConfig::default() };
8810            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
8811            assert_eq!(report.entries, expected_report.entries, "{order:?}/{threads}");
8812            assert_eq!(report.dirs_read, expected_report.dirs_read, "{order:?}/{threads}");
8813            // Public roll-ups resolve internal ids, so their named maps are stable even
8814            // when traversal order changes id assignment.
8815            let (totals, expected_totals) = (index.total(), expected.total());
8816            assert_eq!(
8817                (totals.files, totals.dirs, totals.bytes, totals.allocated),
8818                (
8819                    expected_totals.files,
8820                    expected_totals.dirs,
8821                    expected_totals.bytes,
8822                    expected_totals.allocated
8823                ),
8824                "{order:?}/{threads} roll-up"
8825            );
8826            assert_eq!(
8827                totals.newest_mtime_ns, expected_totals.newest_mtime_ns,
8828                "{order:?}/{threads} newest mtime"
8829            );
8830            assert_eq!(
8831                totals.by_ext, expected_totals.by_ext,
8832                "{order:?}/{threads} extension tallies"
8833            );
8834            assert_eq!(
8835                index_fingerprint(&index),
8836                index_fingerprint(&expected),
8837                "{order:?}/{threads} produced a different index"
8838            );
8839        }
8840    }
8841
8842    #[test]
8843    fn a_single_worker_breadth_first_walk_is_strictly_level_ordered() {
8844        // The strict guarantee, which holds only with one worker. With several, the
8845        // queue is ordered but the claims are not: a fast worker can enqueue and claim
8846        // depth d+2 while a slow worker still holds depth d+1. See
8847        // `breadth_first_starts_every_top_level_subtree_early` for the property the
8848        // default configuration actually provides, which is the one consumers rely on.
8849        let dir = branching_tree();
8850        let config = ScanConfig {
8851            order: ScanOrder::BreadthFirst,
8852            threads: Some(1),
8853            batch_size: 1,
8854            ..ScanConfig::default()
8855        };
8856        let mut depths_in_order: Vec<usize> = Vec::new();
8857        scan(dir.path(), &config, &mut |observation| {
8858            for op in &observation.ops {
8859                if let Op::Upsert { path, kind, .. } = &op.op {
8860                    if kind.is_dir() {
8861                        depths_in_order.push(path.components().count());
8862                    }
8863                }
8864            }
8865        })
8866        .expect("scan");
8867
8868        assert!(depths_in_order.len() > 10, "fixture should have many directories");
8869        assert!(
8870            depths_in_order.windows(2).all(|pair| pair[0] <= pair[1]),
8871            "directory depths were not non-decreasing: {depths_in_order:?}"
8872        );
8873    }
8874
8875    /// How many of the fixture's twelve top-level subtrees have received any file by
8876    /// the time half the files have been emitted.
8877    ///
8878    /// This is the product metric — "is a mid-scan ranking meaningful?" — rather than
8879    /// first-touch, which cannot distinguish the orders at all: reading the root
8880    /// enumerates all twelve children at once either way. What a ranking needs is that
8881    /// the subtrees grow *together*.
8882    fn subtrees_started_at_halfway(order: ScanOrder, threads: usize, dir: &Path) -> usize {
8883        let config =
8884            ScanConfig { order, batch_size: 1, threads: Some(threads), ..ScanConfig::default() };
8885
8886        let mut files: Vec<PathBuf> = Vec::new();
8887        scan(dir, &config, &mut |observation| {
8888            for op in &observation.ops {
8889                if let Op::Upsert { path, kind, .. } = &op.op {
8890                    if !kind.is_dir() {
8891                        files.push(path.clone());
8892                    }
8893                }
8894            }
8895        })
8896        .expect("scan");
8897
8898        let halfway = files.len() / 2;
8899        let mut started: BTreeSet<PathBuf> = BTreeSet::new();
8900        for path in files.iter().take(halfway) {
8901            if let Some(top) = path.components().next() {
8902                started.insert(PathBuf::from(top.as_os_str()));
8903            }
8904        }
8905        started.len()
8906    }
8907
8908    #[test]
8909    fn a_parallel_walk_accounts_for_where_its_time_went() {
8910        // The attribution identity: every named cause is a disjoint slice of worker
8911        // wall time, so the parts can never exceed the whole, and the counters that
8912        // amortization depends on are actually incremented. This is the instrument
8913        // the scheduler experiments will read; if it drifts, they measure noise.
8914        let dir = branching_tree();
8915        let config = ScanConfig { threads: Some(4), batch_size: 64, ..ScanConfig::default() };
8916        let report = scan(dir.path(), &config, &mut |_| {}).expect("scan");
8917        let a = report.attribution;
8918
8919        assert!(a.claims > 0, "a parallel walk claims chunks: {a:?}");
8920        assert!(a.work_ns > 0, "reading directories takes time: {a:?}");
8921        assert!(a.wall_ns > 0);
8922        // claim() locks at least once per successful claim, and release() locks once
8923        // per claim cycle too.
8924        assert!(a.lock_ops >= a.claims * 2, "lock ops out of step with claims: {a:?}");
8925        assert!(
8926            a.accounted_ns() <= a.wall_ns,
8927            "attributed slices are disjoint intervals inside worker wall: {a:?}"
8928        );
8929    }
8930
8931    #[test]
8932    fn a_serial_walk_has_no_coordination_to_attribute() {
8933        // Serial semantics: wall is the loop, "send" is the inline sink (the consumer
8934        // actually running), work is the rest — and the coordination counters stay
8935        // zero because there is no queue lock and no channel.
8936        let dir = branching_tree();
8937        let config = ScanConfig { threads: Some(1), batch_size: 64, ..ScanConfig::default() };
8938        let mut observations = 0usize;
8939        let report = scan(dir.path(), &config, &mut |_| observations += 1).expect("scan");
8940        let a = report.attribution;
8941
8942        assert!(observations > 0, "the sink ran, so send_ns measured something real");
8943        assert!(a.work_ns > 0 && a.wall_ns >= a.work_ns);
8944        assert_eq!(
8945            (a.claims, a.lock_ops, a.lock_contended, a.starved_ns, a.lock_wait_ns),
8946            (0, 0, 0, 0, 0),
8947            "no queue, no lock, nothing to wait on: {a:?}"
8948        );
8949    }
8950
8951    /// Twelve top-level subtrees, each a branching tree several levels deep.
8952    ///
8953    /// Branching matters: an earlier fixture gave every level exactly one child, which
8954    /// pinned the frontier at twelve directories and made both orders behave
8955    /// identically — a LIFO cannot dive when there is nothing to dive into. With two
8956    /// children per level, depth-first pushes siblings and immediately descends into
8957    /// the last one, which is the behaviour that leaves other subtrees behind.
8958    ///
8959    /// It is also deliberately uniform. A version using one deep spur beside shallow
8960    /// siblings made the result depend on whether `readdir` returned the spur early:
8961    /// it passed on APFS and failed on ext4.
8962    fn deep_forest() -> tempfile::TempDir {
8963        let dir = tempfile::tempdir().expect("tempdir");
8964        for top in 0..12 {
8965            let mut level: Vec<PathBuf> = vec![dir.path().join(format!("t{top}"))];
8966            for _ in 0..5 {
8967                let mut next = Vec::new();
8968                for parent in &level {
8969                    for child in 0..2 {
8970                        let path = parent.join(format!("c{child}"));
8971                        for file in 0..3 {
8972                            write_file(&path.join(format!("f{file}.dat")), b"xxxxxxxxxx");
8973                        }
8974                        next.push(path);
8975                    }
8976                }
8977                level = next;
8978            }
8979        }
8980        dir
8981    }
8982
8983    /// Files accumulated by the *least advanced* top-level subtree in the first
8984    /// quarter of the walk.
8985    ///
8986    /// Counting subtrees merely *started* cannot discriminate on a tree whose root
8987    /// fans out twelve ways: every scheduler touches all twelve immediately, because
8988    /// reading the root enumerates them. What differs is whether they then advance
8989    /// together, so the question is how far behind the laggard is.
8990    fn leanest_subtree_early(order: ScanOrder, threads: usize, dir: &Path) -> usize {
8991        let config =
8992            ScanConfig { order, batch_size: 1, threads: Some(threads), ..ScanConfig::default() };
8993        let mut files: Vec<PathBuf> = Vec::new();
8994        scan(dir, &config, &mut |observation| {
8995            for op in &observation.ops {
8996                if let Op::Upsert { path, kind, .. } = &op.op {
8997                    if !kind.is_dir() {
8998                        files.push(path.clone());
8999                    }
9000                }
9001            }
9002        })
9003        .expect("scan");
9004
9005        let quarter = files.len() / 4;
9006        let mut per_top: BTreeMap<PathBuf, usize> = BTreeMap::new();
9007        for path in files.iter().take(quarter) {
9008            if let Some(top) = path.components().next() {
9009                *per_top.entry(PathBuf::from(top.as_os_str())).or_default() += 1;
9010            }
9011        }
9012        (0..12)
9013            .map(|top| per_top.get(&PathBuf::from(format!("t{top}"))).copied().unwrap_or(0))
9014            .min()
9015            .unwrap_or(0)
9016    }
9017
9018    #[test]
9019    fn deep_subtrees_do_not_delay_their_siblings() {
9020        // The orientation property, and the reason breadth-first is the default: when
9021        // every top-level subtree is deep, depth-first pours its early effort down
9022        // whichever ones it picked up and leaves the rest at zero, while the region
9023        // scheduler advances all twelve together. A user watching the top level fill
9024        // in sees a meaningful ranking in the first case and a misleading one in the
9025        // second.
9026        //
9027        // Asserted at one worker only, and that bound is deliberate. This metric reads
9028        // *emission* order, and under several workers emission reflects which worker
9029        // finished first as much as which region was claimed — so it varies with core
9030        // count. Measured on a six-core machine the margin is wide (33-37 files against
9031        // 6); on a CI runner with fewer cores both orders can report zero. That makes it
9032        // a benchmark-grade observation, recorded in exp-013, not a unit-test assertion.
9033        //
9034        // The scheduling property itself *is* asserted deterministically, against the
9035        // queue rather than through a walk, by
9036        // `the_region_scheduler_spreads_workers_over_distinct_subtrees`.
9037        let dir = deep_forest();
9038        let breadth = leanest_subtree_early(ScanOrder::BreadthFirst, 1, dir.path());
9039        let depth = leanest_subtree_early(ScanOrder::DepthFirst, 1, dir.path());
9040        assert!(
9041            breadth > depth,
9042            "breadth-first should leave its least advanced top-level subtree further \
9043             along: {breadth} files against {depth}"
9044        );
9045    }
9046
9047    #[test]
9048    fn the_region_scheduler_spreads_workers_over_distinct_subtrees() {
9049        // The scheduler invariant, checked directly on the queue rather than through a
9050        // walk: consecutive claims by *different* workers must land in different
9051        // regions while several regions have work. This is what the round-robin ready
9052        // ring buys, and it is the thing a global FIFO could not promise.
9053        let queue = DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, None, None);
9054        let mut timing = WalkAttribution::default();
9055
9056        // Bootstrap: drain the root, then seed four top-level regions.
9057        let mut claimed = Vec::new();
9058        let root = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
9059        claimed.clear();
9060        queue.extend(
9061            (0..4).map(|top| (PathBuf::from(format!("t{top}")), 1, RegionId::UNASSIGNED)),
9062            &mut timing,
9063        );
9064        assert!(root.release(0, 0, &mut timing).is_none());
9065
9066        // Four workers with no affinity must each be handed a different region. The
9067        // claims are held for the whole loop, as four concurrent workers would hold
9068        // them, because releasing between them would let one worker take every region.
9069        let mut regions = BTreeSet::new();
9070        let mut held = Vec::new();
9071        for _ in 0..4 {
9072            let mut claimed = Vec::new();
9073            held.push(queue.claim(&mut claimed, &mut timing).expect("a region has work"));
9074            regions.insert(claimed[0].2.0);
9075            assert_eq!(claimed.len(), 1, "one directory per region so far");
9076        }
9077        assert_eq!(regions.len(), 4, "each claim took a distinct region: {regions:?}");
9078    }
9079
9080    #[test]
9081    fn breadth_first_spreads_early_work_across_top_level_subtrees() {
9082        // The justification for making breadth-first the default: at the halfway point
9083        // more of the tree's top-level subtrees have started filling, so a consumer
9084        // ranking by size mid-scan is comparing partial values rather than a mix of
9085        // final values and zeros.
9086        //
9087        // Pinned with one worker, where the ordering guarantee is strict and the result
9088        // is deterministic. The multi-worker case is deliberately NOT asserted here:
9089        // measured on this fixture the advantage disappears under the default worker
9090        // count (both orders start 7-8 subtrees, run to run), because emission order is
9091        // then dominated by worker scheduling rather than by queue order. That is a
9092        // real limitation of the current design, recorded in the plan and tracked
9093        // rather than papered over with a test tuned until it passed.
9094        let dir = branching_tree();
9095        let breadth = subtrees_started_at_halfway(ScanOrder::BreadthFirst, 1, dir.path());
9096        let depth = subtrees_started_at_halfway(ScanOrder::DepthFirst, 1, dir.path());
9097
9098        assert!(
9099            breadth > depth,
9100            "breadth-first should have more top-level subtrees underway at the halfway \
9101             point, but started {breadth} against depth-first's {depth}"
9102        );
9103    }
9104
9105    #[test]
9106    fn scan_order_does_not_change_the_cache_scope() {
9107        // Order is operational, like the worker count: it changes when observations
9108        // appear, never which ones, so it must not be able to invalidate a snapshot.
9109        let breadth = ScanConfig { order: ScanOrder::BreadthFirst, ..ScanConfig::default() };
9110        let depth = ScanConfig { order: ScanOrder::DepthFirst, ..ScanConfig::default() };
9111        assert_eq!(breadth.scope(), depth.scope());
9112    }
9113
9114    #[test]
9115    fn worker_threads_are_bounded_and_never_zero() {
9116        let zero = ScanConfig { threads: Some(0), ..ScanConfig::default() };
9117        assert_eq!(zero.worker_threads(), 1, "zero threads must fall back to the serial walk");
9118        let absurd = ScanConfig { threads: Some(usize::MAX), ..ScanConfig::default() };
9119        assert_eq!(absurd.worker_threads(), MAX_SCAN_THREADS);
9120        // The automatic choice is capped well below what a caller may request, because
9121        // the measured knee is far below the core count on a large machine.
9122        let automatic = ScanConfig { threads: None, ..ScanConfig::default() };
9123        assert!((1..=DEFAULT_SCAN_THREADS_CAP).contains(&automatic.worker_threads()));
9124    }
9125
9126    #[test]
9127    fn automatic_worker_pool_keeps_a_conservative_start_and_bounded_reserve() {
9128        assert_eq!(automatic_worker_pool(1), WorkerPool::fixed(1));
9129        assert_eq!(
9130            automatic_worker_pool(4),
9131            WorkerPool {
9132                initial: 4,
9133                maximum: 8,
9134                calibration: Some(WorkerCalibration::new(
9135                    ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
9136                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
9137                )),
9138            }
9139        );
9140        assert_eq!(
9141            automatic_worker_pool(10),
9142            WorkerPool {
9143                initial: DEFAULT_SCAN_THREADS_CAP,
9144                maximum: ADAPTIVE_SCAN_THREADS_CAP,
9145                calibration: Some(WorkerCalibration::new(
9146                    ADAPTIVE_SCAN_CALIBRATION_ENTRIES,
9147                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
9148                )),
9149            }
9150        );
9151    }
9152
9153    #[test]
9154    fn an_abandoned_claim_does_not_strand_the_other_workers() {
9155        // The liveness property behind `DirectoryClaim`. A worker that stops mid-chunk
9156        // — consumer gone, or a panic unwinding through the directory read — still owes
9157        // the queue its claim, and `claim` parks everyone else until `outstanding`
9158        // reaches zero. Before the claim was an RAII guard both of those exits skipped
9159        // the release, and every remaining worker waited on the condvar forever while
9160        // the scoped join waited on them.
9161        let queue = std::sync::Arc::new(DirectoryQueue::new(
9162            (PathBuf::new(), 0),
9163            ScanOrder::BreadthFirst,
9164            None,
9165            None,
9166        ));
9167        let mut timing = WalkAttribution::default();
9168        let mut claimed = Vec::new();
9169
9170        // One worker takes the root and abandons it without publishing anything.
9171        drop(queue.claim(&mut claimed, &mut timing).expect("root is claimable"));
9172
9173        // A second worker must now be told the walk is over rather than parking.
9174        let waiter = queue.clone();
9175        let (done, finished) = std::sync::mpsc::sync_channel(1);
9176        std::thread::spawn(move || {
9177            let mut timing = WalkAttribution::default();
9178            let mut claimed = Vec::new();
9179            let outcome = waiter.claim(&mut claimed, &mut timing).is_some();
9180            done.send(outcome).expect("publish the claim outcome");
9181        });
9182
9183        assert_eq!(
9184            finished.recv_timeout(std::time::Duration::from_secs(5)),
9185            Ok(false),
9186            "the queue must report the walk finished instead of parking the worker"
9187        );
9188    }
9189
9190    #[test]
9191    fn automatic_queue_activates_its_reserve_only_for_slow_initial_work() {
9192        let slow = WorkerCalibration::new(3, 10);
9193        let queue =
9194            DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, Some(slow), None);
9195        let mut timing = WalkAttribution::default();
9196        let mut claimed = Vec::new();
9197
9198        let claim = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
9199        queue.extend([(PathBuf::from("child"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
9200        assert!(claim.release(2, 20, &mut timing).is_none());
9201
9202        claimed.clear();
9203        let claim = queue.claim(&mut claimed, &mut timing).expect("child is claimable");
9204        queue.extend([(PathBuf::from("grandchild"), 2, claimed[0].2)].into_iter(), &mut timing);
9205        assert_eq!(claim.release(1, 10, &mut timing), Some(2));
9206
9207        let fast = WorkerCalibration::new(3, 11);
9208        let queue =
9209            DirectoryQueue::new((PathBuf::new(), 0), ScanOrder::BreadthFirst, Some(fast), None);
9210        let mut timing = WalkAttribution::default();
9211        let mut claimed = Vec::new();
9212        let claim = queue.claim(&mut claimed, &mut timing).expect("root is claimable");
9213        assert!(claim.release(3, 30, &mut timing).is_none());
9214        assert!(queue.lock().controller.is_none(), "calibration decides only once");
9215    }
9216
9217    #[test]
9218    fn repeated_windows_reconsider_a_late_slow_phase_in_entry_order() {
9219        let calibration = WorkerCalibration::new(4, 10);
9220        let mut controller =
9221            WorkerController::new(calibration, WorkerPolicyExperiment::RepeatedWindows);
9222
9223        let fast = controller.observe(4, 20).expect("first complete window");
9224        let slow = controller.observe(4, 80).expect("second complete window");
9225
9226        assert!(!fast.slow);
9227        assert!(slow.slow);
9228        assert_eq!((fast.start_entry_ordinal, fast.end_entry_ordinal), (0, 4));
9229        assert_eq!((slow.start_entry_ordinal, slow.end_entry_ordinal), (4, 8));
9230    }
9231
9232    #[test]
9233    fn shipped_trace_retains_post_decision_windows_without_changing_policy() {
9234        let calibration = WorkerCalibration::new(2, 10);
9235        let pool = WorkerPool { initial: 2, maximum: 4, calibration: Some(calibration) };
9236        let recorder =
9237            ScanDiagnosticsRecorder::new(pool, 2, WorkerPolicyExperiment::ShippedOneShot);
9238        let queue = DirectoryQueue::new_with_policy(
9239            (PathBuf::new(), 0),
9240            ScanOrder::BreadthFirst,
9241            Some(calibration),
9242            Some(recorder.clone()),
9243            pool.initial,
9244            pool.maximum,
9245            WorkerPolicyExperiment::ShippedOneShot,
9246        );
9247        let mut timing = WalkAttribution::default();
9248        let mut claimed = Vec::new();
9249
9250        let first = queue.claim(&mut claimed, &mut timing).expect("first window");
9251        queue.extend([(PathBuf::from("late"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
9252        assert_eq!(first.release(2, 10, &mut timing), None, "fast prefix holds");
9253
9254        claimed.clear();
9255        let late = queue.claim(&mut claimed, &mut timing).expect("late phase");
9256        queue.extend([(PathBuf::from("tail"), 2, RegionId::UNASSIGNED)].into_iter(), &mut timing);
9257        assert_eq!(late.release(2, 40, &mut timing), None, "shadow cannot scale");
9258
9259        let diagnostics = recorder.finish();
9260        assert_eq!(diagnostics.worker_policy.outcome, WorkerPolicyOutcome::Held);
9261        assert_eq!(
9262            diagnostics
9263                .worker_policy
9264                .windows
9265                .iter()
9266                .map(|window| window.decision)
9267                .collect::<Vec<_>>(),
9268            vec![WorkerPolicyDecision::Hold, WorkerPolicyDecision::ObserveSlow]
9269        );
9270    }
9271
9272    #[test]
9273    fn staged_controller_requires_a_useful_frontier_then_stays_bounded() {
9274        let calibration = WorkerCalibration::new(1, 10);
9275        let pool = WorkerPool { initial: 2, maximum: 8, calibration: Some(calibration) };
9276        let recorder =
9277            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::StagedGatedWindows);
9278        let queue = DirectoryQueue::new_with_policy(
9279            (PathBuf::new(), 0),
9280            ScanOrder::BreadthFirst,
9281            Some(calibration),
9282            Some(recorder.clone()),
9283            pool.initial,
9284            pool.maximum,
9285            WorkerPolicyExperiment::StagedGatedWindows,
9286        );
9287        let mut timing = WalkAttribution::default();
9288        let mut claimed = Vec::new();
9289
9290        let root = queue.claim(&mut claimed, &mut timing).expect("root");
9291        queue.extend([(PathBuf::from("narrow"), 1, RegionId::UNASSIGNED)].into_iter(), &mut timing);
9292        assert_eq!(root.release(1, 20, &mut timing), None);
9293
9294        claimed.clear();
9295        let narrow = queue.claim(&mut claimed, &mut timing).expect("narrow child");
9296        queue.extend(
9297            (0..9).map(|index| (PathBuf::from(format!("wide-{index}")), 2, RegionId::UNASSIGNED)),
9298            &mut timing,
9299        );
9300        assert_eq!(narrow.release(1, 20, &mut timing), Some(4));
9301
9302        claimed.clear();
9303        let wide = queue.claim(&mut claimed, &mut timing).expect("wide claim");
9304        queue.extend(
9305            (0..9).map(|index| (PathBuf::from(format!("wider-{index}")), 3, RegionId::UNASSIGNED)),
9306            &mut timing,
9307        );
9308        assert_eq!(wide.release(1, 20, &mut timing), Some(8));
9309
9310        let diagnostics = recorder.finish();
9311        let decisions: Vec<_> = diagnostics
9312            .worker_policy
9313            .windows
9314            .iter()
9315            .map(|window| (window.decision, window.requested_workers))
9316            .collect();
9317        assert_eq!(
9318            decisions,
9319            vec![
9320                (WorkerPolicyDecision::HoldInsufficientFrontier, None),
9321                (WorkerPolicyDecision::ScaleUp, Some(4)),
9322                (WorkerPolicyDecision::ScaleUp, Some(8)),
9323            ]
9324        );
9325        assert!(
9326            diagnostics
9327                .worker_policy
9328                .windows
9329                .windows(2)
9330                .all(|pair| pair[0].end_entry_ordinal <= pair[1].start_entry_ordinal)
9331        );
9332        assert!(diagnostics.worker_policy.windows.iter().all(|window| {
9333            window.requested_workers.is_none_or(|workers| workers <= pool.maximum)
9334        }));
9335    }
9336
9337    #[test]
9338    fn staged_controller_does_not_add_producers_to_a_delayed_handoff() {
9339        let calibration = WorkerCalibration::new(1, 10);
9340        let pool = WorkerPool { initial: 2, maximum: 8, calibration: Some(calibration) };
9341        let recorder =
9342            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::StagedGatedWindows);
9343        recorder.handoff_sent();
9344        recorder.handoff_sent();
9345        let queue = DirectoryQueue::new_with_policy(
9346            (PathBuf::new(), 0),
9347            ScanOrder::BreadthFirst,
9348            Some(calibration),
9349            Some(recorder.clone()),
9350            pool.initial,
9351            pool.maximum,
9352            WorkerPolicyExperiment::StagedGatedWindows,
9353        );
9354        let mut timing = WalkAttribution::default();
9355        let mut claimed = Vec::new();
9356        let claim = queue.claim(&mut claimed, &mut timing).expect("root");
9357        queue.extend(
9358            (0..9).map(|index| (PathBuf::from(format!("ready-{index}")), 1, RegionId::UNASSIGNED)),
9359            &mut timing,
9360        );
9361
9362        assert_eq!(claim.release(1, 20, &mut timing), None);
9363        recorder.handoff_received();
9364        recorder.handoff_received();
9365        let diagnostics = recorder.finish();
9366        assert_eq!(
9367            diagnostics.worker_policy.windows[0].decision,
9368            WorkerPolicyDecision::HoldHandoffBacklog
9369        );
9370    }
9371
9372    #[test]
9373    fn candidate_retains_post_expansion_shadow_history() {
9374        let calibration = WorkerCalibration::new(1, 10);
9375        let pool = WorkerPool { initial: 2, maximum: 4, calibration: Some(calibration) };
9376        let recorder =
9377            ScanDiagnosticsRecorder::new(pool, 4, WorkerPolicyExperiment::RepeatedWindows);
9378        let queue = DirectoryQueue::new_with_policy(
9379            (PathBuf::new(), 0),
9380            ScanOrder::BreadthFirst,
9381            Some(calibration),
9382            Some(recorder.clone()),
9383            pool.initial,
9384            pool.maximum,
9385            WorkerPolicyExperiment::RepeatedWindows,
9386        );
9387        let mut timing = WalkAttribution::default();
9388        let mut claimed = Vec::new();
9389
9390        let slow = queue.claim(&mut claimed, &mut timing).expect("slow prefix");
9391        queue.extend(
9392            (0..4).map(|index| (PathBuf::from(format!("fast-{index}")), 1, RegionId::UNASSIGNED)),
9393            &mut timing,
9394        );
9395        assert_eq!(slow.release(1, 20, &mut timing), Some(4));
9396
9397        claimed.clear();
9398        let fast = queue.claim(&mut claimed, &mut timing).expect("fast suffix");
9399        queue.extend([(PathBuf::from("tail"), 2, RegionId::UNASSIGNED)].into_iter(), &mut timing);
9400        assert_eq!(fast.release(1, 1, &mut timing), None);
9401
9402        let diagnostics = recorder.finish();
9403        assert_eq!(
9404            diagnostics
9405                .worker_policy
9406                .windows
9407                .iter()
9408                .map(|window| window.decision)
9409                .collect::<Vec<_>>(),
9410            vec![WorkerPolicyDecision::ScaleUp, WorkerPolicyDecision::ObserveFast]
9411        );
9412    }
9413
9414    #[test]
9415    fn every_experimental_controller_preserves_exactness_and_shutdown() {
9416        let dir = branching_tree();
9417        let serial = ScanConfig { threads: Some(1), ..ScanConfig::default() };
9418        let (reference, _) = scan_into_index(dir.path(), &serial).expect("serial reference");
9419        let automatic = ScanConfig { threads: None, ..ScanConfig::default() };
9420
9421        for policy in [
9422            WorkerPolicyExperiment::ShippedOneShot,
9423            WorkerPolicyExperiment::RepeatedWindows,
9424            WorkerPolicyExperiment::StagedGatedWindows,
9425        ] {
9426            let (index, report, diagnostics) =
9427                scan_into_index_with_policy_diagnostics(dir.path(), &automatic, policy)
9428                    .expect("candidate scan finishes");
9429            assert!(report.is_complete(), "{policy:?}: {:?}", report.errors);
9430            assert_eq!(index_fingerprint(&reference), index_fingerprint(&index), "{policy:?}");
9431            assert_eq!(diagnostics.worker_policy.ready_directories_at_finish, 0);
9432            assert_eq!(diagnostics.worker_policy.in_flight_directories_at_finish, 0);
9433            assert_eq!(diagnostics.worker_policy.handoff_backlog_at_finish, 0);
9434            assert!(
9435                diagnostics.worker_policy.workers_spawned
9436                    <= diagnostics.worker_policy.maximum_workers
9437            );
9438        }
9439    }
9440
9441    /// A deterministic model of the automatic worker policy under *completion* order.
9442    ///
9443    /// The scaling decision is driven by chunk releases, and chunks complete in whatever
9444    /// order the filesystem and the workers produce them — not in traversal order. On a
9445    /// homogeneous tree that distinction is invisible, because every prefix looks like
9446    /// every other. On a heterogeneous one it decides the answer.
9447    ///
9448    /// These tests exist because the alternative is a stopwatch on a real tree, which
9449    /// measures one host on one day and cannot separate a policy defect from ambient
9450    /// noise. Replaying an explicit completion order through the shipped calibration
9451    /// isolates the policy exactly, and does so identically on every platform.
9452    ///
9453    /// They characterize behavior; they do not endorse a replacement. Which controller
9454    /// is *faster* is a question only the held-out Apple Silicon/APFS matrix can answer.
9455    mod completion_order {
9456        use super::{ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, WorkerCalibration};
9457
9458        /// One chunk release: entries observed and worker time spent observing them.
9459        #[derive(Clone, Copy, Debug)]
9460        struct Chunk {
9461            entries: u64,
9462            work_ns: u64,
9463        }
9464
9465        impl Chunk {
9466            /// A run of `entries` entries costing `per_entry_ns` each.
9467            const fn at(entries: u64, per_entry_ns: u64) -> Self {
9468                Self { entries, work_ns: entries.saturating_mul(per_entry_ns) }
9469            }
9470        }
9471
9472        /// What a policy concluded over one completion order.
9473        #[derive(Debug, PartialEq, Eq)]
9474        enum Outcome {
9475            /// The policy found the filesystem slow and expanded the reserve.
9476            ScaledUp { after_chunks: usize },
9477            /// The policy found the filesystem fast and held the initial pool.
9478            Held { after_chunks: usize },
9479            /// The walk ended before the policy observed enough to conclude anything.
9480            ///
9481            /// Distinct from [`Outcome::Held`] on purpose: nothing was measured, so a
9482            /// held pool here is an absence of evidence rather than a decision.
9483            Undecided,
9484        }
9485
9486        /// Mean cost per entry over a whole trace, which is what the threshold *means*.
9487        fn whole_trace_ns_per_entry(trace: &[Chunk]) -> u64 {
9488            let entries: u64 = trace.iter().map(|chunk| chunk.entries).sum();
9489            let work_ns: u64 = trace.iter().map(|chunk| chunk.work_ns).sum();
9490            assert!(entries > 0, "a trace must observe entries");
9491            work_ns / entries
9492        }
9493
9494        /// Replay a completion order through the *shipped* calibration.
9495        ///
9496        /// This drives [`WorkerCalibration::observe`] itself rather than restating its
9497        /// arithmetic, so the model cannot quietly drift from the policy it is evidence
9498        /// about. The loop mirrors `DirectoryQueue::release`: fold each chunk in, and
9499        /// stop at the first one that produces a verdict.
9500        fn shipped(window: u64, threshold_ns: u64, trace: &[Chunk]) -> Outcome {
9501            let mut calibration = WorkerCalibration::new(window, threshold_ns);
9502            for (index, chunk) in trace.iter().enumerate() {
9503                if let Some(slow) = calibration.observe(chunk.entries, chunk.work_ns) {
9504                    let after_chunks = index + 1;
9505                    return if slow {
9506                        Outcome::ScaledUp { after_chunks }
9507                    } else {
9508                        Outcome::Held { after_chunks }
9509                    };
9510                }
9511            }
9512            Outcome::Undecided
9513        }
9514
9515        /// Entries per chunk in the traces below. Four fill the 16,384-entry window.
9516        const CHUNK: u64 = 4_096;
9517        /// A shallow, cache-warm phase: metadata already resident.
9518        const FAST: Chunk = Chunk::at(CHUNK, 2_000);
9519        /// A deep, cold phase: the latency-bound regime the reserve exists to hide.
9520        const SLOW: Chunk = Chunk::at(CHUNK, 90_000);
9521
9522        #[test]
9523        fn completion_order_alone_flips_the_shipped_decision() {
9524            // The defect, stated as an experiment: hold the *tree* constant and vary
9525            // only the order its chunks complete in. Both traces contain the same four
9526            // fast and four slow chunks, so they describe the same filesystem work.
9527            let fast_phase_first = [FAST, FAST, FAST, FAST, SLOW, SLOW, SLOW, SLOW];
9528            let interleaved = [SLOW, FAST, SLOW, FAST, SLOW, FAST, SLOW, FAST];
9529
9530            let window = 4 * CHUNK;
9531            let threshold = ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY;
9532
9533            // Whole-walk truth is identical, and by the policy's own threshold both
9534            // walks are latency-bound: 46 µs per entry against a 30 µs trigger.
9535            let truth = whole_trace_ns_per_entry(&fast_phase_first);
9536            assert_eq!(truth, whole_trace_ns_per_entry(&interleaved));
9537            assert!(
9538                truth >= threshold,
9539                "both traces are slow walks by the shipped threshold: {truth} < {threshold}"
9540            );
9541
9542            // Yet the decision depends entirely on which chunks happened to finish
9543            // first. One walk hides latency; the other runs the whole slow phase on the
9544            // starting pool, having concluded from an unrepresentative prefix.
9545            assert_eq!(
9546                shipped(window, threshold, &fast_phase_first),
9547                Outcome::Held { after_chunks: 4 },
9548                "a fast prefix holds the pool for a walk that is slow overall"
9549            );
9550            assert_eq!(
9551                shipped(window, threshold, &interleaved),
9552                Outcome::ScaledUp { after_chunks: 4 },
9553                "the same tree scales up when its slow chunks land in the window"
9554            );
9555        }
9556
9557        #[test]
9558        fn a_slow_phase_after_the_window_is_never_reconsidered() {
9559            // The heterogeneous-tree case from the field report. A small fast region
9560            // fills the window, and everything after it is slow — but the calibration
9561            // is already gone, so no amount of later evidence can reopen the decision.
9562            let mut trace = vec![FAST; 4];
9563            trace.extend(std::iter::repeat_n(SLOW, 400));
9564
9565            let observed = whole_trace_ns_per_entry(&trace);
9566            assert!(
9567                observed >= 2 * ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
9568                "the walk is overwhelmingly latency-bound: {observed} ns per entry"
9569            );
9570
9571            assert_eq!(
9572                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
9573                Outcome::Held { after_chunks: 4 },
9574                "1% of the walk decided the worker policy for the other 99%"
9575            );
9576        }
9577
9578        #[test]
9579        fn slow_in_flight_work_is_censored_by_fast_completions() {
9580            // Four slow chunks have already been claimed, but their filesystem calls
9581            // remain in flight while four cache-warm chunks complete. Completion-order
9582            // calibration cannot see owed work: the fast completions close the window
9583            // and permanently hold before any slow claim returns.
9584            let completed_before_slow_returns = [FAST, FAST, FAST, FAST];
9585            assert_eq!(
9586                shipped(
9587                    4 * CHUNK,
9588                    ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY,
9589                    &completed_before_slow_returns,
9590                ),
9591                Outcome::Held { after_chunks: 4 }
9592            );
9593
9594            let mut eventual_completions = completed_before_slow_returns.to_vec();
9595            eventual_completions.extend([SLOW, SLOW, SLOW, SLOW]);
9596            assert!(
9597                whole_trace_ns_per_entry(&eventual_completions)
9598                    >= ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY
9599            );
9600            assert_eq!(
9601                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &eventual_completions,),
9602                Outcome::Held { after_chunks: 4 }
9603            );
9604        }
9605
9606        #[test]
9607        fn a_slow_prefix_can_scale_a_walk_that_is_fast_overall() {
9608            // The mirror-image error. A one-way expansion reacts correctly to the
9609            // prefix by its local threshold, but the prefix is under 1% of this walk
9610            // and the whole trace is firmly in the fast regime. A repeated trigger
9611            // alone cannot undo an expansion; staged growth limits exposure but does
9612            // not make reversible parking unnecessary.
9613            let mut trace = vec![SLOW; 4];
9614            trace.extend(std::iter::repeat_n(FAST, 400));
9615            let observed = whole_trace_ns_per_entry(&trace);
9616            assert!(observed < ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY);
9617            assert_eq!(
9618                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
9619                Outcome::ScaledUp { after_chunks: 4 }
9620            );
9621            assert_eq!(
9622                sliding(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
9623                Outcome::ScaledUp { after_chunks: 4 }
9624            );
9625        }
9626
9627        #[test]
9628        fn a_walk_shorter_than_the_window_decides_nothing() {
9629            // Fails closed rather than reporting a held pool: a walk this short never
9630            // observed enough to have an opinion, and an artifact that recorded `Held`
9631            // would claim a measurement that was never taken.
9632            let trace = [FAST, SLOW];
9633            assert_eq!(
9634                shipped(4 * CHUNK, ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY, &trace),
9635                Outcome::Undecided
9636            );
9637        }
9638
9639        /// A screening-only candidate: a window that slides instead of closing once.
9640        ///
9641        /// Present as *evidence about a design*, not as a proposed change. It keeps the
9642        /// shipped trigger and pool bounds and alters only when the question is asked,
9643        /// which is the narrowest edit that could address the order sensitivity above.
9644        /// Whether it is faster on a real tree is unmeasured here and unmeasurable in a
9645        /// virtualized non-APFS environment; selecting it would need the held-out Apple
9646        /// Silicon matrix that this workstream has not yet been able to run.
9647        struct SlidingWindow {
9648            window_entries: u64,
9649            threshold_ns: u64,
9650            recent: std::collections::VecDeque<Chunk>,
9651            entries: u64,
9652            work_ns: u64,
9653        }
9654
9655        impl SlidingWindow {
9656            fn new(window_entries: u64, threshold_ns: u64) -> Self {
9657                Self {
9658                    window_entries,
9659                    threshold_ns,
9660                    recent: std::collections::VecDeque::new(),
9661                    entries: 0,
9662                    work_ns: 0,
9663                }
9664            }
9665
9666            /// Fold in a chunk and re-ask the question over the trailing window.
9667            fn observe(&mut self, chunk: Chunk) -> Option<bool> {
9668                self.recent.push_back(chunk);
9669                self.entries = self.entries.saturating_add(chunk.entries);
9670                self.work_ns = self.work_ns.saturating_add(chunk.work_ns);
9671
9672                // Drop from the front while the window stays full without the oldest
9673                // chunk, so the answer describes recent work rather than the whole walk.
9674                while let Some(oldest) = self.recent.front().copied() {
9675                    if self.entries - oldest.entries < self.window_entries {
9676                        break;
9677                    }
9678                    self.recent.pop_front();
9679                    self.entries -= oldest.entries;
9680                    self.work_ns -= oldest.work_ns;
9681                }
9682
9683                (self.entries >= self.window_entries)
9684                    .then(|| self.work_ns / self.entries >= self.threshold_ns)
9685            }
9686        }
9687
9688        /// Replay a completion order through the candidate, stopping at its first
9689        /// scale-up. The shipped pool only grows, so a later verdict cannot undo one.
9690        fn sliding(window: u64, threshold_ns: u64, trace: &[Chunk]) -> Outcome {
9691            let mut policy = SlidingWindow::new(window, threshold_ns);
9692            let mut decided = None;
9693            for (index, chunk) in trace.iter().enumerate() {
9694                if let Some(slow) = policy.observe(*chunk) {
9695                    let after_chunks = index + 1;
9696                    if slow {
9697                        return Outcome::ScaledUp { after_chunks };
9698                    }
9699                    decided.get_or_insert(Outcome::Held { after_chunks });
9700                }
9701            }
9702            decided.unwrap_or(Outcome::Undecided)
9703        }
9704
9705        #[test]
9706        fn screening_a_sliding_window_against_the_order_sensitivity() {
9707            let window = 4 * CHUNK;
9708            let threshold = ADAPTIVE_SCAN_SLOW_WORK_NS_PER_ENTRY;
9709
9710            // The pair that splits the shipped policy reaches one answer here, and it
9711            // is the answer the whole-trace mean supports in both orders.
9712            let fast_phase_first = [FAST, FAST, FAST, FAST, SLOW, SLOW, SLOW, SLOW];
9713            let interleaved = [SLOW, FAST, SLOW, FAST, SLOW, FAST, SLOW, FAST];
9714            // Both reach the same verdict; they differ only in how long the fast prefix
9715            // delays it, which is the behavior a trailing window is supposed to have.
9716            assert_eq!(
9717                sliding(window, threshold, &fast_phase_first),
9718                Outcome::ScaledUp { after_chunks: 6 }
9719            );
9720            assert_eq!(
9721                sliding(window, threshold, &interleaved),
9722                Outcome::ScaledUp { after_chunks: 4 }
9723            );
9724
9725            // And the late slow phase is reached rather than missed: two slow chunks
9726            // after the window closes are enough to pull the trailing mean over.
9727            let mut late = vec![FAST; 4];
9728            late.extend(std::iter::repeat_n(SLOW, 400));
9729            assert_eq!(sliding(window, threshold, &late), Outcome::ScaledUp { after_chunks: 6 });
9730
9731            // A genuinely fast tree must still hold the pool: the candidate has to keep
9732            // the property the shipped policy gets right, or it is not a candidate.
9733            let uniformly_fast = vec![FAST; 40];
9734            assert_eq!(
9735                sliding(window, threshold, &uniformly_fast),
9736                Outcome::Held { after_chunks: 4 }
9737            );
9738
9739            // A short walk still decides nothing, for the same reason as above.
9740            assert_eq!(sliding(window, threshold, &[FAST, SLOW]), Outcome::Undecided);
9741        }
9742    }
9743
9744    #[test]
9745    fn thread_count_does_not_change_the_cache_scope() {
9746        // Threads are an operational choice. If they leaked into the scope, changing
9747        // the pool size would invalidate every snapshot on disk.
9748        let serial = ScanConfig { threads: Some(1), ..ScanConfig::default() };
9749        let parallel = ScanConfig { threads: Some(8), ..ScanConfig::default() };
9750        assert_eq!(serial.scope(), parallel.scope());
9751    }
9752
9753    #[test]
9754    fn scan_populates_an_index_end_to_end() {
9755        let dir = sample_tree();
9756        let (index, report) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
9757
9758        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
9759        let total = index.total();
9760        assert_eq!(total.files, 3);
9761        assert_eq!(total.dirs, 2);
9762        assert_eq!(total.bytes, 5 + 12 + 9);
9763        assert_eq!(total.by_ext[".rs"].files, 2);
9764        assert_eq!(total.by_ext[".txt"].files, 1);
9765
9766        let src = index.rollup(Path::new("src")).expect("src");
9767        assert_eq!(src.files, 2);
9768        assert_eq!(src.dirs, 1);
9769    }
9770
9771    #[test]
9772    fn cold_scan_routes_control_sources_through_both_walkers() {
9773        let dir = tempfile::tempdir().expect("tempdir");
9774        write_file(&dir.path().join(".gitignore"), b"*.log\n");
9775        write_file(&dir.path().join("debug.log"), b"ignored");
9776        write_file(&dir.path().join("keep.rs"), b"visible");
9777
9778        for threads in [1, 4] {
9779            let config =
9780                ScanConfig { read_controls: true, threads: Some(threads), ..ScanConfig::default() };
9781            let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
9782
9783            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
9784            assert!(
9785                index
9786                    .controls()
9787                    .expect("control state observed")
9788                    .source_is(Path::new(".gitignore"), b"*.log\n")
9789            );
9790            assert_eq!(
9791                index.is_ignored(Path::new("debug.log")).expect("control state observed"),
9792                Some(true)
9793            );
9794            assert_eq!(
9795                index.is_ignored(Path::new("keep.rs")).expect("control state observed"),
9796                Some(false)
9797            );
9798            let partitions = index.partition_total().expect("control state observed");
9799            assert_eq!(partitions.all.files, 3);
9800            assert_eq!(partitions.unignored.files, 2);
9801        }
9802    }
9803
9804    #[cfg(unix)]
9805    #[test]
9806    fn raced_fifo_control_source_is_rejected_without_blocking() {
9807        let dir = tempfile::tempdir().expect("tempdir");
9808        let control = dir.path().join(".gitignore");
9809        let status = match std::process::Command::new("mkfifo").arg(&control).status() {
9810            Ok(status) => status,
9811            Err(error) if error.kind() == std::io::ErrorKind::NotFound => return,
9812            Err(error) => panic!("create fifo: {error}"),
9813        };
9814        assert!(status.success(), "mkfifo exited with {status}");
9815
9816        let root = dir.path().to_path_buf();
9817        let (sender, receiver) = std::sync::mpsc::channel();
9818        std::thread::spawn(move || {
9819            let result = read_control_op_unconditional(
9820                &root,
9821                Path::new(".gitignore"),
9822                EntryKind::File,
9823                Some(crate::control::DEFAULT_CONTROL_BUDGET),
9824            );
9825            sender.send(result).ok();
9826        });
9827        let result = receiver
9828            .recv_timeout(std::time::Duration::from_secs(1))
9829            .expect("a raced FIFO must not block the scan worker")
9830            .expect("the non-regular replacement is a normal control removal");
9831
9832        assert!(matches!(result, Some(Op::ControlRemove { .. })));
9833    }
9834
9835    #[test]
9836    fn hidden_admission_keeps_exact_allowlist_and_control_signals_only() {
9837        let dir = tempfile::tempdir().expect("tempdir");
9838        write_file(&dir.path().join(".gitignore"), b"*.log\n");
9839        write_file(&dir.path().join("debug.log"), b"ignored");
9840        write_file(&dir.path().join(".secret/token"), b"hidden");
9841        write_file(&dir.path().join(".github/workflows/check.yml"), b"visible");
9842        let hidden = std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([".github"]));
9843
9844        for threads in [1, 4] {
9845            let config = ScanConfig {
9846                hidden: Some(std::sync::Arc::clone(&hidden)),
9847                threads: Some(threads),
9848                read_controls: true,
9849                ..ScanConfig::default()
9850            };
9851            let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
9852
9853            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
9854            assert!(index.lookup(Path::new(".gitignore")).is_none());
9855            assert!(index.lookup(Path::new(".secret")).is_none());
9856            assert!(index.lookup(Path::new(".secret/token")).is_none());
9857            assert!(index.lookup(Path::new(".github/workflows/check.yml")).is_some());
9858            assert!(
9859                index
9860                    .controls()
9861                    .expect("control state observed")
9862                    .source_is(Path::new(".gitignore"), b"*.log\n")
9863            );
9864            assert_eq!(
9865                index.is_ignored(Path::new("debug.log")).expect("control state observed"),
9866                Some(true)
9867            );
9868
9869            fs::remove_file(dir.path().join(".gitignore")).expect("remove control");
9870            if threads > 1 {
9871                fs::create_dir(dir.path().join(".gitignore")).expect("replace with directory");
9872            }
9873            let reconciled = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
9874            assert!(reconciled.is_complete());
9875            assert!(index.controls().expect("control state observed").is_empty());
9876            if threads > 1 {
9877                fs::remove_dir(dir.path().join(".gitignore")).expect("remove directory");
9878            }
9879            write_file(&dir.path().join(".gitignore"), b"*.log\n");
9880        }
9881    }
9882
9883    #[test]
9884    fn excluded_ignored_directory_is_not_enumerated_and_rule_edits_reconcile_it() {
9885        let root = tempfile::tempdir().expect("root");
9886        write_file(&root.path().join(".gitignore"), b"target/\n");
9887        write_file(&root.path().join("target/deep/file.rs"), b"code");
9888        write_file(&root.path().join("keep.rs"), b"kept");
9889        let config = ScanConfig {
9890            population: crate::query::IgnoredEntries::Exclude,
9891            threads: Some(4),
9892            ..ScanConfig::default()
9893        };
9894
9895        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
9896        assert!(cold.is_complete(), "{:?}", cold.errors);
9897        assert_eq!(cold.dirs_read, 1, "ignored target was not opened");
9898        assert!(index.lookup(Path::new("target")).is_none());
9899        assert!(index.lookup(Path::new("keep.rs")).is_some());
9900
9901        write_file(&root.path().join(".gitignore"), b"");
9902        let exposed = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile exposure");
9903        assert!(exposed.is_complete(), "{:?}", exposed.scan.errors);
9904        assert!(index.lookup(Path::new("target/deep/file.rs")).is_some());
9905        assert!(exposed.scan.dirs_read >= 3);
9906
9907        write_file(&root.path().join(".gitignore"), b"target/\n");
9908        let excluded = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile exclusion");
9909        assert!(excluded.is_complete(), "{:?}", excluded.scan.errors);
9910        assert!(index.lookup(Path::new("target")).is_none());
9911        assert_eq!(excluded.scan.dirs_read, 1, "ignored target was not reopened");
9912    }
9913
9914    #[test]
9915    fn excluded_subtree_refresh_keeps_ignored_file_and_directory_out_of_scope() {
9916        let root = tempfile::tempdir().expect("root");
9917        write_file(&root.path().join(".gitignore"), b"target/\n*.log\n");
9918        write_file(&root.path().join("target/deep/file.rs"), b"code");
9919        write_file(&root.path().join("debug.log"), b"ignored");
9920        let config = ScanConfig {
9921            population: crate::query::IgnoredEntries::Exclude,
9922            ..ScanConfig::default()
9923        };
9924        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
9925        assert!(cold.is_complete());
9926        assert!(index.lookup(Path::new("target")).is_none());
9927        assert!(index.lookup(Path::new("debug.log")).is_none());
9928
9929        for path in ["target", "debug.log"] {
9930            let refresh = reconcile_subtree(&mut index, Path::new(path), &config, &mut |_| {})
9931                .expect("subtree refresh");
9932            assert!(refresh.is_complete(), "{path}: {:?}", refresh.scan.errors);
9933            assert!(index.lookup(Path::new(path)).is_none(), "{path} is outside scope");
9934        }
9935        assert_eq!(index.total().dirs, 0);
9936    }
9937
9938    #[test]
9939    fn handled_subtree_refresh_recovers_pruned_ancestry_after_control_edit() {
9940        let root = tempfile::tempdir().expect("root");
9941        write_file(&root.path().join(".gitignore"), b"target/\n");
9942        write_file(&root.path().join("target/deep/file.rs"), b"code");
9943        let config = ScanConfig {
9944            population: crate::query::IgnoredEntries::Exclude,
9945            ..ScanConfig::default()
9946        };
9947        let (index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
9948        assert!(cold.is_complete());
9949        let handle = IndexHandle::new(index);
9950
9951        let hidden = reconcile_subtree_handle(
9952            &handle,
9953            Path::new("target/deep/file.rs"),
9954            &config,
9955            &mut |_| {},
9956        )
9957        .expect("refresh pruned descendant");
9958        assert!(hidden.is_complete());
9959        assert!(
9960            !handle
9961                .read_with(|index| index.lookup(Path::new("target")).is_some())
9962                .expect("read after hidden refresh")
9963        );
9964
9965        write_file(&root.path().join(".gitignore"), b"");
9966        let exposed = reconcile_subtree_handle(&handle, Path::new("target"), &config, &mut |_| {})
9967            .expect("refresh changed control");
9968        assert!(exposed.is_complete());
9969        assert!(
9970            handle
9971                .read_with(|index| index.lookup(Path::new("target/deep/file.rs")).is_some())
9972                .expect("read after control recovery")
9973        );
9974
9975        write_file(&root.path().join(".gitignore"), b"target/\n");
9976        let hidden_again =
9977            reconcile_subtree_handle(&handle, Path::new("target"), &config, &mut |_| {})
9978                .expect("refresh restored control");
9979        assert!(hidden_again.is_complete());
9980        assert!(
9981            !handle
9982                .read_with(|index| index.lookup(Path::new("target")).is_some())
9983                .expect("read after restored control")
9984        );
9985    }
9986
9987    #[cfg(unix)]
9988    #[test]
9989    fn targeted_refresh_keeps_unknown_population_below_unreadable_ancestor_control() {
9990        use std::os::unix::fs::PermissionsExt;
9991
9992        if !crate::test_support::require_permission_bits() {
9993            return;
9994        }
9995        let root = tempfile::tempdir().expect("root");
9996        let control = root.path().join(".gitignore");
9997        write_file(&control, b"*.log\n");
9998        write_file(&root.path().join("a/keep.rs"), b"unknown membership");
9999        let config =
10000            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
10001        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("unreadable");
10002        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold scan");
10003        assert!(!cold.is_complete());
10004        assert!(index.lookup(Path::new("a/keep.rs")).is_some());
10005        assert_eq!(index.ignored_classification(Path::new("a/keep.rs")), None);
10006
10007        let refreshed = reconcile_subtree(&mut index, Path::new("a/keep.rs"), &config, &mut |_| {});
10008        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore control");
10009        let refreshed = refreshed.expect("targeted refresh");
10010        assert!(!refreshed.is_complete(), "unreadable governing control was not visited");
10011        assert!(index.lookup(Path::new("a/keep.rs")).is_some(), "unknown must stay retained");
10012        assert_eq!(index.ignored_classification(Path::new("a/keep.rs")), None);
10013    }
10014
10015    #[test]
10016    fn exclusion_honors_nested_negation_and_refused_rule_changes() {
10017        let root = tempfile::tempdir().expect("root");
10018        write_file(&root.path().join("nested/.gitignore"), b"*.log\n!keep.log\n");
10019        write_file(&root.path().join("nested/keep.log"), b"negated");
10020        write_file(&root.path().join("nested/drop.log"), b"ignored");
10021        let config = ScanConfig {
10022            population: crate::query::IgnoredEntries::Exclude,
10023            control_limits: crate::control::ControlLimits {
10024                line_limit: Some(20),
10025                ..crate::control::ControlLimits::default()
10026            },
10027            ..ScanConfig::default()
10028        };
10029        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold");
10030        assert!(cold.is_complete(), "{:?}", cold.errors);
10031        assert!(index.lookup(Path::new("nested/keep.log")).is_some());
10032        assert!(index.lookup(Path::new("nested/drop.log")).is_none());
10033
10034        write_file(&root.path().join("nested/.gitignore"), b"this-line-is-over-the-limit\n");
10035        let changed =
10036            reconcile(&mut index, &config, &mut |_| {}).expect("reconcile refused source");
10037        assert!(changed.is_complete(), "{:?}", changed.scan.errors);
10038        assert!(index.lookup(Path::new("nested/drop.log")).is_some(), "unknown cannot be pruned");
10039        assert_eq!(index.ignored_classification(Path::new("nested/drop.log")), None);
10040    }
10041
10042    #[test]
10043    fn only_population_skips_nonignored_content_candidates_and_refusals_are_unknown() {
10044        let root = tempfile::tempdir().expect("root");
10045        write_file(&root.path().join(".gitignore"), b"*.log\n");
10046        write_file(&root.path().join("keep.rs"), b"code");
10047        write_file(&root.path().join("debug.log"), b"ignored");
10048        write_file(&root.path().join("nested/keep.rs"), b"kept");
10049        write_file(&root.path().join("nested/debug.log"), b"ignored below nonignored dir");
10050        let only =
10051            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
10052        let (index, report) = scan_into_index(root.path(), &only).expect("scan");
10053        assert!(report.is_complete());
10054        assert!(index.lookup(Path::new("keep.rs")).is_none());
10055        assert!(index.lookup(Path::new("nested")).is_some());
10056        assert!(index.lookup(Path::new("nested/keep.rs")).is_none());
10057        assert!(index.lookup(Path::new("nested/debug.log")).is_some());
10058        let candidates = index.analysis_candidates(crate::content::AnalysisSet::NONE.with_lines());
10059        assert_eq!(candidates.len(), 2);
10060        assert!(
10061            candidates.iter().any(|candidate| candidate.relative_path == Path::new("debug.log"))
10062        );
10063        assert!(
10064            candidates
10065                .iter()
10066                .any(|candidate| candidate.relative_path == Path::new("nested/debug.log"))
10067        );
10068
10069        let refused = ScanConfig {
10070            population: crate::query::IgnoredEntries::Exclude,
10071            control_limits: crate::control::ControlLimits {
10072                line_limit: Some(1),
10073                ..crate::control::ControlLimits::default()
10074            },
10075            ..ScanConfig::default()
10076        };
10077        let (index, report) = scan_into_index(root.path(), &refused).expect("refused scan");
10078        assert!(report.is_complete());
10079        assert!(index.lookup(Path::new("debug.log")).is_some(), "unknown is not pruned");
10080        assert_eq!(index.ignored_classification(Path::new("debug.log")), None);
10081        assert!(
10082            index.analysis_candidates(crate::content::AnalysisSet::NONE.with_lines()).is_empty()
10083        );
10084    }
10085
10086    #[test]
10087    fn only_population_content_is_complete_with_retained_control_file() {
10088        let root = tempfile::tempdir().expect("root");
10089        write_file(&root.path().join(".gitignore"), b"vendor/\n");
10090        write_file(&root.path().join("main.rs"), b"fn main() {}\n");
10091        write_file(&root.path().join("vendor/lib.rs"), b"fn lib() {}\n");
10092        let config =
10093            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
10094        let (mut index, scan) = scan_into_index(root.path(), &config).expect("scan");
10095        assert!(scan.is_complete());
10096        let profile = crate::content::AnalysisSet::NONE.with_lines().with_code();
10097        let analyzed = crate::content::analyze_index(
10098            &mut index,
10099            crate::content::AnalysisRequest {
10100                profile,
10101                ..crate::content::AnalysisRequest::default()
10102            },
10103        );
10104        assert!(analyzed.is_complete(), "{analyzed:?}");
10105        assert_eq!(analyzed.candidates, 1);
10106        assert!(!index.content_has_pending(profile));
10107    }
10108
10109    #[test]
10110    fn only_population_reconciles_rule_changes_without_losing_traversal() {
10111        let root = tempfile::tempdir().expect("root");
10112        write_file(&root.path().join("nested/.gitignore"), b"*.log\n");
10113        write_file(&root.path().join("nested/first.log"), b"first");
10114        write_file(&root.path().join("nested/second.txt"), b"second");
10115        let config =
10116            ScanConfig { population: crate::query::IgnoredEntries::Only, ..ScanConfig::default() };
10117        let (mut index, cold) = scan_into_index(root.path(), &config).expect("cold");
10118        assert!(cold.is_complete());
10119        assert!(index.lookup(Path::new("nested")).is_some());
10120        assert!(index.lookup(Path::new("nested/first.log")).is_some());
10121        assert!(index.lookup(Path::new("nested/second.txt")).is_none());
10122
10123        write_file(&root.path().join("nested/.gitignore"), b"*.txt\n");
10124        let changed = reconcile(&mut index, &config, &mut |_| {}).expect("rule change");
10125        assert!(changed.is_complete(), "{:?}", changed.scan.errors);
10126        assert!(index.lookup(Path::new("nested/first.log")).is_none());
10127        assert!(index.lookup(Path::new("nested/second.txt")).is_some());
10128    }
10129
10130    #[cfg(unix)]
10131    #[test]
10132    fn excluded_special_objects_never_enter_cold_or_reconciled_facts() {
10133        use std::os::unix::net::UnixListener;
10134
10135        let dir = tempfile::tempdir().expect("tempdir");
10136        let socket_path = dir.path().join("service.sock");
10137        let _listener = UnixListener::bind(&socket_path).expect("bind socket");
10138        write_file(&dir.path().join("replacement"), b"ordinary");
10139        let (kept, kept_report) =
10140            scan_into_index(dir.path(), &ScanConfig::default()).expect("default scan");
10141        assert!(kept_report.is_complete());
10142        assert_eq!(kept.kind(Path::new("service.sock")), Some(EntryKind::Other));
10143
10144        let serial_config =
10145            ScanConfig { exclude_special: true, threads: Some(1), ..ScanConfig::default() };
10146        let parallel_config =
10147            ScanConfig { exclude_special: true, threads: Some(4), ..ScanConfig::default() };
10148        let (mut serial, serial_report) =
10149            scan_into_index(dir.path(), &serial_config).expect("serial scan");
10150        let (mut parallel, parallel_report) =
10151            scan_into_index(dir.path(), &parallel_config).expect("parallel scan");
10152
10153        for (index, report) in [(&serial, &serial_report), (&parallel, &parallel_report)] {
10154            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10155            assert!(index.lookup(Path::new("service.sock")).is_none());
10156            assert!(index.lookup(Path::new("replacement")).is_some());
10157        }
10158
10159        fs::remove_file(dir.path().join("replacement")).expect("remove file");
10160        let _replacement =
10161            UnixListener::bind(dir.path().join("replacement")).expect("bind replacement socket");
10162        let serial_reconciled =
10163            reconcile(&mut serial, &serial_config, &mut |_| {}).expect("serial reconcile");
10164        let parallel_reconciled =
10165            reconcile(&mut parallel, &parallel_config, &mut |_| {}).expect("parallel reconcile");
10166
10167        assert!(serial_reconciled.is_complete());
10168        assert!(parallel_reconciled.is_complete());
10169        assert!(serial.lookup(Path::new("replacement")).is_none());
10170        assert!(parallel.lookup(Path::new("replacement")).is_none());
10171        assert_eq!(index_fingerprint(&serial), index_fingerprint(&parallel));
10172    }
10173
10174    #[test]
10175    fn control_sources_respect_a_single_operation_batch_bound() {
10176        let dir = tempfile::tempdir().expect("tempdir");
10177        write_file(&dir.path().join(".gitignore"), b"*.log\n");
10178        write_file(&dir.path().join("debug.log"), b"ignored");
10179
10180        for threads in [1, 4] {
10181            let config = ScanConfig {
10182                read_controls: true,
10183                threads: Some(threads),
10184                batch_size: 1,
10185                ..ScanConfig::default()
10186            };
10187            let mut largest = 0;
10188            let report = scan(dir.path(), &config, &mut |observation| {
10189                largest = largest.max(observation.len());
10190            })
10191            .expect("scan");
10192
10193            assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10194            assert_eq!(largest, 1);
10195        }
10196    }
10197
10198    #[test]
10199    fn cold_scan_matches_the_metabrowser_nested_control_fixture() {
10200        let dir = tempfile::tempdir().expect("tempdir");
10201        write_file(&dir.path().join(".gitignore"), b"node_modules/\n*.pyc\n");
10202        write_file(&dir.path().join("src/app.py"), b"x");
10203        write_file(&dir.path().join("src/thing.pyc"), b"x");
10204        write_file(&dir.path().join("src/generated/.gitignore"), b"*.gen\n");
10205        write_file(&dir.path().join("src/generated/out.gen"), b"x");
10206        write_file(&dir.path().join("node_modules/.gitignore"), b"!keep-me.py\n");
10207        write_file(&dir.path().join("node_modules/keep-me.py"), b"x");
10208
10209        let (index, report) = scan_into_index(
10210            dir.path(),
10211            &ScanConfig { read_controls: true, threads: Some(4), ..ScanConfig::default() },
10212        )
10213        .expect("scan fixture");
10214
10215        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10216        assert_eq!(
10217            index.is_ignored(Path::new("src/app.py")).expect("control state observed"),
10218            Some(false)
10219        );
10220        assert_eq!(
10221            index.is_ignored(Path::new("src/thing.pyc")).expect("control state observed"),
10222            Some(true)
10223        );
10224        assert_eq!(
10225            index.is_ignored(Path::new("src/generated")).expect("control state observed"),
10226            Some(false)
10227        );
10228        assert_eq!(
10229            index.is_ignored(Path::new("src/generated/out.gen")).expect("control state observed"),
10230            Some(true)
10231        );
10232        assert_eq!(
10233            index.is_ignored(Path::new("node_modules")).expect("control state observed"),
10234            Some(true)
10235        );
10236        assert_eq!(
10237            index.is_ignored(Path::new("node_modules/keep-me.py")).expect("control state observed"),
10238            Some(true)
10239        );
10240    }
10241
10242    #[test]
10243    fn reconciliation_observes_same_metadata_control_edits_and_last_deletion() {
10244        let dir = tempfile::tempdir().expect("tempdir");
10245        write_file(&dir.path().join(".gitignore"), b"*.log\n");
10246        write_file(&dir.path().join("debug.log"), b"ignored");
10247        let config = ScanConfig { read_controls: true, threads: Some(1), ..ScanConfig::default() };
10248        let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
10249        assert!(report.is_complete());
10250        assert_eq!(
10251            index.is_ignored(Path::new("debug.log")).expect("control state observed"),
10252            Some(true)
10253        );
10254
10255        // Same-length content proves control identity is not inferred from stat-tier
10256        // metadata, which can remain unchanged on coarse filesystems.
10257        write_file(&dir.path().join(".gitignore"), b"*.tmp\n");
10258        let edited = reconcile(&mut index, &config, &mut |_| {}).expect("edit reconcile");
10259        assert!(edited.is_complete());
10260        assert_eq!(edited.apply.controls, 1);
10261        assert_eq!(edited.apply.reclassified, 1);
10262        assert!(
10263            index
10264                .controls()
10265                .expect("control state observed")
10266                .source_is(Path::new(".gitignore"), b"*.tmp\n")
10267        );
10268        assert_eq!(
10269            index.is_ignored(Path::new("debug.log")).expect("control state observed"),
10270            Some(false)
10271        );
10272
10273        fs::remove_file(dir.path().join(".gitignore")).expect("remove control");
10274        let removed = reconcile(&mut index, &config, &mut |_| {}).expect("remove reconcile");
10275        assert!(removed.is_complete());
10276        assert_eq!(removed.apply.controls, 1);
10277        assert!(index.controls().expect("control state observed").is_empty());
10278        let partitions = index.partition_total().expect("control state observed");
10279        assert_eq!(partitions.all, partitions.unignored);
10280    }
10281
10282    #[cfg(unix)]
10283    #[test]
10284    fn directory_entry_metadata_does_not_follow_symlinks() {
10285        use std::os::unix::fs::symlink;
10286
10287        let root = tempfile::tempdir().expect("root");
10288        let outside = tempfile::tempdir().expect("outside");
10289        write_file(&outside.path().join("must-not-be-scanned.txt"), b"outside");
10290        symlink(outside.path(), root.path().join("link")).expect("symlink");
10291
10292        let (index, report) = scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
10293
10294        assert!(report.is_complete(), "unexpected errors: {:?}", report.errors);
10295        assert_eq!(index.kind(Path::new("link")), Some(EntryKind::Symlink));
10296        assert!(index.lookup(Path::new("link/must-not-be-scanned.txt")).is_none());
10297        assert_eq!(index.total().files, 0);
10298    }
10299
10300    #[test]
10301    fn cold_scan_establishes_a_baseline_without_change_history() {
10302        let dir = sample_tree();
10303        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10304
10305        assert_eq!(index.clock(), crate::Clock::ZERO);
10306        assert!(index.since(crate::Clock::ZERO).commits.is_empty());
10307    }
10308
10309    #[test]
10310    fn max_depth_stops_descent() {
10311        let dir = sample_tree();
10312        let config = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
10313        let (index, _) = scan_into_index(dir.path(), &config).expect("scan");
10314
10315        assert!(index.lookup(Path::new("src")).is_some());
10316        assert!(index.lookup(Path::new("src/main.rs")).is_none());
10317    }
10318
10319    #[test]
10320    fn zero_max_depth_keeps_only_the_index_root() {
10321        let dir = sample_tree();
10322        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
10323        let (index, report) = scan_into_index(dir.path(), &config).expect("scan");
10324
10325        assert!(index.is_empty());
10326        assert_eq!(report.entries, 0);
10327        assert_eq!(report.dirs_read, 0);
10328    }
10329
10330    #[test]
10331    fn direct_scan_records_the_canonical_root() {
10332        let dir = sample_tree();
10333        let aliased = dir.path().join(".");
10334        let (index, _) = scan_into_index(&aliased, &ScanConfig::default()).expect("scan");
10335
10336        assert_eq!(index.root_path(), dir.path().canonicalize().expect("canonical root"));
10337    }
10338
10339    #[test]
10340    fn unsupported_symlink_following_is_rejected_on_cold_and_warm_paths() {
10341        let dir = sample_tree();
10342        let unsupported = ScanConfig { follow_symlinks: true, ..ScanConfig::default() };
10343        assert!(matches!(
10344            scan_into_index(dir.path(), &unsupported),
10345            Err(Error::UnsupportedScanConfig(_))
10346        ));
10347
10348        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10349        assert!(matches!(
10350            revalidate(&index, &unsupported, &mut |_| {}),
10351            Err(Error::UnsupportedScanConfig(_))
10352        ));
10353    }
10354
10355    #[test]
10356    fn revalidation_uses_the_same_depth_boundary_as_cold_scan() {
10357        let dir = sample_tree();
10358        let config = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
10359        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
10360        write_file(&dir.path().join("src/added-after-scan.txt"), b"new");
10361
10362        let mut observations = Vec::new();
10363        revalidate(&index, &config, &mut |observation| observations.push(observation))
10364            .expect("revalidate");
10365        for observation in &observations {
10366            index.apply_ok(observation);
10367        }
10368
10369        assert!(index.lookup(Path::new("src/added-after-scan.txt")).is_none());
10370    }
10371
10372    #[test]
10373    fn zero_depth_revalidation_prunes_cached_root_children() {
10374        let dir = tempfile::tempdir().expect("tempdir");
10375        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
10376        let mut index = Index::new_with_scope(dir.path(), config.scope());
10377        index.apply_baseline_ok(&Observation::new(vec![Op::Upsert {
10378            path: PathBuf::from("stale.txt"),
10379            kind: EntryKind::File,
10380            attrs: Attrs::default(),
10381        }]));
10382
10383        let mut observations = Vec::new();
10384        let report = revalidate(&index, &config, &mut |observation| {
10385            observations.push(observation);
10386        })
10387        .expect("revalidate");
10388        for observation in &observations {
10389            index.apply_ok(observation);
10390        }
10391
10392        assert!(index.is_empty());
10393        assert_eq!(report.dirs_read, 0);
10394    }
10395
10396    #[test]
10397    fn zero_depth_applying_reconciliation_prunes_cached_root_children() {
10398        let dir = tempfile::tempdir().expect("tempdir");
10399        let config = ScanConfig { max_depth: Some(0), ..ScanConfig::default() };
10400        let mut index = Index::new_with_scope(dir.path(), config.scope());
10401        index.apply_baseline_ok(&Observation::new(vec![Op::Upsert {
10402            path: PathBuf::from("stale.txt"),
10403            kind: EntryKind::File,
10404            attrs: Attrs::default(),
10405        }]));
10406
10407        let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
10408
10409        assert!(index.is_empty());
10410        assert_eq!(report.scan.dirs_read, 0);
10411    }
10412
10413    #[test]
10414    fn filesystem_boundary_is_part_of_the_shared_descent_policy() {
10415        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
10416        let attrs = Attrs { dev: 22, ..Attrs::default() };
10417        assert!(!should_descend(EntryKind::Dir, attrs, 0, 11, &config));
10418        assert!(should_descend(EntryKind::Dir, Attrs { dev: 11, ..attrs }, 0, 11, &config,));
10419    }
10420
10421    /// A cold scan's index records its own pass start, the stamp a snapshot of it writes:
10422    /// never earlier than an instant taken before the scan, so it is not a stale or zero
10423    /// stamp, and never later than one taken after it. The builder constructs the index,
10424    /// and so takes the stamp, before the walk begins.
10425    #[test]
10426    fn a_cold_scan_stamps_its_own_pass_start() {
10427        let nanos = || {
10428            i64::try_from(
10429                std::time::SystemTime::now()
10430                    .duration_since(std::time::UNIX_EPOCH)
10431                    .expect("after the epoch")
10432                    .as_nanos(),
10433            )
10434            .expect("nanoseconds")
10435        };
10436        let dir = sample_tree();
10437        let before = nanos();
10438        let (index, report) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10439        let after = nanos();
10440        assert!(report.is_complete() && report.entries > 0, "{report:?}");
10441        let stamp = index.writing_pass_started_at_ns();
10442        assert!(before <= stamp && stamp <= after, "{before} <= {stamp} <= {after}");
10443    }
10444
10445    #[test]
10446    fn scanning_a_file_is_an_error_not_a_panic() {
10447        let dir = sample_tree();
10448        let err = scan_into_index(&dir.path().join("a.txt"), &ScanConfig::default());
10449        assert!(err.is_err());
10450    }
10451
10452    #[test]
10453    fn deltas_arrive_in_batches_of_the_configured_size() {
10454        let dir = tempfile::tempdir().expect("tempdir");
10455        for i in 0..25 {
10456            write_file(&dir.path().join(format!("f{i}.txt")), b"x");
10457        }
10458        let config = ScanConfig { batch_size: 10, ..ScanConfig::default() };
10459        let mut sizes = Vec::new();
10460        scan(dir.path(), &config, &mut |d| sizes.push(d.len())).expect("scan");
10461
10462        assert!(sizes.len() >= 3, "expected several batches, got {sizes:?}");
10463        assert!(sizes.iter().all(|&n| n <= 10));
10464        assert_eq!(sizes.iter().sum::<usize>(), 25);
10465    }
10466
10467    #[test]
10468    fn invalid_batch_sizes_are_rejected_before_allocation() {
10469        let zero = ScanConfig { batch_size: 0, ..ScanConfig::default() };
10470        let unbounded = ScanConfig { batch_size: usize::MAX, ..ScanConfig::default() };
10471
10472        assert!(matches!(zero.validate(), Err(Error::UnsupportedScanConfig(_))));
10473        assert!(matches!(unbounded.validate(), Err(Error::UnsupportedScanConfig(_))));
10474    }
10475
10476    #[test]
10477    fn reconciliation_scope_budget_publishes_before_returning_retry() {
10478        let directory = tempfile::tempdir().expect("root");
10479        let config = ScanConfig::default();
10480        let (index, _) = scan_into_index(directory.path(), &config).expect("scan root-only tree");
10481        let handle = IndexHandle::new(index);
10482        let mut started = false;
10483        let mut published_partial = false;
10484        let report = reconcile_handle(&handle, &config, &mut |commit| {
10485            if !started {
10486                started = true;
10487                // No entries are added: distinct absent children must not grow history
10488                // for the paused root pass without bound.
10489                for child in ["missing-a", "missing-b", "missing-c"] {
10490                    let nested = reconcile_subtree_handle(&handle, Path::new(child), &config, &mut |_| {}).expect("newer absent scope");
10491                    assert!(nested.is_complete());
10492                }
10493            }
10494            if commit.state.iter().any(|state| matches!(state,
10495                crate::StateTransition::IndexState { current, .. }
10496                    if current.coverage == crate::Coverage::Partial(crate::CoverageReason::Inaccessible))) {
10497                assert_eq!(handle.read_with(Index::state).expect("coherent state").coverage,
10498                    crate::Coverage::Partial(crate::CoverageReason::Inaccessible));
10499                published_partial = true;
10500            }
10501        }).expect("interrupted pass returns retryable report");
10502        assert!(!report.is_complete());
10503        assert!(report.retry_required);
10504        assert!(published_partial, "the transition precedes the caller's retry result");
10505        let recovered = reconcile_handle(&handle, &config, &mut |_| {}).expect("retry");
10506        assert!(recovered.is_complete());
10507        assert_eq!(
10508            handle.read_with(Index::state).expect("recovered state").coverage,
10509            crate::Coverage::Complete
10510        );
10511    }
10512
10513    #[test]
10514    fn stale_arbitration_keeps_a_reconciliation_incomplete() {
10515        let report = ReconcileReport {
10516            scan: ScanReport::default(),
10517            apply: ApplyStats { stale: 1, ..ApplyStats::default() },
10518            observations: 1,
10519            ..ReconcileReport::default()
10520        };
10521
10522        assert!(!report.is_complete());
10523    }
10524
10525    #[test]
10526    fn portable_system_time_conversion_preserves_pre_epoch_values() {
10527        let before_epoch = std::time::UNIX_EPOCH
10528            // Windows timestamps have 100 ns granularity, so use a duration that every
10529            // supported platform can represent without rounding back to the epoch.
10530            .checked_sub(std::time::Duration::from_secs(1))
10531            .expect("represent pre-epoch fixture");
10532
10533        assert_eq!(system_time_ns(before_epoch), -1_000_000_000);
10534        assert_eq!(system_time_ns(std::time::UNIX_EPOCH), 0);
10535    }
10536
10537    #[cfg(not(unix))]
10538    #[test]
10539    fn one_filesystem_fails_when_device_identity_is_unavailable() {
10540        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
10541
10542        assert!(matches!(config.validate(), Err(Error::UnsupportedScanConfig(_))));
10543    }
10544
10545    #[test]
10546    fn revalidate_is_a_no_op_against_an_unchanged_tree() {
10547        let dir = sample_tree();
10548        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10549        let before = index.total();
10550
10551        let mut deltas = Vec::new();
10552        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
10553        let mut unchanged = 0;
10554        for delta in &deltas {
10555            unchanged += index.apply_ok(delta).unchanged;
10556        }
10557
10558        assert_eq!(unchanged, 5, "3 files + 2 dirs all already known");
10559        assert_eq!(index.total(), before);
10560    }
10561
10562    #[cfg(windows)]
10563    #[test]
10564    fn windows_reconcile_detects_same_size_rewrite_with_preserved_mtime() {
10565        let root = tempfile::tempdir().expect("tempdir");
10566        let path = root.path().join("same.txt");
10567        write_file(&path, b"first");
10568        let modified = fs::metadata(&path).expect("metadata").modified().expect("mtime");
10569        let (mut index, _) =
10570            scan_into_index(root.path(), &ScanConfig::default()).expect("initial scan");
10571        let before = *index.attrs(Path::new("same.txt")).expect("initial attrs");
10572
10573        // NTFS stamps change time from the system clock, which advances in ticks of up to
10574        // 15.625 ms, so a rewrite stamped in the same tick as the first write is
10575        // indistinguishable from it. Wait for the clock to leave that tick rather than for
10576        // a fixed interval; the precise clock `SystemTime` reads runs at most one tick ahead.
10577        let stamped = std::time::UNIX_EPOCH
10578            + std::time::Duration::from_nanos(
10579                u64::try_from(before.ctime_ns).expect("change time after the epoch"),
10580            );
10581        while std::time::SystemTime::now() <= stamped + std::time::Duration::from_millis(20) {
10582            std::thread::sleep(std::time::Duration::from_millis(5));
10583        }
10584        write_file(&path, b"other");
10585        File::options()
10586            .write(true)
10587            .open(&path)
10588            .expect("open rewritten file")
10589            .set_times(std::fs::FileTimes::new().set_modified(modified))
10590            .expect("restore mtime");
10591        reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
10592
10593        let after = *index.attrs(Path::new("same.txt")).expect("rewritten attrs");
10594        assert_eq!((after.size, after.mtime_ns), (before.size, before.mtime_ns));
10595        assert_ne!(after.ctime_ns, before.ctime_ns, "change time detects the rewrite");
10596        assert_ne!(after.fingerprint(), before.fingerprint());
10597    }
10598
10599    #[cfg(windows)]
10600    #[test]
10601    fn windows_reconcile_detects_path_identity_replacement() {
10602        let root = tempfile::tempdir().expect("tempdir");
10603        let path = root.path().join("replace.txt");
10604        let displaced = root.path().join("displaced.txt");
10605        write_file(&path, b"first");
10606        let modified = fs::metadata(&path).expect("metadata").modified().expect("mtime");
10607        let (mut index, _) =
10608            scan_into_index(root.path(), &ScanConfig::default()).expect("initial scan");
10609        let before = *index.attrs(Path::new("replace.txt")).expect("initial attrs");
10610
10611        fs::rename(&path, &displaced).expect("retain old file identity");
10612        write_file(&path, b"other");
10613        File::options()
10614            .write(true)
10615            .open(&path)
10616            .expect("open replacement")
10617            .set_times(std::fs::FileTimes::new().set_modified(modified))
10618            .expect("restore mtime");
10619        reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
10620
10621        let after = *index.attrs(Path::new("replace.txt")).expect("replacement attrs");
10622        assert_eq!((after.size, after.mtime_ns), (before.size, before.mtime_ns));
10623        assert_ne!(
10624            (after.dev, after.inode),
10625            (before.dev, before.inode),
10626            "volume serial and file index identify the replacement"
10627        );
10628        assert_ne!(after.fingerprint(), before.fingerprint());
10629    }
10630
10631    #[test]
10632    fn direct_reconciliation_counts_unchanged_entries_and_publishes_state_commits() {
10633        let dir = sample_tree();
10634        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
10635        let before_total = index.total();
10636        let before_clock = index.clock();
10637        let mut commits = Vec::new();
10638
10639        let report = reconcile(&mut index, &ScanConfig::default(), &mut |commit| {
10640            commits.push(commit.clone());
10641        })
10642        .expect("reconcile");
10643
10644        assert!(report.is_complete());
10645        assert_eq!(report.apply.unchanged, 5, "3 files + 2 dirs all already known");
10646        assert_eq!(commits.len(), 2);
10647        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
10648        assert_eq!(
10649            index.clock(),
10650            crate::Clock(before_clock.0 + 2),
10651            "start and finish are state commits"
10652        );
10653        let commits = index.since(before_clock).commits;
10654        assert_eq!(commits.len(), 2);
10655        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
10656        assert_eq!(index.total(), before_total);
10657    }
10658
10659    #[test]
10660    fn parallel_and_serial_reconciliation_produce_the_same_index() {
10661        let dir = sample_tree();
10662        let portable_config =
10663            ScanConfig { threads: Some(1), batch_size: 2, ..ScanConfig::default() };
10664        let (mut portable, _) =
10665            scan_into_index(dir.path(), &portable_config).expect("portable baseline");
10666        let mut bulk = portable.clone();
10667
10668        fs::remove_file(dir.path().join("a.txt")).expect("remove file");
10669        write_file(&dir.path().join("src/main.rs"), b"fn main() { much longer }");
10670        write_file(&dir.path().join("src/added.md"), b"new file");
10671
10672        let portable_report = reconcile(&mut portable, &portable_config, &mut |_| {})
10673            .expect("portable reconciliation");
10674        let bulk_config = ScanConfig { threads: Some(2), ..portable_config };
10675        let bulk_report =
10676            reconcile(&mut bulk, &bulk_config, &mut |_| {}).expect("bulk reconciliation");
10677
10678        assert!(portable_report.is_complete());
10679        assert!(bulk_report.is_complete());
10680        assert_eq!(bulk_report.scan.attribution, WalkAttribution::default());
10681        assert_eq!(bulk_report.scan.entries, portable_report.scan.entries);
10682        assert_eq!(bulk_report.scan.dirs_read, portable_report.scan.dirs_read);
10683        assert_eq!(bulk_report.apply, portable_report.apply);
10684        assert_eq!(index_fingerprint(&bulk), index_fingerprint(&portable));
10685        assert_eq!(bulk.total(), portable.total());
10686    }
10687
10688    #[test]
10689    fn parallel_reconciliation_workers_publish_directory_counters() {
10690        let _serial = crate::counters::test_serial();
10691        let dir = sample_tree();
10692        let baseline = ScanConfig { threads: Some(1), ..ScanConfig::default() };
10693        let (mut index, scan) = scan_into_index(dir.path(), &baseline).expect("baseline scan");
10694        assert!(scan.is_complete());
10695
10696        crate::counters::enable(true);
10697        crate::counters::reset();
10698        let config = ScanConfig { threads: Some(4), ..baseline };
10699        let report = reconcile(&mut index, &config, &mut |_| {}).expect("reconciliation");
10700        crate::counters::flush_thread();
10701        let counts = crate::counters::snapshot();
10702        crate::counters::reset();
10703        crate::counters::enable(false);
10704
10705        assert!(report.is_complete());
10706        assert!(
10707            counts.dir_opens >= report.scan.dirs_read,
10708            "parallel worker directory opens were folded: {counts:?}, report={report:?}"
10709        );
10710    }
10711
10712    #[test]
10713    fn parallel_reconciliation_matches_serial_across_structural_transitions() {
10714        for max_depth in [None, Some(1), Some(2)] {
10715            for order in [ScanOrder::BreadthFirst, ScanOrder::DepthFirst] {
10716                let dir = reconciliation_transition_tree();
10717                let reference_config = ScanConfig {
10718                    order,
10719                    max_depth,
10720                    threads: Some(1),
10721                    batch_size: 2,
10722                    ..ScanConfig::default()
10723                };
10724                let (baseline, baseline_report) =
10725                    scan_into_index(dir.path(), &reference_config).expect("baseline scan");
10726                assert!(baseline_report.is_complete());
10727                mutate_reconciliation_transition_tree(dir.path());
10728
10729                let mut serial = baseline.clone();
10730                let mut serial_commits = Vec::new();
10731                let serial_report = reconcile(&mut serial, &reference_config, &mut |commit| {
10732                    serial_commits.push(commit.clone());
10733                })
10734                .expect("serial reconciliation");
10735                let (fresh, fresh_report) =
10736                    scan_into_index(dir.path(), &reference_config).expect("fresh oracle");
10737                assert!(serial_report.is_complete(), "serial {order:?}/{max_depth:?}");
10738                assert!(fresh_report.is_complete(), "fresh {order:?}/{max_depth:?}");
10739                assert_eq!(
10740                    index_fingerprint(&serial),
10741                    index_fingerprint(&fresh),
10742                    "serial did not converge to a fresh scan for {order:?}/{max_depth:?}"
10743                );
10744
10745                for workers in [2, 4] {
10746                    let mut parallel = baseline.clone();
10747                    let config = ScanConfig { threads: Some(workers), ..reference_config.clone() };
10748                    let mut parallel_commits = Vec::new();
10749                    let report = reconcile(&mut parallel, &config, &mut |commit| {
10750                        parallel_commits.push(commit.clone());
10751                    })
10752                    .expect("parallel reconciliation");
10753                    let context = format!("{order:?}/{max_depth:?}/{workers} workers");
10754
10755                    assert!(report.is_complete(), "{context}: unexpected partial report");
10756                    assert_eq!(report.scan.entries, serial_report.scan.entries, "{context}");
10757                    assert_eq!(report.scan.dirs_read, serial_report.scan.dirs_read, "{context}");
10758                    assert_eq!(report.apply, serial_report.apply, "{context}");
10759                    assert_eq!(
10760                        effective_ops(&parallel_commits),
10761                        effective_ops(&serial_commits),
10762                        "{context}: effective delta differs"
10763                    );
10764                    assert_eq!(
10765                        index_fingerprint(&parallel),
10766                        index_fingerprint(&serial),
10767                        "{context}: final index differs"
10768                    );
10769                    let (parallel_total, serial_total) = (parallel.total(), serial.total());
10770                    assert_eq!(
10771                        (
10772                            parallel_total.files,
10773                            parallel_total.dirs,
10774                            parallel_total.bytes,
10775                            parallel_total.allocated,
10776                            parallel_total.newest_mtime_ns,
10777                        ),
10778                        (
10779                            serial_total.files,
10780                            serial_total.dirs,
10781                            serial_total.bytes,
10782                            serial_total.allocated,
10783                            serial_total.newest_mtime_ns,
10784                        ),
10785                        "{context}: roll-up differs"
10786                    );
10787                    assert_eq!(
10788                        parallel_total.by_ext, serial_total.by_ext,
10789                        "{context}: extension roll-up differs"
10790                    );
10791                }
10792            }
10793        }
10794    }
10795
10796    #[cfg(unix)]
10797    #[test]
10798    fn revalidation_metadata_errors_do_not_delete_enumerated_entries() {
10799        use std::os::unix::fs::PermissionsExt;
10800
10801        if !crate::test_support::require_permission_bits() {
10802            return;
10803        }
10804
10805        let dir = sample_tree();
10806        let config = ScanConfig::default();
10807        let (mut index, baseline_report) =
10808            scan_into_index(dir.path(), &config).expect("baseline scan");
10809        assert!(baseline_report.is_complete());
10810        let before = index_fingerprint(&index);
10811        let original_permissions = fs::metadata(dir.path()).expect("root metadata").permissions();
10812
10813        fs::set_permissions(dir.path(), fs::Permissions::from_mode(0o400))
10814            .expect("remove search permission");
10815        let mut observations = Vec::new();
10816        let outcome = revalidate(&index, &config, &mut |observation| {
10817            observations.push(observation);
10818        });
10819        fs::set_permissions(dir.path(), original_permissions).expect("restore permissions");
10820
10821        let report = outcome.expect("operational metadata errors are a partial report");
10822        assert!(!report.errors.is_empty(), "the fixture did not induce metadata errors");
10823        for observation in &observations {
10824            index.apply_ok(observation);
10825        }
10826        assert_eq!(index_fingerprint(&index), before);
10827        assert!(index.attrs(Path::new("a.txt")).is_some(), "existing entry was removed");
10828    }
10829
10830    #[cfg(unix)]
10831    #[test]
10832    fn reconciliation_metadata_errors_drop_unverified_entries_like_a_cold_scan() {
10833        use std::os::unix::fs::PermissionsExt;
10834
10835        if !crate::test_support::require_permission_bits() {
10836            return;
10837        }
10838
10839        // macOS's parallel path may satisfy the whole directory through
10840        // getattrlistbulk even without search permission. One worker pins the portable
10841        // fallback there; Linux also exercises the parallel portable worker.
10842        let worker_counts = if cfg!(target_os = "macos") { vec![1] } else { vec![1, 2] };
10843        for workers in worker_counts {
10844            let dir = sample_tree();
10845            let config =
10846                ScanConfig { threads: Some(workers), batch_size: 2, ..ScanConfig::default() };
10847            let (mut index, baseline_report) =
10848                scan_into_index(dir.path(), &config).expect("baseline scan");
10849            assert!(baseline_report.is_complete());
10850            let before = index_fingerprint(&index);
10851            let original_permissions =
10852                fs::metadata(dir.path()).expect("root metadata").permissions();
10853
10854            // Reading names requires read permission; looking up their metadata also
10855            // requires search permission. This makes enumeration succeed and each
10856            // metadata lookup fail, the boundary where an encountered name used to be
10857            // misclassified as a deletion.
10858            fs::set_permissions(dir.path(), fs::Permissions::from_mode(0o400))
10859                .expect("remove search permission");
10860            let outcome = reconcile(&mut index, &config, &mut |_| {});
10861            let cold = scan_into_index(dir.path(), &config);
10862            fs::set_permissions(dir.path(), original_permissions).expect("restore permissions");
10863
10864            let report = outcome.expect("operational metadata errors are a partial report");
10865            let (cold, cold_report) = cold.expect("cold partial scan");
10866            assert!(!report.scan.errors.is_empty(), "the fixture did not induce metadata errors");
10867            assert!(!report.is_complete());
10868            assert!(!cold_report.is_complete());
10869            assert_eq!(index_fingerprint(&index), index_fingerprint(&cold));
10870            assert!(
10871                index_fingerprint(&index).is_empty(),
10872                "neither warm nor cold may retain attributes it could not verify"
10873            );
10874            assert_eq!(index.directory_complete(Path::new("")), Some(false));
10875            assert_eq!(cold.directory_complete(Path::new("")), Some(false));
10876            assert!(!before.is_empty(), "the fixture began with retained facts");
10877        }
10878    }
10879
10880    #[cfg(unix)]
10881    #[test]
10882    fn failed_listing_withdraws_retained_completeness_and_recovers() {
10883        use std::os::unix::fs::PermissionsExt;
10884
10885        if !crate::test_support::require_permission_bits() {
10886            return;
10887        }
10888        for workers in [1, 2] {
10889            let dir = tempfile::tempdir().expect("root");
10890            write_file(&dir.path().join("ancestor/blocked/unknown.txt"), b"unknown");
10891            write_file(&dir.path().join("healthy/known.txt"), b"known");
10892            let config = ScanConfig { threads: Some(workers), ..ScanConfig::default() };
10893            let (mut warm, baseline) = scan_into_index(dir.path(), &config).expect("baseline");
10894            assert!(baseline.is_complete());
10895            let blocked = dir.path().join("ancestor/blocked");
10896            fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny listing");
10897            let probe = fs::read_dir(&blocked);
10898            let mut commits = Vec::new();
10899            let refreshed =
10900                reconcile(&mut warm, &config, &mut |commit| commits.push(commit.clone()));
10901            let cold = scan_into_index(dir.path(), &config);
10902            fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore");
10903            assert_eq!(
10904                probe.expect_err("real denied listing").kind(),
10905                std::io::ErrorKind::PermissionDenied
10906            );
10907            assert!(!refreshed.expect("partial refresh").is_complete());
10908            let (cold, report) = cold.expect("partial cold scan");
10909            assert!(!report.is_complete());
10910            for index in [&warm, &cold] {
10911                for path in ["", "ancestor", "healthy"] {
10912                    assert_eq!(
10913                        index.directory_complete(Path::new(path)),
10914                        Some(true),
10915                        "workers={workers}: {path}"
10916                    );
10917                }
10918                assert_eq!(index.directory_complete(Path::new("ancestor/blocked")), Some(false));
10919                assert_eq!(index.freshness_at(Path::new("ancestor")), crate::Freshness::Partial);
10920            }
10921            assert!(commits.iter().flat_map(|commit| &commit.state).any(|state| matches!(state,
10922                crate::StateTransition::IndexState { current, .. } if current.coverage != crate::Coverage::Complete
10923            )), "failure is published");
10924            assert!(reconcile(&mut warm, &config, &mut |_| {}).expect("recovery").is_complete());
10925            assert_eq!(warm.directory_complete(Path::new("ancestor/blocked")), Some(true));
10926            assert_eq!(warm.state().coverage, crate::Coverage::Complete);
10927        }
10928    }
10929
10930    #[test]
10931    fn unreadable_directory_warm_answer_matches_cold_verified_tree() {
10932        for workers in [1, 2] {
10933            let dir = tempfile::tempdir().expect("tempdir");
10934            write_file(&dir.path().join("blocked/old.txt"), b"old");
10935            write_file(&dir.path().join("verified.txt"), b"verified");
10936            let config = ScanConfig { threads: Some(workers), ..ScanConfig::default() };
10937            let (mut warm, baseline) = scan_into_index(dir.path(), &config).expect("baseline");
10938            assert!(baseline.is_complete());
10939
10940            let blocked = dir.path().join("blocked").canonicalize().expect("blocked path");
10941            let hook = install_child_metadata_hook(dir.path(), move |path| {
10942                (path == blocked)
10943                    .then(|| std::io::Error::from(std::io::ErrorKind::PermissionDenied))
10944            });
10945            let warm_report = reconcile(&mut warm, &config, &mut |_| {}).expect("warm partial");
10946            let (cold, cold_report) = scan_into_index(dir.path(), &config).expect("cold partial");
10947            drop(hook);
10948
10949            assert!(!warm_report.is_complete(), "workers={workers}");
10950            assert!(!cold_report.is_complete(), "workers={workers}");
10951            assert!(warm.lookup(Path::new("blocked")).is_none(), "workers={workers}");
10952            assert!(warm.lookup(Path::new("blocked/old.txt")).is_none(), "workers={workers}");
10953            assert_eq!(index_fingerprint(&warm), index_fingerprint(&cold), "workers={workers}");
10954            assert!(warm.lookup(Path::new("verified.txt")).is_some(), "workers={workers}");
10955        }
10956    }
10957
10958    #[test]
10959    fn deferred_change_overflow_retries_without_applying_a_partial_wave() {
10960        let dir = sample_tree();
10961        let config = ScanConfig { threads: Some(2), batch_size: 2, ..ScanConfig::default() };
10962        let (mut index, _) = scan_into_index(dir.path(), &config).expect("baseline");
10963        let before = index_fingerprint(&index);
10964
10965        fs::remove_file(dir.path().join("a.txt")).expect("remove file");
10966        write_file(&dir.path().join("added.md"), b"new file");
10967        write_file(&dir.path().join("src/main.rs"), b"fn main() { much longer }");
10968
10969        let root = index.root_path().to_path_buf();
10970        let root_meta = {
10971            crate::counters::bump(|c| c.stats += 1);
10972            fs::symlink_metadata(&root)
10973        }
10974        .expect("root metadata");
10975        let mut commits = Vec::new();
10976        let outcome = reconcile_direct_parallel(
10977            &mut index,
10978            &root,
10979            root_device(&root, &root_meta).expect("root device"),
10980            &config,
10981            1,
10982            &mut |commit| commits.push(commit.clone()),
10983        )
10984        .expect("parallel attempt");
10985        let DirectParallelOutcome::RetrySerial { prefix, remaining } = outcome else {
10986            panic!("the deliberately tiny deferred budget must trigger the retry");
10987        };
10988
10989        assert_eq!(prefix.apply, ApplyStats::default());
10990        assert_eq!(remaining, VecDeque::from([(PathBuf::new(), 0)]));
10991        assert!(commits.is_empty());
10992        assert_eq!(index_fingerprint(&index), before);
10993
10994        let serial = ScanConfig { threads: Some(1), ..config };
10995        let report = reconcile(&mut index, &serial, &mut |_| {}).expect("serial retry");
10996        let (expected, expected_report) = scan_into_index(dir.path(), &serial).expect("oracle");
10997        assert!(report.is_complete());
10998        assert!(expected_report.is_complete());
10999        assert_eq!(index_fingerprint(&index), index_fingerprint(&expected));
11000        assert_eq!(index.total().files, expected.total().files);
11001        assert_eq!(index.total().dirs, expected.total().dirs);
11002        assert_eq!(index.total().bytes, expected.total().bytes);
11003        assert_eq!(index.total().allocated, expected.total().allocated);
11004        assert_eq!(index.total().newest_mtime_ns, expected.total().newest_mtime_ns);
11005        assert_eq!(index.total().by_ext, expected.total().by_ext);
11006    }
11007
11008    #[test]
11009    fn late_overflow_resumes_without_double_counting_completed_waves() {
11010        let dir = tempfile::tempdir().expect("tempdir");
11011        // The root wave discovers more than one full wave of directories. A change in
11012        // the second wave then forces the serial fallback only after the first wave's
11013        // unchanged entries have already been counted.
11014        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
11015            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), b"unchanged");
11016        }
11017        let parallel = ScanConfig { threads: Some(2), ..ScanConfig::default() };
11018        let (baseline, _) = scan_into_index(dir.path(), &parallel).expect("baseline");
11019        let mut candidate = baseline.clone();
11020        let mut serial_oracle = baseline;
11021
11022        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
11023            write_file(
11024                &dir.path().join(format!("d{directory:04}/file.txt")),
11025                b"changed after the first wave",
11026            );
11027        }
11028
11029        let candidate_report = reconcile_target_inner(
11030            &mut ReconcileTarget::Direct(&mut candidate),
11031            Path::new(""),
11032            0,
11033            &parallel,
11034            0,
11035            &mut |_| {},
11036        )
11037        .expect("late-overflow reconciliation");
11038        let serial = ScanConfig { threads: Some(1), ..parallel };
11039        let oracle_report =
11040            reconcile(&mut serial_oracle, &serial, &mut |_| {}).expect("serial oracle");
11041
11042        assert_eq!(candidate_report.apply, oracle_report.apply);
11043        assert_eq!(candidate_report.scan.entries, oracle_report.scan.entries);
11044        assert_eq!(candidate_report.scan.dirs_read, oracle_report.scan.dirs_read);
11045        assert_eq!(index_fingerprint(&candidate), index_fingerprint(&serial_oracle));
11046    }
11047
11048    /// The four counts a walk reports, in the order [`crate::ProgressSnapshot`] shows them.
11049    fn walked(report: &ScanReport) -> (u64, u64, u64, u64) {
11050        (report.dirs_read, report.files_walked, report.bytes_walked, report.allocated_walked)
11051    }
11052
11053    fn reported(progress: &crate::Progress) -> (u64, u64, u64, u64) {
11054        let snapshot = progress.snapshot();
11055        (snapshot.directories, snapshot.files, snapshot.bytes, snapshot.allocated)
11056    }
11057
11058    /// Each walker is a separate loop with its own reporting sites, so each is checked:
11059    /// the detached cold walk, the streaming walk, the transient summary fold, the
11060    /// reference revalidation, exclusive reconciliation serial and in parallel waves,
11061    /// and shared-handle reconciliation. The tree is wide enough that every parallel
11062    /// walker claims several chunks and the small batch size fills several batches, so a
11063    /// walker that reported only its final state would still fail on the counts a
11064    /// mid-walk chunk added twice or not at all.
11065    #[test]
11066    fn every_walker_reports_exactly_what_its_report_counts() {
11067        let dir = tempfile::tempdir().expect("tempdir");
11068        for directory in 0..12 {
11069            for file in 0..5 {
11070                write_file(
11071                    &dir.path().join(format!("d{directory}/f{file}.txt")),
11072                    &vec![b'x'; directory * 5 + file + 1],
11073                );
11074            }
11075        }
11076        for threads in [1, 4] {
11077            let context = format!("threads={threads}");
11078            let progress = crate::Progress::new();
11079            let cold_config = ScanConfig {
11080                threads: Some(threads),
11081                batch_size: 4,
11082                progress: Some(progress.clone()),
11083                ..ScanConfig::default()
11084            };
11085            let (mut index, cold) = scan_into_index(dir.path(), &cold_config).expect("cold scan");
11086            assert_eq!(cold.dirs_read, 13, "{context}: the root and twelve children");
11087            assert_eq!(
11088                progress.snapshot().phase,
11089                crate::ProgressPhase::Indexing,
11090                "{context}: the detached walk ends by assembling the index"
11091            );
11092            assert_eq!(reported(&progress), walked(&cold), "{context}: detached cold walk");
11093
11094            let progress = crate::Progress::new();
11095            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
11096            let streamed = scan(dir.path(), &config, &mut |_| {}).expect("streaming scan");
11097            assert_eq!(walked(&streamed), walked(&cold), "{context}");
11098            assert_eq!(reported(&progress), walked(&streamed), "{context}: streaming walk");
11099
11100            let progress = crate::Progress::new();
11101            let config = ScanConfig {
11102                read_controls: false,
11103                progress: Some(progress.clone()),
11104                ..cold_config.clone()
11105            };
11106            let folded = scan_summary_fold(dir.path(), &config, &mut |_| {}).expect("fold");
11107            assert_eq!(walked(&folded), walked(&cold), "{context}");
11108            assert_eq!(reported(&progress), walked(&folded), "{context}: summary fold");
11109
11110            let progress = crate::Progress::new();
11111            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
11112            let revalidated = revalidate(&index, &config, &mut |_| {}).expect("revalidate");
11113            assert_eq!(progress.snapshot().phase, crate::ProgressPhase::Revalidating, "{context}");
11114            assert_eq!(reported(&progress), walked(&revalidated), "{context}: revalidate");
11115
11116            // Changes, so reconciliation defers and applies operations rather than
11117            // discarding every entry as unchanged.
11118            write_file(&dir.path().join(format!("d0/new{threads}.txt")), b"added");
11119            fs::remove_file(dir.path().join(format!("d1/f{}.txt", threads - 1))).expect("remove");
11120            write_file(&dir.path().join("d2/f0.txt"), &vec![b'y'; 40 + threads]);
11121            // An independent walk of the changed tree. The handle and the reconcile's own
11122            // report both come from the walker's counts, so agreeing with each other
11123            // would not show that the walker counted anything; agreeing with this does.
11124            let (_, fresh) = scan_into_index(dir.path(), &ScanConfig::default()).expect("fresh");
11125            let progress = crate::Progress::new();
11126            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config.clone() };
11127            let reconciled = reconcile(&mut index, &config, &mut |_| {}).expect("reconcile");
11128            assert!(reconciled.apply.mutated(), "{context}: the changes were applied");
11129            assert_eq!(progress.snapshot().phase, crate::ProgressPhase::Revalidating, "{context}");
11130            assert_eq!(reported(&progress), walked(&reconciled.scan), "{context}: reconcile");
11131            assert_eq!(walked(&reconciled.scan), walked(&fresh), "{context}: the whole tree");
11132
11133            let handle = crate::IndexHandle::new(index);
11134            let progress = crate::Progress::new();
11135            let config = ScanConfig { progress: Some(progress.clone()), ..cold_config };
11136            let shared = reconcile_handle(&handle, &config, &mut |_| {}).expect("shared");
11137            assert_eq!(reported(&progress), walked(&shared.scan), "{context}: shared handle");
11138        }
11139    }
11140
11141    /// A wave that overflows its deferred-operation budget is thrown away and rewalked
11142    /// serially. The report counts each directory once, as the logical pass does, and
11143    /// progress counts the wave's reads both times, as the filesystem did them: work
11144    /// done, not the answer. This is the one walker relation that is not equality, and
11145    /// the difference is exactly the rewalked wave.
11146    #[test]
11147    fn progress_counts_a_rewalked_wave_twice_where_the_report_counts_it_once() {
11148        let dir = tempfile::tempdir().expect("tempdir");
11149        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
11150            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), b"unchanged");
11151        }
11152        // A file in the wave that completes, so the serial rewalk has counts of that wave
11153        // to carry forward as already added rather than add again.
11154        write_file(&dir.path().join("root.txt"), b"counted by the wave that completes");
11155        let parallel = ScanConfig { threads: Some(2), ..ScanConfig::default() };
11156        let (mut index, _) = scan_into_index(dir.path(), &parallel).expect("baseline");
11157        // Larger than any filesystem stores inline in the inode, so each copy occupies
11158        // blocks of its own wherever the test runs.
11159        let changed = &vec![b'c'; 8_193];
11160        for directory in 0..=RECONCILE_WAVE_DIRECTORIES {
11161            write_file(&dir.path().join(format!("d{directory:04}/file.txt")), changed);
11162        }
11163
11164        let progress = crate::Progress::new();
11165        let observed = ScanConfig { progress: Some(progress.clone()), ..parallel };
11166        let report = reconcile_target_inner(
11167            &mut ReconcileTarget::Direct(&mut index),
11168            Path::new(""),
11169            0,
11170            &observed,
11171            0,
11172            &mut |_| {},
11173        )
11174        .expect("late-overflow reconciliation");
11175
11176        // The root wave changes nothing and completes; the second wave holds exactly
11177        // one full wave of changed directories, overflows, and is rewalked with the one
11178        // directory the wave left behind.
11179        let rewalked = u64::try_from(RECONCILE_WAVE_DIRECTORIES).expect("fits");
11180        let snapshot = progress.snapshot();
11181        assert_eq!(snapshot.directories, report.scan.dirs_read + rewalked);
11182        assert_eq!(snapshot.files, report.scan.files_walked + rewalked);
11183        assert_eq!(
11184            snapshot.bytes,
11185            report.scan.bytes_walked + rewalked * u64::try_from(changed.len()).expect("fits")
11186        );
11187        let allocated = index.attrs(Path::new("d0000/file.txt")).expect("indexed").allocated;
11188        assert!(allocated > 0, "a file with content occupies blocks");
11189        assert_eq!(snapshot.allocated, report.scan.allocated_walked + rewalked * allocated);
11190    }
11191
11192    /// Several invalidated roots are reconciled one at a time and their reports summed,
11193    /// so every walked count has to survive the sum, allocated bytes included.
11194    #[test]
11195    fn a_multi_root_reconcile_sums_every_walked_count() {
11196        let dir = tempfile::tempdir().expect("tempdir");
11197        for directory in ["a", "b", "c"] {
11198            for file in 0..3 {
11199                write_file(&dir.path().join(format!("{directory}/f{file}.txt")), b"before");
11200            }
11201        }
11202        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11203        for directory in ["a", "b"] {
11204            write_file(&dir.path().join(format!("{directory}/f0.txt")), &vec![b'x'; 5_000]);
11205        }
11206        index.apply_ok(&Observation::new(
11207            ["a", "b"]
11208                .into_iter()
11209                .map(|directory| Op::InvalidateSubtree {
11210                    path: PathBuf::from(directory),
11211                    reason: crate::InvalidateReason::Requested,
11212                })
11213                .collect(),
11214        ));
11215        let report =
11216            reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {}).expect("reconcile");
11217
11218        let attrs: Vec<Attrs> = ["a", "b"]
11219            .into_iter()
11220            .flat_map(|directory| (0..3).map(move |file| format!("{directory}/f{file}.txt")))
11221            .map(|path| *index.attrs(Path::new(&path)).expect("indexed"))
11222            .collect();
11223        assert_eq!(report.scan.files_walked, 6, "the two roots' files, and not c's");
11224        assert_eq!(report.scan.bytes_walked, attrs.iter().map(|attrs| attrs.size).sum::<u64>());
11225        assert_eq!(
11226            report.scan.allocated_walked,
11227            attrs.iter().map(|attrs| attrs.allocated).sum::<u64>()
11228        );
11229    }
11230
11231    #[test]
11232    fn shared_reconciliation_retains_conditional_no_op_arbitration() {
11233        let dir = sample_tree();
11234        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11235        let handle = crate::IndexHandle::new(index);
11236        let before_clock = handle.clock().expect("clock");
11237        let mut commits = Vec::new();
11238
11239        let report = reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
11240            commits.push(commit.clone());
11241        })
11242        .expect("reconcile");
11243
11244        assert!(report.is_complete());
11245        assert_eq!(report.apply.unchanged, 5, "3 files + 2 dirs all already known");
11246        assert_eq!(commits.len(), 2);
11247        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
11248        assert_eq!(
11249            handle.clock().expect("clock"),
11250            crate::Clock(before_clock.0 + 2),
11251            "start and finish are state commits"
11252        );
11253        let commits = handle.since(before_clock).expect("state commits").commits;
11254        assert_eq!(commits.len(), 2);
11255        assert!(commits.iter().all(|commit| commit.changes.is_empty()));
11256    }
11257
11258    #[test]
11259    fn revalidate_detects_additions_edits_and_deletions() {
11260        let dir = sample_tree();
11261        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11262
11263        fs::remove_file(dir.path().join("a.txt")).expect("remove");
11264        write_file(&dir.path().join("src/main.rs"), b"fn main() { longer }");
11265        write_file(&dir.path().join("added.md"), b"new");
11266
11267        let mut deltas = Vec::new();
11268        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
11269        let mut stats = crate::index::ApplyStats::default();
11270        for delta in &deltas {
11271            let s = index.apply_ok(delta);
11272            stats.inserted += s.inserted;
11273            stats.updated += s.updated;
11274            stats.removed += s.removed;
11275        }
11276
11277        assert_eq!(stats.inserted, 1, "added.md");
11278        assert_eq!(stats.updated, 1, "main.rs grew");
11279        assert_eq!(stats.removed, 1, "a.txt is gone");
11280
11281        let total = index.total();
11282        assert_eq!(total.files, 3);
11283        assert_eq!(total.bytes, 20 + 9 + 3);
11284        assert!(!total.by_ext.contains_key(".txt"));
11285        assert_eq!(total.by_ext[".md"].files, 1);
11286    }
11287
11288    #[test]
11289    fn revalidate_removes_a_whole_vanished_directory() {
11290        let dir = sample_tree();
11291        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11292        fs::remove_dir_all(dir.path().join("src")).expect("remove dir");
11293
11294        let mut deltas = Vec::new();
11295        revalidate(&index, &ScanConfig::default(), &mut |d| deltas.push(d)).expect("revalidate");
11296        for delta in &deltas {
11297            index.apply_ok(delta);
11298        }
11299
11300        let total = index.total();
11301        assert_eq!(total.files, 1);
11302        assert_eq!(total.dirs, 0);
11303        assert!(index.lookup(Path::new("src")).is_none());
11304    }
11305
11306    #[test]
11307    fn pending_invalidation_reconciles_the_requested_subtree() {
11308        let dir = sample_tree();
11309        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11310        write_file(&dir.path().join("src/added.rs"), b"new");
11311        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
11312            path: PathBuf::from("src"),
11313            reason: crate::InvalidateReason::Requested,
11314        }]));
11315        assert_eq!(index.freshness_at(Path::new("src")), crate::Freshness::Stale);
11316
11317        let mut applied = Vec::new();
11318        let report = reconcile_pending(&mut index, &ScanConfig::default(), &mut |delta| {
11319            applied.push(delta.clone());
11320        })
11321        .expect("reconcile pending");
11322
11323        assert!(report.is_complete());
11324        assert!(index.lookup(Path::new("src/added.rs")).is_some());
11325        assert_eq!(index.freshness_at(Path::new("src")), crate::Freshness::Fresh);
11326        assert!(index.take_pending_invalidations().is_empty());
11327        assert!(applied.iter().any(|commit| commit_touches(commit, Path::new("src/added.rs"))));
11328    }
11329
11330    /// A retained `.gitignore` reconciled as the root of its own walk re-reads its rules.
11331    /// A file does not descend, so the subtree-root branch was the only place that could
11332    /// read them, and it did not: the table kept `*.log` while the pass reported complete.
11333    #[test]
11334    fn reconciling_a_retained_control_file_as_the_subtree_root_rereads_its_rules() {
11335        let dir = tempfile::tempdir().expect("tempdir");
11336        write_file(&dir.path().join(".gitignore"), b"*.log\n");
11337        write_file(&dir.path().join("a.log"), b"log");
11338        let config = ScanConfig { read_controls: true, ..ScanConfig::default() };
11339        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
11340        assert_eq!(
11341            index.is_ignored(Path::new("a.log")).expect("control state observed"),
11342            Some(true)
11343        );
11344
11345        write_file(&dir.path().join(".gitignore"), b"# nothing is ignored now\n");
11346        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
11347            path: PathBuf::from(".gitignore"),
11348            reason: crate::InvalidateReason::Requested,
11349        }]));
11350        let report = reconcile_pending(&mut index, &config, &mut |_| {}).expect("reconcile");
11351
11352        assert!(report.is_complete(), "{:?}", report.scan.errors);
11353        assert_eq!(index.freshness_at(Path::new(".gitignore")), crate::Freshness::Fresh);
11354        assert_eq!(
11355            index.is_ignored(Path::new("a.log")).expect("control state observed"),
11356            Some(false)
11357        );
11358    }
11359
11360    /// A control file the pass cannot verify contributes neither stale rules nor an entry.
11361    #[cfg(unix)]
11362    #[test]
11363    fn reconciling_an_unreadable_control_file_root_drops_its_rules_and_stays_partial() {
11364        use std::os::unix::fs::PermissionsExt;
11365
11366        if !crate::test_support::require_permission_bits() {
11367            return;
11368        }
11369
11370        let dir = tempfile::tempdir().expect("tempdir");
11371        let control = dir.path().join(".gitignore");
11372        write_file(&control, b"*.log\n");
11373        write_file(&dir.path().join("a.log"), b"log");
11374        let config = ScanConfig { read_controls: true, ..ScanConfig::default() };
11375        let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
11376
11377        write_file(&control, b"# rewritten, then made unreadable\n");
11378        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("chmod");
11379        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
11380            path: PathBuf::from(".gitignore"),
11381            reason: crate::InvalidateReason::Requested,
11382        }]));
11383        let report = reconcile_pending(&mut index, &config, &mut |_| {});
11384        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore");
11385        let report = report.expect("reconcile");
11386
11387        assert!(!report.is_complete());
11388        assert_eq!(report.scan.errors.len(), 1, "{:?}", report.scan.errors);
11389        assert_eq!(index.freshness_at(Path::new(".gitignore")), crate::Freshness::Partial);
11390        assert!(
11391            !index.controls().expect("control state observed").contains(Path::new(".gitignore"))
11392        );
11393        assert_eq!(index.is_ignored(Path::new("a.log")).expect("control state observed"), None);
11394    }
11395
11396    #[cfg(unix)]
11397    #[test]
11398    fn unreadable_control_keeps_new_excluded_file_unknown_and_out_of_analysis() {
11399        use std::os::unix::fs::PermissionsExt;
11400
11401        if !crate::test_support::require_permission_bits() {
11402            return;
11403        }
11404        let dir = tempfile::tempdir().expect("tempdir");
11405        let control = dir.path().join(".gitignore");
11406        write_file(&control, b"*.log\n");
11407        write_file(&dir.path().join("keep.rs"), b"code");
11408        let config = ScanConfig {
11409            population: crate::query::IgnoredEntries::Exclude,
11410            ..ScanConfig::default()
11411        };
11412        let (mut index, cold) = scan_into_index(dir.path(), &config).expect("cold");
11413        assert!(cold.is_complete());
11414        write_file(&dir.path().join("debug.log"), b"must not analyze");
11415        fs::set_permissions(&control, fs::Permissions::from_mode(0o000)).expect("chmod");
11416        let report = reconcile(&mut index, &config, &mut |_| {});
11417        fs::set_permissions(&control, fs::Permissions::from_mode(0o644)).expect("restore");
11418        let report = report.expect("reconcile");
11419        assert!(!report.is_complete());
11420        assert!(index.lookup(Path::new("debug.log")).is_some(), "unknown is retained");
11421        assert_eq!(index.ignored_classification(Path::new("debug.log")), None);
11422        assert!(!index.ignored_classification_complete_below(Path::new("")));
11423        let candidates = index.analysis_candidates(crate::content::AnalysisSet::NONE.with_lines());
11424        assert!(
11425            candidates.iter().all(|candidate| candidate.relative_path != Path::new("debug.log"))
11426        );
11427        let repaired =
11428            reconcile(&mut index, &config, &mut |_| {}).expect("reconcile repaired control");
11429        assert!(repaired.is_complete(), "{:?}", repaired.scan.errors);
11430        assert!(index.ignored_classification_complete_below(Path::new("")));
11431        assert!(index.lookup(Path::new("debug.log")).is_none(), "known ignored file is pruned");
11432    }
11433
11434    #[test]
11435    fn handle_reconciliation_publishes_after_each_delta_is_applied() {
11436        let dir = sample_tree();
11437        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11438        let handle = crate::IndexHandle::new(index);
11439        let reader = handle.clone();
11440        write_file(&dir.path().join("added.md"), b"new");
11441
11442        let mut observed_after_apply = false;
11443        reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
11444            if commit_touches(commit, Path::new("added.md")) {
11445                observed_after_apply =
11446                    reader.kind(Path::new("added.md")).expect("query index").is_some();
11447            }
11448        })
11449        .expect("reconcile handle");
11450
11451        assert!(observed_after_apply);
11452    }
11453
11454    /// Delete `name` under `root` from inside its own metadata lookup, after the listing
11455    /// returned it, on whichever thread performs the lookup.
11456    fn delete_between_listing_and_stat(root: &Path, name: &'static str) -> WalkHookGuard {
11457        install_child_metadata_hook(root, move |path| {
11458            delete_if_named(path, name);
11459            None
11460        })
11461    }
11462
11463    /// As [`delete_between_listing_and_stat`], and every reconciliation listing under `root`
11464    /// also ends in an error, so none of them is complete.
11465    fn delete_between_listing_and_stat_in_a_failing_listing(
11466        root: &Path,
11467        name: &'static str,
11468    ) -> WalkHookGuard {
11469        install_walk_hook(root, move |point| match point {
11470            WalkHookPoint::ChildMetadata(path) => {
11471                delete_if_named(path, name);
11472                None
11473            }
11474            WalkHookPoint::ListingEnd => Some(std::io::Error::other("injected listing error")),
11475            WalkHookPoint::ControlLookup(_) => None,
11476        })
11477    }
11478
11479    fn delete_if_named(path: &Path, name: &str) {
11480        if path.file_name() == Some(OsStr::new(name)) {
11481            fs::remove_file(path).expect("delete between listing and stat");
11482        }
11483    }
11484
11485    /// A name the listing returned that is gone by the time it is stat'd was deleted, on
11486    /// the serial path and on the parallel waves a full-root pass takes by default. The
11487    /// walk removes it; recorded as an error, it would settle as a phantom entry with
11488    /// permanent partial freshness.
11489    #[test]
11490    fn a_child_deleted_between_listing_and_stat_is_removed_rather_than_reported() {
11491        for threads in [1, 4] {
11492            let dir = tempfile::tempdir().expect("tempdir");
11493            write_file(&dir.path().join("keep.txt"), b"keep");
11494            write_file(&dir.path().join("gone.txt"), b"gone");
11495            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
11496            let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
11497            assert!(index.lookup(Path::new("gone.txt")).is_some());
11498
11499            index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
11500                path: PathBuf::new(),
11501                reason: crate::InvalidateReason::Requested,
11502            }]));
11503            let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
11504            let report = reconcile_pending(&mut index, &config, &mut |_| {});
11505            drop(hook);
11506            let report = report.expect("reconcile");
11507
11508            assert!(report.is_complete(), "threads {threads}: {:?}", report.scan.errors);
11509            assert_eq!(report.apply.removed, 1, "threads {threads}");
11510            assert!(index.lookup(Path::new("gone.txt")).is_none(), "threads {threads}");
11511            assert!(index.lookup(Path::new("keep.txt")).is_some(), "threads {threads}");
11512            assert_eq!(index.freshness(), crate::Freshness::Fresh, "threads {threads}");
11513        }
11514    }
11515
11516    /// A vanished control file takes its rules with it on both reconcile paths, even when the
11517    /// rest of its listing fails: the stat's `NotFound` is the evidence. A retained file's
11518    /// rules go with its entry's removal; a hidden-pruned one has no entry to remove, so
11519    /// without its own removal its rules would go on ignoring its siblings.
11520    #[test]
11521    fn a_control_file_deleted_between_listing_and_stat_takes_its_rules_with_it() {
11522        for prune_hidden in [false, true] {
11523            for (threads, listing_fails) in [(1, false), (4, false), (1, true), (4, true)] {
11524                let case = format!(
11525                    "prune hidden {prune_hidden}, threads {threads}, listing fails {listing_fails}"
11526                );
11527                let dir = tempfile::tempdir().expect("tempdir");
11528                write_file(&dir.path().join(".gitignore"), b"*.log\n");
11529                write_file(&dir.path().join("a.log"), b"log");
11530                let config = ScanConfig {
11531                    read_controls: true,
11532                    threads: Some(threads),
11533                    hidden: prune_hidden
11534                        .then(|| std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([""; 0]))),
11535                    ..ScanConfig::default()
11536                };
11537                let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
11538                assert_eq!(
11539                    index.is_ignored(Path::new("a.log")).expect("control state observed"),
11540                    Some(true),
11541                    "{case}"
11542                );
11543
11544                index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
11545                    path: PathBuf::new(),
11546                    reason: crate::InvalidateReason::Requested,
11547                }]));
11548                let hook = if listing_fails {
11549                    delete_between_listing_and_stat_in_a_failing_listing(dir.path(), ".gitignore")
11550                } else {
11551                    delete_between_listing_and_stat(dir.path(), ".gitignore")
11552                };
11553                let report = reconcile_pending(&mut index, &config, &mut |_| {});
11554                drop(hook);
11555                let report = report.expect("reconcile");
11556
11557                assert_eq!(
11558                    report.is_complete(),
11559                    !listing_fails,
11560                    "{case}: {:?}",
11561                    report.scan.errors
11562                );
11563                assert!(index.lookup(Path::new(".gitignore")).is_none(), "{case}");
11564                assert!(
11565                    !index
11566                        .controls()
11567                        .expect("control state observed")
11568                        .contains(Path::new(".gitignore")),
11569                    "{case}"
11570                );
11571                assert_eq!(
11572                    index.is_ignored(Path::new("a.log")).expect("control state observed"),
11573                    Some(false),
11574                    "{case}"
11575                );
11576            }
11577        }
11578    }
11579
11580    /// A cold walk records a name gone by its stat as it records a name the listing never
11581    /// returned: not at all, and without an error that would make the walk partial.
11582    #[test]
11583    fn a_cold_walk_omits_a_child_deleted_between_listing_and_stat() {
11584        for threads in [1, 4] {
11585            let dir = tempfile::tempdir().expect("tempdir");
11586            write_file(&dir.path().join("keep.txt"), b"keep");
11587            write_file(&dir.path().join("gone.txt"), b"gone");
11588            let config = ScanConfig { threads: Some(threads), ..ScanConfig::default() };
11589
11590            let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
11591            let scanned = scan_into_index_via_scanner(dir.path(), &config);
11592            drop(hook);
11593            let (index, report) = scanned.expect("scan");
11594
11595            assert!(report.is_complete(), "threads {threads}: {:?}", report.errors);
11596            assert!(index.lookup(Path::new("gone.txt")).is_none(), "threads {threads}");
11597            assert!(index.lookup(Path::new("keep.txt")).is_some(), "threads {threads}");
11598        }
11599    }
11600
11601    /// Revalidation emits the removal a reconciliation would apply.
11602    #[test]
11603    fn revalidation_removes_a_child_deleted_between_listing_and_stat() {
11604        let dir = tempfile::tempdir().expect("tempdir");
11605        write_file(&dir.path().join("keep.txt"), b"keep");
11606        write_file(&dir.path().join("gone.txt"), b"gone");
11607        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11608
11609        let hook = delete_between_listing_and_stat(dir.path(), "gone.txt");
11610        let mut observations = Vec::new();
11611        let report = revalidate(&index, &ScanConfig::default(), &mut |observation| {
11612            observations.push(observation);
11613        });
11614        drop(hook);
11615        let report = report.expect("revalidate");
11616        for observation in &observations {
11617            index.apply_ok(observation);
11618        }
11619
11620        assert!(report.is_complete(), "{:?}", report.errors);
11621        assert!(index.lookup(Path::new("gone.txt")).is_none());
11622        assert!(index.lookup(Path::new("keep.txt")).is_some());
11623    }
11624
11625    /// Revalidation emits the same rules removal when the rest of the listing fails.
11626    #[test]
11627    fn revalidation_removes_the_rules_of_a_control_file_deleted_in_a_failing_listing() {
11628        for prune_hidden in [false, true] {
11629            let dir = tempfile::tempdir().expect("tempdir");
11630            write_file(&dir.path().join(".gitignore"), b"*.log\n");
11631            write_file(&dir.path().join("a.log"), b"log");
11632            let config = ScanConfig {
11633                read_controls: true,
11634                hidden: prune_hidden
11635                    .then(|| std::sync::Arc::new(crate::HiddenPolicy::prune_hidden([""; 0]))),
11636                ..ScanConfig::default()
11637            };
11638            let (mut index, _) = scan_into_index(dir.path(), &config).expect("scan");
11639            assert_eq!(
11640                index.is_ignored(Path::new("a.log")).expect("control state observed"),
11641                Some(true)
11642            );
11643
11644            let hook =
11645                delete_between_listing_and_stat_in_a_failing_listing(dir.path(), ".gitignore");
11646            let mut observations = Vec::new();
11647            let report = revalidate(&index, &config, &mut |observation| {
11648                observations.push(observation);
11649            });
11650            drop(hook);
11651            let report = report.expect("revalidate");
11652            for observation in &observations {
11653                index.apply_ok(observation);
11654            }
11655
11656            assert!(!report.is_complete(), "prune hidden {prune_hidden}");
11657            assert!(index.lookup(Path::new(".gitignore")).is_none(), "prune hidden {prune_hidden}");
11658            assert!(
11659                !index
11660                    .controls()
11661                    .expect("control state observed")
11662                    .contains(Path::new(".gitignore")),
11663                "prune hidden {prune_hidden}"
11664            );
11665            assert_eq!(
11666                index.is_ignored(Path::new("a.log")).expect("control state observed"),
11667                Some(false),
11668                "prune hidden {prune_hidden}"
11669            );
11670        }
11671    }
11672
11673    #[test]
11674    fn reconciliation_does_not_clear_a_newer_invalidation() {
11675        let dir = sample_tree();
11676        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11677        let handle = crate::IndexHandle::new(index);
11678        write_file(&dir.path().join("added.md"), b"new");
11679
11680        let invalidator = handle.clone();
11681        let mut saw_reconciling = false;
11682        reconcile_handle(&handle, &ScanConfig::default(), &mut |commit| {
11683            if commit_touches(commit, Path::new("added.md")) {
11684                saw_reconciling =
11685                    invalidator.freshness().expect("query") == crate::Freshness::Reconciling;
11686                invalidator
11687                    .apply(&Observation::new(vec![Op::InvalidateSubtree {
11688                        path: PathBuf::new(),
11689                        reason: crate::InvalidateReason::WatchOverflow,
11690                    }]))
11691                    .expect("new invalidation");
11692            }
11693        })
11694        .expect("reconcile handle");
11695
11696        assert!(saw_reconciling);
11697        assert_eq!(handle.freshness().expect("query"), crate::Freshness::Stale);
11698    }
11699
11700    #[test]
11701    fn failed_reconciliation_marks_the_scope_partial() {
11702        let dir = sample_tree();
11703        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11704        fs::remove_dir_all(dir.path()).expect("remove root");
11705
11706        assert!(reconcile(&mut index, &ScanConfig::default(), &mut |_| {}).is_err());
11707        assert_eq!(index.freshness(), crate::Freshness::Partial);
11708    }
11709
11710    #[test]
11711    fn successful_subtree_retry_restores_complete_root_coverage() {
11712        let dir = tempfile::tempdir().expect("tempdir");
11713        write_file(&dir.path().join("blocked/known.txt"), b"known");
11714        let config = ScanConfig::default();
11715        let (mut index, report) = scan_into_index(dir.path(), &config).expect("scan");
11716        assert!(report.is_complete());
11717        let blocked = dir.path().join("blocked");
11718        let fault = install_walk_hook(&blocked, |_| {
11719            Some(std::io::Error::new(
11720                std::io::ErrorKind::PermissionDenied,
11721                "deterministic subtree refusal",
11722            ))
11723        });
11724
11725        let failed = reconcile_subtree(&mut index, Path::new("blocked"), &config, &mut |_| {})
11726            .expect("partial");
11727        assert!(!failed.scan.is_complete());
11728        assert_eq!(
11729            index.state().coverage,
11730            crate::Coverage::Partial(crate::CoverageReason::Inaccessible)
11731        );
11732        drop(fault);
11733
11734        let recovered = reconcile_subtree(&mut index, Path::new("blocked"), &config, &mut |_| {})
11735            .expect("retry");
11736
11737        assert!(recovered.scan.is_complete());
11738        assert_eq!(index.state().coverage, crate::Coverage::Complete);
11739    }
11740
11741    #[test]
11742    fn failed_pending_reconciliation_remains_queued_for_retry() {
11743        let dir = tempfile::tempdir().expect("tempdir");
11744        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11745        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
11746            path: PathBuf::new(),
11747            reason: crate::InvalidateReason::Requested,
11748        }]));
11749        fs::remove_dir_all(dir.path()).expect("remove root");
11750
11751        assert!(reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {}).is_err());
11752        assert_eq!(
11753            index.take_pending_invalidations(),
11754            vec![(PathBuf::new(), crate::InvalidateReason::Requested)]
11755        );
11756        assert_eq!(index.freshness(), crate::Freshness::Partial);
11757    }
11758
11759    #[cfg(unix)]
11760    #[test]
11761    fn partial_cold_scan_keeps_verified_siblings_complete() {
11762        use std::os::unix::fs::PermissionsExt;
11763        if !crate::test_support::require_permission_bits() {
11764            return;
11765        }
11766        let root = tempfile::tempdir().expect("root");
11767        write_file(&root.path().join("blocked/unknown.txt"), b"unread");
11768        write_file(&root.path().join("healthy/nested/known.txt"), b"known");
11769        let blocked = root.path().join("blocked");
11770        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
11771        let scan = ScanConfig::default();
11772        let detached = scan_into_index(root.path(), &scan);
11773        let streamed = scan_into_index_via_scanner(root.path(), &scan);
11774        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
11775        for result in [detached, streamed] {
11776            let (index, report) = result.expect("partial scan still returns its facts");
11777            assert!(!report.is_complete(), "permission fixture must fail the blocked listing");
11778            assert!(
11779                index
11780                    .issues()
11781                    .iter()
11782                    .any(|issue| issue.path.as_deref() == Some(Path::new("blocked")))
11783            );
11784            assert_eq!(index.freshness_at(Path::new("")), crate::Freshness::Partial);
11785            assert_eq!(index.directory_complete(Path::new("")), Some(true));
11786            assert_eq!(index.directory_complete(Path::new("blocked")), Some(false));
11787            assert_eq!(index.freshness_at(Path::new("blocked")), crate::Freshness::Partial);
11788            assert!(index.lookup(Path::new("blocked/unknown.txt")).is_none());
11789            for sibling in ["healthy", "healthy/nested"] {
11790                assert_eq!(index.directory_complete(Path::new(sibling)), Some(true), "{sibling}");
11791                assert_eq!(
11792                    index.freshness_at(Path::new(sibling)),
11793                    crate::Freshness::Fresh,
11794                    "{sibling}"
11795                );
11796            }
11797            assert!(index.lookup(Path::new("healthy/nested/known.txt")).is_some());
11798            assert!(
11799                !crate::stored_state::entries_writable(&index),
11800                "partial root cannot persist metadata"
11801            );
11802        }
11803    }
11804
11805    #[cfg(unix)]
11806    #[test]
11807    fn partial_pending_reconciliation_remains_queued_for_retry() {
11808        use std::os::unix::fs::PermissionsExt;
11809
11810        if !crate::test_support::require_permission_bits() {
11811            return;
11812        }
11813        let dir = tempfile::tempdir().expect("tempdir");
11814        write_file(&dir.path().join("blocked/known.txt"), b"known");
11815        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11816        let blocked = dir.path().join("blocked");
11817        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
11818        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
11819            path: PathBuf::from("blocked"),
11820            reason: crate::InvalidateReason::VerificationFailed,
11821        }]));
11822
11823        let report = reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {})
11824            .expect("permission failure is a partial report");
11825        let pending = index.take_pending_invalidations();
11826        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
11827        assert!(!report.is_complete(), "permission fixture must make reconciliation partial");
11828        assert_eq!(
11829            pending,
11830            vec![(PathBuf::from("blocked"), crate::InvalidateReason::VerificationFailed)]
11831        );
11832        assert_eq!(index.freshness_at(Path::new("blocked")), crate::Freshness::Partial);
11833    }
11834
11835    /// The shared API settles an unreadable subtree instead of queueing it again.
11836    ///
11837    /// Its per-event driver, `Watcher::apply_next`, drains after every event, so a retry
11838    /// re-walked the same unreadable subtree on each unrelated event, forever. The subtree
11839    /// stays partial and the report still names the error, once.
11840    #[cfg(unix)]
11841    #[test]
11842    fn partial_shared_pending_reconciliation_settles_instead_of_retrying() {
11843        use std::os::unix::fs::PermissionsExt;
11844
11845        if !crate::test_support::require_permission_bits() {
11846            return;
11847        }
11848        let dir = tempfile::tempdir().expect("tempdir");
11849        write_file(&dir.path().join("blocked/known.txt"), b"known");
11850        let (index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
11851        let handle = crate::IndexHandle::new(index);
11852        let blocked = dir.path().join("blocked");
11853        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o000)).expect("deny reads");
11854        handle
11855            .apply(&Observation::new(vec![Op::InvalidateSubtree {
11856                path: PathBuf::from("blocked"),
11857                reason: crate::InvalidateReason::VerificationFailed,
11858            }]))
11859            .expect("invalidate");
11860
11861        let report = reconcile_pending_handle(&handle, &ScanConfig::default(), &mut |_| {})
11862            .expect("permission failure is a partial report");
11863        let pending = handle.take_pending_invalidations().expect("pending");
11864        let freshness = handle.freshness_at(Path::new("blocked")).expect("freshness");
11865        fs::set_permissions(&blocked, fs::Permissions::from_mode(0o700)).expect("restore reads");
11866        assert!(!report.is_complete(), "permission fixture must make reconciliation partial");
11867        assert!(pending.is_empty(), "{pending:?}");
11868        assert_eq!(freshness, crate::Freshness::Partial);
11869        assert!(!report.scan.errors.is_empty());
11870    }
11871
11872    #[test]
11873    fn pending_scope_mismatch_does_not_drain_the_retry_queue() {
11874        let dir = tempfile::tempdir().expect("tempdir");
11875        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
11876        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
11877        index.apply_ok(&Observation::new(vec![Op::InvalidateSubtree {
11878            path: PathBuf::new(),
11879            reason: crate::InvalidateReason::Requested,
11880        }]));
11881
11882        let error = reconcile_pending(&mut index, &ScanConfig::default(), &mut |_| {})
11883            .expect_err("mismatched scope must fail");
11884
11885        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
11886        assert_eq!(
11887            index.take_pending_invalidations(),
11888            vec![(PathBuf::new(), crate::InvalidateReason::Requested)]
11889        );
11890    }
11891
11892    #[test]
11893    fn reconciliation_rejects_a_scope_mismatch_before_mutating() {
11894        let dir = tempfile::tempdir().expect("tempdir");
11895        write_file(&dir.path().join("deep/nested.txt"), b"nested");
11896        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
11897        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
11898        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
11899
11900        let error = reconcile(&mut index, &ScanConfig::default(), &mut |_| {})
11901            .expect_err("mismatched scope must fail");
11902
11903        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
11904        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
11905        assert_eq!(index.freshness(), crate::Freshness::Fresh);
11906    }
11907
11908    #[test]
11909    fn subtree_reconciliation_rejects_a_path_beyond_the_depth_scope() {
11910        let dir = tempfile::tempdir().expect("tempdir");
11911        write_file(&dir.path().join("deep/nested.txt"), b"nested");
11912        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
11913        let (mut index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
11914
11915        let result =
11916            reconcile_subtree(&mut index, Path::new("deep/nested.txt"), &shallow, &mut |_| {});
11917
11918        assert!(matches!(result, Err(Error::SubtreeOutsideScanScope { .. })));
11919        assert!(index.lookup(Path::new("deep/nested.txt")).is_none());
11920        assert_eq!(index.freshness(), crate::Freshness::Fresh);
11921    }
11922
11923    #[cfg(unix)]
11924    #[test]
11925    fn subtree_reconciliation_does_not_follow_an_ancestor_symlink() {
11926        use std::os::unix::fs::symlink;
11927
11928        let root = tempfile::tempdir().expect("root");
11929        let outside = tempfile::tempdir().expect("outside");
11930        write_file(&outside.path().join("secret.txt"), b"secret");
11931        symlink(outside.path(), root.path().join("link")).expect("symlink");
11932        let config = ScanConfig::default();
11933        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
11934
11935        let result =
11936            reconcile_subtree(&mut index, Path::new("link/secret.txt"), &config, &mut |_| {});
11937
11938        assert!(matches!(result, Err(Error::SubtreeOutsideScanScope { .. })));
11939        assert!(index.lookup(Path::new("link/secret.txt")).is_none());
11940        assert_eq!(index.freshness(), crate::Freshness::Fresh);
11941    }
11942
11943    #[test]
11944    fn subtree_reconciliation_widens_to_a_non_directory_ancestor() {
11945        let root = tempfile::tempdir().expect("root");
11946        write_file(&root.path().join("parent/child.txt"), b"old");
11947        let config = ScanConfig::default();
11948        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
11949        fs::remove_dir_all(root.path().join("parent")).expect("remove directory");
11950        write_file(&root.path().join("parent"), b"replacement");
11951
11952        let report =
11953            reconcile_subtree(&mut index, Path::new("parent/child.txt"), &config, &mut |_| {})
11954                .expect("reconcile widened ancestor");
11955
11956        assert!(report.is_complete());
11957        assert_eq!(index.kind(Path::new("parent")), Some(EntryKind::File));
11958        assert!(index.lookup(Path::new("parent/child.txt")).is_none());
11959        assert_eq!(index.freshness(), crate::Freshness::Fresh);
11960    }
11961
11962    #[test]
11963    fn subtree_reconciliation_widens_to_a_missing_ancestor() {
11964        let root = tempfile::tempdir().expect("root");
11965        write_file(&root.path().join("parent/child.txt"), b"old");
11966        let config = ScanConfig::default();
11967        let (mut index, _) = scan_into_index(root.path(), &config).expect("scan");
11968        fs::remove_dir_all(root.path().join("parent")).expect("remove directory");
11969
11970        let report =
11971            reconcile_subtree(&mut index, Path::new("parent/child.txt"), &config, &mut |_| {})
11972                .expect("reconcile widened ancestor");
11973
11974        assert!(report.is_complete());
11975        assert!(index.lookup(Path::new("parent")).is_none());
11976        assert_eq!(index.freshness(), crate::Freshness::Fresh);
11977    }
11978
11979    #[test]
11980    fn observation_only_revalidation_rejects_a_scope_mismatch() {
11981        let dir = tempfile::tempdir().expect("tempdir");
11982        let shallow = ScanConfig { max_depth: Some(1), ..ScanConfig::default() };
11983        let (index, _) = scan_into_index(dir.path(), &shallow).expect("scan");
11984        let mut observations = Vec::new();
11985
11986        let error = revalidate(&index, &ScanConfig::default(), &mut |observation| {
11987            observations.push(observation);
11988        })
11989        .expect_err("mismatched scope must fail");
11990
11991        assert!(matches!(error, Error::ScanScopeMismatch { .. }));
11992        assert!(observations.is_empty());
11993    }
11994
11995    #[cfg(unix)]
11996    #[test]
11997    fn a_new_filesystem_boundary_prunes_cached_descendants() {
11998        use std::os::unix::fs::MetadataExt;
11999
12000        let root = Path::new("/");
12001        let root_dev = {
12002            crate::counters::bump(|c| c.stats += 1);
12003            fs::symlink_metadata(root)
12004        }
12005        .expect("stat root")
12006        .dev();
12007        let Some(mount) = [Path::new("/dev"), Path::new("/proc"), Path::new("/sys")]
12008            .into_iter()
12009            .find(|candidate| {
12010                fs::symlink_metadata(candidate)
12011                    .is_ok_and(|metadata| metadata.is_dir() && metadata.dev() != root_dev)
12012            })
12013        else {
12014            return; // This host exposes no convenient cross-device directory.
12015        };
12016        let relative = mount.strip_prefix(root).expect("mount is below root");
12017        let stale_child = relative.join(".fdu-stale-snapshot-entry");
12018        let config = ScanConfig { one_filesystem: true, ..ScanConfig::default() };
12019        let mount_meta = fs::symlink_metadata(mount).expect("stat mount");
12020        let mut index = Index::new_with_scope(root, config.scope());
12021        index.apply_baseline_ok(&Observation::new(vec![
12022            Op::Upsert {
12023                path: relative.to_path_buf(),
12024                kind: EntryKind::Dir,
12025                attrs: attrs_from(mount, &mount_meta).expect("mount attrs"),
12026            },
12027            Op::Upsert {
12028                path: stale_child.clone(),
12029                kind: EntryKind::File,
12030                attrs: Attrs { size: 10, allocated: 10, ..Attrs::default() },
12031            },
12032        ]));
12033
12034        let error = reconcile_subtree(&mut index, &stale_child, &config, &mut |_| {})
12035            .expect_err("a descendant below the mount boundary is outside scope");
12036        assert!(matches!(error, Error::SubtreeOutsideScanScope { .. }));
12037        assert!(index.lookup(&stale_child).is_some());
12038
12039        reconcile_subtree(&mut index, relative, &config, &mut |_| {}).expect("reconcile mount");
12040
12041        assert!(index.lookup(relative).is_some(), "the mount point itself stays visible");
12042        assert!(index.lookup(&stale_child).is_none(), "out-of-scope descendants are pruned");
12043    }
12044
12045    #[test]
12046    fn subtree_reconciliation_rejects_paths_outside_the_root() {
12047        let dir = sample_tree();
12048        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12049
12050        assert!(matches!(
12051            reconcile_subtree(
12052                &mut index,
12053                Path::new("../outside"),
12054                &ScanConfig::default(),
12055                &mut |_| {},
12056            ),
12057            Err(Error::PathEscapesRoot(_))
12058        ));
12059        assert_eq!(index.freshness(), crate::Freshness::Fresh);
12060    }
12061
12062    #[test]
12063    fn normalized_walk_errors_keep_index_and_one_shot_status_in_lockstep() {
12064        let root = Path::new("/root");
12065        let mut order: Vec<_> = (0..66).rev().collect();
12066        order.push(65);
12067        let mut errors = order
12068            .into_iter()
12069            .map(|number| {
12070                Error::io(
12071                    root.join(format!("file-{number:02}")),
12072                    std::io::Error::new(std::io::ErrorKind::PermissionDenied, "denied"),
12073                )
12074            })
12075            .collect::<Vec<_>>();
12076
12077        normalize_walk_errors(root, &mut errors);
12078        assert_eq!(errors.len(), 66, "the repeated cause is removed once");
12079
12080        let mut index = crate::Index::new(root);
12081        index.record_walk_errors(&mut errors);
12082        let status = crate::query::TreeStatus::of_walk(
12083            root,
12084            &mut ScanReport { errors, ..ScanReport::default() },
12085        );
12086
12087        assert_eq!(status.errors, index.issues());
12088        assert_eq!(status.errors_omitted, index.state().issues.omitted);
12089        assert_eq!(status.errors.len(), crate::MAX_RETAINED_ISSUES);
12090        assert_eq!(status.errors_omitted, 2);
12091    }
12092
12093    #[cfg(target_os = "linux")]
12094    #[test]
12095    fn scan_and_revalidate_keep_non_utf8_names_distinct() {
12096        use std::ffi::OsString;
12097        use std::os::unix::ffi::OsStringExt;
12098
12099        let dir = tempfile::tempdir().expect("tempdir");
12100        let first = PathBuf::from(OsString::from_vec(vec![b'n', 0x80]));
12101        let second = PathBuf::from(OsString::from_vec(vec![b'n', 0x81]));
12102        write_file(&dir.path().join(&first), b"a");
12103        write_file(&dir.path().join(&second), b"bb");
12104
12105        let (mut index, _) = scan_into_index(dir.path(), &ScanConfig::default()).expect("scan");
12106        assert_eq!(index.total().files, 2);
12107        assert_eq!(index.total().bytes, 3);
12108
12109        fs::remove_file(dir.path().join(&first)).expect("remove first");
12110        let mut observations = Vec::new();
12111        revalidate(&index, &ScanConfig::default(), &mut |observation| {
12112            observations.push(observation);
12113        })
12114        .expect("revalidate");
12115        for observation in &observations {
12116            index.apply_ok(observation);
12117        }
12118        assert!(index.lookup(&first).is_none());
12119        assert!(index.lookup(&second).is_some());
12120        assert_eq!(index.total().bytes, 2);
12121    }
12122}