Skip to main content

fdu_core/content/
content_analysis.rs

1//! Parallel streaming file reads and conditional analysis commits.
2
3use std::fs::File;
4use std::io::Read;
5use std::num::NonZeroUsize;
6use std::sync::{Arc, mpsc};
7
8use crate::Index;
9use crate::classify::{ContentFamily, TypeRegistry, classify_with};
10
11use super::{
12    AnalysisApplyOutcome, AnalysisCandidate, AnalysisObservation, AnalysisRequest, AnalyzerOutcome,
13    BasicAccumulator, BasicMetrics, CodeAccumulator, CodeMetrics, ContentProvenance,
14    CoverageReason, FileAnalysis, TextAdmission, WordMetrics,
15    content_markdown_metrics::analyze_markdown,
16};
17
18const READ_CHUNK_BYTES: usize = 64 * 1024;
19const CLASSIFICATION_PREFIX_BYTES: usize = 16 * 1024;
20const MAX_ERROR_BYTES: usize = 512;
21/// Candidates walked at once, and the most queued for the workers
22/// ([`analyze_index_in_batches`]): enough to keep every worker reading while the next
23/// batch is walked, and a few megabytes of paths and classifications at most, whatever
24/// the tree holds.
25const ANALYSIS_BATCH_CANDIDATES: usize = 4096;
26/// The largest Markdown file the words unit renders exactly, in bytes.
27///
28/// Rendering needs the whole source and a document-wide parse beside it, so a worker's
29/// memory grows with the file. Up to this size that is a few hundred megabytes at most
30/// per worker; above it the file is counted as plain text and its record says so
31/// ([`CoverageReason::TextOnly`]). No Markdown document anyone writes is near it: the
32/// largest in common corpora are a few megabytes, so every real file is rendered
33/// exactly, and the bound exists for the generated or concatenated file that would
34/// otherwise exhaust memory (fdu-b2qz).
35///
36/// Fixed: no request option lifts it, which makes it the one recorded exception to the
37/// rule that every bound is liftable (the design principles' "Truncate Freely; Never
38/// Truncate Silently" says why). Lifting it needs an analyzer option in the content
39/// identity, so that a lifted bound re-analyzes what it counted as text; [`AnalysisLimits`]
40/// is the seam such an option would fill.
41pub(crate) const MARKDOWN_EXACT_BYTES: u64 = 64 * 1024 * 1024;
42
43/// Resource limits an analysis pass runs under.
44#[derive(Clone, Copy, Debug)]
45pub(crate) struct AnalysisLimits {
46    /// [`MARKDOWN_EXACT_BYTES`], or a smaller bound in a test.
47    pub(crate) markdown_exact_bytes: u64,
48}
49
50impl Default for AnalysisLimits {
51    fn default() -> Self {
52        Self { markdown_exact_bytes: MARKDOWN_EXACT_BYTES }
53    }
54}
55
56/// The most bytes [`analyze_open_file`] held of each file beyond its read chunk, by
57/// absolute path: the classification prefix, a deferred code buffer, and a Markdown
58/// source. A test seam for the retention bounds.
59#[cfg(test)]
60static PEAK_RETAINED: std::sync::Mutex<std::collections::BTreeMap<std::path::PathBuf, usize>> =
61    std::sync::Mutex::new(std::collections::BTreeMap::new());
62
63/// Operational counters from one content-analysis pass.
64#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
65pub struct AnalysisReport {
66    /// Regular files considered by the requested profile.
67    pub candidates: u64,
68    /// Bytes actually returned by fresh file reads, including partial binary probes.
69    pub bytes_read: u64,
70    /// Wall time spent processing fresh candidates.
71    pub elapsed_ns: u64,
72    /// Results accepted by the index's conditional mutation boundary.
73    pub applied: u64,
74    /// Results discarded because indexed metadata changed while workers ran.
75    pub stale: u64,
76    /// Coverage of the shared lines unit.
77    pub lines: AnalyzerCoverage,
78    /// Coverage of the code unit when requested.
79    pub code: Option<AnalyzerCoverage>,
80    /// Coverage of the words unit when requested.
81    pub words: Option<AnalyzerCoverage>,
82}
83
84/// Operational and semantic outcomes for one requested analyzer unit.
85#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
86pub struct AnalyzerCoverage {
87    /// Files for which the unit produced metrics.
88    pub analyzed: u64,
89    /// Known or observed binary files.
90    pub binary: u64,
91    /// Files whose byte stream was not valid UTF-8.
92    pub invalid_utf8: u64,
93    /// Files with a recognized but unsupported text encoding.
94    pub unsupported_encoding: u64,
95    /// Files that changed during their read.
96    pub changed_during_read: u64,
97    /// File-open, metadata, or read failures.
98    pub io_errors: u64,
99    /// Files for which this analyzer was unavailable.
100    pub unsupported: u64,
101}
102
103impl AnalysisReport {
104    /// Whether content analysis completed without an operational failure.
105    ///
106    /// Binary data, invalid UTF-8, and unsupported analyzers are coverage outcomes. They
107    /// remain visible in roll-ups but do not mean the filesystem operation failed.
108    pub fn is_complete(&self) -> bool {
109        self.stale == 0
110            && self.lines.is_complete()
111            && self.code.is_none_or(AnalyzerCoverage::is_complete)
112            && self.words.is_none_or(AnalyzerCoverage::is_complete)
113    }
114
115    /// Explain operational failures without presenting expected coverage as an error.
116    pub fn failure_message(&self) -> Option<String> {
117        (!self.is_complete()).then(|| {
118            format!(
119                "content analysis had operational failures (I/O errors: {}; changed during read: {}; stale results: {}). File and byte totals remain complete; content metrics omit affected files",
120                self.io_errors(), self.changed_during_read(), self.stale
121            )
122        })
123    }
124
125    fn io_errors(&self) -> u64 {
126        self.lines.io_errors
127    }
128
129    fn changed_during_read(&self) -> u64 {
130        self.lines.changed_during_read
131    }
132}
133
134impl AnalyzerCoverage {
135    const fn is_complete(self) -> bool {
136        self.changed_during_read == 0 && self.io_errors == 0
137    }
138
139    fn count(&mut self, reason: CoverageReason) {
140        let counter = match reason {
141            // Counted as plain text is still a value the unit produced.
142            CoverageReason::Analyzed | CoverageReason::TextOnly => &mut self.analyzed,
143            CoverageReason::Binary => &mut self.binary,
144            CoverageReason::InvalidUtf8 => &mut self.invalid_utf8,
145            CoverageReason::UnsupportedEncoding => &mut self.unsupported_encoding,
146            CoverageReason::IoError => &mut self.io_errors,
147            CoverageReason::ChangedDuringRead => &mut self.changed_during_read,
148            CoverageReason::Unsupported => &mut self.unsupported,
149        };
150        *counter = counter.saturating_add(1);
151    }
152}
153
154/// Analyze all regular files selected by `request` with a fixed-size worker pool.
155///
156/// Workers own immutable candidates and never retain an index borrow during I/O. The
157/// caller thread applies observations afterward, so metadata changes remain serialized
158/// through the index's own apply step, which is crate-private until the request model
159/// decides the public analysis surface.
160pub fn analyze_index(index: &mut Index, request: AnalysisRequest) -> AnalysisReport {
161    analyze_index_observed(index, request, None)
162}
163
164/// [`analyze_index`], reporting the pass through `progress` when one is attached.
165///
166/// The candidate total is known before the first file is read, so this is the one
167/// phase with an exact denominator. Each result is counted as it reaches the caller's
168/// thread, whether the index applies it or discards it as stale: the file was read
169/// either way, and progress counts work done. At return, `analysis` is
170/// `(candidates, candidates)`.
171pub(crate) fn analyze_index_observed(
172    index: &mut Index,
173    request: AnalysisRequest,
174    progress: Option<&crate::Progress>,
175) -> AnalysisReport {
176    analyze_index_in_batches(
177        index,
178        request,
179        progress,
180        ANALYSIS_BATCH_CANDIDATES,
181        AnalysisLimits::default(),
182    )
183}
184
185/// [`analyze_index_observed`], walking `batch` candidates at a time.
186///
187/// The candidates are walked in batches rather than materialized whole: each holds two
188/// paths and a classification, so a million-file tree cost hundreds of megabytes before
189/// the first read (fdu-xjfk). The count is taken first, without building any, so the
190/// denominator is exact before the first read. Every result is still applied
191/// conditionally on the revision and fingerprint its candidate carried, and nothing else
192/// changes ([`analyze_candidates`]).
193pub(crate) fn analyze_index_in_batches(
194    index: &mut Index,
195    request: AnalysisRequest,
196    progress: Option<&crate::Progress>,
197    batch: usize,
198    limits: AnalysisLimits,
199) -> AnalysisReport {
200    if !request.profile.is_enabled() {
201        return AnalysisReport::default();
202    }
203    let pass_started_at_ns =
204        crate::query::system_time_to_nanos(std::time::SystemTime::now()).unwrap_or(0);
205    let previous_state = index.content().and_then(super::ContentIndex::state);
206    index.prepare_content_analysis(request);
207    let mut report = AnalysisReport {
208        candidates: index.count_pending_analysis_candidates(request),
209        code: request.profile.includes_code().then(AnalyzerCoverage::default),
210        words: request.profile.includes_words().then(AnalyzerCoverage::default),
211        ..AnalysisReport::default()
212    };
213    if let Some(progress) = progress {
214        progress.enter(crate::ProgressPhase::Analyzing);
215        progress.begin_analysis(report.candidates);
216    }
217    if report.candidates == 0 {
218        finish_content_tier(index, &report, previous_state, pass_started_at_ns);
219        return report;
220    }
221
222    let started = std::time::Instant::now();
223    // Cloned out before the scope: the workers need the index's rules while the caller's
224    // thread holds the index mutably, and an `Arc` is what lets both be true.
225    let types = index.types_shared();
226    let workers =
227        worker_count(request.workers, usize::try_from(report.candidates).unwrap_or(usize::MAX));
228    let pass = Pass { request, limits, workers, batch: batch.max(1) };
229    analyze_candidates(index, &types, pass, &mut report, progress);
230    report.elapsed_ns = elapsed_ns(started);
231    finish_content_tier(index, &report, previous_state, pass_started_at_ns);
232    report
233}
234
235/// What one analysis pass reads under, and how it schedules the reads.
236#[derive(Clone, Copy)]
237struct Pass {
238    request: AnalysisRequest,
239    limits: AnalysisLimits,
240    workers: usize,
241    /// Candidates walked at once, and the most queued for the workers, up to
242    /// [`ANALYSIS_BATCH_CANDIDATES`].
243    batch: usize,
244}
245
246/// Read every candidate on the pass's workers, in one scope for the whole pass, and
247/// apply each result as it arrives.
248///
249/// The caller's thread does everything that touches the index. It walks the next batch
250/// once the last is queued, queues candidates as the workers take them, and applies
251/// results in between, so at most two batches of candidates are alive at once: the one
252/// queued and the one walked. The workers pull from the queue and never see the index.
253///
254/// One scope rather than one per batch (R164-6): a scope per batch joined every worker at
255/// each batch's end, so the workers that finished first waited for the slowest file of the
256/// batch, a multi-gigabyte file near a batch's end serialized the pool, and the next batch
257/// was walked only then.
258///
259/// The caller's thread waits for a result only while the queue is full, when a worker
260/// always has a candidate to read, so the pass cannot stall on itself. A worker that
261/// panics stops taking candidates; the others finish the queue, and the scope then
262/// re-raises the panic, as it did when each batch had its own scope.
263fn analyze_candidates(
264    index: &mut Index,
265    types: &Arc<TypeRegistry>,
266    pass: Pass,
267    report: &mut AnalysisReport,
268    progress: Option<&crate::Progress>,
269) {
270    // The channel allocates its bound up front, so a caller's larger batch is not its size.
271    let (work, queue) =
272        mpsc::sync_channel::<AnalysisCandidate>(pass.batch.min(ANALYSIS_BATCH_CANDIDATES));
273    let queue = std::sync::Mutex::new(queue);
274    let (sender, results) = mpsc::sync_channel(pass.workers.saturating_mul(2).max(1));
275
276    std::thread::scope(|scope| {
277        for _ in 0..pass.workers {
278            let sender = sender.clone();
279            let queue = &queue;
280            scope.spawn(move || {
281                let _counter_guard = crate::counters::thread_flush_guard();
282                loop {
283                    // The lock is released at the end of this statement, before the read.
284                    let next =
285                        queue.lock().unwrap_or_else(std::sync::PoisonError::into_inner).recv();
286                    let Ok(candidate) = next else { break };
287                    let analyzed = analyze_candidate(types, candidate, pass.request, pass.limits);
288                    if sender.send(analyzed).is_err() {
289                        break;
290                    }
291                }
292            });
293        }
294        drop(sender);
295
296        let mut walk = crate::index::AnalysisWalk::start();
297        let mut walked = std::collections::VecDeque::new();
298        loop {
299            if walked.is_empty() {
300                walked.extend(index.next_analysis_candidates(pass.request, &mut walk, pass.batch));
301                if walked.is_empty() {
302                    break;
303                }
304            }
305            while let Some(candidate) = walked.pop_front() {
306                match work.try_send(candidate) {
307                    Ok(()) => {}
308                    Err(mpsc::TrySendError::Full(candidate)) => {
309                        walked.push_front(candidate);
310                        break;
311                    }
312                    Err(mpsc::TrySendError::Disconnected(_)) => {
313                        unreachable!("the queue's receiver outlives the scope")
314                    }
315                }
316            }
317            if !walked.is_empty() {
318                // The queue is full, so a worker has a candidate: wait for a result. None
319                // arrives only once every worker has stopped, which a panic alone does.
320                let Ok(result) = results.recv() else { break };
321                apply_result(index, report, progress, result);
322            }
323        }
324        // Closing the queue lets the workers finish what it holds and stop, and the
325        // results end when the last of them has.
326        drop(work);
327        for result in results {
328            apply_result(index, report, progress, result);
329        }
330    });
331}
332
333/// Count one worker result and apply it conditionally, on the caller's thread.
334fn apply_result(
335    index: &mut Index,
336    report: &mut AnalysisReport,
337    progress: Option<&crate::Progress>,
338    (observation, bytes_read): (AnalysisObservation, u64),
339) {
340    report.bytes_read = report.bytes_read.saturating_add(bytes_read);
341    count_coverage(report, &observation.analysis);
342    match index.apply_analysis(observation) {
343        AnalysisApplyOutcome::Applied => report.applied = report.applied.saturating_add(1),
344        AnalysisApplyOutcome::Stale => report.stale = report.stale.saturating_add(1),
345    }
346    // Per result, on this thread, beside the index apply each result already costs: the
347    // workers never touch the handle.
348    if let Some(progress) = progress {
349        progress.add_analyzed(1);
350    }
351}
352
353fn finish_content_tier(
354    index: &mut Index,
355    report: &AnalysisReport,
356    previous: Option<super::ContentTierState>,
357    pass_started_at_ns: i64,
358) {
359    let entry = index.state();
360    let source = match entry.source {
361        crate::Source::Cached => crate::Source::Cached,
362        _ if previous.is_some_and(|state| state.source == crate::Source::Cached) => {
363            crate::Source::Revalidated
364        }
365        _ => crate::Source::Scanned,
366    };
367    let freshness = if report.is_complete() {
368        match source {
369            crate::Source::Cached => crate::Freshness::Stale,
370            _ => entry.freshness,
371        }
372    } else {
373        crate::Freshness::Partial
374    };
375    let observed_at_ns = if source == crate::Source::Cached {
376        previous.and_then(|state| state.observed_at_ns)
377    } else {
378        Some(pass_started_at_ns)
379    };
380    index.set_content_tier_state(source, freshness, observed_at_ns);
381}
382
383fn elapsed_ns(started: std::time::Instant) -> u64 {
384    u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)
385}
386
387fn worker_count(requested: usize, candidates: usize) -> usize {
388    let available = std::thread::available_parallelism().map_or(1, NonZeroUsize::get);
389    let requested = if requested == 0 { available } else { requested };
390    requested.clamp(1, candidates.max(1))
391}
392
393fn analyze_candidate(
394    types: &TypeRegistry,
395    candidate: AnalysisCandidate,
396    request: AnalysisRequest,
397    limits: AnalysisLimits,
398) -> (AnalysisObservation, u64) {
399    let (analysis, bytes_read) = if candidate.classification.family == ContentFamily::Binary {
400        (
401            record(
402                types,
403                &candidate,
404                request,
405                candidate.classification.clone(),
406                CoverageReason::Binary,
407                None,
408            ),
409            0,
410        )
411    } else {
412        analyze_open_file(types, &candidate, request, limits)
413    };
414    let provenance = ContentProvenance::for_request(request, types.fingerprint());
415    (AnalysisObservation { candidate, profile: request.profile, provenance, analysis }, bytes_read)
416}
417
418fn analyze_open_file(
419    types: &TypeRegistry,
420    candidate: &AnalysisCandidate,
421    request: AnalysisRequest,
422    limits: AnalysisLimits,
423) -> (FileAnalysis, u64) {
424    crate::counters::bump(|c| c.file_opens += 1);
425    let mut file = match File::open(&candidate.absolute_path) {
426        Ok(file) => file,
427        Err(error) => return (io_record(types, candidate, request, &error), 0),
428    };
429    let before = match file.metadata() {
430        Ok(metadata) => match crate::scan::attrs_from_file(&file, &metadata) {
431            Ok(attrs) => attrs.fingerprint(),
432            Err(error) => return (io_record(types, candidate, request, &error), 0),
433        },
434        Err(error) => return (io_record(types, candidate, request, &error), 0),
435    };
436    if before != candidate.attrs.fingerprint() {
437        return (
438            record(
439                types,
440                candidate,
441                request,
442                candidate.classification.clone(),
443                CoverageReason::ChangedDuringRead,
444                None,
445            ),
446            0,
447        );
448    }
449
450    let mut accumulator = BasicAccumulator::with_logical_metrics(request.profile.includes_words());
451    let mut code_accumulator = request
452        .profile
453        .includes_code()
454        .then(|| CodeAccumulator::for_type(candidate.classification.file_type.as_str()))
455        .flatten();
456    let mut deferred_code = (request.profile.includes_code()
457        && candidate.classification.family == ContentFamily::Unknown)
458        .then(Vec::new);
459    // A Markdown file over the exact bound is never held: it is counted as plain text.
460    let markdown_exact = candidate.attrs.size <= limits.markdown_exact_bytes;
461    let mut markdown_source = (request.profile.includes_words()
462        && markdown_exact
463        && (candidate.classification.file_type.as_str() == "markdown"
464            || candidate.classification.family == ContentFamily::Unknown))
465        .then(Vec::new);
466    #[cfg(test)]
467    let mut peak_retained = 0_usize;
468    let mut prefix = Vec::with_capacity(CLASSIFICATION_PREFIX_BYTES);
469    let mut chunk = vec![0_u8; READ_CHUNK_BYTES];
470    let mut read_failure = None;
471    let mut early_binary = None;
472    let mut encoding_prefix = [0_u8; 4];
473    let mut encoding_prefix_len = 0_usize;
474    let mut encoding_decided = false;
475    let mut unsupported_encoding = false;
476    let mut bytes_read = 0_u64;
477    loop {
478        crate::counters::bump(|c| c.file_reads += 1);
479        match file.read(&mut chunk) {
480            Ok(0) => break,
481            Ok(count) => {
482                crate::counters::bump(|c| c.bytes_read += count as u64);
483                bytes_read = bytes_read.saturating_add(u64::try_from(count).unwrap_or(u64::MAX));
484                if prefix.len() < CLASSIFICATION_PREFIX_BYTES {
485                    let take = count.min(CLASSIFICATION_PREFIX_BYTES - prefix.len());
486                    prefix.extend_from_slice(&chunk[..take]);
487                }
488                let mut body = &chunk[..count];
489                if !encoding_decided {
490                    let take = body.len().min(encoding_prefix.len() - encoding_prefix_len);
491                    encoding_prefix[encoding_prefix_len..encoding_prefix_len + take]
492                        .copy_from_slice(&body[..take]);
493                    encoding_prefix_len += take;
494                    body = &body[take..];
495                    if has_unsupported_encoding_bom(&encoding_prefix[..encoding_prefix_len]) {
496                        unsupported_encoding = true;
497                        break;
498                    }
499                    if encoding_prefix_len < encoding_prefix.len() {
500                        continue;
501                    }
502                    encoding_decided = true;
503                }
504                if prefix.len() >= 8 && candidate.classification.family == ContentFamily::Unknown {
505                    let classification =
506                        classify_with(types, &candidate.relative_path, Some(&prefix));
507                    if classification.family == ContentFamily::Binary {
508                        early_binary = Some(classification);
509                        break;
510                    }
511                    if prefix.len() == CLASSIFICATION_PREFIX_BYTES {
512                        if let Some(deferred) = deferred_code.take() {
513                            if classification.family == ContentFamily::Code {
514                                if let Some(mut code) =
515                                    CodeAccumulator::for_type(classification.file_type.as_str())
516                                {
517                                    code.push(&deferred);
518                                    code_accumulator = Some(code);
519                                }
520                            }
521                        }
522                        if classification.file_type.as_str() != "markdown" {
523                            markdown_source = None;
524                        }
525                    }
526                }
527                if encoding_decided && encoding_prefix_len != 0 {
528                    push_analysis_bytes(
529                        &mut accumulator,
530                        &mut code_accumulator,
531                        &mut deferred_code,
532                        &mut markdown_source,
533                        &encoding_prefix[..encoding_prefix_len],
534                    );
535                    encoding_prefix_len = 0;
536                }
537                push_analysis_bytes(
538                    &mut accumulator,
539                    &mut code_accumulator,
540                    &mut deferred_code,
541                    &mut markdown_source,
542                    body,
543                );
544                #[cfg(test)]
545                {
546                    peak_retained = peak_retained.max(
547                        prefix.len()
548                            + deferred_code.as_ref().map_or(0, Vec::len)
549                            + markdown_source.as_ref().map_or(0, Vec::len),
550                    );
551                }
552            }
553            Err(error) => {
554                read_failure = Some(error);
555                break;
556            }
557        }
558    }
559    #[cfg(test)]
560    PEAK_RETAINED
561        .lock()
562        .unwrap_or_else(std::sync::PoisonError::into_inner)
563        .insert(candidate.absolute_path.clone(), peak_retained);
564
565    let after = match file.metadata() {
566        Ok(metadata) => match crate::scan::attrs_from_file(&file, &metadata) {
567            Ok(attrs) => attrs.fingerprint(),
568            Err(error) => return (io_record(types, candidate, request, &error), bytes_read),
569        },
570        Err(error) => return (io_record(types, candidate, request, &error), bytes_read),
571    };
572    if before != after {
573        return (
574            record(
575                types,
576                candidate,
577                request,
578                candidate.classification.clone(),
579                CoverageReason::ChangedDuringRead,
580                None,
581            ),
582            bytes_read,
583        );
584    }
585    if let Some(error) = read_failure {
586        return (io_record(types, candidate, request, &error), bytes_read);
587    }
588    if unsupported_encoding {
589        return (
590            record(
591                types,
592                candidate,
593                request,
594                candidate.classification.clone(),
595                CoverageReason::UnsupportedEncoding,
596                None,
597            ),
598            bytes_read,
599        );
600    }
601    if encoding_prefix_len != 0 {
602        push_analysis_bytes(
603            &mut accumulator,
604            &mut code_accumulator,
605            &mut deferred_code,
606            &mut markdown_source,
607            &encoding_prefix[..encoding_prefix_len],
608        );
609    }
610    if let Some(classification) = early_binary {
611        return (
612            record(types, candidate, request, classification, CoverageReason::Binary, None),
613            bytes_read,
614        );
615    }
616    let classification = classify_with(types, &candidate.relative_path, Some(&prefix));
617    if classification.family == ContentFamily::Binary {
618        return (
619            record(types, candidate, request, classification, CoverageReason::Binary, None),
620            bytes_read,
621        );
622    }
623    let analysis = match accumulator.finish() {
624        TextAdmission::Accepted(mut metrics) => {
625            // Word volume is meaningful for any text, and the accumulator has already
626            // counted it during the streaming read — zeroing it outside prose and markup
627            // discarded finished work and, with it, the answer to "how much text is in
628            // this tree", which is the cheap proxy for context-window sizing that agent
629            // consumers ask for.
630            //
631            let mut code_supported = !request.profile.includes_code();
632            if request.profile.includes_code() {
633                if code_accumulator.is_none() && classification.family == ContentFamily::Code {
634                    if let Some(deferred) = deferred_code {
635                        if let Some(mut code) =
636                            CodeAccumulator::for_type(classification.file_type.as_str())
637                        {
638                            code.push(&deferred);
639                            code_accumulator = Some(code);
640                        }
641                    }
642                }
643                code_supported = code_accumulator.is_some();
644                if let Some(code) = code_accumulator.take() {
645                    let code_metrics = code.finish();
646                    debug_assert_eq!(metrics.physical_lines, code_metrics.physical_lines);
647                    metrics.code_lines = code_metrics.code_lines;
648                    metrics.comment_lines = code_metrics.comment_lines;
649                    metrics.code_blank_lines = code_metrics.code_blank_lines;
650                }
651            }
652            if request.profile.includes_words() && classification.file_type.as_str() == "markdown" {
653                if let Some(source) = markdown_source {
654                    let source = std::str::from_utf8(&source)
655                        .expect("basic admission already established valid UTF-8");
656                    let visible = analyze_markdown(source);
657                    metrics.visible_words = visible.visible_words;
658                    metrics.visible_logical_word_stats = visible.visible_logical_word_stats;
659                    metrics.paragraphs = visible.paragraphs;
660                }
661            }
662            let lines = BasicMetrics {
663                physical_lines: metrics.physical_lines,
664                blank_lines: metrics.blank_lines,
665                nonblank_lines: metrics.nonblank_lines,
666                raw_words: metrics.raw_words,
667            };
668            let code = request.profile.includes_code().then(|| {
669                if code_supported {
670                    AnalyzerOutcome::analyzed(CodeMetrics {
671                        code_lines: metrics.code_lines,
672                        comment_lines: metrics.comment_lines,
673                        code_blank_lines: metrics.code_blank_lines,
674                    })
675                } else {
676                    AnalyzerOutcome::unavailable(CoverageReason::Unsupported)
677                }
678            });
679            let words = request.profile.includes_words().then(|| {
680                let words = WordMetrics {
681                    paragraphs: metrics.paragraphs,
682                    visible_words: metrics.visible_words,
683                    logical_word_stats: metrics.logical_word_stats,
684                    visible_logical_word_stats: metrics.visible_logical_word_stats,
685                };
686                if classification.file_type.as_str() == "markdown" && !markdown_exact {
687                    // Counted as plain text, as a `.txt` of the same bytes is: every
688                    // word visible, the paragraphs its blank-line runs.
689                    AnalyzerOutcome::text_only(WordMetrics {
690                        visible_words: metrics.raw_words,
691                        visible_logical_word_stats: metrics.logical_word_stats,
692                        ..words
693                    })
694                } else {
695                    AnalyzerOutcome::analyzed(words)
696                }
697            });
698            analyzed_record(candidate, classification, lines, code, words)
699        }
700        TextAdmission::Binary => {
701            record(types, candidate, request, classification, CoverageReason::Binary, None)
702        }
703        TextAdmission::InvalidUtf8 => {
704            record(types, candidate, request, classification, CoverageReason::InvalidUtf8, None)
705        }
706    };
707    (analysis, bytes_read)
708}
709
710fn has_unsupported_encoding_bom(prefix: &[u8]) -> bool {
711    prefix.starts_with(&[0xff, 0xfe])
712        || prefix.starts_with(&[0xfe, 0xff])
713        || prefix.starts_with(&[0x00, 0x00, 0xfe, 0xff])
714}
715
716fn push_analysis_bytes(
717    accumulator: &mut BasicAccumulator,
718    code_accumulator: &mut Option<CodeAccumulator>,
719    deferred_code: &mut Option<Vec<u8>>,
720    markdown_source: &mut Option<Vec<u8>>,
721    bytes: &[u8],
722) {
723    accumulator.push(bytes);
724    if let Some(code) = code_accumulator {
725        code.push(bytes);
726    }
727    if let Some(deferred) = deferred_code {
728        deferred.extend_from_slice(bytes);
729    }
730    if let Some(source) = markdown_source {
731        source.extend_from_slice(bytes);
732    }
733}
734
735fn analyzed_record(
736    candidate: &AnalysisCandidate,
737    classification: crate::classify::Classification,
738    lines: BasicMetrics,
739    code: Option<AnalyzerOutcome<CodeMetrics>>,
740    words: Option<AnalyzerOutcome<WordMetrics>>,
741) -> FileAnalysis {
742    FileAnalysis {
743        fingerprint: candidate.attrs.fingerprint(),
744        bytes: candidate.attrs.size,
745        detection: classification.into(),
746        lines: AnalyzerOutcome::analyzed(lines),
747        code,
748        words,
749        error: None,
750    }
751}
752
753fn io_record(
754    types: &TypeRegistry,
755    candidate: &AnalysisCandidate,
756    request: AnalysisRequest,
757    error: &std::io::Error,
758) -> FileAnalysis {
759    let mut detail = error.to_string();
760    detail.truncate(char_boundary_at_or_before(&detail, MAX_ERROR_BYTES));
761    record(
762        types,
763        candidate,
764        request,
765        candidate.classification.clone(),
766        CoverageReason::IoError,
767        Some(detail),
768    )
769}
770
771fn char_boundary_at_or_before(value: &str, limit: usize) -> usize {
772    let mut boundary = limit.min(value.len());
773    while !value.is_char_boundary(boundary) {
774        boundary -= 1;
775    }
776    boundary
777}
778
779fn record(
780    _types: &TypeRegistry,
781    candidate: &AnalysisCandidate,
782    request: AnalysisRequest,
783    classification: crate::classify::Classification,
784    coverage: CoverageReason,
785    error: Option<String>,
786) -> FileAnalysis {
787    FileAnalysis {
788        fingerprint: candidate.attrs.fingerprint(),
789        bytes: candidate.attrs.size,
790        detection: classification.into(),
791        lines: AnalyzerOutcome::unavailable(coverage),
792        code: request.profile.includes_code().then_some(AnalyzerOutcome::unavailable(coverage)),
793        words: request.profile.includes_words().then_some(AnalyzerOutcome::unavailable(coverage)),
794        error,
795    }
796}
797
798fn count_coverage(report: &mut AnalysisReport, analysis: &FileAnalysis) {
799    report.lines.count(analysis.lines.coverage());
800    if let (Some(coverage), Some(outcome)) = (&mut report.code, analysis.code) {
801        coverage.count(outcome.coverage());
802    }
803    if let (Some(coverage), Some(outcome)) = (&mut report.words, analysis.words) {
804        coverage.count(outcome.coverage());
805    }
806}
807
808#[cfg(test)]
809mod tests {
810    use std::collections::BTreeMap;
811    use std::fs;
812    use std::path::Path;
813
814    use crate::content::AnalysisSet;
815    use crate::scan::ScanConfig;
816
817    use super::*;
818
819    /// Exercises many streaming chunks with a realistically large generated source file.
820    const LARGE_CODE_FILE_BYTES: usize = 17 * 1024 * 1024;
821
822    #[test]
823    fn expected_coverage_gaps_are_not_operational_failures() {
824        let report = AnalysisReport {
825            lines: AnalyzerCoverage { invalid_utf8: 1, ..AnalyzerCoverage::default() },
826            code: Some(AnalyzerCoverage { unsupported: 3, ..AnalyzerCoverage::default() }),
827            ..AnalysisReport::default()
828        };
829
830        assert!(report.is_complete());
831        assert_eq!(report.failure_message(), None);
832
833        let failed = AnalysisReport {
834            lines: AnalyzerCoverage {
835                io_errors: 1,
836                changed_during_read: 2,
837                ..AnalyzerCoverage::default()
838            },
839            stale: 3,
840            ..AnalysisReport::default()
841        };
842        assert!(!failed.is_complete());
843        assert_eq!(
844            failed.failure_message().as_deref(),
845            Some(
846                "content analysis had operational failures (I/O errors: 1; changed during read: 2; stale results: 3). File and byte totals remain complete; content metrics omit affected files"
847            )
848        );
849    }
850
851    fn limits(markdown_exact_bytes: u64) -> AnalysisLimits {
852        AnalysisLimits { markdown_exact_bytes }
853    }
854
855    fn analyzed_under(
856        root: &Path,
857        request: AnalysisRequest,
858        limits: AnalysisLimits,
859    ) -> (Index, AnalysisReport) {
860        let (mut index, _) =
861            crate::scan::scan_into_index(root, &ScanConfig::default()).expect("scan");
862        let report = analyze_index_in_batches(&mut index, request, None, 4096, limits);
863        (index, report)
864    }
865
866    fn record(index: &Index, name: &str) -> FileAnalysis {
867        index.content().and_then(|content| content.file(Path::new(name))).cloned().expect(name)
868    }
869
870    fn peak_retained(root: &Path, name: &str) -> usize {
871        let path = root.canonicalize().expect("canonical root").join(name);
872        *PEAK_RETAINED
873            .lock()
874            .unwrap_or_else(std::sync::PoisonError::into_inner)
875            .get(&path)
876            .unwrap_or_else(|| panic!("{name} was analyzed"))
877    }
878
879    /// A Markdown file over the exact bound is counted as plain text, as a `.txt` of the
880    /// same bytes is, and every surface says so: the record's words coverage, the
881    /// documents row, the machine coverage map, and a report note (fdu-b2qz). At and
882    /// below the bound the answer is the rendered one it always was.
883    #[test]
884    fn a_markdown_file_over_the_exact_bound_is_counted_as_plain_text_and_says_so() {
885        let root = tempfile::tempdir().expect("tempdir");
886        let paragraph = "# Title\n\nRead [the label](https://example.test/very/long/url) and \
887                         ![alt](image.png).\n\n```rust\nlet fenced = \"hidden\";\n```\n\n\
888                         plain words here\n\n";
889        let markdown: Vec<u8> = paragraph.repeat(40).into_bytes();
890        fs::write(root.path().join("doc.md"), &markdown).expect("markdown");
891        fs::write(root.path().join("doc.txt"), &markdown).expect("text twin");
892        let size = u64::try_from(markdown.len()).expect("fits");
893        let request = AnalysisRequest { profile: AnalysisSet::ALL, workers: 2 };
894
895        let (exact, exact_report) = analyzed_under(root.path(), request, limits(size));
896        let rendered = record(&exact, "doc.md");
897        let rendered_words = rendered.words.expect("words").value().expect("rendered value");
898        assert_eq!(rendered.words.expect("words").coverage(), CoverageReason::Analyzed);
899        let twin_words = record(&exact, "doc.txt").words.expect("words").value().expect("twin");
900        assert!(
901            rendered_words.visible_words < rendered.lines.value().expect("lines").raw_words,
902            "rendering hides link destinations and code: {rendered_words:?}"
903        );
904
905        let (bounded, bounded_report) = analyzed_under(root.path(), request, limits(size - 1));
906        let counted = record(&bounded, "doc.md");
907        let words = counted.words.expect("words");
908        assert_eq!(words.coverage(), CoverageReason::TextOnly);
909        let counted_words = words.value().expect("a text-only outcome carries its value");
910        let raw_words = counted.lines.value().expect("lines").raw_words;
911        assert_eq!(counted_words.visible_words, raw_words, "every word is visible");
912        assert_eq!(counted_words.visible_logical_word_stats, twin_words.logical_word_stats);
913        assert_eq!(
914            counted_words.paragraphs, twin_words.paragraphs,
915            "paragraphs are blank-line runs"
916        );
917        assert_eq!(counted_words.logical_word_stats, twin_words.logical_word_stats);
918        assert_eq!(counted.lines, rendered.lines, "the lines unit is unchanged");
919        assert_eq!(counted.code, rendered.code, "the code unit is unchanged");
920        assert_eq!(bounded_report.candidates, exact_report.candidates);
921        assert_eq!(bounded_report.words, exact_report.words, "counted as text is still a value");
922        assert!(counted.is_reusable());
923        assert_eq!(counted.operational_failure(), None);
924
925        // The documents view says so, in text, in the machine coverage map, and in a note.
926        let request = crate::query::Request::new(
927            crate::query::Basis::held_by(&bounded),
928            crate::query::Query {
929                views: vec![crate::query::ViewSpec::Documents],
930                ..crate::query::Query::default()
931            },
932            std::time::UNIX_EPOCH,
933        );
934        let report =
935            crate::query::report(&bounded, &request, std::time::UNIX_EPOCH).expect("report");
936        assert!(
937            report.notes.iter().any(|note| note
938                == "note: 1 Markdown file over 64 MiB counted as plain text: every word counted \
939                    visible, paragraphs are blank-line runs"),
940            "{:?}",
941            report.notes
942        );
943        let text = crate::report_format::render(&report, crate::report_format::Format::Text, false)
944            .expect("text");
945        assert!(text.contains("1 counted as text"), "{text}");
946        let json = crate::report_format::render(&report, crate::report_format::Format::Json, false)
947            .expect("json");
948        assert!(json.contains("\"text_only\": 1"), "{json}");
949
950        let exact_request = crate::query::Request::new(
951            crate::query::Basis::held_by(&exact),
952            crate::query::Query {
953                views: vec![crate::query::ViewSpec::Documents],
954                ..crate::query::Query::default()
955            },
956            std::time::UNIX_EPOCH,
957        );
958        let exact_report =
959            crate::query::report(&exact, &exact_request, std::time::UNIX_EPOCH).expect("report");
960        assert!(exact_report.notes.iter().all(|note| !note.contains("counted as plain text")));
961
962        // The note speaks for the report's selection, not for every record the index
963        // holds: a selection that leaves the Markdown file out says nothing of it, and two
964        // views of one selection count it once.
965        let notes_for = |views: Vec<crate::query::ViewSpec>, include: &[&str]| {
966            let selection = crate::query::Selection {
967                include: include
968                    .iter()
969                    .map(|source| crate::query::Pattern::parse(source).expect("pattern"))
970                    .collect(),
971                ..crate::query::Selection::default()
972            };
973            let query = crate::query::Query { selection, views, ..crate::query::Query::default() };
974            let request = crate::query::Request::new(
975                crate::query::Basis::held_by(&bounded),
976                query,
977                std::time::UNIX_EPOCH,
978            );
979            crate::query::report(&bounded, &request, std::time::UNIX_EPOCH).expect("report").notes
980        };
981        let text_only = |notes: &[String]| -> Vec<String> {
982            notes.iter().filter(|note| note.contains("counted as plain text")).cloned().collect()
983        };
984        let documents = crate::query::ViewSpec::Documents;
985        let types = crate::query::ViewSpec::Types;
986        assert_eq!(
987            text_only(&notes_for(vec![documents], &["*.txt"])),
988            Vec::<String>::new(),
989            "no selected row was counted as text"
990        );
991        assert_eq!(
992            text_only(&notes_for(vec![documents, types], &[])),
993            ["note: 1 Markdown file over 64 MiB counted as plain text: every word counted \
994              visible, paragraphs are blank-line runs"],
995            "two views of one selection count the file once"
996        );
997    }
998
999    /// The other half of the bound (fdu-b2qz): a file of unknown type retains at most
1000    /// its classification prefix and one read chunk once the prefix has settled its
1001    /// type, a Markdown file over the exact bound retains the same, and only a Markdown
1002    /// file within the bound holds its source, which the exact rendering needs.
1003    #[test]
1004    fn an_unknown_type_file_retains_at_most_the_prefix_and_a_chunk() {
1005        let root = tempfile::tempdir().expect("tempdir");
1006        let line = b"2026-09-30T00:00:00Z info the quick brown fox jumps over the lazy dog\n";
1007        let mut body = Vec::with_capacity(1024 * 1024 + line.len());
1008        while body.len() < 1024 * 1024 {
1009            body.extend_from_slice(line);
1010        }
1011        fs::write(root.path().join("service.log"), &body).expect("log");
1012        fs::write(root.path().join("dump"), &body).expect("no extension");
1013        fs::write(root.path().join("notes.md"), &body).expect("markdown");
1014        let request = AnalysisRequest { profile: AnalysisSet::ALL, workers: 1 };
1015        let bound = 2 * CLASSIFICATION_PREFIX_BYTES + READ_CHUNK_BYTES;
1016
1017        let (_, _) = analyzed_under(root.path(), request, limits(512 * 1024));
1018        for name in ["service.log", "dump", "notes.md"] {
1019            let peak = peak_retained(root.path(), name);
1020            assert!(peak <= bound, "{name} retained {peak} bytes of a {}-byte file", body.len());
1021        }
1022
1023        let (_, _) = analyzed_under(root.path(), request, AnalysisLimits::default());
1024        assert!(peak_retained(root.path(), "service.log") <= bound);
1025        assert!(
1026            peak_retained(root.path(), "notes.md") >= body.len(),
1027            "within the bound the rendered answer needs the whole source"
1028        );
1029    }
1030
1031    /// Batched scheduling changes what is alive at once and nothing else (fdu-xjfk):
1032    /// one candidate at a time and every candidate at once leave the same records,
1033    /// counts, and denominator.
1034    #[test]
1035    fn scheduling_in_batches_of_one_leaves_the_same_records_as_all_at_once() {
1036        let root = tempfile::tempdir().expect("tempdir");
1037        fs::create_dir_all(root.path().join("src/deep")).expect("directories");
1038        fs::write(root.path().join("src/main.rs"), b"fn main() {\n    // hi\n}\n").expect("write");
1039        fs::write(root.path().join("src/deep/lib.rs"), b"pub fn f() {}\n").expect("write");
1040        fs::write(root.path().join("notes.md"), b"one two\n\nthree\n").expect("write");
1041        fs::write(root.path().join("blob.bin"), b"\x00\x01\x02").expect("write");
1042        fs::write(root.path().join("latin.txt"), b"caf\xe9\n").expect("write");
1043        let request = AnalysisRequest { profile: AnalysisSet::ALL, workers: 3 };
1044        let records = |batch: usize| {
1045            let (mut index, _) =
1046                crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1047            let report = analyze_index_in_batches(
1048                &mut index,
1049                request,
1050                None,
1051                batch,
1052                AnalysisLimits::default(),
1053            );
1054            let content = index.content().expect("content");
1055            let records: BTreeMap<_, _> = content
1056                .records()
1057                .map(|(path, analysis)| (path.to_path_buf(), format!("{analysis:?}")))
1058                .collect();
1059            (report, records)
1060        };
1061        let (one_at_a_time, singly) = records(1);
1062        let (all_at_once, whole) = records(usize::MAX);
1063        assert_eq!(singly.len(), 5);
1064        assert_eq!(singly, whole);
1065        assert_eq!(one_at_a_time.candidates, 5);
1066        assert_eq!(all_at_once.candidates, 5);
1067        assert_eq!(one_at_a_time.applied, all_at_once.applied);
1068        assert_eq!(one_at_a_time.stale, all_at_once.stale);
1069        assert_eq!(one_at_a_time.bytes_read, all_at_once.bytes_read);
1070        assert_eq!(one_at_a_time.lines, all_at_once.lines);
1071        assert_eq!(one_at_a_time.code, all_at_once.code);
1072        assert_eq!(one_at_a_time.words, all_at_once.words);
1073        // Batches that split the tree unevenly, so the queue fills and the next batch is
1074        // walked while candidates are still being read.
1075        for batch in [2, 3] {
1076            let (split, records) = records(batch);
1077            assert_eq!(records, whole, "batches of {batch}");
1078            assert_eq!(
1079                (split.candidates, split.applied, split.stale, split.bytes_read),
1080                (
1081                    all_at_once.candidates,
1082                    all_at_once.applied,
1083                    all_at_once.stale,
1084                    all_at_once.bytes_read
1085                ),
1086                "batches of {batch}"
1087            );
1088            assert_eq!(
1089                (split.lines, split.code, split.words),
1090                (all_at_once.lines, all_at_once.code, all_at_once.words),
1091                "batches of {batch}"
1092            );
1093        }
1094    }
1095
1096    /// Every words-unit metric is independent of whether code was also requested.
1097    #[test]
1098    fn code_carries_word_volume_and_paragraphs() {
1099        let root = tempfile::tempdir().expect("tempdir");
1100        std::fs::write(root.path().join("main.rs"), b"fn main() {\n\n    let x = 1;\n}\n")
1101            .expect("write");
1102        std::fs::write(root.path().join("notes.md"), b"one two\n\nthree four\n").expect("write");
1103        let (mut index, _) =
1104            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1105        analyze_index(
1106            &mut index,
1107            AnalysisRequest { profile: super::super::AnalysisSet::ALL, workers: 2 },
1108        );
1109
1110        let content = index.content().expect("content");
1111        let code = content.file(std::path::Path::new("main.rs")).expect("code record");
1112        assert!(code.lines.value().expect("line metrics").raw_words > 0);
1113        assert_eq!(code.code.expect("code outcome").coverage(), CoverageReason::Analyzed);
1114        assert!(code.words.and_then(AnalyzerOutcome::value).expect("word metrics").paragraphs > 0);
1115
1116        let prose = content.file(std::path::Path::new("notes.md")).expect("prose record");
1117        assert!(prose.lines.value().expect("line metrics").raw_words > 0);
1118        assert_eq!(prose.code.expect("code outcome").coverage(), CoverageReason::Unsupported);
1119        assert!(prose.words.and_then(AnalyzerOutcome::value).expect("word metrics").paragraphs > 0);
1120    }
1121
1122    #[test]
1123    fn every_metric_is_independent_of_other_requested_units_and_grouping_is_name_only() {
1124        let root = tempfile::tempdir().expect("tempdir");
1125        fs::create_dir_all(root.path().join("generated")).expect("generated directory");
1126        for (path, bytes) in [
1127            ("main.rs", b"// comment\nfn main() { println!(\"hello world\"); }\n".as_slice()),
1128            ("tool.py", b"# comment\nprint('hello world')\n"),
1129            ("Main.hs", b"-- comment\nmain = putStrLn \"hello world\"\n"),
1130            ("guide.md", b"# Hello\n\nVisible [words](https://example.test).\n"),
1131            ("notes.txt", b"first paragraph words\n\nsecond paragraph words\n"),
1132            ("ambiguous.h", b"// generated fixture\nnamespace demo { int value; }\n"),
1133            ("script", b"#!/usr/bin/env python3\nprint('from shebang')\n"),
1134            ("document", b"%PDF-1.7\nfixture"),
1135            ("nul.unknown", b"text before\0binary"),
1136            ("invalid.unknown", &[b't', b'e', b'x', b't', 0xff]),
1137            ("generated/output.rs", b"// Code generated; DO NOT EDIT.\nfn output() {}\n"),
1138        ] {
1139            fs::write(root.path().join(path), bytes).expect("write fixture");
1140        }
1141        let (baseline, scan) =
1142            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1143        assert!(scan.is_complete());
1144
1145        let profiles = [
1146            AnalysisSet::LINES_ONLY,
1147            AnalysisSet::CODE_ONLY,
1148            AnalysisSet::WORDS_ONLY,
1149            AnalysisSet::ALL,
1150        ];
1151        let mut observed = Vec::new();
1152        for profile in profiles {
1153            let mut index = baseline.clone();
1154            let analysis = analyze_index(&mut index, AnalysisRequest { profile, workers: 2 });
1155            assert!(analysis.is_complete(), "{profile:?}: {analysis:?}");
1156            let query = crate::query::Query {
1157                views: vec![crate::query::ViewSpec::Types],
1158                ..crate::query::Query::default()
1159            };
1160            let report = crate::query::report(
1161                &index,
1162                &crate::test_support::read_of(&index, query),
1163                std::time::UNIX_EPOCH,
1164            )
1165            .expect("report");
1166            let crate::query::Section::Metrics { summary, .. } = &report.sections[0] else {
1167                panic!("expected type metrics")
1168            };
1169            let mut groups: Vec<_> = summary.rows.iter().map(|row| row.id.clone()).collect();
1170            groups.sort();
1171            let values: std::collections::BTreeMap<_, _> = std::iter::once(&summary.total)
1172                .chain(&summary.rows)
1173                .map(|row| {
1174                    let metrics: Vec<_> = crate::content::METRICS
1175                        .iter()
1176                        .map(|metric| row.metric_value(metric))
1177                        .collect();
1178                    (row.id.clone(), metrics)
1179                })
1180                .collect();
1181            observed.push((profile, groups, values));
1182        }
1183
1184        for (_, groups, _) in &observed[1..] {
1185            assert_eq!(groups, &observed[0].1, "requested analyzers must not change type groups");
1186        }
1187        for (metric_index, metric) in crate::content::METRICS.iter().enumerate() {
1188            let (_, _, baseline) = observed
1189                .iter()
1190                .find(|(profile, _, _)| profile.contains(metric.owner))
1191                .expect("an owning profile exists");
1192            for (profile, _, rows) in &observed {
1193                assert_eq!(rows.keys().collect::<Vec<_>>(), baseline.keys().collect::<Vec<_>>());
1194                for (id, values) in rows {
1195                    let expected = if profile.contains(metric.owner) {
1196                        Some(baseline[id][metric_index].expect("owning profile exposes metric"))
1197                    } else {
1198                        None
1199                    };
1200                    assert_eq!(
1201                        values[metric_index], expected,
1202                        "{} changed or leaked in row {id} under {profile:?}",
1203                        metric.name
1204                    );
1205                }
1206            }
1207        }
1208    }
1209
1210    #[test]
1211    fn prefix_classification_handoff_neither_drops_nor_double_counts_large_files() {
1212        let root = tempfile::tempdir().expect("tempdir");
1213        for size in [20 * 1024, 80 * 1024] {
1214            let mut python = b"#!/usr/bin/env python3\n".to_vec();
1215            while python.len() < size {
1216                python.extend_from_slice(b"value = 1  # one comment\n");
1217            }
1218            fs::write(root.path().join(format!("script-{size}")), &python).expect("shebang");
1219            fs::write(root.path().join(format!("script-{size}.py")), &python).expect("python");
1220
1221            let mut plain = Vec::new();
1222            while plain.len() < size {
1223                plain.extend_from_slice(b"ordinary words in a plain text line\n");
1224            }
1225            fs::write(root.path().join(format!("plain-{size}.unknown")), &plain)
1226                .expect("unknown text");
1227            fs::write(root.path().join(format!("plain-{size}.txt")), &plain).expect("text");
1228        }
1229        let (baseline, scan) =
1230            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1231        assert!(scan.is_complete());
1232
1233        for profile in [AnalysisSet::CODE_ONLY, AnalysisSet::WORDS_ONLY, AnalysisSet::ALL] {
1234            let mut index = baseline.clone();
1235            let report = analyze_index(&mut index, AnalysisRequest { profile, workers: 1 });
1236            assert!(report.is_complete(), "{profile:?}: {report:?}");
1237            let content = index.content().expect("content");
1238            for size in [20 * 1024, 80 * 1024] {
1239                let script = content
1240                    .file(Path::new(&format!("script-{size}")))
1241                    .expect("extensionless script");
1242                let python =
1243                    content.file(Path::new(&format!("script-{size}.py"))).expect("named Python");
1244                assert_eq!(script.lines.value(), python.lines.value(), "{profile:?}, {size} bytes");
1245                if profile.includes_code() {
1246                    assert_eq!(script.code, python.code, "code handoff at {size} bytes");
1247                }
1248                if profile.includes_words() {
1249                    assert_eq!(script.words, python.words, "word handoff at {size} bytes");
1250                }
1251
1252                let unknown = content
1253                    .file(Path::new(&format!("plain-{size}.unknown")))
1254                    .expect("unknown text");
1255                let text =
1256                    content.file(Path::new(&format!("plain-{size}.txt"))).expect("named text");
1257                assert_eq!(unknown.lines.value(), text.lines.value(), "plain lines at {size}");
1258                if profile.includes_words() {
1259                    assert_eq!(unknown.words, text.words, "plain words at {size}");
1260                }
1261            }
1262        }
1263    }
1264
1265    #[test]
1266    fn pool_analyzes_text_and_skips_known_binary_files() {
1267        let root = tempfile::tempdir().expect("tempdir");
1268        fs::write(root.path().join("notes.md"), "one two\n\nthree\n").expect("write text");
1269        fs::write(root.path().join("image.png"), b"not opened as text").expect("write binary");
1270        let (mut index, scan) =
1271            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1272        assert!(scan.is_complete());
1273
1274        let report = analyze_index(
1275            &mut index,
1276            AnalysisRequest { profile: super::super::AnalysisSet::NONE.with_lines(), workers: 2 },
1277        );
1278
1279        assert_eq!(report.candidates, 2);
1280        assert_eq!(report.lines.analyzed, 1);
1281        assert_eq!(report.lines.binary, 1);
1282        assert_eq!(report.bytes_read, 15, "known binary files must not be opened");
1283        assert!(report.elapsed_ns > 0);
1284        let root_rollup = index.content_rollup(std::path::Path::new("")).expect("content root");
1285        assert_eq!(root_rollup.total.files, 2);
1286        assert_eq!(root_rollup.total.lines.metrics.physical_lines, 3);
1287        assert_eq!(root_rollup.total.lines.metrics.raw_words, 3);
1288    }
1289
1290    #[test]
1291    fn content_workers_publish_their_file_io_counters() {
1292        let _serial = crate::counters::test_serial();
1293        let root = tempfile::tempdir().expect("tempdir");
1294        fs::write(root.path().join("notes.md"), "one two\n\nthree\n").expect("write text");
1295        let (mut index, scan) =
1296            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1297        assert!(scan.is_complete());
1298
1299        crate::counters::enable(true);
1300        crate::counters::reset();
1301        let report = analyze_index(
1302            &mut index,
1303            AnalysisRequest { profile: super::super::AnalysisSet::NONE.with_lines(), workers: 2 },
1304        );
1305        crate::counters::flush_thread();
1306        let counts = crate::counters::snapshot();
1307        crate::counters::reset();
1308        crate::counters::enable(false);
1309
1310        assert_eq!(report.lines.analyzed, 1);
1311        assert!(counts.file_opens >= 1, "content worker file open was folded: {counts:?}");
1312        assert!(counts.file_reads >= 2, "data and EOF reads were folded: {counts:?}");
1313        assert!(counts.bytes_read >= 15, "content bytes were folded: {counts:?}");
1314    }
1315
1316    #[test]
1317    fn large_generated_code_is_analyzed_through_eof() {
1318        let root = tempfile::tempdir().expect("tempdir");
1319        let mut source = b"/* generated */\n".to_vec();
1320        source.resize(LARGE_CODE_FILE_BYTES - 1, b'x');
1321        source.push(b'\n');
1322        fs::write(root.path().join("generated.c"), source).expect("write generated C");
1323        let (mut index, _) =
1324            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1325
1326        let report = analyze_index(
1327            &mut index,
1328            AnalysisRequest { profile: super::super::AnalysisSet::NONE.with_code(), workers: 1 },
1329        );
1330
1331        assert_eq!(report.lines.analyzed, 1);
1332        assert_eq!(report.bytes_read, LARGE_CODE_FILE_BYTES as u64);
1333        assert!(report.elapsed_ns > 0);
1334        assert!(report.is_complete());
1335        let metrics = &index.content_rollup(std::path::Path::new("")).expect("content root").total;
1336        assert_eq!(metrics.bytes, LARGE_CODE_FILE_BYTES as u64);
1337        assert_eq!(metrics.lines.metrics.physical_lines, 2);
1338        assert_eq!(metrics.code.metrics.code_lines, 1);
1339        assert_eq!(metrics.code.metrics.comment_lines, 1);
1340    }
1341
1342    #[test]
1343    fn nul_and_invalid_utf8_are_coverage_not_provisional_metrics() {
1344        let root = tempfile::tempdir().expect("tempdir");
1345        fs::write(root.path().join("nul.unknown"), b"line\nlate\0nul").expect("write nul");
1346        fs::write(root.path().join("bad.unknown"), [b'a', 0xff]).expect("write invalid");
1347        let (mut index, _) =
1348            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1349
1350        let report = analyze_index(
1351            &mut index,
1352            AnalysisRequest {
1353                profile: super::super::AnalysisSet::NONE.with_lines(),
1354                ..AnalysisRequest::default()
1355            },
1356        );
1357        assert_eq!(report.lines.binary, 1);
1358        assert_eq!(report.lines.invalid_utf8, 1);
1359        let metrics = &index.content_rollup(std::path::Path::new("")).expect("root").total.lines;
1360        assert_eq!(metrics.metrics, BasicMetrics::default());
1361    }
1362
1363    #[test]
1364    fn deep_detection_drives_named_consumers_and_report_evidence() {
1365        let root = tempfile::tempdir().expect("tempdir");
1366        fs::create_dir_all(root.path().join("vendor/docs")).expect("directories");
1367        fs::write(
1368            root.path().join("vendor/docs/generated.h"),
1369            "// Code generated by fixture; DO NOT EDIT.\nnamespace demo { int value; }\n",
1370        )
1371        .expect("write header");
1372        fs::write(
1373            root.path().join("script.inc"),
1374            "# vim: set filetype=rust:\n// comment\nfn main() {}\n",
1375        )
1376        .expect("write modeline");
1377        fs::write(root.path().join("download"), b"%PDF-1.7\nfixture payload")
1378            .expect("write signature");
1379        let (mut index, _) =
1380            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1381
1382        let report = analyze_index(
1383            &mut index,
1384            AnalysisRequest {
1385                profile: super::super::AnalysisSet::NONE.with_code(),
1386                ..AnalysisRequest::default()
1387            },
1388        );
1389        assert_eq!(report.lines.analyzed, 2);
1390        assert_eq!(report.lines.binary, 1);
1391
1392        let content = index.content().expect("content");
1393        let header = content.file(std::path::Path::new("vendor/docs/generated.h")).expect("header");
1394        assert_eq!(header.detection.file_type.as_str(), "cpp");
1395        assert_eq!(header.detection.source, crate::classify::DetectionSource::AmbiguousContent);
1396        assert!(header.detection.flags.generated);
1397        assert!(header.detection.flags.vendored);
1398        assert!(header.detection.flags.documentation);
1399
1400        let script = content.file(std::path::Path::new("script.inc")).expect("script");
1401        assert_eq!(script.detection.file_type.as_str(), "rust");
1402        let script_code = script.code.and_then(AnalyzerOutcome::value).expect("code metrics");
1403        assert_eq!(script_code.comment_lines, 1);
1404        assert_eq!(script_code.code_lines, 2);
1405
1406        let pdf = content.file(std::path::Path::new("download")).expect("pdf");
1407        assert_eq!(pdf.detection.file_type.as_str(), "pdf");
1408        assert_eq!(pdf.lines.coverage(), CoverageReason::Binary);
1409
1410        let query = crate::query::Query {
1411            views: vec![crate::query::ViewSpec::Types],
1412            ..crate::query::Query::default()
1413        };
1414        let report = crate::query::report(
1415            &index,
1416            &crate::test_support::read_of(&index, query),
1417            std::time::UNIX_EPOCH,
1418        )
1419        .expect("report");
1420        let crate::query::Section::Metrics { summary, .. } = &report.sections[0] else {
1421            panic!("expected metric summary")
1422        };
1423        assert_eq!(summary.total.generated_files, 1);
1424        assert_eq!(summary.total.vendored_files, 1);
1425        assert_eq!(summary.total.documentation_files, 1);
1426        assert_eq!(
1427            summary.total.detection_sources.get(&crate::classify::DetectionSource::FormatSignature),
1428            Some(&1)
1429        );
1430        let header_row = summary.rows.iter().find(|row| row.id == "c").expect("name group");
1431        assert_eq!(
1432            header_row.detection_sources.get(&crate::classify::DetectionSource::AmbiguousContent),
1433            Some(&1),
1434            "content evidence remains visible without reassigning the .h name group"
1435        );
1436        let script_row = summary
1437            .rows
1438            .iter()
1439            .find(|row| row.id == "unknown:.inc")
1440            .expect("unresolved name group");
1441        assert_eq!(
1442            script_row.detection_sources.get(&crate::classify::DetectionSource::Modeline),
1443            Some(&1),
1444            "the modeline is evidence even though grouping follows the filename"
1445        );
1446    }
1447
1448    #[test]
1449    fn code_profile_partitions_supported_languages_and_marks_others_unsupported() {
1450        let root = tempfile::tempdir().expect("tempdir");
1451        fs::write(root.path().join("main.rs"), "// comment\nfn main() {} // mixed\n\n")
1452            .expect("write rust");
1453        fs::write(root.path().join("Main.hs"), "-- not claimed\nmain = pure ()\n")
1454            .expect("write haskell");
1455        let (mut index, _) =
1456            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1457
1458        let report = analyze_index(
1459            &mut index,
1460            AnalysisRequest {
1461                profile: super::super::AnalysisSet::NONE.with_code(),
1462                ..AnalysisRequest::default()
1463            },
1464        );
1465
1466        assert_eq!(report.lines.analyzed, 2);
1467        assert_eq!(report.code.expect("code coverage").unsupported, 1);
1468        let rust = index.content().expect("content").file(std::path::Path::new("main.rs"));
1469        let rust = rust.expect("rust record");
1470        let lines = rust.lines.value().expect("line metrics");
1471        let metrics = rust.code.and_then(AnalyzerOutcome::value).expect("code metrics");
1472        assert_eq!(lines.physical_lines, 3);
1473        assert_eq!(metrics.code_lines, 1);
1474        assert_eq!(metrics.comment_lines, 1);
1475        assert_eq!(metrics.code_blank_lines, 1);
1476        assert_eq!(
1477            lines.physical_lines,
1478            metrics.code_lines + metrics.comment_lines + metrics.code_blank_lines
1479        );
1480        let haskell = index.content().expect("content").file(std::path::Path::new("Main.hs"));
1481        let haskell = haskell.expect("haskell record");
1482        assert_eq!(haskell.lines.coverage(), CoverageReason::Analyzed);
1483        assert_eq!(haskell.code.expect("code outcome").coverage(), CoverageReason::Unsupported);
1484
1485        let query = crate::query::Query {
1486            views: vec![crate::query::ViewSpec::Languages],
1487            ..crate::query::Query::default()
1488        };
1489        let rendered = crate::query::report(
1490            &index,
1491            &crate::test_support::read_of(&index, query),
1492            std::time::UNIX_EPOCH,
1493        )
1494        .expect("report");
1495        let crate::query::Section::Metrics { summary, .. } = &rendered.sections[0] else {
1496            panic!("expected language metrics")
1497        };
1498        assert_eq!(summary.share_metric, crate::query::ShareMetric::CodeLines);
1499        assert_eq!((summary.total.share.numerator, summary.total.share.denominator), (1, 1));
1500        assert_eq!(summary.total.coverage.get(&CoverageReason::Unsupported), Some(&1));
1501    }
1502
1503    #[test]
1504    fn unicode_boms_are_nonoperational_unsupported_encoding_outcomes_per_unit() {
1505        let root = tempfile::tempdir().expect("tempdir");
1506        fs::write(root.path().join("little.rs"), [0xff, 0xfe, b'f', 0, b'n', 0])
1507            .expect("UTF-16 LE");
1508        fs::write(root.path().join("big.txt"), [0xfe, 0xff, 0, b'w']).expect("UTF-16 BE");
1509        fs::write(root.path().join("wide"), [0, 0, 0xfe, 0xff, 0, 0, 0, b'w']).expect("UTF-32 BE");
1510        fs::write(root.path().join("notes.md"), b"ordinary words\n").expect("UTF-8");
1511        let (mut index, _) =
1512            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1513
1514        let report = analyze_index(
1515            &mut index,
1516            AnalysisRequest { profile: AnalysisSet::ALL, ..AnalysisRequest::default() },
1517        );
1518
1519        assert!(report.is_complete(), "unsupported encodings are not operational failures");
1520        assert_eq!(report.lines.unsupported_encoding, 3);
1521        assert_eq!(report.code.expect("code coverage").unsupported_encoding, 3);
1522        assert_eq!(report.words.expect("word coverage").unsupported_encoding, 3);
1523        assert_eq!(report.lines.binary, 0);
1524        assert_eq!(report.lines.invalid_utf8, 0);
1525        let query = crate::query::Query {
1526            views: vec![crate::query::ViewSpec::Types],
1527            ..crate::query::Query::default()
1528        };
1529        let rendered = crate::query::report(
1530            &index,
1531            &crate::test_support::read_of(&index, query),
1532            std::time::UNIX_EPOCH,
1533        )
1534        .expect("report");
1535        for format in [crate::report_format::Format::Json, crate::report_format::Format::Yaml] {
1536            let output = crate::report_format::render(&rendered, format, false)
1537                .expect("compatible report format");
1538            let total = match format {
1539                crate::report_format::Format::Json => {
1540                    output.split("\"rows\":").next().expect("metrics total")
1541                }
1542                crate::report_format::Format::Yaml => {
1543                    output.split("\n      rows:").next().expect("metrics total")
1544                }
1545                _ => unreachable!("only machine document formats are tested"),
1546            };
1547            assert_eq!(
1548                total.matches("unsupported_encoding").count(),
1549                3,
1550                "each requested analyzer unit has its own total coverage map: {output}"
1551            );
1552        }
1553        let text =
1554            crate::report_format::render(&rendered, crate::report_format::Format::Text, false)
1555                .expect("compatible report format");
1556        assert!(
1557            text.contains("1 lines (1 nonblank, 0 blank), 2 words"),
1558            "unsupported code coverage cannot turn prose lines into a zero code partition: {text}"
1559        );
1560        let content = index.content().expect("content");
1561        for path in ["little.rs", "big.txt", "wide"] {
1562            let record = content.file(Path::new(path)).expect("encoding record");
1563            assert_eq!(record.lines.coverage(), CoverageReason::UnsupportedEncoding);
1564            assert_eq!(
1565                record.code.expect("code outcome").coverage(),
1566                CoverageReason::UnsupportedEncoding
1567            );
1568            assert_eq!(
1569                record.words.expect("word outcome").coverage(),
1570                CoverageReason::UnsupportedEncoding
1571            );
1572            assert!(record.error.is_none());
1573        }
1574        assert_eq!(
1575            content
1576                .file(Path::new("notes.md"))
1577                .expect("markdown record")
1578                .code
1579                .expect("code outcome")
1580                .coverage(),
1581            CoverageReason::Unsupported,
1582            "an unsupported analyzer remains distinct from an unsupported encoding"
1583        );
1584    }
1585
1586    #[test]
1587    fn unsupported_encoding_boms_are_recognized_from_progressive_prefixes() {
1588        for (bom, recognized_at) in [
1589            (&[0xff, 0xfe][..], 2_usize),
1590            (&[0xfe, 0xff][..], 2_usize),
1591            (&[0xff, 0xfe, 0, 0][..], 2_usize),
1592            (&[0, 0, 0xfe, 0xff][..], 4_usize),
1593        ] {
1594            for length in 0..=bom.len() {
1595                assert_eq!(
1596                    has_unsupported_encoding_bom(&bom[..length]),
1597                    length >= recognized_at,
1598                    "{length} byte prefix of {bom:?}"
1599                );
1600            }
1601        }
1602        assert!(!has_unsupported_encoding_bom(&[0xef, 0xbb, 0xbf, b'x']));
1603    }
1604
1605    #[test]
1606    fn document_profile_uses_visible_markdown_and_logical_plain_text() {
1607        let root = tempfile::tempdir().expect("tempdir");
1608        fs::write(
1609            root.path().join("guide.md"),
1610            "# Read [the label](https://example.test)\n\n`hidden code` 中文\n",
1611        )
1612        .expect("write markdown");
1613        fs::write(root.path().join("notes.txt"), "oneverylongtoken\n\nplain words\n")
1614            .expect("write text");
1615        let (mut index, _) =
1616            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1617
1618        let report = analyze_index(
1619            &mut index,
1620            AnalysisRequest {
1621                profile: super::super::AnalysisSet::NONE.with_words(),
1622                ..AnalysisRequest::default()
1623            },
1624        );
1625        assert!(report.is_complete());
1626
1627        let query = crate::query::Query {
1628            views: vec![crate::query::ViewSpec::Documents],
1629            ..crate::query::Query::default()
1630        };
1631        let rendered = crate::query::report(
1632            &index,
1633            &crate::test_support::read_of(&index, query),
1634            std::time::UNIX_EPOCH,
1635        )
1636        .expect("report");
1637        let crate::query::Section::Metrics { summary, .. } = &rendered.sections[0] else {
1638            panic!("expected document metrics")
1639        };
1640        assert_eq!(summary.share_metric, crate::query::ShareMetric::DocumentWords);
1641        let rows =
1642            summary.rows.iter().map(|row| (row.id.as_str(), row)).collect::<BTreeMap<_, _>>();
1643        let markdown = rows["markdown"];
1644        assert!(markdown.metrics.raw_words > markdown.metrics.visible_words);
1645        assert_eq!(markdown.metrics.visible_words, Some(4));
1646        assert_eq!(markdown.metrics.paragraphs, Some(2));
1647        let text = rows["text"];
1648        assert!(text.metrics.logical_words > text.metrics.raw_words);
1649        assert_eq!(crate::query::document_words(&summary.total), Some(7));
1650    }
1651
1652    #[test]
1653    fn an_empty_analysis_still_retains_the_requested_identity() {
1654        let root = tempfile::tempdir().expect("tempdir");
1655        let (mut index, _) =
1656            crate::scan::scan_into_index(root.path(), &ScanConfig::default()).expect("scan");
1657        let request = AnalysisRequest {
1658            profile: super::super::AnalysisSet::NONE.with_lines(),
1659            ..AnalysisRequest::default()
1660        };
1661
1662        let analysis = analyze_index(&mut index, request);
1663
1664        assert_eq!(analysis.candidates, 0);
1665        let content = index.content().expect("the requested derived tier remains explicit");
1666        assert_eq!(content.profile(), Some(request.profile));
1667        assert_eq!(
1668            content.provenance(),
1669            Some(ContentProvenance::for_request(request, crate::classify::type_rule_fingerprint()))
1670        );
1671    }
1672
1673    #[test]
1674    fn content_reports_preserve_empty_profiles_and_unavailable_shares() {
1675        let empty = tempfile::tempdir().expect("empty tempdir");
1676        let (mut empty_index, _) =
1677            crate::scan::scan_into_index(empty.path(), &ScanConfig::default()).expect("scan");
1678        analyze_index(
1679            &mut empty_index,
1680            AnalysisRequest {
1681                profile: super::super::AnalysisSet::NONE.with_lines(),
1682                ..AnalysisRequest::default()
1683            },
1684        );
1685        let summary = crate::query::report(
1686            &empty_index,
1687            &crate::test_support::read_of(
1688                &empty_index,
1689                crate::query::Query {
1690                    views: vec![crate::query::ViewSpec::Summary],
1691                    ..crate::query::Query::default()
1692                },
1693            ),
1694            std::time::UNIX_EPOCH,
1695        )
1696        .expect("report");
1697        let json =
1698            crate::report_format::render(&summary, crate::report_format::Format::Json, false)
1699                .expect("compatible report format");
1700        assert!(json.contains("\"schema\": \"fdu.report/10\""), "{json}");
1701        assert!(json.contains("\"analyze\": [\"lines\"]"), "{json}");
1702
1703        let unsupported = tempfile::tempdir().expect("unsupported tempdir");
1704        fs::write(unsupported.path().join("Main.hs"), "main = pure ()\n").expect("write");
1705        let (mut unsupported_index, _) =
1706            crate::scan::scan_into_index(unsupported.path(), &ScanConfig::default()).expect("scan");
1707        analyze_index(
1708            &mut unsupported_index,
1709            AnalysisRequest {
1710                profile: super::super::AnalysisSet::NONE.with_code(),
1711                ..AnalysisRequest::default()
1712            },
1713        );
1714        let languages = crate::query::report(
1715            &unsupported_index,
1716            &crate::test_support::read_of(
1717                &unsupported_index,
1718                crate::query::Query {
1719                    views: vec![crate::query::ViewSpec::Languages],
1720                    ..crate::query::Query::default()
1721                },
1722            ),
1723            std::time::UNIX_EPOCH,
1724        )
1725        .expect("report");
1726        let text =
1727            crate::report_format::render(&languages, crate::report_format::Format::Text, false)
1728                .expect("compatible report format");
1729        assert!(text.contains("—"), "an unavailable 0/0 share needs a distinct marker: {text}");
1730        assert!(
1731            text.contains("1 lines (1 nonblank, 0 blank), 1 unsupported"),
1732            "unsupported code falls back to the valid line partition: {text}"
1733        );
1734        assert!(!text.contains("0.0%"), "unmeasured is not a zero percentage: {text}");
1735    }
1736}