Skip to main content

fdu_core/content/
content_model.rs

1//! Versioned content-analysis contracts shared across the engine and report layers.
2
3use std::path::PathBuf;
4
5use crate::classify::{
6    Classification, ClassificationFlags, ContentFamily, DetectionConfidence, DetectionSource,
7    FileTypeId,
8};
9use crate::query::Rejection;
10use crate::{Attrs, EntryId, Fingerprint};
11
12/// Stable analyzer identity.
13#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
14pub struct AnalyzerId(pub &'static str);
15
16/// Version of an analyzer's counting semantics.
17#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
18pub struct AnalyzerVersion(pub u16);
19
20/// Definition of one measured value exposed by content reports.
21#[derive(Clone, Copy, PartialEq, Eq, Debug)]
22pub struct MetricDef {
23    /// Stable report key.
24    pub name: &'static str,
25    /// Requestable unit that owns the value's presence.
26    pub owner: AnalysisSet,
27    /// Analyzer dialect that defines the value.
28    pub analyzer: AnalyzerId,
29    /// Short semantic definition.
30    pub doc: &'static str,
31}
32
33/// Fingerprint of semantic analyzer options; operational worker count is excluded.
34#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
35pub struct OptionsFingerprint(pub u64);
36
37/// Fused physical-line and raw-word analyzer.
38pub const CONTENT_BASIC: AnalyzerId = AnalyzerId("content-basic-v1");
39/// Common-language code/comment/blank analyzer.
40pub const CODE_SLOC: AnalyzerId = AnalyzerId("code-sloc-v1");
41/// Plain-text logical word and paragraph analyzer.
42pub const TEXT_LOGICAL: AnalyzerId = AnalyzerId("text-logical-v1");
43/// Reader-visible Markdown prose analyzer.
44pub const MARKDOWN_PROSE: AnalyzerId = AnalyzerId("markdown-prose-v1");
45
46/// The single registry of content metric names, owners, and definitions.
47pub const METRICS: &[MetricDef] = &[
48    MetricDef {
49        name: "physical_lines",
50        owner: AnalysisSet::LINES_ONLY,
51        analyzer: CONTENT_BASIC,
52        doc: "Logical physical lines across admitted text files.",
53    },
54    MetricDef {
55        name: "blank_lines",
56        owner: AnalysisSet::LINES_ONLY,
57        analyzer: CONTENT_BASIC,
58        doc: "Whitespace-only physical lines.",
59    },
60    MetricDef {
61        name: "nonblank_lines",
62        owner: AnalysisSet::LINES_ONLY,
63        analyzer: CONTENT_BASIC,
64        doc: "Physical lines containing non-whitespace text.",
65    },
66    MetricDef {
67        name: "raw_words",
68        owner: AnalysisSet::LINES_ONLY,
69        analyzer: CONTENT_BASIC,
70        doc: "Whitespace-delimited words before document projection.",
71    },
72    MetricDef {
73        name: "code_lines",
74        owner: AnalysisSet::CODE_ONLY,
75        analyzer: CODE_SLOC,
76        doc: "Code-bearing lines in supported source languages.",
77    },
78    MetricDef {
79        name: "comment_lines",
80        owner: AnalysisSet::CODE_ONLY,
81        analyzer: CODE_SLOC,
82        doc: "Comment-only lines in supported source languages.",
83    },
84    MetricDef {
85        name: "code_blank_lines",
86        owner: AnalysisSet::CODE_ONLY,
87        analyzer: CODE_SLOC,
88        doc: "Blank lines under the code analyzer's syntax.",
89    },
90    MetricDef {
91        name: "logical_words",
92        owner: AnalysisSet::WORDS_ONLY,
93        analyzer: TEXT_LOGICAL,
94        doc: "Normalized logical word volume.",
95    },
96    MetricDef {
97        name: "paragraphs",
98        owner: AnalysisSet::WORDS_ONLY,
99        analyzer: TEXT_LOGICAL,
100        doc: "Plain-text runs or reader-visible Markdown paragraphs.",
101    },
102    MetricDef {
103        name: "visible_words",
104        owner: AnalysisSet::WORDS_ONLY,
105        analyzer: MARKDOWN_PROSE,
106        doc: "Reader-visible Markdown words.",
107    },
108    MetricDef {
109        name: "visible_logical_words",
110        owner: AnalysisSet::WORDS_ONLY,
111        analyzer: MARKDOWN_PROSE,
112        doc: "Normalized reader-visible Markdown words.",
113    },
114    MetricDef {
115        name: "document_words",
116        owner: AnalysisSet::WORDS_ONLY,
117        analyzer: TEXT_LOGICAL,
118        doc: "Logical words after the document-type projection.",
119    },
120];
121
122/// The set of content analyzers a request enables.
123///
124/// A set rather than a ladder, because the analyzers are independent: `code` and `words`
125/// measure different things over different families and either is useful without the
126/// other.  An ordered enum could name only the combinations somebody thought to
127/// enumerate — four of the eight this registry already permits — and it made
128/// `text-logical-v1` without `markdown-prose-v1` unreachable.
129///
130/// `lines` is the base every analyzer shares: any analyzer that runs has already
131/// streamed the file, so line counts cost nothing extra.  It is therefore implicit in
132/// every non-empty set rather than something a caller must remember to request, which is
133/// why each `with_*` constructor sets it.
134#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug, Default)]
135pub struct AnalysisSet(u8);
136
137impl AnalysisSet {
138    const LINES: u8 = 1 << 0;
139    const CODE: u8 = 1 << 1;
140    const WORDS: u8 = 1 << 2;
141    const KNOWN: u8 = Self::LINES | Self::CODE | Self::WORDS;
142
143    /// Open no file; preserve the metadata-only behavior.
144    pub const NONE: Self = Self(0);
145    /// Physical-line and raw-word unit.
146    pub const LINES_ONLY: Self = Self(Self::LINES);
147    /// Code unit, including its shared line pass.
148    pub const CODE_ONLY: Self = Self(Self::LINES | Self::CODE);
149    /// Word unit, including its shared line pass.
150    pub const WORDS_ONLY: Self = Self(Self::LINES | Self::WORDS);
151    /// Every registered analyzer.
152    pub const ALL: Self = Self(Self::KNOWN);
153
154    /// Add physical, blank, and nonblank line counts.
155    #[must_use]
156    pub const fn with_lines(self) -> Self {
157        Self(self.0 | Self::LINES)
158    }
159
160    /// Add common-language standard SLOC over the `code` family.
161    #[must_use]
162    pub const fn with_code(self) -> Self {
163        Self(self.0 | Self::LINES | Self::CODE)
164    }
165
166    /// Add raw, normalized, and reader-visible word volume.
167    #[must_use]
168    pub const fn with_words(self) -> Self {
169        Self(self.0 | Self::LINES | Self::WORDS)
170    }
171
172    /// Whether any source file may be opened.
173    pub const fn is_enabled(self) -> bool {
174        self.0 != 0
175    }
176
177    /// Whether standard SLOC is requested.
178    pub const fn includes_code(self) -> bool {
179        self.0 & Self::CODE != 0
180    }
181
182    /// Whether logical and visible word metrics are requested.
183    pub const fn includes_words(self) -> bool {
184        self.0 & Self::WORDS != 0
185    }
186
187    /// Whether this request includes every unit in `other`.
188    pub const fn contains(self, other: Self) -> bool {
189        self.0 & other.0 == other.0
190    }
191
192    /// Every unit in either set.
193    #[must_use]
194    pub const fn union(self, other: Self) -> Self {
195        Self(self.0 | other.0)
196    }
197
198    /// The analyzers a caller names to request this set, in canonical order.
199    ///
200    /// [`Self::labels`] without `lines` beside another analyzer, since every analyzer
201    /// includes it: a caller types `code`, not `lines,code`. What a note or a tip says when
202    /// it names a set, so the words it shows are the words a caller would type.
203    pub fn named(self) -> Vec<&'static str> {
204        let labels = self.labels();
205        if labels.len() > 1 {
206            labels.into_iter().filter(|label| *label != "lines").collect()
207        } else {
208            labels
209        }
210    }
211
212    /// The shortest value of this axis's grammar that requests exactly this set: `none`,
213    /// `all`, or [`Self::named`] joined by commas.
214    pub fn request_label(self) -> String {
215        if !self.is_enabled() {
216            Self::NONE_LABEL.to_owned()
217        } else if self == Self::ALL {
218            "all".to_owned()
219        } else {
220            self.named().join(",")
221        }
222    }
223
224    /// Stable on-disk and fingerprint encoding.
225    pub const fn bits(self) -> u8 {
226        self.0
227    }
228
229    /// Decode [`Self::bits`], rejecting any analyzer this build does not know.
230    ///
231    /// How the grammar spells the empty set, and the one spelling of an analyzer set a
232    /// `const` can state: every other set is a list [`Self::labels`] builds.
233    ///
234    /// Named because a surface whose help text states the default analyzer set must read
235    /// that spelling rather than write the word again.
236    pub const NONE_LABEL: &'static str = "none";
237
238    /// Unknown bits mean a record written by a newer build whose extra analyzers cannot
239    /// be honored, so it is refused rather than silently under-reported.
240    pub const fn from_bits(bits: u8) -> Option<Self> {
241        if bits & !Self::KNOWN == 0 { Some(Self(bits)) } else { None }
242    }
243
244    /// Parse the comma-delimited vocabulary both front ends accept, naming the axis as
245    /// the library and the Python API spell it.
246    ///
247    /// Lives here rather than in either front end because it is the axis's grammar, not
248    /// one surface's flag parsing: the CLI and the Python binding must accept exactly the
249    /// same words or the two surfaces disagree about what a request means.
250    ///
251    /// `none` and `all` are totals and cannot be combined with anything, including each
252    /// other — `none,code` has no coherent reading, and silently letting one win is how a
253    /// caller ends up with analysis they did not ask for or did not get.
254    pub fn parse(value: &str) -> Result<Self, String> {
255        Self::parse_labeled(value, "analyze")
256    }
257
258    /// `label` is how the calling surface names this axis in its diagnostics: `--analyze`
259    /// for the CLI, `analyze` for the Python API. Passed in rather than rewritten
260    /// afterwards, because the CLI used to relabel by substring replace and that hit the
261    /// user's own token: `--analyze analyzer` reported `invalid --analyze "--analyzer"`,
262    /// misquoting the very value it was rejecting (fdu-7j6z).
263    pub fn parse_labeled(value: &str, label: &str) -> Result<Self, String> {
264        Self::parse_rejecting(value).map_err(|rejection| rejection.labeled(label))
265    }
266
267    /// [`Self::parse_labeled`], refusing with the value and expectation rather than a
268    /// sentence, so the request model can name the axis in a typed refusal.
269    pub(crate) fn parse_rejecting(value: &str) -> Result<Self, Rejection> {
270        let mut set = Self::NONE;
271        let mut seen: Vec<String> = Vec::new();
272        let mut total: Option<&'static str> = None;
273        for raw in value.split(',') {
274            let token = raw.trim().to_ascii_lowercase();
275            if token.is_empty() {
276                return Err(Rejection::new(value, "empty entry in the list"));
277            }
278            if seen.contains(&token) {
279                return Err(Rejection::new(value, format!("{token:?} appears more than once")));
280            }
281            seen.push(token.clone());
282            match token.as_str() {
283                "none" => total = Some(Self::NONE_LABEL),
284                "all" => {
285                    total = Some("all");
286                    set = Self::ALL;
287                }
288                "lines" => set = set.with_lines(),
289                "code" => set = set.with_code(),
290                "words" => set = set.with_words(),
291                other => {
292                    return Err(Rejection::new(
293                        other,
294                        "expected one of none, lines, code, words, all",
295                    ));
296                }
297            }
298        }
299        if let Some(total) = total {
300            if seen.len() > 1 {
301                return Err(Rejection::new(
302                    value,
303                    format!("{total:?} names the whole axis and cannot be combined"),
304                ));
305            }
306            if total == Self::NONE_LABEL {
307                return Ok(Self::NONE);
308            }
309        }
310        Ok(set)
311    }
312
313    /// Requested analyzers in canonical order, as the CLI and reports spell them.
314    pub fn labels(self) -> Vec<&'static str> {
315        let mut labels = Vec::new();
316        if self.0 & Self::LINES != 0 {
317            labels.push("lines");
318        }
319        if self.includes_code() {
320            labels.push("code");
321        }
322        if self.includes_words() {
323            labels.push("words");
324        }
325        labels
326    }
327}
328
329/// Content-derived classification evidence retained separately from name grouping.
330#[derive(Clone, PartialEq, Eq, Debug)]
331pub struct ContentDetection {
332    /// Type suggested by the bounded content probe.
333    pub file_type: FileTypeId,
334    /// Broad family suggested by the bounded content probe.
335    pub family: ContentFamily,
336    /// Evidence source.
337    pub source: DetectionSource,
338    /// Strength of the evidence.
339    pub confidence: DetectionConfidence,
340    /// Orthogonal generated, vendored, and documentation markers.
341    pub flags: ClassificationFlags,
342}
343
344impl From<Classification> for ContentDetection {
345    fn from(value: Classification) -> Self {
346        Self {
347            file_type: value.file_type,
348            family: value.family,
349            source: value.source,
350            confidence: value.confidence,
351            flags: value.flags,
352        }
353    }
354}
355
356impl From<ContentDetection> for Classification {
357    fn from(value: ContentDetection) -> Self {
358        Self {
359            file_type: value.file_type,
360            family: value.family,
361            source: value.source,
362            confidence: value.confidence,
363            flags: value.flags,
364        }
365    }
366}
367
368/// Settings for one analysis pass.
369#[derive(Clone, Copy, PartialEq, Eq, Debug)]
370pub struct AnalysisRequest {
371    /// Analyzer bundle to run.
372    pub profile: AnalysisSet,
373    /// Maximum worker count; zero selects the available parallelism.
374    pub workers: usize,
375}
376
377impl Default for AnalysisRequest {
378    fn default() -> Self {
379        Self { profile: AnalysisSet::NONE, workers: 0 }
380    }
381}
382
383impl AnalysisRequest {
384    /// Fingerprint only settings that can change a stored answer.
385    pub fn options_fingerprint(self) -> OptionsFingerprint {
386        const OFFSET: u64 = 0xcbf2_9ce4_8422_2325;
387        const PRIME: u64 = 0x0000_0100_0000_01b3;
388        let hash = [self.profile.bits()]
389            .into_iter()
390            .fold(OFFSET, |hash, byte| (hash ^ u64::from(byte)).wrapping_mul(PRIME));
391        OptionsFingerprint(hash)
392    }
393}
394
395/// Analyzer/rule/options identity attached to cached and reported content.
396#[derive(Clone, PartialEq, Eq, Debug)]
397pub struct ContentProvenance {
398    /// Compiled file-type rule identity.
399    pub type_rules_fingerprint: u64,
400    /// Semantic option identity.
401    pub options_fingerprint: OptionsFingerprint,
402    /// Analyzer dialects enabled by the profile.
403    pub analyzers: Vec<(AnalyzerId, AnalyzerVersion)>,
404}
405
406impl ContentProvenance {
407    /// Resolve analyzer dialects implied by a request, under a given set of type rules.
408    ///
409    /// The fingerprint is passed rather than read from a global: a caller may run two
410    /// indexes under different taxonomies in one process, and a record must record the
411    /// rules that actually produced it.
412    pub fn for_request(request: AnalysisRequest, type_rules_fingerprint: u64) -> Self {
413        const VERSION_ONE: AnalyzerVersion = AnalyzerVersion(1);
414        let mut analyzers = Vec::new();
415        if request.profile.is_enabled() {
416            analyzers.push((CONTENT_BASIC, VERSION_ONE));
417        }
418        if request.profile.includes_code() {
419            analyzers.push((CODE_SLOC, AnalyzerVersion(3)));
420        }
421        if request.profile.includes_words() {
422            analyzers.push((TEXT_LOGICAL, VERSION_ONE));
423            // 2: a Markdown file over the exact bound is counted as plain text and its
424            // record says so (fdu-b2qz); a record from 1 held the rendered count.
425            analyzers.push((MARKDOWN_PROSE, AnalyzerVersion(2)));
426        }
427        Self {
428            type_rules_fingerprint,
429            options_fingerprint: request.options_fingerprint(),
430            analyzers,
431        }
432    }
433}
434
435/// Additive sufficient statistics for FlexDoc-style logical word volume.
436#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
437pub struct LogicalWordStats {
438    /// Non-whitespace wide/fullwidth characters, each worth half a logical word.
439    pub wide_chars: u64,
440    /// Whitespace-delimited non-wide tokens.
441    pub nonwide_tokens: u64,
442    /// Non-whitespace non-wide characters used by the 3..6 clamp.
443    pub nonwide_chars: u64,
444}
445
446/// Metrics owned by the always-present line analyzer unit.
447#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
448pub struct BasicMetrics {
449    /// Logical physical lines across admitted text files.
450    pub physical_lines: u64,
451    /// Whitespace-only lines.
452    pub blank_lines: u64,
453    /// Lines containing at least one non-whitespace character.
454    pub nonblank_lines: u64,
455    /// Whitespace-delimited words before document projection.
456    pub raw_words: u64,
457}
458
459/// Metrics owned by the code analyzer unit.
460#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
461pub struct CodeMetrics {
462    /// Code-bearing lines.
463    pub code_lines: u64,
464    /// Comment-only lines.
465    pub comment_lines: u64,
466    /// Blank lines under the code analyzer's syntax.
467    pub code_blank_lines: u64,
468}
469
470/// Metrics owned by the word analyzer unit.
471#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
472pub struct WordMetrics {
473    /// Plain-text paragraph runs or visible Markdown paragraphs.
474    pub paragraphs: u64,
475    /// Reader-visible Markdown words.
476    pub visible_words: u64,
477    /// Additive logical-word sufficient statistics.
478    pub logical_word_stats: LogicalWordStats,
479    /// Reader-visible Markdown logical-word sufficient statistics.
480    pub visible_logical_word_stats: LogicalWordStats,
481}
482
483/// One analyzer unit's explicit coverage and optional successful value.
484#[derive(Clone, Copy, PartialEq, Eq, Debug)]
485pub struct AnalyzerOutcome<T> {
486    /// Why the unit did or did not produce a value.
487    coverage: CoverageReason,
488    /// Successful measured value; absent for every non-analyzed outcome.
489    value: Option<T>,
490}
491
492impl<T> AnalyzerOutcome<T> {
493    /// A successful analyzer result.
494    pub(crate) const fn analyzed(value: T) -> Self {
495        Self { coverage: CoverageReason::Analyzed, value: Some(value) }
496    }
497
498    /// A unit that could not produce a value for the named reason.
499    pub(crate) fn unavailable(coverage: CoverageReason) -> Self {
500        assert!(
501            !matches!(coverage, CoverageReason::Analyzed | CoverageReason::TextOnly),
502            "an outcome with a value must carry one"
503        );
504        Self { coverage, value: None }
505    }
506
507    /// A value counted from the file as plain text ([`CoverageReason::TextOnly`]).
508    pub(crate) const fn text_only(value: T) -> Self {
509        Self { coverage: CoverageReason::TextOnly, value: Some(value) }
510    }
511
512    /// Coverage outcome for this unit.
513    pub const fn coverage(&self) -> CoverageReason {
514        self.coverage
515    }
516
517    /// Measured value: present for an analyzed outcome and for one counted as plain
518    /// text, absent for every unavailable outcome.
519    pub const fn value(self) -> Option<T>
520    where
521        T: Copy,
522    {
523        self.value
524    }
525
526    pub(crate) fn from_parts(coverage: CoverageReason, value: Option<T>) -> Option<Self> {
527        if matches!(coverage, CoverageReason::Analyzed | CoverageReason::TextOnly)
528            == value.is_some()
529        {
530            Some(Self { coverage, value })
531        } else {
532            None
533        }
534    }
535
536    /// Operational failures must be retried rather than treated as cache hits.
537    pub const fn is_reusable(&self) -> bool {
538        !matches!(self.coverage, CoverageReason::IoError | CoverageReason::ChangedDuringRead)
539    }
540}
541
542impl LogicalWordStats {
543    pub(crate) fn add_assign(&mut self, other: Self) {
544        self.wide_chars = self.wide_chars.saturating_add(other.wide_chars);
545        self.nonwide_tokens = self.nonwide_tokens.saturating_add(other.nonwide_tokens);
546        self.nonwide_chars = self.nonwide_chars.saturating_add(other.nonwide_chars);
547    }
548
549    /// Derive rounded logical words after aggregation.
550    pub fn logical_words(self) -> u64 {
551        let chars = u128::from(self.nonwide_chars);
552        let tokens = u128::from(self.nonwide_tokens);
553        let wide = u128::from(self.wide_chars);
554        let (numerator, denominator) = if tokens.saturating_mul(6) < chars {
555            (chars.saturating_add(wide.saturating_mul(3)), 6)
556        } else if tokens.saturating_mul(3) > chars {
557            (chars.saturating_mul(2).saturating_add(wide.saturating_mul(3)), 6)
558        } else {
559            (tokens.saturating_mul(2).saturating_add(wide), 2)
560        };
561        let rounded = numerator.saturating_add(denominator / 2) / denominator;
562        u64::try_from(rounded).unwrap_or(u64::MAX)
563    }
564}
565
566/// Fixed additive metric slots shipped by the first content schema.
567#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
568pub struct MetricValues {
569    /// Logical physical lines across accepted text files.
570    pub physical_lines: u64,
571    /// Whitespace-only lines.
572    pub blank_lines: u64,
573    /// Lines containing at least one non-whitespace character.
574    pub nonblank_lines: u64,
575    /// Whitespace-delimited prose words before markup projection.
576    pub raw_words: u64,
577    /// Code-bearing lines under `code-sloc-v1`.
578    pub code_lines: u64,
579    /// Comment-only lines under `code-sloc-v1`.
580    pub comment_lines: u64,
581    /// Blank lines under `code-sloc-v1`, distinct from whitespace-only source lines.
582    pub code_blank_lines: u64,
583    /// Plain-text paragraph runs.
584    pub paragraphs: u64,
585    /// Reader-visible Markdown words.
586    pub visible_words: u64,
587    /// Additive logical-word sufficient statistics.
588    pub logical_word_stats: LogicalWordStats,
589    /// Reader-visible Markdown logical-word sufficient statistics.
590    pub visible_logical_word_stats: LogicalWordStats,
591}
592
593/// Why a requested file did or did not produce metrics.
594#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
595pub enum CoverageReason {
596    /// Requested analyzers completed.
597    Analyzed,
598    /// Known binary type or a NUL byte made text metrics inapplicable.
599    Binary,
600    /// Input was not valid UTF-8.
601    InvalidUtf8,
602    /// A recognized Unicode byte-order mark names an encoding no analyzer decodes.
603    UnsupportedEncoding,
604    /// No shipped analyzer accepts this type.
605    Unsupported,
606    /// File I/O failed; the human error is retained separately.
607    IoError,
608    /// Metadata changed while the file was being read.
609    ChangedDuringRead,
610    /// The unit produced a value by counting the file as plain text rather than as the
611    /// document it is: a Markdown file over the exact bound under the words unit, whose
612    /// every word is counted visible and whose paragraphs are its blank-line runs.
613    TextOnly,
614}
615
616/// Sparse analysis record for one regular file.
617#[derive(Clone, PartialEq, Eq, Debug)]
618pub struct FileAnalysis {
619    /// Metadata fingerprint this result describes.
620    pub fingerprint: Fingerprint,
621    /// Apparent bytes represented by the record.
622    pub bytes: u64,
623    /// Bounded content evidence, kept separate from name-based grouping.
624    pub detection: ContentDetection,
625    /// Shared physical-line and raw-word outcome.
626    pub lines: AnalyzerOutcome<BasicMetrics>,
627    /// Code outcome when the request included the code unit.
628    pub code: Option<AnalyzerOutcome<CodeMetrics>>,
629    /// Word outcome when the request included the word unit.
630    pub words: Option<AnalyzerOutcome<WordMetrics>>,
631    /// Optional path-specific failure detail.
632    pub error: Option<String>,
633}
634
635impl FileAnalysis {
636    /// Whether optional unit slots exactly match the tier's requested analyzer set.
637    pub const fn matches_profile(&self, profile: AnalysisSet) -> bool {
638        self.code.is_some() == profile.includes_code()
639            && self.words.is_some() == profile.includes_words()
640    }
641
642    /// File-level operational failure, counted once even though it affects every unit.
643    pub const fn operational_failure(&self) -> Option<CoverageReason> {
644        match self.lines.coverage() {
645            CoverageReason::IoError => Some(CoverageReason::IoError),
646            CoverageReason::ChangedDuringRead => Some(CoverageReason::ChangedDuringRead),
647            CoverageReason::Analyzed
648            | CoverageReason::TextOnly
649            | CoverageReason::Binary
650            | CoverageReason::InvalidUtf8
651            | CoverageReason::UnsupportedEncoding
652            | CoverageReason::Unsupported => None,
653        }
654    }
655
656    /// Whether every retained requested unit is safe to reuse.
657    pub const fn is_reusable(&self) -> bool {
658        self.operational_failure().is_none()
659            && self.lines.is_reusable()
660            && match self.code {
661                Some(outcome) => outcome.is_reusable(),
662                None => true,
663            }
664            && match self.words {
665                Some(outcome) => outcome.is_reusable(),
666                None => true,
667            }
668    }
669}
670
671/// Owned immutable candidate captured before worker execution.
672///
673/// Crate-private with [`Index::analysis_candidates`] and [`Index::apply_analysis`] until
674/// the request model (P1.3) decides whether an out-of-crate analyzer is a supported
675/// surface (`fdu-5upj`): the tier must be prepared for a candidate's identity before a
676/// result for it can commit, and preparation is crate-private.
677///
678/// [`Index::analysis_candidates`]: crate::Index::analysis_candidates
679/// [`Index::apply_analysis`]: crate::Index::apply_analysis
680#[derive(Clone, Debug)]
681pub(crate) struct AnalysisCandidate {
682    /// Generation-safe index identity.
683    pub entry_id: EntryId,
684    /// Entry revision at capture time.
685    pub revision: u64,
686    /// Path relative to the index root.
687    pub relative_path: PathBuf,
688    /// Absolute filesystem path.
689    pub absolute_path: PathBuf,
690    /// Last observed attributes.
691    pub attrs: Attrs,
692    /// Metadata-only classification.
693    pub classification: Classification,
694}
695
696/// Live file identity used to match a sidecar record on cache-only restore.
697///
698/// Restore does not classify and does not open the file. The sidecar already stores the
699/// classification that `apply_analysis` would have committed, and the apply-path
700/// classify self-check is a crate-private consistency guard, not part of the answer.
701#[derive(Clone, Debug)]
702pub(crate) struct RestoreCandidate {
703    /// Generation-safe index identity.
704    pub entry_id: EntryId,
705    /// Entry revision at capture time.
706    pub revision: u64,
707    /// Path relative to the index root.
708    pub relative_path: PathBuf,
709    /// Last observed attributes.
710    pub attrs: Attrs,
711}
712
713/// Worker result submitted to the index's derived-data mutation boundary.
714#[derive(Clone, Debug)]
715pub(crate) struct AnalysisObservation {
716    /// Candidate identity and expectation.
717    pub candidate: AnalysisCandidate,
718    /// Analyzer set whose tier may accept the result.
719    pub profile: AnalysisSet,
720    /// Analyzer identity whose tier may accept the result.
721    pub provenance: ContentProvenance,
722    /// Completed or skipped analysis record.
723    pub analysis: FileAnalysis,
724}
725
726/// Result of conditionally committing one worker observation.
727#[derive(Clone, Copy, PartialEq, Eq, Debug)]
728pub(crate) enum AnalysisApplyOutcome {
729    /// The sparse record and ancestor rollups changed.
730    Applied,
731    /// The result was discarded: metadata changed after candidate capture, or the content
732    /// tier holds another identity than the one the result was produced under, which is
733    /// the same answer because both mean the result describes something else.
734    Stale,
735}
736
737#[cfg(test)]
738mod tests {
739    use std::collections::HashSet;
740
741    use super::{
742        AnalysisRequest, AnalysisSet, AnalyzerOutcome, AnalyzerVersion, CODE_SLOC,
743        ContentProvenance, CoverageReason, LogicalWordStats, METRICS,
744    };
745
746    #[test]
747    #[should_panic(expected = "an outcome with a value must carry one")]
748    fn unavailable_outcome_cannot_claim_success() {
749        let _: AnalyzerOutcome<()> = AnalyzerOutcome::unavailable(CoverageReason::Analyzed);
750    }
751
752    #[test]
753    fn metric_registry_has_unique_names_and_one_requestable_owner_each() {
754        let mut names = HashSet::new();
755        for metric in METRICS {
756            assert!(names.insert(metric.name), "duplicate metric name {}", metric.name);
757            assert!(
758                matches!(
759                    metric.owner,
760                    AnalysisSet::LINES_ONLY | AnalysisSet::CODE_ONLY | AnalysisSet::WORDS_ONLY
761                ),
762                "{} has a non-unit owner {:?}",
763                metric.name,
764                metric.owner
765            );
766        }
767        assert_eq!(names.len(), 12);
768    }
769
770    #[test]
771    fn logical_words_derive_only_after_additive_stats_are_combined() {
772        let first = LogicalWordStats { wide_chars: 3, nonwide_tokens: 1, nonwide_chars: 12 };
773        let second = LogicalWordStats { wide_chars: 1, nonwide_tokens: 9, nonwide_chars: 6 };
774        let combined = LogicalWordStats {
775            wide_chars: first.wide_chars + second.wide_chars,
776            nonwide_tokens: first.nonwide_tokens + second.nonwide_tokens,
777            nonwide_chars: first.nonwide_chars + second.nonwide_chars,
778        };
779        assert_eq!(combined.logical_words(), 8);
780    }
781
782    #[test]
783    fn logical_words_match_the_pinned_rational_clamp_and_half_up_rounding() {
784        let logical = |wide_chars, nonwide_tokens, nonwide_chars| {
785            LogicalWordStats { wide_chars, nonwide_tokens, nonwide_chars }.logical_words()
786        };
787        assert_eq!(logical(0, 0, 0), 0);
788        assert_eq!(logical(0, 2, 9), 2, "ordinary prose passes through");
789        assert_eq!(logical(0, 1, 12), 2, "long tokens use the six-character floor");
790        assert_eq!(logical(0, 4, 4), 1, "short tokens use the three-character ceiling");
791        assert_eq!(logical(3, 0, 0), 2, "wide halves round up once");
792        assert_eq!(logical(1, 1, 1), 1, "mixed fractions combine before rounding");
793    }
794
795    #[test]
796    fn the_analyzer_vocabulary_parses_every_accepted_spelling() {
797        let cases = [
798            ("none", AnalysisSet::NONE),
799            ("lines", AnalysisSet::NONE.with_lines()),
800            ("code", AnalysisSet::NONE.with_code()),
801            ("words", AnalysisSet::NONE.with_words()),
802            ("all", AnalysisSet::ALL),
803            // Order is irrelevant: a set has no order, so neither spelling is preferred.
804            ("code,words", AnalysisSet::ALL),
805            ("words,code", AnalysisSet::ALL),
806            // `lines` is already implied by any analyzer, so naming it adds nothing.
807            ("lines,code", AnalysisSet::NONE.with_code()),
808            // Whitespace and case are incidental, as in every other list flag.
809            (" CODE , Words ", AnalysisSet::ALL),
810        ];
811        for (input, expected) in cases {
812            assert_eq!(AnalysisSet::parse(input), Ok(expected), "parsing {input:?}");
813        }
814    }
815
816    #[test]
817    fn the_analyzer_vocabulary_rejects_every_incoherent_request() {
818        // Each rejection names what was wrong; a set flag that silently drops a token is
819        // how a caller ends up believing it measured something it did not.
820        let cases = [
821            ("", "empty entry"),
822            ("code,,words", "empty entry"),
823            ("code,code", "more than once"),
824            ("basic", "expected one of"),
825            ("documents", "expected one of"),
826            ("full", "expected one of"),
827            ("none,code", "cannot be combined"),
828            ("all,code", "cannot be combined"),
829            ("none,all", "cannot be combined"),
830        ];
831        for (input, needle) in cases {
832            let error = AnalysisSet::parse(input).expect_err(&format!("{input:?} must fail"));
833            assert!(error.contains(needle), "parsing {input:?} said {error:?}, wanted {needle:?}");
834        }
835    }
836
837    /// The CLI used to relabel by substring replace, which rewrote the user's own token:
838    /// `--analyze analyzer` reported `invalid --analyze "--analyzer"`, misquoting the very
839    /// value it was rejecting. The label is a parameter now, so the value is untouched.
840    #[test]
841    fn a_label_never_rewrites_the_value_it_is_reporting() {
842        for value in ["analyzer", "reanalyze", "analyze-all"] {
843            let error = AnalysisSet::parse_labeled(value, "--analyze")
844                .expect_err("must reject an unknown analyzer");
845            assert!(
846                error.contains(&format!("{value:?}")),
847                "{error} must quote {value:?} exactly as typed"
848            );
849            assert!(error.starts_with("invalid --analyze "), "{error} must carry the label");
850        }
851    }
852
853    #[test]
854    fn the_on_disk_encoding_round_trips_and_refuses_unknown_analyzers() {
855        for set in [
856            AnalysisSet::NONE,
857            AnalysisSet::NONE.with_lines(),
858            AnalysisSet::NONE.with_code(),
859            AnalysisSet::NONE.with_words(),
860            AnalysisSet::ALL,
861        ] {
862            assert_eq!(AnalysisSet::from_bits(set.bits()), Some(set));
863        }
864        // A record written by a build with an analyzer this one lacks cannot be honored,
865        // so it is refused rather than silently under-reported as absent metrics.
866        assert_eq!(AnalysisSet::from_bits(0b1000_0000), None);
867    }
868
869    #[test]
870    fn code_metrics_use_the_updated_analyzer_version() {
871        let request = AnalysisRequest { profile: AnalysisSet::NONE.with_code(), workers: 1 };
872        let provenance = ContentProvenance::for_request(request, 42);
873        assert!(provenance.analyzers.contains(&(CODE_SLOC, AnalyzerVersion(3))));
874    }
875
876    #[test]
877    fn labels_are_the_vocabulary_parse_accepts() {
878        for set in [
879            AnalysisSet::NONE.with_lines(),
880            AnalysisSet::NONE.with_code(),
881            AnalysisSet::NONE.with_words(),
882            AnalysisSet::ALL,
883        ] {
884            let spelled = set.labels().join(",");
885            assert_eq!(AnalysisSet::parse(&spelled), Ok(set), "round trip through {spelled:?}");
886        }
887        assert!(AnalysisSet::NONE.labels().is_empty());
888    }
889
890    /// The shortest spelling of a set is what a caller types, and parses back to the set.
891    #[test]
892    fn a_request_label_is_the_shortest_spelling_that_parses_back() {
893        for (set, named, label) in [
894            (AnalysisSet::NONE, &[][..], "none"),
895            (AnalysisSet::LINES_ONLY, &["lines"][..], "lines"),
896            (AnalysisSet::CODE_ONLY, &["code"][..], "code"),
897            (AnalysisSet::WORDS_ONLY, &["words"][..], "words"),
898            (AnalysisSet::ALL, &["code", "words"][..], "all"),
899        ] {
900            assert_eq!(set.named(), named, "{set:?}");
901            assert_eq!(set.request_label(), label, "{set:?}");
902            assert_eq!(AnalysisSet::parse(label), Ok(set), "round trip through {label:?}");
903        }
904        assert_eq!(AnalysisSet::CODE_ONLY.union(AnalysisSet::WORDS_ONLY), AnalysisSet::ALL);
905        assert_eq!(AnalysisSet::NONE.union(AnalysisSet::LINES_ONLY), AnalysisSet::LINES_ONLY);
906    }
907}