Skip to main content

fdu_core/content/
content_model.rs

1//! Versioned content-analysis contracts shared across the engine and report layers.
2
3use std::path::PathBuf;
4
5use crate::classify::{
6    Classification, ClassificationFlags, ContentFamily, DetectionConfidence, DetectionSource,
7    FileTypeId,
8};
9use crate::query::Rejection;
10use crate::{Attrs, EntryId, Fingerprint};
11
12/// Stable analyzer identity.
13#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
14pub struct AnalyzerId(pub &'static str);
15
16/// Version of an analyzer's counting semantics.
17#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
18pub struct AnalyzerVersion(pub u16);
19
20/// Definition of one measured value exposed by content reports.
21#[derive(Clone, Copy, PartialEq, Eq, Debug)]
22pub struct MetricDef {
23    /// Stable report key.
24    pub name: &'static str,
25    /// Requestable unit that owns the value's presence.
26    pub owner: AnalysisSet,
27    /// Analyzer dialect that defines the value.
28    pub analyzer: AnalyzerId,
29    /// Short semantic definition.
30    pub doc: &'static str,
31}
32
33/// Fingerprint of semantic analyzer options; operational worker count is excluded.
34#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
35pub struct OptionsFingerprint(pub u64);
36
37/// Fused physical-line and raw-word analyzer.
38pub const CONTENT_BASIC: AnalyzerId = AnalyzerId("content-basic-v1");
39/// Common-language code/comment/blank analyzer.
40pub const CODE_SLOC: AnalyzerId = AnalyzerId("code-sloc-v1");
41/// Plain-text logical word and paragraph analyzer.
42pub const TEXT_LOGICAL: AnalyzerId = AnalyzerId("text-logical-v1");
43/// Reader-visible Markdown prose analyzer.
44pub const MARKDOWN_PROSE: AnalyzerId = AnalyzerId("markdown-prose-v1");
45
46/// The single registry of content metric names, owners, and definitions.
47pub const METRICS: &[MetricDef] = &[
48    MetricDef {
49        name: "physical_lines",
50        owner: AnalysisSet::LINES_ONLY,
51        analyzer: CONTENT_BASIC,
52        doc: "Logical physical lines across admitted text files.",
53    },
54    MetricDef {
55        name: "blank_lines",
56        owner: AnalysisSet::LINES_ONLY,
57        analyzer: CONTENT_BASIC,
58        doc: "Whitespace-only physical lines.",
59    },
60    MetricDef {
61        name: "nonblank_lines",
62        owner: AnalysisSet::LINES_ONLY,
63        analyzer: CONTENT_BASIC,
64        doc: "Physical lines containing non-whitespace text.",
65    },
66    MetricDef {
67        name: "raw_words",
68        owner: AnalysisSet::LINES_ONLY,
69        analyzer: CONTENT_BASIC,
70        doc: "Whitespace-delimited words before document projection.",
71    },
72    MetricDef {
73        name: "code_lines",
74        owner: AnalysisSet::CODE_ONLY,
75        analyzer: CODE_SLOC,
76        doc: "Code-bearing lines in supported source languages.",
77    },
78    MetricDef {
79        name: "comment_lines",
80        owner: AnalysisSet::CODE_ONLY,
81        analyzer: CODE_SLOC,
82        doc: "Comment-only lines in supported source languages.",
83    },
84    MetricDef {
85        name: "code_blank_lines",
86        owner: AnalysisSet::CODE_ONLY,
87        analyzer: CODE_SLOC,
88        doc: "Blank lines under the code analyzer's syntax.",
89    },
90    MetricDef {
91        name: "logical_words",
92        owner: AnalysisSet::WORDS_ONLY,
93        analyzer: TEXT_LOGICAL,
94        doc: "Normalized logical word volume.",
95    },
96    MetricDef {
97        name: "paragraphs",
98        owner: AnalysisSet::WORDS_ONLY,
99        analyzer: TEXT_LOGICAL,
100        doc: "Plain-text runs or reader-visible Markdown paragraphs.",
101    },
102    MetricDef {
103        name: "visible_words",
104        owner: AnalysisSet::WORDS_ONLY,
105        analyzer: MARKDOWN_PROSE,
106        doc: "Reader-visible Markdown words.",
107    },
108    MetricDef {
109        name: "visible_logical_words",
110        owner: AnalysisSet::WORDS_ONLY,
111        analyzer: MARKDOWN_PROSE,
112        doc: "Normalized reader-visible Markdown words.",
113    },
114    MetricDef {
115        name: "document_words",
116        owner: AnalysisSet::WORDS_ONLY,
117        analyzer: TEXT_LOGICAL,
118        doc: "Logical words after the document-type projection.",
119    },
120];
121
122/// The set of content analyzers a request enables.
123///
124/// A set rather than a ladder, because the analyzers are independent: `code` and `words`
125/// measure different things over different families and either is useful without the
126/// other.  An ordered enum could name only the combinations somebody thought to
127/// enumerate — four of the eight this registry already permits — and it made
128/// `text-logical-v1` without `markdown-prose-v1` unreachable.
129///
130/// `lines` is the base every analyzer shares: any analyzer that runs has already
131/// streamed the file, so line counts cost nothing extra.  It is therefore implicit in
132/// every non-empty set rather than something a caller must remember to request, which is
133/// why each `with_*` constructor sets it.
134#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug, Default)]
135pub struct AnalysisSet(u8);
136
137impl AnalysisSet {
138    const LINES: u8 = 1 << 0;
139    const CODE: u8 = 1 << 1;
140    const WORDS: u8 = 1 << 2;
141    const KNOWN: u8 = Self::LINES | Self::CODE | Self::WORDS;
142
143    /// Open no file; preserve the metadata-only behavior.
144    pub const NONE: Self = Self(0);
145    /// Physical-line and raw-word unit.
146    pub const LINES_ONLY: Self = Self(Self::LINES);
147    /// Code unit, including its shared line pass.
148    pub const CODE_ONLY: Self = Self(Self::LINES | Self::CODE);
149    /// Word unit, including its shared line pass.
150    pub const WORDS_ONLY: Self = Self(Self::LINES | Self::WORDS);
151    /// Every registered analyzer.
152    pub const ALL: Self = Self(Self::KNOWN);
153
154    /// Add physical, blank, and nonblank line counts.
155    #[must_use]
156    pub const fn with_lines(self) -> Self {
157        Self(self.0 | Self::LINES)
158    }
159
160    /// Add common-language standard SLOC over the `code` family.
161    #[must_use]
162    pub const fn with_code(self) -> Self {
163        Self(self.0 | Self::LINES | Self::CODE)
164    }
165
166    /// Add raw, normalized, and reader-visible word volume.
167    #[must_use]
168    pub const fn with_words(self) -> Self {
169        Self(self.0 | Self::LINES | Self::WORDS)
170    }
171
172    /// Whether any source file may be opened.
173    pub const fn is_enabled(self) -> bool {
174        self.0 != 0
175    }
176
177    /// Whether standard SLOC is requested.
178    pub const fn includes_code(self) -> bool {
179        self.0 & Self::CODE != 0
180    }
181
182    /// Whether logical and visible word metrics are requested.
183    pub const fn includes_words(self) -> bool {
184        self.0 & Self::WORDS != 0
185    }
186
187    /// Whether this request includes every unit in `other`.
188    pub const fn contains(self, other: Self) -> bool {
189        self.0 & other.0 == other.0
190    }
191
192    /// Stable on-disk and fingerprint encoding.
193    pub const fn bits(self) -> u8 {
194        self.0
195    }
196
197    /// Decode [`Self::bits`], rejecting any analyzer this build does not know.
198    ///
199    /// How the grammar spells the empty set, and the one spelling of an analyzer set a
200    /// `const` can state: every other set is a list [`Self::labels`] builds.
201    ///
202    /// Named because a surface whose help text states the default analyzer set must read
203    /// that spelling rather than write the word again.
204    pub const NONE_LABEL: &'static str = "none";
205
206    /// Unknown bits mean a record written by a newer build whose extra analyzers cannot
207    /// be honored, so it is refused rather than silently under-reported.
208    pub const fn from_bits(bits: u8) -> Option<Self> {
209        if bits & !Self::KNOWN == 0 { Some(Self(bits)) } else { None }
210    }
211
212    /// Parse the comma-delimited vocabulary both front ends accept, naming the axis as
213    /// the library and the Python API spell it.
214    ///
215    /// Lives here rather than in either front end because it is the axis's grammar, not
216    /// one surface's flag parsing: the CLI and the Python binding must accept exactly the
217    /// same words or the two surfaces disagree about what a request means.
218    ///
219    /// `none` and `all` are totals and cannot be combined with anything, including each
220    /// other — `none,code` has no coherent reading, and silently letting one win is how a
221    /// caller ends up with analysis they did not ask for or did not get.
222    pub fn parse(value: &str) -> Result<Self, String> {
223        Self::parse_labeled(value, "analyze")
224    }
225
226    /// `label` is how the calling surface names this axis in its diagnostics: `--analyze`
227    /// for the CLI, `analyze` for the Python API. Passed in rather than rewritten
228    /// afterwards, because the CLI used to relabel by substring replace and that hit the
229    /// user's own token: `--analyze analyzer` reported `invalid --analyze "--analyzer"`,
230    /// misquoting the very value it was rejecting (fdu-7j6z).
231    pub fn parse_labeled(value: &str, label: &str) -> Result<Self, String> {
232        Self::parse_rejecting(value).map_err(|rejection| rejection.labeled(label))
233    }
234
235    /// [`Self::parse_labeled`], refusing with the value and expectation rather than a
236    /// sentence, so the request model can name the axis in a typed refusal.
237    pub(crate) fn parse_rejecting(value: &str) -> Result<Self, Rejection> {
238        let mut set = Self::NONE;
239        let mut seen: Vec<String> = Vec::new();
240        let mut total: Option<&'static str> = None;
241        for raw in value.split(',') {
242            let token = raw.trim().to_ascii_lowercase();
243            if token.is_empty() {
244                return Err(Rejection::new(value, "empty entry in the list"));
245            }
246            if seen.contains(&token) {
247                return Err(Rejection::new(value, format!("{token:?} appears more than once")));
248            }
249            seen.push(token.clone());
250            match token.as_str() {
251                "none" => total = Some(Self::NONE_LABEL),
252                "all" => {
253                    total = Some("all");
254                    set = Self::ALL;
255                }
256                "lines" => set = set.with_lines(),
257                "code" => set = set.with_code(),
258                "words" => set = set.with_words(),
259                other => {
260                    return Err(Rejection::new(
261                        other,
262                        "expected one of none, lines, code, words, all",
263                    ));
264                }
265            }
266        }
267        if let Some(total) = total {
268            if seen.len() > 1 {
269                return Err(Rejection::new(
270                    value,
271                    format!("{total:?} names the whole axis and cannot be combined"),
272                ));
273            }
274            if total == Self::NONE_LABEL {
275                return Ok(Self::NONE);
276            }
277        }
278        Ok(set)
279    }
280
281    /// Requested analyzers in canonical order, as the CLI and reports spell them.
282    pub fn labels(self) -> Vec<&'static str> {
283        let mut labels = Vec::new();
284        if self.0 & Self::LINES != 0 {
285            labels.push("lines");
286        }
287        if self.includes_code() {
288            labels.push("code");
289        }
290        if self.includes_words() {
291            labels.push("words");
292        }
293        labels
294    }
295}
296
297/// Content-derived classification evidence retained separately from name grouping.
298#[derive(Clone, PartialEq, Eq, Debug)]
299pub struct ContentDetection {
300    /// Type suggested by the bounded content probe.
301    pub file_type: FileTypeId,
302    /// Broad family suggested by the bounded content probe.
303    pub family: ContentFamily,
304    /// Evidence source.
305    pub source: DetectionSource,
306    /// Strength of the evidence.
307    pub confidence: DetectionConfidence,
308    /// Orthogonal generated, vendored, and documentation markers.
309    pub flags: ClassificationFlags,
310}
311
312impl From<Classification> for ContentDetection {
313    fn from(value: Classification) -> Self {
314        Self {
315            file_type: value.file_type,
316            family: value.family,
317            source: value.source,
318            confidence: value.confidence,
319            flags: value.flags,
320        }
321    }
322}
323
324impl From<ContentDetection> for Classification {
325    fn from(value: ContentDetection) -> Self {
326        Self {
327            file_type: value.file_type,
328            family: value.family,
329            source: value.source,
330            confidence: value.confidence,
331            flags: value.flags,
332        }
333    }
334}
335
336/// Settings for one analysis pass.
337#[derive(Clone, Copy, PartialEq, Eq, Debug)]
338pub struct AnalysisRequest {
339    /// Analyzer bundle to run.
340    pub profile: AnalysisSet,
341    /// Maximum worker count; zero selects the available parallelism.
342    pub workers: usize,
343}
344
345impl Default for AnalysisRequest {
346    fn default() -> Self {
347        Self { profile: AnalysisSet::NONE, workers: 0 }
348    }
349}
350
351impl AnalysisRequest {
352    /// Fingerprint only settings that can change a stored answer.
353    pub fn options_fingerprint(self) -> OptionsFingerprint {
354        const OFFSET: u64 = 0xcbf2_9ce4_8422_2325;
355        const PRIME: u64 = 0x0000_0100_0000_01b3;
356        let hash = [self.profile.bits()]
357            .into_iter()
358            .fold(OFFSET, |hash, byte| (hash ^ u64::from(byte)).wrapping_mul(PRIME));
359        OptionsFingerprint(hash)
360    }
361}
362
363/// Analyzer/rule/options identity attached to cached and reported content.
364#[derive(Clone, PartialEq, Eq, Debug)]
365pub struct ContentProvenance {
366    /// Compiled file-type rule identity.
367    pub type_rules_fingerprint: u64,
368    /// Semantic option identity.
369    pub options_fingerprint: OptionsFingerprint,
370    /// Analyzer dialects enabled by the profile.
371    pub analyzers: Vec<(AnalyzerId, AnalyzerVersion)>,
372}
373
374impl ContentProvenance {
375    /// Resolve analyzer dialects implied by a request, under a given set of type rules.
376    ///
377    /// The fingerprint is passed rather than read from a global: a caller may run two
378    /// indexes under different taxonomies in one process, and a record must record the
379    /// rules that actually produced it.
380    pub fn for_request(request: AnalysisRequest, type_rules_fingerprint: u64) -> Self {
381        const VERSION_ONE: AnalyzerVersion = AnalyzerVersion(1);
382        let mut analyzers = Vec::new();
383        if request.profile.is_enabled() {
384            analyzers.push((CONTENT_BASIC, VERSION_ONE));
385        }
386        if request.profile.includes_code() {
387            analyzers.push((CODE_SLOC, AnalyzerVersion(3)));
388        }
389        if request.profile.includes_words() {
390            analyzers.push((TEXT_LOGICAL, VERSION_ONE));
391            // 2: a Markdown file over the exact bound is counted as plain text and its
392            // record says so (fdu-b2qz); a record from 1 held the rendered count.
393            analyzers.push((MARKDOWN_PROSE, AnalyzerVersion(2)));
394        }
395        Self {
396            type_rules_fingerprint,
397            options_fingerprint: request.options_fingerprint(),
398            analyzers,
399        }
400    }
401}
402
403/// Additive sufficient statistics for FlexDoc-style logical word volume.
404#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
405pub struct LogicalWordStats {
406    /// Non-whitespace wide/fullwidth characters, each worth half a logical word.
407    pub wide_chars: u64,
408    /// Whitespace-delimited non-wide tokens.
409    pub nonwide_tokens: u64,
410    /// Non-whitespace non-wide characters used by the 3..6 clamp.
411    pub nonwide_chars: u64,
412}
413
414/// Metrics owned by the always-present line analyzer unit.
415#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
416pub struct BasicMetrics {
417    /// Logical physical lines across admitted text files.
418    pub physical_lines: u64,
419    /// Whitespace-only lines.
420    pub blank_lines: u64,
421    /// Lines containing at least one non-whitespace character.
422    pub nonblank_lines: u64,
423    /// Whitespace-delimited words before document projection.
424    pub raw_words: u64,
425}
426
427/// Metrics owned by the code analyzer unit.
428#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
429pub struct CodeMetrics {
430    /// Code-bearing lines.
431    pub code_lines: u64,
432    /// Comment-only lines.
433    pub comment_lines: u64,
434    /// Blank lines under the code analyzer's syntax.
435    pub code_blank_lines: u64,
436}
437
438/// Metrics owned by the word analyzer unit.
439#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
440pub struct WordMetrics {
441    /// Plain-text paragraph runs or visible Markdown paragraphs.
442    pub paragraphs: u64,
443    /// Reader-visible Markdown words.
444    pub visible_words: u64,
445    /// Additive logical-word sufficient statistics.
446    pub logical_word_stats: LogicalWordStats,
447    /// Reader-visible Markdown logical-word sufficient statistics.
448    pub visible_logical_word_stats: LogicalWordStats,
449}
450
451/// One analyzer unit's explicit coverage and optional successful value.
452#[derive(Clone, Copy, PartialEq, Eq, Debug)]
453pub struct AnalyzerOutcome<T> {
454    /// Why the unit did or did not produce a value.
455    coverage: CoverageReason,
456    /// Successful measured value; absent for every non-analyzed outcome.
457    value: Option<T>,
458}
459
460impl<T> AnalyzerOutcome<T> {
461    /// A successful analyzer result.
462    pub(crate) const fn analyzed(value: T) -> Self {
463        Self { coverage: CoverageReason::Analyzed, value: Some(value) }
464    }
465
466    /// A unit that could not produce a value for the named reason.
467    pub(crate) fn unavailable(coverage: CoverageReason) -> Self {
468        assert!(
469            !matches!(coverage, CoverageReason::Analyzed | CoverageReason::TextOnly),
470            "an outcome with a value must carry one"
471        );
472        Self { coverage, value: None }
473    }
474
475    /// A value counted from the file as plain text ([`CoverageReason::TextOnly`]).
476    pub(crate) const fn text_only(value: T) -> Self {
477        Self { coverage: CoverageReason::TextOnly, value: Some(value) }
478    }
479
480    /// Coverage outcome for this unit.
481    pub const fn coverage(&self) -> CoverageReason {
482        self.coverage
483    }
484
485    /// Measured value: present for an analyzed outcome and for one counted as plain
486    /// text, absent for every unavailable outcome.
487    pub const fn value(self) -> Option<T>
488    where
489        T: Copy,
490    {
491        self.value
492    }
493
494    pub(crate) fn from_parts(coverage: CoverageReason, value: Option<T>) -> Option<Self> {
495        if matches!(coverage, CoverageReason::Analyzed | CoverageReason::TextOnly)
496            == value.is_some()
497        {
498            Some(Self { coverage, value })
499        } else {
500            None
501        }
502    }
503
504    /// Operational failures must be retried rather than treated as cache hits.
505    pub const fn is_reusable(&self) -> bool {
506        !matches!(self.coverage, CoverageReason::IoError | CoverageReason::ChangedDuringRead)
507    }
508}
509
510impl LogicalWordStats {
511    pub(crate) fn add_assign(&mut self, other: Self) {
512        self.wide_chars = self.wide_chars.saturating_add(other.wide_chars);
513        self.nonwide_tokens = self.nonwide_tokens.saturating_add(other.nonwide_tokens);
514        self.nonwide_chars = self.nonwide_chars.saturating_add(other.nonwide_chars);
515    }
516
517    /// Derive rounded logical words after aggregation.
518    pub fn logical_words(self) -> u64 {
519        let chars = u128::from(self.nonwide_chars);
520        let tokens = u128::from(self.nonwide_tokens);
521        let wide = u128::from(self.wide_chars);
522        let (numerator, denominator) = if tokens.saturating_mul(6) < chars {
523            (chars.saturating_add(wide.saturating_mul(3)), 6)
524        } else if tokens.saturating_mul(3) > chars {
525            (chars.saturating_mul(2).saturating_add(wide.saturating_mul(3)), 6)
526        } else {
527            (tokens.saturating_mul(2).saturating_add(wide), 2)
528        };
529        let rounded = numerator.saturating_add(denominator / 2) / denominator;
530        u64::try_from(rounded).unwrap_or(u64::MAX)
531    }
532}
533
534/// Fixed additive metric slots shipped by the first content schema.
535#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
536pub struct MetricValues {
537    /// Logical physical lines across accepted text files.
538    pub physical_lines: u64,
539    /// Whitespace-only lines.
540    pub blank_lines: u64,
541    /// Lines containing at least one non-whitespace character.
542    pub nonblank_lines: u64,
543    /// Whitespace-delimited prose words before markup projection.
544    pub raw_words: u64,
545    /// Code-bearing lines under `code-sloc-v1`.
546    pub code_lines: u64,
547    /// Comment-only lines under `code-sloc-v1`.
548    pub comment_lines: u64,
549    /// Blank lines under `code-sloc-v1`, distinct from whitespace-only source lines.
550    pub code_blank_lines: u64,
551    /// Plain-text paragraph runs.
552    pub paragraphs: u64,
553    /// Reader-visible Markdown words.
554    pub visible_words: u64,
555    /// Additive logical-word sufficient statistics.
556    pub logical_word_stats: LogicalWordStats,
557    /// Reader-visible Markdown logical-word sufficient statistics.
558    pub visible_logical_word_stats: LogicalWordStats,
559}
560
561/// Why a requested file did or did not produce metrics.
562#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
563pub enum CoverageReason {
564    /// Requested analyzers completed.
565    Analyzed,
566    /// Known binary type or a NUL byte made text metrics inapplicable.
567    Binary,
568    /// Input was not valid UTF-8.
569    InvalidUtf8,
570    /// A recognized Unicode byte-order mark names an encoding no analyzer decodes.
571    UnsupportedEncoding,
572    /// No shipped analyzer accepts this type.
573    Unsupported,
574    /// File I/O failed; the human error is retained separately.
575    IoError,
576    /// Metadata changed while the file was being read.
577    ChangedDuringRead,
578    /// The unit produced a value by counting the file as plain text rather than as the
579    /// document it is: a Markdown file over the exact bound under the words unit, whose
580    /// every word is counted visible and whose paragraphs are its blank-line runs.
581    TextOnly,
582}
583
584/// Sparse analysis record for one regular file.
585#[derive(Clone, PartialEq, Eq, Debug)]
586pub struct FileAnalysis {
587    /// Metadata fingerprint this result describes.
588    pub fingerprint: Fingerprint,
589    /// Apparent bytes represented by the record.
590    pub bytes: u64,
591    /// Bounded content evidence, kept separate from name-based grouping.
592    pub detection: ContentDetection,
593    /// Shared physical-line and raw-word outcome.
594    pub lines: AnalyzerOutcome<BasicMetrics>,
595    /// Code outcome when the request included the code unit.
596    pub code: Option<AnalyzerOutcome<CodeMetrics>>,
597    /// Word outcome when the request included the word unit.
598    pub words: Option<AnalyzerOutcome<WordMetrics>>,
599    /// Optional path-specific failure detail.
600    pub error: Option<String>,
601}
602
603impl FileAnalysis {
604    /// Whether optional unit slots exactly match the tier's requested analyzer set.
605    pub const fn matches_profile(&self, profile: AnalysisSet) -> bool {
606        self.code.is_some() == profile.includes_code()
607            && self.words.is_some() == profile.includes_words()
608    }
609
610    /// File-level operational failure, counted once even though it affects every unit.
611    pub const fn operational_failure(&self) -> Option<CoverageReason> {
612        match self.lines.coverage() {
613            CoverageReason::IoError => Some(CoverageReason::IoError),
614            CoverageReason::ChangedDuringRead => Some(CoverageReason::ChangedDuringRead),
615            CoverageReason::Analyzed
616            | CoverageReason::TextOnly
617            | CoverageReason::Binary
618            | CoverageReason::InvalidUtf8
619            | CoverageReason::UnsupportedEncoding
620            | CoverageReason::Unsupported => None,
621        }
622    }
623
624    /// Whether every retained requested unit is safe to reuse.
625    pub const fn is_reusable(&self) -> bool {
626        self.operational_failure().is_none()
627            && self.lines.is_reusable()
628            && match self.code {
629                Some(outcome) => outcome.is_reusable(),
630                None => true,
631            }
632            && match self.words {
633                Some(outcome) => outcome.is_reusable(),
634                None => true,
635            }
636    }
637}
638
639/// Owned immutable candidate captured before worker execution.
640///
641/// Crate-private with [`Index::analysis_candidates`] and [`Index::apply_analysis`] until
642/// the request model (P1.3) decides whether an out-of-crate analyzer is a supported
643/// surface (`fdu-5upj`): the tier must be prepared for a candidate's identity before a
644/// result for it can commit, and preparation is crate-private.
645///
646/// [`Index::analysis_candidates`]: crate::Index::analysis_candidates
647/// [`Index::apply_analysis`]: crate::Index::apply_analysis
648#[derive(Clone, Debug)]
649pub(crate) struct AnalysisCandidate {
650    /// Generation-safe index identity.
651    pub entry_id: EntryId,
652    /// Entry revision at capture time.
653    pub revision: u64,
654    /// Path relative to the index root.
655    pub relative_path: PathBuf,
656    /// Absolute filesystem path.
657    pub absolute_path: PathBuf,
658    /// Last observed attributes.
659    pub attrs: Attrs,
660    /// Metadata-only classification.
661    pub classification: Classification,
662}
663
664/// Live file identity used to match a sidecar record on cache-only restore.
665///
666/// Restore does not classify and does not open the file. The sidecar already stores the
667/// classification that `apply_analysis` would have committed, and the apply-path
668/// classify self-check is a crate-private consistency guard, not part of the answer.
669#[derive(Clone, Debug)]
670pub(crate) struct RestoreCandidate {
671    /// Generation-safe index identity.
672    pub entry_id: EntryId,
673    /// Entry revision at capture time.
674    pub revision: u64,
675    /// Path relative to the index root.
676    pub relative_path: PathBuf,
677    /// Last observed attributes.
678    pub attrs: Attrs,
679}
680
681/// Worker result submitted to the index's derived-data mutation boundary.
682#[derive(Clone, Debug)]
683pub(crate) struct AnalysisObservation {
684    /// Candidate identity and expectation.
685    pub candidate: AnalysisCandidate,
686    /// Analyzer set whose tier may accept the result.
687    pub profile: AnalysisSet,
688    /// Analyzer identity whose tier may accept the result.
689    pub provenance: ContentProvenance,
690    /// Completed or skipped analysis record.
691    pub analysis: FileAnalysis,
692}
693
694/// Result of conditionally committing one worker observation.
695#[derive(Clone, Copy, PartialEq, Eq, Debug)]
696pub(crate) enum AnalysisApplyOutcome {
697    /// The sparse record and ancestor rollups changed.
698    Applied,
699    /// The result was discarded: metadata changed after candidate capture, or the content
700    /// tier holds another identity than the one the result was produced under, which is
701    /// the same answer because both mean the result describes something else.
702    Stale,
703}
704
705#[cfg(test)]
706mod tests {
707    use std::collections::HashSet;
708
709    use super::{
710        AnalysisRequest, AnalysisSet, AnalyzerOutcome, AnalyzerVersion, CODE_SLOC,
711        ContentProvenance, CoverageReason, LogicalWordStats, METRICS,
712    };
713
714    #[test]
715    #[should_panic(expected = "an outcome with a value must carry one")]
716    fn unavailable_outcome_cannot_claim_success() {
717        let _: AnalyzerOutcome<()> = AnalyzerOutcome::unavailable(CoverageReason::Analyzed);
718    }
719
720    #[test]
721    fn metric_registry_has_unique_names_and_one_requestable_owner_each() {
722        let mut names = HashSet::new();
723        for metric in METRICS {
724            assert!(names.insert(metric.name), "duplicate metric name {}", metric.name);
725            assert!(
726                matches!(
727                    metric.owner,
728                    AnalysisSet::LINES_ONLY | AnalysisSet::CODE_ONLY | AnalysisSet::WORDS_ONLY
729                ),
730                "{} has a non-unit owner {:?}",
731                metric.name,
732                metric.owner
733            );
734        }
735        assert_eq!(names.len(), 12);
736    }
737
738    #[test]
739    fn logical_words_derive_only_after_additive_stats_are_combined() {
740        let first = LogicalWordStats { wide_chars: 3, nonwide_tokens: 1, nonwide_chars: 12 };
741        let second = LogicalWordStats { wide_chars: 1, nonwide_tokens: 9, nonwide_chars: 6 };
742        let combined = LogicalWordStats {
743            wide_chars: first.wide_chars + second.wide_chars,
744            nonwide_tokens: first.nonwide_tokens + second.nonwide_tokens,
745            nonwide_chars: first.nonwide_chars + second.nonwide_chars,
746        };
747        assert_eq!(combined.logical_words(), 8);
748    }
749
750    #[test]
751    fn logical_words_match_the_pinned_rational_clamp_and_half_up_rounding() {
752        let logical = |wide_chars, nonwide_tokens, nonwide_chars| {
753            LogicalWordStats { wide_chars, nonwide_tokens, nonwide_chars }.logical_words()
754        };
755        assert_eq!(logical(0, 0, 0), 0);
756        assert_eq!(logical(0, 2, 9), 2, "ordinary prose passes through");
757        assert_eq!(logical(0, 1, 12), 2, "long tokens use the six-character floor");
758        assert_eq!(logical(0, 4, 4), 1, "short tokens use the three-character ceiling");
759        assert_eq!(logical(3, 0, 0), 2, "wide halves round up once");
760        assert_eq!(logical(1, 1, 1), 1, "mixed fractions combine before rounding");
761    }
762
763    #[test]
764    fn the_analyzer_vocabulary_parses_every_accepted_spelling() {
765        let cases = [
766            ("none", AnalysisSet::NONE),
767            ("lines", AnalysisSet::NONE.with_lines()),
768            ("code", AnalysisSet::NONE.with_code()),
769            ("words", AnalysisSet::NONE.with_words()),
770            ("all", AnalysisSet::ALL),
771            // Order is irrelevant: a set has no order, so neither spelling is preferred.
772            ("code,words", AnalysisSet::ALL),
773            ("words,code", AnalysisSet::ALL),
774            // `lines` is already implied by any analyzer, so naming it adds nothing.
775            ("lines,code", AnalysisSet::NONE.with_code()),
776            // Whitespace and case are incidental, as in every other list flag.
777            (" CODE , Words ", AnalysisSet::ALL),
778        ];
779        for (input, expected) in cases {
780            assert_eq!(AnalysisSet::parse(input), Ok(expected), "parsing {input:?}");
781        }
782    }
783
784    #[test]
785    fn the_analyzer_vocabulary_rejects_every_incoherent_request() {
786        // Each rejection names what was wrong; a set flag that silently drops a token is
787        // how a caller ends up believing it measured something it did not.
788        let cases = [
789            ("", "empty entry"),
790            ("code,,words", "empty entry"),
791            ("code,code", "more than once"),
792            ("basic", "expected one of"),
793            ("documents", "expected one of"),
794            ("full", "expected one of"),
795            ("none,code", "cannot be combined"),
796            ("all,code", "cannot be combined"),
797            ("none,all", "cannot be combined"),
798        ];
799        for (input, needle) in cases {
800            let error = AnalysisSet::parse(input).expect_err(&format!("{input:?} must fail"));
801            assert!(error.contains(needle), "parsing {input:?} said {error:?}, wanted {needle:?}");
802        }
803    }
804
805    /// The CLI used to relabel by substring replace, which rewrote the user's own token:
806    /// `--analyze analyzer` reported `invalid --analyze "--analyzer"`, misquoting the very
807    /// value it was rejecting. The label is a parameter now, so the value is untouched.
808    #[test]
809    fn a_label_never_rewrites_the_value_it_is_reporting() {
810        for value in ["analyzer", "reanalyze", "analyze-all"] {
811            let error = AnalysisSet::parse_labeled(value, "--analyze")
812                .expect_err("must reject an unknown analyzer");
813            assert!(
814                error.contains(&format!("{value:?}")),
815                "{error} must quote {value:?} exactly as typed"
816            );
817            assert!(error.starts_with("invalid --analyze "), "{error} must carry the label");
818        }
819    }
820
821    #[test]
822    fn the_on_disk_encoding_round_trips_and_refuses_unknown_analyzers() {
823        for set in [
824            AnalysisSet::NONE,
825            AnalysisSet::NONE.with_lines(),
826            AnalysisSet::NONE.with_code(),
827            AnalysisSet::NONE.with_words(),
828            AnalysisSet::ALL,
829        ] {
830            assert_eq!(AnalysisSet::from_bits(set.bits()), Some(set));
831        }
832        // A record written by a build with an analyzer this one lacks cannot be honored,
833        // so it is refused rather than silently under-reported as absent metrics.
834        assert_eq!(AnalysisSet::from_bits(0b1000_0000), None);
835    }
836
837    #[test]
838    fn code_metrics_use_the_updated_analyzer_version() {
839        let request = AnalysisRequest { profile: AnalysisSet::NONE.with_code(), workers: 1 };
840        let provenance = ContentProvenance::for_request(request, 42);
841        assert!(provenance.analyzers.contains(&(CODE_SLOC, AnalyzerVersion(3))));
842    }
843
844    #[test]
845    fn labels_are_the_vocabulary_parse_accepts() {
846        for set in [
847            AnalysisSet::NONE.with_lines(),
848            AnalysisSet::NONE.with_code(),
849            AnalysisSet::NONE.with_words(),
850            AnalysisSet::ALL,
851        ] {
852            let spelled = set.labels().join(",");
853            assert_eq!(AnalysisSet::parse(&spelled), Ok(set), "round trip through {spelled:?}");
854        }
855        assert!(AnalysisSet::NONE.labels().is_empty());
856    }
857}