Skip to main content

fdu_core/content/
content_model.rs

1//! Versioned content-analysis contracts shared across the engine and report layers.
2
3use std::path::PathBuf;
4
5use crate::classify::{
6    Classification, ClassificationFlags, ContentFamily, DetectionConfidence, DetectionSource,
7    FileTypeId,
8};
9use crate::query::Rejection;
10use crate::{Attrs, EntryId, Fingerprint};
11
12/// Stable analyzer identity.
13#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
14pub struct AnalyzerId(pub &'static str);
15
16/// Version of an analyzer's counting semantics.
17#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
18pub struct AnalyzerVersion(pub u16);
19
20/// Definition of one measured value exposed by content reports.
21#[derive(Clone, Copy, PartialEq, Eq, Debug)]
22pub struct MetricDef {
23    /// Stable report key.
24    pub name: &'static str,
25    /// Requestable unit that owns the value's presence.
26    pub owner: AnalysisSet,
27    /// Analyzer dialect that defines the value.
28    pub analyzer: AnalyzerId,
29    /// Short semantic definition.
30    pub doc: &'static str,
31}
32
33/// Fingerprint of semantic analyzer options; operational worker count is excluded.
34#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
35pub struct OptionsFingerprint(pub u64);
36
37/// Fused physical-line and raw-word analyzer.
38pub const CONTENT_BASIC: AnalyzerId = AnalyzerId("content-basic-v1");
39/// Common-language code/comment/blank analyzer.
40pub const CODE_SLOC: AnalyzerId = AnalyzerId("code-sloc-v1");
41/// Plain-text logical word and paragraph analyzer.
42pub const TEXT_LOGICAL: AnalyzerId = AnalyzerId("text-logical-v1");
43/// Reader-visible Markdown prose analyzer.
44pub const MARKDOWN_PROSE: AnalyzerId = AnalyzerId("markdown-prose-v1");
45
46/// The single registry of content metric names, owners, and definitions.
47pub const METRICS: &[MetricDef] = &[
48    MetricDef {
49        name: "physical_lines",
50        owner: AnalysisSet::LINES_ONLY,
51        analyzer: CONTENT_BASIC,
52        doc: "Logical physical lines across admitted text files.",
53    },
54    MetricDef {
55        name: "blank_lines",
56        owner: AnalysisSet::LINES_ONLY,
57        analyzer: CONTENT_BASIC,
58        doc: "Whitespace-only physical lines.",
59    },
60    MetricDef {
61        name: "nonblank_lines",
62        owner: AnalysisSet::LINES_ONLY,
63        analyzer: CONTENT_BASIC,
64        doc: "Physical lines containing non-whitespace text.",
65    },
66    MetricDef {
67        name: "raw_words",
68        owner: AnalysisSet::LINES_ONLY,
69        analyzer: CONTENT_BASIC,
70        doc: "Whitespace-delimited words before document projection.",
71    },
72    MetricDef {
73        name: "code_lines",
74        owner: AnalysisSet::CODE_ONLY,
75        analyzer: CODE_SLOC,
76        doc: "Code-bearing lines in supported source languages.",
77    },
78    MetricDef {
79        name: "comment_lines",
80        owner: AnalysisSet::CODE_ONLY,
81        analyzer: CODE_SLOC,
82        doc: "Comment-only lines in supported source languages.",
83    },
84    MetricDef {
85        name: "code_blank_lines",
86        owner: AnalysisSet::CODE_ONLY,
87        analyzer: CODE_SLOC,
88        doc: "Blank lines under the code analyzer's syntax.",
89    },
90    MetricDef {
91        name: "logical_words",
92        owner: AnalysisSet::WORDS_ONLY,
93        analyzer: TEXT_LOGICAL,
94        doc: "Normalized logical word volume.",
95    },
96    MetricDef {
97        name: "paragraphs",
98        owner: AnalysisSet::WORDS_ONLY,
99        analyzer: TEXT_LOGICAL,
100        doc: "Plain-text runs or reader-visible Markdown paragraphs.",
101    },
102    MetricDef {
103        name: "visible_words",
104        owner: AnalysisSet::WORDS_ONLY,
105        analyzer: MARKDOWN_PROSE,
106        doc: "Reader-visible Markdown words.",
107    },
108    MetricDef {
109        name: "visible_logical_words",
110        owner: AnalysisSet::WORDS_ONLY,
111        analyzer: MARKDOWN_PROSE,
112        doc: "Normalized reader-visible Markdown words.",
113    },
114    MetricDef {
115        name: "document_words",
116        owner: AnalysisSet::WORDS_ONLY,
117        analyzer: TEXT_LOGICAL,
118        doc: "Logical words after the document-type projection.",
119    },
120];
121
122/// The set of content analyzers a request enables.
123///
124/// A set rather than a ladder, because the analyzers are independent: `code` and `words`
125/// measure different things over different families and either is useful without the
126/// other.  An ordered enum could name only the combinations somebody thought to
127/// enumerate — four of the eight this registry already permits — and it made
128/// `text-logical-v1` without `markdown-prose-v1` unreachable.
129///
130/// `lines` is the base every analyzer shares: any analyzer that runs has already
131/// streamed the file, so line counts cost nothing extra.  It is therefore implicit in
132/// every non-empty set rather than something a caller must remember to request, which is
133/// why each `with_*` constructor sets it.
134#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug, Default)]
135pub struct AnalysisSet(u8);
136
137impl AnalysisSet {
138    const LINES: u8 = 1 << 0;
139    const CODE: u8 = 1 << 1;
140    const WORDS: u8 = 1 << 2;
141    const KNOWN: u8 = Self::LINES | Self::CODE | Self::WORDS;
142
143    /// Open no file; preserve the metadata-only behavior.
144    pub const NONE: Self = Self(0);
145    /// Physical-line and raw-word unit.
146    pub const LINES_ONLY: Self = Self(Self::LINES);
147    /// Code unit, including its shared line pass.
148    pub const CODE_ONLY: Self = Self(Self::LINES | Self::CODE);
149    /// Word unit, including its shared line pass.
150    pub const WORDS_ONLY: Self = Self(Self::LINES | Self::WORDS);
151    /// Every registered analyzer.
152    pub const ALL: Self = Self(Self::KNOWN);
153
154    /// Add physical, blank, and nonblank line counts.
155    #[must_use]
156    pub const fn with_lines(self) -> Self {
157        Self(self.0 | Self::LINES)
158    }
159
160    /// Add common-language standard SLOC over the `code` family.
161    #[must_use]
162    pub const fn with_code(self) -> Self {
163        Self(self.0 | Self::LINES | Self::CODE)
164    }
165
166    /// Add raw, normalized, and reader-visible word volume.
167    #[must_use]
168    pub const fn with_words(self) -> Self {
169        Self(self.0 | Self::LINES | Self::WORDS)
170    }
171
172    /// Whether any source file may be opened.
173    pub const fn is_enabled(self) -> bool {
174        self.0 != 0
175    }
176
177    /// Whether standard SLOC is requested.
178    pub const fn includes_code(self) -> bool {
179        self.0 & Self::CODE != 0
180    }
181
182    /// Whether logical and visible word metrics are requested.
183    pub const fn includes_words(self) -> bool {
184        self.0 & Self::WORDS != 0
185    }
186
187    /// Whether this request includes every unit in `other`.
188    pub const fn contains(self, other: Self) -> bool {
189        self.0 & other.0 == other.0
190    }
191
192    /// Stable on-disk and fingerprint encoding.
193    pub const fn bits(self) -> u8 {
194        self.0
195    }
196
197    /// Decode [`Self::bits`], rejecting any analyzer this build does not know.
198    ///
199    /// How the grammar spells the empty set, and the one spelling of an analyzer set a
200    /// `const` can state: every other set is a list [`Self::labels`] builds.
201    ///
202    /// Named because a surface whose help text states the default analyzer set must read
203    /// that spelling rather than write the word again.
204    pub const NONE_LABEL: &'static str = "none";
205
206    /// Unknown bits mean a record written by a newer build whose extra analyzers cannot
207    /// be honored, so it is refused rather than silently under-reported.
208    pub const fn from_bits(bits: u8) -> Option<Self> {
209        if bits & !Self::KNOWN == 0 { Some(Self(bits)) } else { None }
210    }
211
212    /// Parse the comma-delimited vocabulary both front ends accept, naming the axis as
213    /// the library and the Python API spell it.
214    ///
215    /// Lives here rather than in either front end because it is the axis's grammar, not
216    /// one surface's flag parsing: the CLI and the Python binding must accept exactly the
217    /// same words or the two surfaces disagree about what a request means.
218    ///
219    /// `none` and `all` are totals and cannot be combined with anything, including each
220    /// other — `none,code` has no coherent reading, and silently letting one win is how a
221    /// caller ends up with analysis they did not ask for or did not get.
222    pub fn parse(value: &str) -> Result<Self, String> {
223        Self::parse_labeled(value, "analyze")
224    }
225
226    /// `label` is how the calling surface names this axis in its diagnostics: `--analyze`
227    /// for the CLI, `analyze` for the Python API. Passed in rather than rewritten
228    /// afterwards, because the CLI used to relabel by substring replace and that hit the
229    /// user's own token: `--analyze analyzer` reported `invalid --analyze "--analyzer"`,
230    /// misquoting the very value it was rejecting (fdu-7j6z).
231    pub fn parse_labeled(value: &str, label: &str) -> Result<Self, String> {
232        Self::parse_rejecting(value).map_err(|rejection| rejection.labeled(label))
233    }
234
235    /// [`Self::parse_labeled`], refusing with the value and expectation rather than a
236    /// sentence, so the request model can name the axis in a typed refusal.
237    pub(crate) fn parse_rejecting(value: &str) -> Result<Self, Rejection> {
238        let mut set = Self::NONE;
239        let mut seen: Vec<String> = Vec::new();
240        let mut total: Option<&'static str> = None;
241        for raw in value.split(',') {
242            let token = raw.trim().to_ascii_lowercase();
243            if token.is_empty() {
244                return Err(Rejection::new(value, "empty entry in the list"));
245            }
246            if seen.contains(&token) {
247                return Err(Rejection::new(value, format!("{token:?} appears more than once")));
248            }
249            seen.push(token.clone());
250            match token.as_str() {
251                "none" => total = Some(Self::NONE_LABEL),
252                "all" => {
253                    total = Some("all");
254                    set = Self::ALL;
255                }
256                "lines" => set = set.with_lines(),
257                "code" => set = set.with_code(),
258                "words" => set = set.with_words(),
259                other => {
260                    return Err(Rejection::new(
261                        other,
262                        "expected one of none, lines, code, words, all",
263                    ));
264                }
265            }
266        }
267        if let Some(total) = total {
268            if seen.len() > 1 {
269                return Err(Rejection::new(
270                    value,
271                    format!("{total:?} names the whole axis and cannot be combined"),
272                ));
273            }
274            if total == Self::NONE_LABEL {
275                return Ok(Self::NONE);
276            }
277        }
278        Ok(set)
279    }
280
281    /// Requested analyzers in canonical order, as the CLI and reports spell them.
282    pub fn labels(self) -> Vec<&'static str> {
283        let mut labels = Vec::new();
284        if self.0 & Self::LINES != 0 {
285            labels.push("lines");
286        }
287        if self.includes_code() {
288            labels.push("code");
289        }
290        if self.includes_words() {
291            labels.push("words");
292        }
293        labels
294    }
295}
296
297/// Content-derived classification evidence retained separately from name grouping.
298#[derive(Clone, PartialEq, Eq, Debug)]
299pub struct ContentDetection {
300    /// Type suggested by the bounded content probe.
301    pub file_type: FileTypeId,
302    /// Broad family suggested by the bounded content probe.
303    pub family: ContentFamily,
304    /// Evidence source.
305    pub source: DetectionSource,
306    /// Strength of the evidence.
307    pub confidence: DetectionConfidence,
308    /// Orthogonal generated, vendored, and documentation markers.
309    pub flags: ClassificationFlags,
310}
311
312impl From<Classification> for ContentDetection {
313    fn from(value: Classification) -> Self {
314        Self {
315            file_type: value.file_type,
316            family: value.family,
317            source: value.source,
318            confidence: value.confidence,
319            flags: value.flags,
320        }
321    }
322}
323
324impl From<ContentDetection> for Classification {
325    fn from(value: ContentDetection) -> Self {
326        Self {
327            file_type: value.file_type,
328            family: value.family,
329            source: value.source,
330            confidence: value.confidence,
331            flags: value.flags,
332        }
333    }
334}
335
336/// Settings for one analysis pass.
337#[derive(Clone, Copy, PartialEq, Eq, Debug)]
338pub struct AnalysisRequest {
339    /// Analyzer bundle to run.
340    pub profile: AnalysisSet,
341    /// Maximum worker count; zero selects the available parallelism.
342    pub workers: usize,
343}
344
345impl Default for AnalysisRequest {
346    fn default() -> Self {
347        Self { profile: AnalysisSet::NONE, workers: 0 }
348    }
349}
350
351impl AnalysisRequest {
352    /// Fingerprint only settings that can change a stored answer.
353    pub fn options_fingerprint(self) -> OptionsFingerprint {
354        const OFFSET: u64 = 0xcbf2_9ce4_8422_2325;
355        const PRIME: u64 = 0x0000_0100_0000_01b3;
356        let hash = [self.profile.bits()]
357            .into_iter()
358            .fold(OFFSET, |hash, byte| (hash ^ u64::from(byte)).wrapping_mul(PRIME));
359        OptionsFingerprint(hash)
360    }
361}
362
363/// Analyzer/rule/options identity attached to cached and reported content.
364#[derive(Clone, PartialEq, Eq, Debug)]
365pub struct ContentProvenance {
366    /// Compiled file-type rule identity.
367    pub type_rules_fingerprint: u64,
368    /// Semantic option identity.
369    pub options_fingerprint: OptionsFingerprint,
370    /// Analyzer dialects enabled by the profile.
371    pub analyzers: Vec<(AnalyzerId, AnalyzerVersion)>,
372}
373
374impl ContentProvenance {
375    /// Resolve analyzer dialects implied by a request, under a given set of type rules.
376    ///
377    /// The fingerprint is passed rather than read from a global: a caller may run two
378    /// indexes under different taxonomies in one process, and a record must record the
379    /// rules that actually produced it.
380    pub fn for_request(request: AnalysisRequest, type_rules_fingerprint: u64) -> Self {
381        const VERSION_ONE: AnalyzerVersion = AnalyzerVersion(1);
382        let mut analyzers = Vec::new();
383        if request.profile.is_enabled() {
384            analyzers.push((CONTENT_BASIC, VERSION_ONE));
385        }
386        if request.profile.includes_code() {
387            analyzers.push((CODE_SLOC, AnalyzerVersion(3)));
388        }
389        if request.profile.includes_words() {
390            analyzers.push((TEXT_LOGICAL, VERSION_ONE));
391            analyzers.push((MARKDOWN_PROSE, VERSION_ONE));
392        }
393        Self {
394            type_rules_fingerprint,
395            options_fingerprint: request.options_fingerprint(),
396            analyzers,
397        }
398    }
399}
400
401/// Additive sufficient statistics for FlexDoc-style logical word volume.
402#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
403pub struct LogicalWordStats {
404    /// Non-whitespace wide/fullwidth characters, each worth half a logical word.
405    pub wide_chars: u64,
406    /// Whitespace-delimited non-wide tokens.
407    pub nonwide_tokens: u64,
408    /// Non-whitespace non-wide characters used by the 3..6 clamp.
409    pub nonwide_chars: u64,
410}
411
412/// Metrics owned by the always-present line analyzer unit.
413#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
414pub struct BasicMetrics {
415    /// Logical physical lines across admitted text files.
416    pub physical_lines: u64,
417    /// Whitespace-only lines.
418    pub blank_lines: u64,
419    /// Lines containing at least one non-whitespace character.
420    pub nonblank_lines: u64,
421    /// Whitespace-delimited words before document projection.
422    pub raw_words: u64,
423}
424
425/// Metrics owned by the code analyzer unit.
426#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
427pub struct CodeMetrics {
428    /// Code-bearing lines.
429    pub code_lines: u64,
430    /// Comment-only lines.
431    pub comment_lines: u64,
432    /// Blank lines under the code analyzer's syntax.
433    pub code_blank_lines: u64,
434}
435
436/// Metrics owned by the word analyzer unit.
437#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
438pub struct WordMetrics {
439    /// Plain-text paragraph runs or visible Markdown paragraphs.
440    pub paragraphs: u64,
441    /// Reader-visible Markdown words.
442    pub visible_words: u64,
443    /// Additive logical-word sufficient statistics.
444    pub logical_word_stats: LogicalWordStats,
445    /// Reader-visible Markdown logical-word sufficient statistics.
446    pub visible_logical_word_stats: LogicalWordStats,
447}
448
449/// One analyzer unit's explicit coverage and optional successful value.
450#[derive(Clone, Copy, PartialEq, Eq, Debug)]
451pub struct AnalyzerOutcome<T> {
452    /// Why the unit did or did not produce a value.
453    coverage: CoverageReason,
454    /// Successful measured value; absent for every non-analyzed outcome.
455    value: Option<T>,
456}
457
458impl<T> AnalyzerOutcome<T> {
459    /// A successful analyzer result.
460    pub(crate) const fn analyzed(value: T) -> Self {
461        Self { coverage: CoverageReason::Analyzed, value: Some(value) }
462    }
463
464    /// A unit that could not produce a value for the named reason.
465    pub(crate) fn unavailable(coverage: CoverageReason) -> Self {
466        assert!(
467            !matches!(coverage, CoverageReason::Analyzed),
468            "an analyzed outcome must carry a value"
469        );
470        Self { coverage, value: None }
471    }
472
473    /// Coverage outcome for this unit.
474    pub const fn coverage(&self) -> CoverageReason {
475        self.coverage
476    }
477
478    /// Successful measured value, absent for every unavailable outcome.
479    pub const fn value(self) -> Option<T>
480    where
481        T: Copy,
482    {
483        self.value
484    }
485
486    pub(crate) fn from_parts(coverage: CoverageReason, value: Option<T>) -> Option<Self> {
487        if matches!(coverage, CoverageReason::Analyzed) == value.is_some() {
488            Some(Self { coverage, value })
489        } else {
490            None
491        }
492    }
493
494    /// Operational failures must be retried rather than treated as cache hits.
495    pub const fn is_reusable(&self) -> bool {
496        !matches!(self.coverage, CoverageReason::IoError | CoverageReason::ChangedDuringRead)
497    }
498}
499
500impl LogicalWordStats {
501    pub(crate) fn add_assign(&mut self, other: Self) {
502        self.wide_chars = self.wide_chars.saturating_add(other.wide_chars);
503        self.nonwide_tokens = self.nonwide_tokens.saturating_add(other.nonwide_tokens);
504        self.nonwide_chars = self.nonwide_chars.saturating_add(other.nonwide_chars);
505    }
506
507    /// Derive rounded logical words after aggregation.
508    pub fn logical_words(self) -> u64 {
509        let chars = u128::from(self.nonwide_chars);
510        let tokens = u128::from(self.nonwide_tokens);
511        let wide = u128::from(self.wide_chars);
512        let (numerator, denominator) = if tokens.saturating_mul(6) < chars {
513            (chars.saturating_add(wide.saturating_mul(3)), 6)
514        } else if tokens.saturating_mul(3) > chars {
515            (chars.saturating_mul(2).saturating_add(wide.saturating_mul(3)), 6)
516        } else {
517            (tokens.saturating_mul(2).saturating_add(wide), 2)
518        };
519        let rounded = numerator.saturating_add(denominator / 2) / denominator;
520        u64::try_from(rounded).unwrap_or(u64::MAX)
521    }
522}
523
524/// Fixed additive metric slots shipped by the first content schema.
525#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
526pub struct MetricValues {
527    /// Logical physical lines across accepted text files.
528    pub physical_lines: u64,
529    /// Whitespace-only lines.
530    pub blank_lines: u64,
531    /// Lines containing at least one non-whitespace character.
532    pub nonblank_lines: u64,
533    /// Whitespace-delimited prose words before markup projection.
534    pub raw_words: u64,
535    /// Code-bearing lines under `code-sloc-v1`.
536    pub code_lines: u64,
537    /// Comment-only lines under `code-sloc-v1`.
538    pub comment_lines: u64,
539    /// Blank lines under `code-sloc-v1`, distinct from whitespace-only source lines.
540    pub code_blank_lines: u64,
541    /// Plain-text paragraph runs.
542    pub paragraphs: u64,
543    /// Reader-visible Markdown words.
544    pub visible_words: u64,
545    /// Additive logical-word sufficient statistics.
546    pub logical_word_stats: LogicalWordStats,
547    /// Reader-visible Markdown logical-word sufficient statistics.
548    pub visible_logical_word_stats: LogicalWordStats,
549}
550
551/// Why a requested file did or did not produce metrics.
552#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
553pub enum CoverageReason {
554    /// Requested analyzers completed.
555    Analyzed,
556    /// Known binary type or a NUL byte made text metrics inapplicable.
557    Binary,
558    /// Input was not valid UTF-8.
559    InvalidUtf8,
560    /// A recognized Unicode byte-order mark names an encoding no analyzer decodes.
561    UnsupportedEncoding,
562    /// No shipped analyzer accepts this type.
563    Unsupported,
564    /// File I/O failed; the human error is retained separately.
565    IoError,
566    /// Metadata changed while the file was being read.
567    ChangedDuringRead,
568}
569
570/// Sparse analysis record for one regular file.
571#[derive(Clone, PartialEq, Eq, Debug)]
572pub struct FileAnalysis {
573    /// Metadata fingerprint this result describes.
574    pub fingerprint: Fingerprint,
575    /// Apparent bytes represented by the record.
576    pub bytes: u64,
577    /// Bounded content evidence, kept separate from name-based grouping.
578    pub detection: ContentDetection,
579    /// Shared physical-line and raw-word outcome.
580    pub lines: AnalyzerOutcome<BasicMetrics>,
581    /// Code outcome when the request included the code unit.
582    pub code: Option<AnalyzerOutcome<CodeMetrics>>,
583    /// Word outcome when the request included the word unit.
584    pub words: Option<AnalyzerOutcome<WordMetrics>>,
585    /// Optional path-specific failure detail.
586    pub error: Option<String>,
587}
588
589impl FileAnalysis {
590    /// Whether optional unit slots exactly match the tier's requested analyzer set.
591    pub const fn matches_profile(&self, profile: AnalysisSet) -> bool {
592        self.code.is_some() == profile.includes_code()
593            && self.words.is_some() == profile.includes_words()
594    }
595
596    /// File-level operational failure, counted once even though it affects every unit.
597    pub const fn operational_failure(&self) -> Option<CoverageReason> {
598        match self.lines.coverage() {
599            CoverageReason::IoError => Some(CoverageReason::IoError),
600            CoverageReason::ChangedDuringRead => Some(CoverageReason::ChangedDuringRead),
601            CoverageReason::Analyzed
602            | CoverageReason::Binary
603            | CoverageReason::InvalidUtf8
604            | CoverageReason::UnsupportedEncoding
605            | CoverageReason::Unsupported => None,
606        }
607    }
608
609    /// Whether every retained requested unit is safe to reuse.
610    pub const fn is_reusable(&self) -> bool {
611        self.operational_failure().is_none()
612            && self.lines.is_reusable()
613            && match self.code {
614                Some(outcome) => outcome.is_reusable(),
615                None => true,
616            }
617            && match self.words {
618                Some(outcome) => outcome.is_reusable(),
619                None => true,
620            }
621    }
622}
623
624/// Owned immutable candidate captured before worker execution.
625///
626/// Crate-private with [`Index::analysis_candidates`] and [`Index::apply_analysis`] until
627/// the request model (P1.3) decides whether an out-of-crate analyzer is a supported
628/// surface (`fdu-5upj`): the tier must be prepared for a candidate's identity before a
629/// result for it can commit, and preparation is crate-private.
630///
631/// [`Index::analysis_candidates`]: crate::Index::analysis_candidates
632/// [`Index::apply_analysis`]: crate::Index::apply_analysis
633#[derive(Clone, Debug)]
634pub(crate) struct AnalysisCandidate {
635    /// Generation-safe index identity.
636    pub entry_id: EntryId,
637    /// Entry revision at capture time.
638    pub revision: u64,
639    /// Path relative to the index root.
640    pub relative_path: PathBuf,
641    /// Absolute filesystem path.
642    pub absolute_path: PathBuf,
643    /// Last observed attributes.
644    pub attrs: Attrs,
645    /// Metadata-only classification.
646    pub classification: Classification,
647}
648
649/// Live file identity used to match a sidecar record on cache-only restore.
650///
651/// Restore does not classify and does not open the file. The sidecar already stores the
652/// classification that `apply_analysis` would have committed, and the apply-path
653/// classify self-check is a crate-private consistency guard, not part of the answer.
654#[derive(Clone, Debug)]
655pub(crate) struct RestoreCandidate {
656    /// Generation-safe index identity.
657    pub entry_id: EntryId,
658    /// Entry revision at capture time.
659    pub revision: u64,
660    /// Path relative to the index root.
661    pub relative_path: PathBuf,
662    /// Last observed attributes.
663    pub attrs: Attrs,
664}
665
666/// Worker result submitted to the index's derived-data mutation boundary.
667#[derive(Clone, Debug)]
668pub(crate) struct AnalysisObservation {
669    /// Candidate identity and expectation.
670    pub candidate: AnalysisCandidate,
671    /// Analyzer set whose tier may accept the result.
672    pub profile: AnalysisSet,
673    /// Analyzer identity whose tier may accept the result.
674    pub provenance: ContentProvenance,
675    /// Completed or skipped analysis record.
676    pub analysis: FileAnalysis,
677}
678
679/// Result of conditionally committing one worker observation.
680#[derive(Clone, Copy, PartialEq, Eq, Debug)]
681pub(crate) enum AnalysisApplyOutcome {
682    /// The sparse record and ancestor rollups changed.
683    Applied,
684    /// The result was discarded: metadata changed after candidate capture, or the content
685    /// tier holds another identity than the one the result was produced under, which is
686    /// the same answer because both mean the result describes something else.
687    Stale,
688}
689
690#[cfg(test)]
691mod tests {
692    use std::collections::HashSet;
693
694    use super::{
695        AnalysisRequest, AnalysisSet, AnalyzerOutcome, AnalyzerVersion, CODE_SLOC,
696        ContentProvenance, CoverageReason, LogicalWordStats, METRICS,
697    };
698
699    #[test]
700    #[should_panic(expected = "an analyzed outcome must carry a value")]
701    fn unavailable_outcome_cannot_claim_success() {
702        let _: AnalyzerOutcome<()> = AnalyzerOutcome::unavailable(CoverageReason::Analyzed);
703    }
704
705    #[test]
706    fn metric_registry_has_unique_names_and_one_requestable_owner_each() {
707        let mut names = HashSet::new();
708        for metric in METRICS {
709            assert!(names.insert(metric.name), "duplicate metric name {}", metric.name);
710            assert!(
711                matches!(
712                    metric.owner,
713                    AnalysisSet::LINES_ONLY | AnalysisSet::CODE_ONLY | AnalysisSet::WORDS_ONLY
714                ),
715                "{} has a non-unit owner {:?}",
716                metric.name,
717                metric.owner
718            );
719        }
720        assert_eq!(names.len(), 12);
721    }
722
723    #[test]
724    fn logical_words_derive_only_after_additive_stats_are_combined() {
725        let first = LogicalWordStats { wide_chars: 3, nonwide_tokens: 1, nonwide_chars: 12 };
726        let second = LogicalWordStats { wide_chars: 1, nonwide_tokens: 9, nonwide_chars: 6 };
727        let combined = LogicalWordStats {
728            wide_chars: first.wide_chars + second.wide_chars,
729            nonwide_tokens: first.nonwide_tokens + second.nonwide_tokens,
730            nonwide_chars: first.nonwide_chars + second.nonwide_chars,
731        };
732        assert_eq!(combined.logical_words(), 8);
733    }
734
735    #[test]
736    fn logical_words_match_the_pinned_rational_clamp_and_half_up_rounding() {
737        let logical = |wide_chars, nonwide_tokens, nonwide_chars| {
738            LogicalWordStats { wide_chars, nonwide_tokens, nonwide_chars }.logical_words()
739        };
740        assert_eq!(logical(0, 0, 0), 0);
741        assert_eq!(logical(0, 2, 9), 2, "ordinary prose passes through");
742        assert_eq!(logical(0, 1, 12), 2, "long tokens use the six-character floor");
743        assert_eq!(logical(0, 4, 4), 1, "short tokens use the three-character ceiling");
744        assert_eq!(logical(3, 0, 0), 2, "wide halves round up once");
745        assert_eq!(logical(1, 1, 1), 1, "mixed fractions combine before rounding");
746    }
747
748    #[test]
749    fn the_analyzer_vocabulary_parses_every_accepted_spelling() {
750        let cases = [
751            ("none", AnalysisSet::NONE),
752            ("lines", AnalysisSet::NONE.with_lines()),
753            ("code", AnalysisSet::NONE.with_code()),
754            ("words", AnalysisSet::NONE.with_words()),
755            ("all", AnalysisSet::ALL),
756            // Order is irrelevant: a set has no order, so neither spelling is preferred.
757            ("code,words", AnalysisSet::ALL),
758            ("words,code", AnalysisSet::ALL),
759            // `lines` is already implied by any analyzer, so naming it adds nothing.
760            ("lines,code", AnalysisSet::NONE.with_code()),
761            // Whitespace and case are incidental, as in every other list flag.
762            (" CODE , Words ", AnalysisSet::ALL),
763        ];
764        for (input, expected) in cases {
765            assert_eq!(AnalysisSet::parse(input), Ok(expected), "parsing {input:?}");
766        }
767    }
768
769    #[test]
770    fn the_analyzer_vocabulary_rejects_every_incoherent_request() {
771        // Each rejection names what was wrong; a set flag that silently drops a token is
772        // how a caller ends up believing it measured something it did not.
773        let cases = [
774            ("", "empty entry"),
775            ("code,,words", "empty entry"),
776            ("code,code", "more than once"),
777            ("basic", "expected one of"),
778            ("documents", "expected one of"),
779            ("full", "expected one of"),
780            ("none,code", "cannot be combined"),
781            ("all,code", "cannot be combined"),
782            ("none,all", "cannot be combined"),
783        ];
784        for (input, needle) in cases {
785            let error = AnalysisSet::parse(input).expect_err(&format!("{input:?} must fail"));
786            assert!(error.contains(needle), "parsing {input:?} said {error:?}, wanted {needle:?}");
787        }
788    }
789
790    /// The CLI used to relabel by substring replace, which rewrote the user's own token:
791    /// `--analyze analyzer` reported `invalid --analyze "--analyzer"`, misquoting the very
792    /// value it was rejecting. The label is a parameter now, so the value is untouched.
793    #[test]
794    fn a_label_never_rewrites_the_value_it_is_reporting() {
795        for value in ["analyzer", "reanalyze", "analyze-all"] {
796            let error = AnalysisSet::parse_labeled(value, "--analyze")
797                .expect_err("must reject an unknown analyzer");
798            assert!(
799                error.contains(&format!("{value:?}")),
800                "{error} must quote {value:?} exactly as typed"
801            );
802            assert!(error.starts_with("invalid --analyze "), "{error} must carry the label");
803        }
804    }
805
806    #[test]
807    fn the_on_disk_encoding_round_trips_and_refuses_unknown_analyzers() {
808        for set in [
809            AnalysisSet::NONE,
810            AnalysisSet::NONE.with_lines(),
811            AnalysisSet::NONE.with_code(),
812            AnalysisSet::NONE.with_words(),
813            AnalysisSet::ALL,
814        ] {
815            assert_eq!(AnalysisSet::from_bits(set.bits()), Some(set));
816        }
817        // A record written by a build with an analyzer this one lacks cannot be honored,
818        // so it is refused rather than silently under-reported as absent metrics.
819        assert_eq!(AnalysisSet::from_bits(0b1000_0000), None);
820    }
821
822    #[test]
823    fn code_metrics_use_the_updated_analyzer_version() {
824        let request = AnalysisRequest { profile: AnalysisSet::NONE.with_code(), workers: 1 };
825        let provenance = ContentProvenance::for_request(request, 42);
826        assert!(provenance.analyzers.contains(&(CODE_SLOC, AnalyzerVersion(3))));
827    }
828
829    #[test]
830    fn labels_are_the_vocabulary_parse_accepts() {
831        for set in [
832            AnalysisSet::NONE.with_lines(),
833            AnalysisSet::NONE.with_code(),
834            AnalysisSet::NONE.with_words(),
835            AnalysisSet::ALL,
836        ] {
837            let spelled = set.labels().join(",");
838            assert_eq!(AnalysisSet::parse(&spelled), Ok(set), "round trip through {spelled:?}");
839        }
840        assert!(AnalysisSet::NONE.labels().is_empty());
841    }
842}