use std::path::PathBuf;
use crate::classify::{
Classification, ClassificationFlags, ContentFamily, DetectionConfidence, DetectionSource,
FileTypeId,
};
use crate::query::Rejection;
use crate::{Attrs, EntryId, Fingerprint};
#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
pub struct AnalyzerId(pub &'static str);
#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
pub struct AnalyzerVersion(pub u16);
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub struct MetricDef {
pub name: &'static str,
pub owner: AnalysisSet,
pub analyzer: AnalyzerId,
pub doc: &'static str,
}
#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
pub struct OptionsFingerprint(pub u64);
pub const CONTENT_BASIC: AnalyzerId = AnalyzerId("content-basic-v1");
pub const CODE_SLOC: AnalyzerId = AnalyzerId("code-sloc-v1");
pub const TEXT_LOGICAL: AnalyzerId = AnalyzerId("text-logical-v1");
pub const MARKDOWN_PROSE: AnalyzerId = AnalyzerId("markdown-prose-v1");
pub const METRICS: &[MetricDef] = &[
MetricDef {
name: "physical_lines",
owner: AnalysisSet::LINES_ONLY,
analyzer: CONTENT_BASIC,
doc: "Logical physical lines across admitted text files.",
},
MetricDef {
name: "blank_lines",
owner: AnalysisSet::LINES_ONLY,
analyzer: CONTENT_BASIC,
doc: "Whitespace-only physical lines.",
},
MetricDef {
name: "nonblank_lines",
owner: AnalysisSet::LINES_ONLY,
analyzer: CONTENT_BASIC,
doc: "Physical lines containing non-whitespace text.",
},
MetricDef {
name: "raw_words",
owner: AnalysisSet::LINES_ONLY,
analyzer: CONTENT_BASIC,
doc: "Whitespace-delimited words before document projection.",
},
MetricDef {
name: "code_lines",
owner: AnalysisSet::CODE_ONLY,
analyzer: CODE_SLOC,
doc: "Code-bearing lines in supported source languages.",
},
MetricDef {
name: "comment_lines",
owner: AnalysisSet::CODE_ONLY,
analyzer: CODE_SLOC,
doc: "Comment-only lines in supported source languages.",
},
MetricDef {
name: "code_blank_lines",
owner: AnalysisSet::CODE_ONLY,
analyzer: CODE_SLOC,
doc: "Blank lines under the code analyzer's syntax.",
},
MetricDef {
name: "logical_words",
owner: AnalysisSet::WORDS_ONLY,
analyzer: TEXT_LOGICAL,
doc: "Normalized logical word volume.",
},
MetricDef {
name: "paragraphs",
owner: AnalysisSet::WORDS_ONLY,
analyzer: TEXT_LOGICAL,
doc: "Plain-text runs or reader-visible Markdown paragraphs.",
},
MetricDef {
name: "visible_words",
owner: AnalysisSet::WORDS_ONLY,
analyzer: MARKDOWN_PROSE,
doc: "Reader-visible Markdown words.",
},
MetricDef {
name: "visible_logical_words",
owner: AnalysisSet::WORDS_ONLY,
analyzer: MARKDOWN_PROSE,
doc: "Normalized reader-visible Markdown words.",
},
MetricDef {
name: "document_words",
owner: AnalysisSet::WORDS_ONLY,
analyzer: TEXT_LOGICAL,
doc: "Logical words after the document-type projection.",
},
];
#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug, Default)]
pub struct AnalysisSet(u8);
impl AnalysisSet {
const LINES: u8 = 1 << 0;
const CODE: u8 = 1 << 1;
const WORDS: u8 = 1 << 2;
const KNOWN: u8 = Self::LINES | Self::CODE | Self::WORDS;
pub const NONE: Self = Self(0);
pub const LINES_ONLY: Self = Self(Self::LINES);
pub const CODE_ONLY: Self = Self(Self::LINES | Self::CODE);
pub const WORDS_ONLY: Self = Self(Self::LINES | Self::WORDS);
pub const ALL: Self = Self(Self::KNOWN);
#[must_use]
pub const fn with_lines(self) -> Self {
Self(self.0 | Self::LINES)
}
#[must_use]
pub const fn with_code(self) -> Self {
Self(self.0 | Self::LINES | Self::CODE)
}
#[must_use]
pub const fn with_words(self) -> Self {
Self(self.0 | Self::LINES | Self::WORDS)
}
pub const fn is_enabled(self) -> bool {
self.0 != 0
}
pub const fn includes_code(self) -> bool {
self.0 & Self::CODE != 0
}
pub const fn includes_words(self) -> bool {
self.0 & Self::WORDS != 0
}
pub const fn contains(self, other: Self) -> bool {
self.0 & other.0 == other.0
}
pub const fn bits(self) -> u8 {
self.0
}
pub const NONE_LABEL: &'static str = "none";
pub const fn from_bits(bits: u8) -> Option<Self> {
if bits & !Self::KNOWN == 0 { Some(Self(bits)) } else { None }
}
pub fn parse(value: &str) -> Result<Self, String> {
Self::parse_labeled(value, "analyze")
}
pub fn parse_labeled(value: &str, label: &str) -> Result<Self, String> {
Self::parse_rejecting(value).map_err(|rejection| rejection.labeled(label))
}
pub(crate) fn parse_rejecting(value: &str) -> Result<Self, Rejection> {
let mut set = Self::NONE;
let mut seen: Vec<String> = Vec::new();
let mut total: Option<&'static str> = None;
for raw in value.split(',') {
let token = raw.trim().to_ascii_lowercase();
if token.is_empty() {
return Err(Rejection::new(value, "empty entry in the list"));
}
if seen.contains(&token) {
return Err(Rejection::new(value, format!("{token:?} appears more than once")));
}
seen.push(token.clone());
match token.as_str() {
"none" => total = Some(Self::NONE_LABEL),
"all" => {
total = Some("all");
set = Self::ALL;
}
"lines" => set = set.with_lines(),
"code" => set = set.with_code(),
"words" => set = set.with_words(),
other => {
return Err(Rejection::new(
other,
"expected one of none, lines, code, words, all",
));
}
}
}
if let Some(total) = total {
if seen.len() > 1 {
return Err(Rejection::new(
value,
format!("{total:?} names the whole axis and cannot be combined"),
));
}
if total == Self::NONE_LABEL {
return Ok(Self::NONE);
}
}
Ok(set)
}
pub fn labels(self) -> Vec<&'static str> {
let mut labels = Vec::new();
if self.0 & Self::LINES != 0 {
labels.push("lines");
}
if self.includes_code() {
labels.push("code");
}
if self.includes_words() {
labels.push("words");
}
labels
}
}
#[derive(Clone, PartialEq, Eq, Debug)]
pub struct ContentDetection {
pub file_type: FileTypeId,
pub family: ContentFamily,
pub source: DetectionSource,
pub confidence: DetectionConfidence,
pub flags: ClassificationFlags,
}
impl From<Classification> for ContentDetection {
fn from(value: Classification) -> Self {
Self {
file_type: value.file_type,
family: value.family,
source: value.source,
confidence: value.confidence,
flags: value.flags,
}
}
}
impl From<ContentDetection> for Classification {
fn from(value: ContentDetection) -> Self {
Self {
file_type: value.file_type,
family: value.family,
source: value.source,
confidence: value.confidence,
flags: value.flags,
}
}
}
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub struct AnalysisRequest {
pub profile: AnalysisSet,
pub workers: usize,
}
impl Default for AnalysisRequest {
fn default() -> Self {
Self { profile: AnalysisSet::NONE, workers: 0 }
}
}
impl AnalysisRequest {
pub fn options_fingerprint(self) -> OptionsFingerprint {
const OFFSET: u64 = 0xcbf2_9ce4_8422_2325;
const PRIME: u64 = 0x0000_0100_0000_01b3;
let hash = [self.profile.bits()]
.into_iter()
.fold(OFFSET, |hash, byte| (hash ^ u64::from(byte)).wrapping_mul(PRIME));
OptionsFingerprint(hash)
}
}
#[derive(Clone, PartialEq, Eq, Debug)]
pub struct ContentProvenance {
pub type_rules_fingerprint: u64,
pub options_fingerprint: OptionsFingerprint,
pub analyzers: Vec<(AnalyzerId, AnalyzerVersion)>,
}
impl ContentProvenance {
pub fn for_request(request: AnalysisRequest, type_rules_fingerprint: u64) -> Self {
const VERSION_ONE: AnalyzerVersion = AnalyzerVersion(1);
let mut analyzers = Vec::new();
if request.profile.is_enabled() {
analyzers.push((CONTENT_BASIC, VERSION_ONE));
}
if request.profile.includes_code() {
analyzers.push((CODE_SLOC, AnalyzerVersion(3)));
}
if request.profile.includes_words() {
analyzers.push((TEXT_LOGICAL, VERSION_ONE));
analyzers.push((MARKDOWN_PROSE, VERSION_ONE));
}
Self {
type_rules_fingerprint,
options_fingerprint: request.options_fingerprint(),
analyzers,
}
}
}
#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
pub struct LogicalWordStats {
pub wide_chars: u64,
pub nonwide_tokens: u64,
pub nonwide_chars: u64,
}
#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
pub struct BasicMetrics {
pub physical_lines: u64,
pub blank_lines: u64,
pub nonblank_lines: u64,
pub raw_words: u64,
}
#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
pub struct CodeMetrics {
pub code_lines: u64,
pub comment_lines: u64,
pub code_blank_lines: u64,
}
#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
pub struct WordMetrics {
pub paragraphs: u64,
pub visible_words: u64,
pub logical_word_stats: LogicalWordStats,
pub visible_logical_word_stats: LogicalWordStats,
}
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub struct AnalyzerOutcome<T> {
coverage: CoverageReason,
value: Option<T>,
}
impl<T> AnalyzerOutcome<T> {
pub(crate) const fn analyzed(value: T) -> Self {
Self { coverage: CoverageReason::Analyzed, value: Some(value) }
}
pub(crate) fn unavailable(coverage: CoverageReason) -> Self {
assert!(
!matches!(coverage, CoverageReason::Analyzed),
"an analyzed outcome must carry a value"
);
Self { coverage, value: None }
}
pub const fn coverage(&self) -> CoverageReason {
self.coverage
}
pub const fn value(self) -> Option<T>
where
T: Copy,
{
self.value
}
pub(crate) fn from_parts(coverage: CoverageReason, value: Option<T>) -> Option<Self> {
if matches!(coverage, CoverageReason::Analyzed) == value.is_some() {
Some(Self { coverage, value })
} else {
None
}
}
pub const fn is_reusable(&self) -> bool {
!matches!(self.coverage, CoverageReason::IoError | CoverageReason::ChangedDuringRead)
}
}
impl LogicalWordStats {
pub(crate) fn add_assign(&mut self, other: Self) {
self.wide_chars = self.wide_chars.saturating_add(other.wide_chars);
self.nonwide_tokens = self.nonwide_tokens.saturating_add(other.nonwide_tokens);
self.nonwide_chars = self.nonwide_chars.saturating_add(other.nonwide_chars);
}
pub fn logical_words(self) -> u64 {
let chars = u128::from(self.nonwide_chars);
let tokens = u128::from(self.nonwide_tokens);
let wide = u128::from(self.wide_chars);
let (numerator, denominator) = if tokens.saturating_mul(6) < chars {
(chars.saturating_add(wide.saturating_mul(3)), 6)
} else if tokens.saturating_mul(3) > chars {
(chars.saturating_mul(2).saturating_add(wide.saturating_mul(3)), 6)
} else {
(tokens.saturating_mul(2).saturating_add(wide), 2)
};
let rounded = numerator.saturating_add(denominator / 2) / denominator;
u64::try_from(rounded).unwrap_or(u64::MAX)
}
}
#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
pub struct MetricValues {
pub physical_lines: u64,
pub blank_lines: u64,
pub nonblank_lines: u64,
pub raw_words: u64,
pub code_lines: u64,
pub comment_lines: u64,
pub code_blank_lines: u64,
pub paragraphs: u64,
pub visible_words: u64,
pub logical_word_stats: LogicalWordStats,
pub visible_logical_word_stats: LogicalWordStats,
}
#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
pub enum CoverageReason {
Analyzed,
Binary,
InvalidUtf8,
UnsupportedEncoding,
Unsupported,
IoError,
ChangedDuringRead,
}
#[derive(Clone, PartialEq, Eq, Debug)]
pub struct FileAnalysis {
pub fingerprint: Fingerprint,
pub bytes: u64,
pub detection: ContentDetection,
pub lines: AnalyzerOutcome<BasicMetrics>,
pub code: Option<AnalyzerOutcome<CodeMetrics>>,
pub words: Option<AnalyzerOutcome<WordMetrics>>,
pub error: Option<String>,
}
impl FileAnalysis {
pub const fn matches_profile(&self, profile: AnalysisSet) -> bool {
self.code.is_some() == profile.includes_code()
&& self.words.is_some() == profile.includes_words()
}
pub const fn operational_failure(&self) -> Option<CoverageReason> {
match self.lines.coverage() {
CoverageReason::IoError => Some(CoverageReason::IoError),
CoverageReason::ChangedDuringRead => Some(CoverageReason::ChangedDuringRead),
CoverageReason::Analyzed
| CoverageReason::Binary
| CoverageReason::InvalidUtf8
| CoverageReason::UnsupportedEncoding
| CoverageReason::Unsupported => None,
}
}
pub const fn is_reusable(&self) -> bool {
self.operational_failure().is_none()
&& self.lines.is_reusable()
&& match self.code {
Some(outcome) => outcome.is_reusable(),
None => true,
}
&& match self.words {
Some(outcome) => outcome.is_reusable(),
None => true,
}
}
}
#[derive(Clone, Debug)]
pub(crate) struct AnalysisCandidate {
pub entry_id: EntryId,
pub revision: u64,
pub relative_path: PathBuf,
pub absolute_path: PathBuf,
pub attrs: Attrs,
pub classification: Classification,
}
#[derive(Clone, Debug)]
pub(crate) struct RestoreCandidate {
pub entry_id: EntryId,
pub revision: u64,
pub relative_path: PathBuf,
pub attrs: Attrs,
}
#[derive(Clone, Debug)]
pub(crate) struct AnalysisObservation {
pub candidate: AnalysisCandidate,
pub profile: AnalysisSet,
pub provenance: ContentProvenance,
pub analysis: FileAnalysis,
}
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub(crate) enum AnalysisApplyOutcome {
Applied,
Stale,
}
#[cfg(test)]
mod tests {
use std::collections::HashSet;
use super::{
AnalysisRequest, AnalysisSet, AnalyzerOutcome, AnalyzerVersion, CODE_SLOC,
ContentProvenance, CoverageReason, LogicalWordStats, METRICS,
};
#[test]
#[should_panic(expected = "an analyzed outcome must carry a value")]
fn unavailable_outcome_cannot_claim_success() {
let _: AnalyzerOutcome<()> = AnalyzerOutcome::unavailable(CoverageReason::Analyzed);
}
#[test]
fn metric_registry_has_unique_names_and_one_requestable_owner_each() {
let mut names = HashSet::new();
for metric in METRICS {
assert!(names.insert(metric.name), "duplicate metric name {}", metric.name);
assert!(
matches!(
metric.owner,
AnalysisSet::LINES_ONLY | AnalysisSet::CODE_ONLY | AnalysisSet::WORDS_ONLY
),
"{} has a non-unit owner {:?}",
metric.name,
metric.owner
);
}
assert_eq!(names.len(), 12);
}
#[test]
fn logical_words_derive_only_after_additive_stats_are_combined() {
let first = LogicalWordStats { wide_chars: 3, nonwide_tokens: 1, nonwide_chars: 12 };
let second = LogicalWordStats { wide_chars: 1, nonwide_tokens: 9, nonwide_chars: 6 };
let combined = LogicalWordStats {
wide_chars: first.wide_chars + second.wide_chars,
nonwide_tokens: first.nonwide_tokens + second.nonwide_tokens,
nonwide_chars: first.nonwide_chars + second.nonwide_chars,
};
assert_eq!(combined.logical_words(), 8);
}
#[test]
fn logical_words_match_the_pinned_rational_clamp_and_half_up_rounding() {
let logical = |wide_chars, nonwide_tokens, nonwide_chars| {
LogicalWordStats { wide_chars, nonwide_tokens, nonwide_chars }.logical_words()
};
assert_eq!(logical(0, 0, 0), 0);
assert_eq!(logical(0, 2, 9), 2, "ordinary prose passes through");
assert_eq!(logical(0, 1, 12), 2, "long tokens use the six-character floor");
assert_eq!(logical(0, 4, 4), 1, "short tokens use the three-character ceiling");
assert_eq!(logical(3, 0, 0), 2, "wide halves round up once");
assert_eq!(logical(1, 1, 1), 1, "mixed fractions combine before rounding");
}
#[test]
fn the_analyzer_vocabulary_parses_every_accepted_spelling() {
let cases = [
("none", AnalysisSet::NONE),
("lines", AnalysisSet::NONE.with_lines()),
("code", AnalysisSet::NONE.with_code()),
("words", AnalysisSet::NONE.with_words()),
("all", AnalysisSet::ALL),
("code,words", AnalysisSet::ALL),
("words,code", AnalysisSet::ALL),
("lines,code", AnalysisSet::NONE.with_code()),
(" CODE , Words ", AnalysisSet::ALL),
];
for (input, expected) in cases {
assert_eq!(AnalysisSet::parse(input), Ok(expected), "parsing {input:?}");
}
}
#[test]
fn the_analyzer_vocabulary_rejects_every_incoherent_request() {
let cases = [
("", "empty entry"),
("code,,words", "empty entry"),
("code,code", "more than once"),
("basic", "expected one of"),
("documents", "expected one of"),
("full", "expected one of"),
("none,code", "cannot be combined"),
("all,code", "cannot be combined"),
("none,all", "cannot be combined"),
];
for (input, needle) in cases {
let error = AnalysisSet::parse(input).expect_err(&format!("{input:?} must fail"));
assert!(error.contains(needle), "parsing {input:?} said {error:?}, wanted {needle:?}");
}
}
#[test]
fn a_label_never_rewrites_the_value_it_is_reporting() {
for value in ["analyzer", "reanalyze", "analyze-all"] {
let error = AnalysisSet::parse_labeled(value, "--analyze")
.expect_err("must reject an unknown analyzer");
assert!(
error.contains(&format!("{value:?}")),
"{error} must quote {value:?} exactly as typed"
);
assert!(error.starts_with("invalid --analyze "), "{error} must carry the label");
}
}
#[test]
fn the_on_disk_encoding_round_trips_and_refuses_unknown_analyzers() {
for set in [
AnalysisSet::NONE,
AnalysisSet::NONE.with_lines(),
AnalysisSet::NONE.with_code(),
AnalysisSet::NONE.with_words(),
AnalysisSet::ALL,
] {
assert_eq!(AnalysisSet::from_bits(set.bits()), Some(set));
}
assert_eq!(AnalysisSet::from_bits(0b1000_0000), None);
}
#[test]
fn code_metrics_use_the_updated_analyzer_version() {
let request = AnalysisRequest { profile: AnalysisSet::NONE.with_code(), workers: 1 };
let provenance = ContentProvenance::for_request(request, 42);
assert!(provenance.analyzers.contains(&(CODE_SLOC, AnalyzerVersion(3))));
}
#[test]
fn labels_are_the_vocabulary_parse_accepts() {
for set in [
AnalysisSet::NONE.with_lines(),
AnalysisSet::NONE.with_code(),
AnalysisSet::NONE.with_words(),
AnalysisSet::ALL,
] {
let spelled = set.labels().join(",");
assert_eq!(AnalysisSet::parse(&spelled), Ok(set), "round trip through {spelled:?}");
}
assert!(AnalysisSet::NONE.labels().is_empty());
}
}