use std::borrow::Cow;
use std::collections::{BTreeMap, BTreeSet};
use std::path::{Path, PathBuf};
use crate::classify::{ContentFamily, DetectionConfidence, DetectionSource};
use crate::content::{AnalysisSet, ContentProvenance, CoverageReason, LogicalWordStats, MetricDef};
use crate::control::ControlCoverage;
use crate::engine_contract::{EntryKind, ScanScope};
use crate::index::{EntryId, ExtTally, Index, RollUpScalars};
use crate::query::query_request::{Basis, Request};
use crate::query::query_selection::{
Bound, IgnoredEntries, NameIdentity, Selection, SizeMetric, SortKey,
};
use crate::query::{Rejection, ReportProvenance, TreeStatus, query_subtrees};
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub enum ViewSpec {
List,
Tree,
Types,
Extensions,
Families,
Languages,
Documents,
Files,
Largest,
Recent,
Summary,
}
impl ViewSpec {
fn default_sort(self) -> SortKey {
match self {
Self::List
| Self::Tree
| Self::Types
| Self::Extensions
| Self::Families
| Self::Languages
| Self::Documents
| Self::Summary
| Self::Largest => SortKey::Size,
Self::Files => SortKey::Name,
Self::Recent => SortKey::Mtime,
}
}
const fn default_limit(self) -> Bound {
match self {
Self::List | Self::Files => Bound::All,
Self::Largest | Self::Recent => Bound::Limit(20),
_ => Bound::Limit(10),
}
}
const fn default_depth(self) -> Bound {
match self {
Self::Tree => Bound::Limit(2),
_ => Bound::All,
}
}
const fn files_only(self) -> bool {
matches!(self, Self::Largest | Self::Recent)
}
pub const ALL: [Self; 11] = [
Self::List,
Self::Summary,
Self::Tree,
Self::Families,
Self::Types,
Self::Extensions,
Self::Languages,
Self::Documents,
Self::Largest,
Self::Recent,
Self::Files,
];
pub fn parse(value: &str) -> Result<Self, String> {
match value.trim().to_ascii_lowercase().as_str() {
"list" => Ok(Self::List),
"tree" => Ok(Self::Tree),
"types" => Ok(Self::Types),
"extensions" => Ok(Self::Extensions),
"families" => Ok(Self::Families),
"languages" => Ok(Self::Languages),
"documents" => Ok(Self::Documents),
"largest" => Ok(Self::Largest),
"recent" => Ok(Self::Recent),
"files" => Ok(Self::Files),
"summary" => Ok(Self::Summary),
_ => Err(format!("expected one of {}", Self::vocabulary())),
}
}
pub fn vocabulary() -> String {
let mut names: Vec<&str> = Self::ALL.iter().map(|view| view.label()).collect();
names.push("full");
names.join(", ")
}
pub const fn label(self) -> &'static str {
match self {
Self::List => "list",
Self::Tree => "tree",
Self::Types => "types",
Self::Extensions => "extensions",
Self::Families => "families",
Self::Languages => "languages",
Self::Documents => "documents",
Self::Largest => "largest",
Self::Recent => "recent",
Self::Files => "files",
Self::Summary => "summary",
}
}
pub const fn default_for(analysis: AnalysisSet) -> Self {
match (analysis.includes_code(), analysis.includes_words()) {
(true, true) => Self::Families,
(true, false) => Self::Languages,
(false, true) => Self::Documents,
(false, false) if analysis.is_enabled() => Self::Families,
(false, false) => Self::List,
}
}
pub fn resolve(
spec: Option<&str>,
analysis: AnalysisSet,
label: &str,
) -> Result<(Vec<Self>, Vec<Self>), String> {
Self::resolve_rejecting(spec, analysis).map_err(|rejection| rejection.labeled(label))
}
pub(crate) fn resolve_rejecting(
spec: Option<&str>,
analysis: AnalysisSet,
) -> Result<(Vec<Self>, Vec<Self>), Rejection> {
let Some(spec) = spec else {
return Ok((vec![Self::default_for(analysis)], Vec::new()));
};
let mut parsed: Vec<Self> = Vec::new();
let mut full_seen = false;
for raw in spec.split(',') {
let token = raw.trim();
if token.is_empty() {
return Err(Rejection::new(spec, "empty entry in the list"));
}
if token.eq_ignore_ascii_case("full") {
if full_seen || !parsed.is_empty() {
return Err(Rejection::new("full", Self::FULL_IS_EXCLUSIVE));
}
full_seen = true;
continue;
}
if full_seen {
return Err(Rejection::new("full", Self::FULL_IS_EXCLUSIVE));
}
let view = Self::parse(token).map_err(|expected| Rejection::new(token, expected))?;
if parsed.contains(&view) {
return Err(Rejection::new(spec, format!("{token:?} appears more than once")));
}
parsed.push(view);
}
if full_seen {
return Ok(Self::full_report(analysis));
}
Ok((parsed, Vec::new()))
}
pub const FULL_IS_EXCLUSIVE: &'static str =
"it names the whole report and cannot be combined with another view";
pub fn full_report(analysis: AnalysisSet) -> (Vec<Self>, Vec<Self>) {
Self::ALL
.into_iter()
.filter(|view| view.is_summary_view())
.partition(|view| !matches!(view, Self::Documents) || analysis.is_enabled())
}
pub const fn is_summary_view(self) -> bool {
!matches!(self, Self::List | Self::Files)
}
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct AxisNames {
pub view: &'static str,
pub format: &'static str,
pub analyze: &'static str,
pub control_budget: &'static str,
pub control_line_limit: &'static str,
pub exclude_ignored: &'static str,
pub only_ignored: &'static str,
pub read_controls: &'static str,
pub scan_depth: &'static str,
pub one_filesystem: &'static str,
pub follow_symlinks: &'static str,
pub include: &'static str,
pub modified_since: &'static str,
pub modified_before: &'static str,
pub kind: &'static str,
pub ignored: &'static str,
pub depth: &'static str,
pub limit: &'static str,
pub sort: &'static str,
pub size: &'static str,
pub words_per_page: &'static str,
pub cache: &'static str,
pub watch: &'static str,
}
impl AxisNames {
pub const FLAGS: Self = Self {
view: "--view",
format: "--format",
analyze: "--analyze",
control_budget: "--gitignore-budget",
control_line_limit: "--gitignore-line-limit",
exclude_ignored: "--exclude-ignored",
only_ignored: "--only-ignored",
read_controls: "--no-gitignore",
scan_depth: "--scan-depth",
one_filesystem: "--one-filesystem",
follow_symlinks: "follow_symlinks",
include: "--include",
modified_since: "--modified-since",
modified_before: "--modified-before",
kind: "--kind",
ignored: "--exclude-ignored/--only-ignored",
depth: "--depth",
limit: "--limit",
sort: "--sort",
size: "--size",
words_per_page: "--words-per-page",
cache: "--cache",
watch: "--watch",
};
pub const FIELDS: Self = Self {
view: "view",
format: "format",
analyze: "analyze",
control_budget: "control_budget",
control_line_limit: "control_line_limit",
exclude_ignored: "ignored=exclude",
only_ignored: "ignored=only",
read_controls: "read_controls",
scan_depth: "max_depth",
one_filesystem: "one_filesystem",
follow_symlinks: "follow_symlinks",
include: "include",
modified_since: "modified_since",
modified_before: "modified_before",
kind: "kind",
ignored: "ignored",
depth: "depth",
limit: "limit",
sort: "sort",
size: "size",
words_per_page: "words_per_page",
cache: "cache policy",
watch: "watch",
};
}
impl Default for AxisNames {
fn default() -> Self {
Self::FIELDS
}
}
#[derive(Clone, Debug)]
pub struct Query {
pub selection: Selection,
pub views: Vec<ViewSpec>,
pub format: crate::report_format::Format,
pub omitted_views: Vec<ViewSpec>,
pub axes: &'static AxisNames,
pub words_per_page: u64,
}
impl Default for Query {
fn default() -> Self {
Self {
selection: Selection::default(),
views: Vec::new(),
format: crate::report_format::Format::Text,
omitted_views: Vec::new(),
axes: &AxisNames::FIELDS,
words_per_page: crate::query::Request::DEFAULTS.words_per_page,
}
}
}
impl Query {
pub fn tree_for(&self, view: ViewSpec) -> bool {
use crate::report_format::Format;
match view {
ViewSpec::List => matches!(self.format, Format::Text | Format::Tree),
ViewSpec::Tree => !matches!(self.format, Format::Paths | Format::Long),
ViewSpec::Files => self.format == Format::Tree,
_ => false,
}
}
pub(crate) fn needs_selection_walk(&self) -> bool {
!self.selection.is_unfiltered()
|| self.views.iter().any(|view| {
matches!(view, ViewSpec::List | ViewSpec::Tree | ViewSpec::Files)
&& !self.tree_for(*view)
})
}
pub fn limit_for(&self, view: ViewSpec) -> Bound {
self.selection.limit.unwrap_or_else(|| {
if self.tree_for(view) {
ViewSpec::Tree.default_limit()
} else if view == ViewSpec::Tree {
Bound::All
} else {
view.default_limit()
}
})
}
pub fn depth_for(&self, view: ViewSpec) -> Bound {
self.selection.depth.unwrap_or_else(|| {
if self.tree_for(view) { ViewSpec::Tree.default_depth() } else { view.default_depth() }
})
}
}
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub enum ReportSource {
ColdScan,
WarmRevalidate,
CacheOnly,
}
#[derive(Clone, Debug)]
pub struct TreeNode {
pub path: PathBuf,
pub name: String,
pub kind: EntryKind,
pub bytes: u64,
pub allocated: u64,
pub files: u64,
pub dirs: u64,
pub ignored: Option<IgnoredTally>,
pub newest_mtime_ns: Option<i64>,
pub children: Vec<TreeNode>,
pub truncated: bool,
}
impl Drop for TreeNode {
fn drop(&mut self) {
let mut pending = std::mem::take(&mut self.children);
while let Some(mut node) = pending.pop() {
pending.extend(std::mem::take(&mut node.children));
}
}
}
#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
pub struct IgnoredTally {
pub files: u64,
pub dirs: u64,
pub bytes: u64,
pub allocated: u64,
}
impl IgnoredTally {
pub(crate) fn between(all: RollUpScalars, unignored: RollUpScalars) -> Self {
Self {
files: all.files.saturating_sub(unignored.files),
dirs: all.dirs.saturating_sub(unignored.dirs),
bytes: all.bytes.saturating_sub(unignored.bytes),
allocated: all.allocated.saturating_sub(unignored.allocated),
}
}
fn add(&mut self, other: Self) {
self.files = self.files.saturating_add(other.files);
self.dirs = self.dirs.saturating_add(other.dirs);
self.bytes = self.bytes.saturating_add(other.bytes);
self.allocated = self.allocated.saturating_add(other.allocated);
}
}
#[derive(Clone, Debug)]
pub struct TypeRow {
pub extension: String,
pub files: u64,
pub bytes: u64,
pub allocated: u64,
pub ignored: Option<IgnoredTally>,
}
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub enum MetricGroup {
Type,
Family,
}
#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
pub struct MetricShare {
pub numerator: u64,
pub denominator: u64,
}
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub enum ShareMetric {
ApparentBytes,
AllocatedBytes,
CodeLines,
DocumentWords,
RawWords,
}
impl ShareMetric {
pub const fn as_str(self) -> &'static str {
match self {
Self::ApparentBytes => "apparent_bytes",
Self::AllocatedBytes => "allocated_bytes",
Self::CodeLines => "code_lines",
Self::DocumentWords => "document_words",
Self::RawWords => "raw_words",
}
}
}
#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
pub struct ReportMetricValues {
pub physical_lines: Option<u64>,
pub blank_lines: Option<u64>,
pub nonblank_lines: Option<u64>,
pub raw_words: Option<u64>,
pub code_lines: Option<u64>,
pub comment_lines: Option<u64>,
pub code_blank_lines: Option<u64>,
pub logical_words: Option<u64>,
pub paragraphs: Option<u64>,
pub visible_words: Option<u64>,
pub visible_logical_words: Option<u64>,
pub document_words: Option<u64>,
}
impl ReportMetricValues {
fn for_analysis(analysis: AnalysisSet) -> Self {
let lines = analysis.is_enabled().then_some(0);
let code = analysis.includes_code().then_some(0);
let words = analysis.includes_words().then_some(0);
Self {
physical_lines: lines,
blank_lines: lines,
nonblank_lines: lines,
raw_words: lines,
code_lines: code,
comment_lines: code,
code_blank_lines: code,
logical_words: words,
paragraphs: words,
visible_words: words,
visible_logical_words: words,
document_words: words,
}
}
fn add_assign(&mut self, other: &Self) {
add_optional(&mut self.physical_lines, other.physical_lines);
add_optional(&mut self.blank_lines, other.blank_lines);
add_optional(&mut self.nonblank_lines, other.nonblank_lines);
add_optional(&mut self.raw_words, other.raw_words);
add_optional(&mut self.code_lines, other.code_lines);
add_optional(&mut self.comment_lines, other.comment_lines);
add_optional(&mut self.code_blank_lines, other.code_blank_lines);
add_optional(&mut self.paragraphs, other.paragraphs);
add_optional(&mut self.visible_words, other.visible_words);
}
}
fn add_optional(total: &mut Option<u64>, value: Option<u64>) {
if let (Some(total), Some(value)) = (total, value) {
*total = total.saturating_add(value);
}
}
#[derive(Clone, Debug)]
pub struct MetricRow {
pub analysis: AnalysisSet,
pub id: String,
pub family: ContentFamily,
pub files: u64,
pub bytes: u64,
pub allocated: u64,
pub analyzed_files: u64,
pub metrics: ReportMetricValues,
pub(crate) logical_word_stats: LogicalWordStats,
pub(crate) visible_logical_word_stats: LogicalWordStats,
pub document_raw_words: u64,
pub document_word_stats: LogicalWordStats,
pub document_metric_files: u64,
pub coverage: BTreeMap<CoverageReason, u64>,
pub lines_coverage: BTreeMap<CoverageReason, u64>,
pub code_coverage: Option<BTreeMap<CoverageReason, u64>>,
pub words_coverage: Option<BTreeMap<CoverageReason, u64>>,
pub detection_sources: BTreeMap<DetectionSource, u64>,
pub detection_confidence: BTreeMap<DetectionConfidence, u64>,
pub generated_files: u64,
pub vendored_files: u64,
pub documentation_files: u64,
pub share: MetricShare,
}
impl MetricRow {
pub fn metric_value(&self, metric: &MetricDef) -> Option<u64> {
if !self.analysis.contains(metric.owner) {
return None;
}
match metric.name {
"physical_lines" => self.metrics.physical_lines,
"blank_lines" => self.metrics.blank_lines,
"nonblank_lines" => self.metrics.nonblank_lines,
"raw_words" => self.metrics.raw_words,
"code_lines" => self.metrics.code_lines,
"comment_lines" => self.metrics.comment_lines,
"code_blank_lines" => self.metrics.code_blank_lines,
"logical_words" => self.metrics.logical_words,
"paragraphs" => self.metrics.paragraphs,
"visible_words" => self.metrics.visible_words,
"visible_logical_words" => self.metrics.visible_logical_words,
"document_words" => self.metrics.document_words,
_ => None,
}
}
fn finish_derived_metrics(&mut self) {
if self.analysis.includes_words() {
self.metrics.logical_words = Some(self.logical_word_stats.logical_words());
self.metrics.visible_logical_words =
Some(self.visible_logical_word_stats.logical_words());
self.metrics.document_words = Some(self.document_word_stats.logical_words());
}
}
}
#[derive(Clone, Debug)]
pub struct MetricSummary {
pub group: MetricGroup,
pub total: MetricRow,
pub rows: Vec<MetricRow>,
pub total_rows: usize,
pub share_metric: ShareMetric,
pub words_per_page: u64,
}
#[derive(Clone, Debug)]
pub struct ContentReportMetadata {
pub profile: AnalysisSet,
pub provenance: ContentProvenance,
}
#[derive(Clone, Debug)]
pub struct FileRow {
pub path: PathBuf,
pub kind: EntryKind,
pub bytes: u64,
pub allocated: u64,
pub mtime_ns: i64,
pub files: Option<u64>,
pub dirs: Option<u64>,
pub complete: Option<bool>,
pub age_ns: Option<i128>,
pub ignored: Option<bool>,
}
#[derive(Clone, Copy, Debug, Default)]
pub struct SummaryRow {
pub files: u64,
pub dirs: u64,
pub bytes: u64,
pub allocated: u64,
pub ignored: Option<IgnoredTally>,
pub newest_mtime_ns: Option<i64>,
}
#[derive(Clone, Debug)]
pub enum Section {
Tree {
view: ViewSpec,
root: TreeNode,
},
Extensions {
rows: Vec<TypeRow>,
total: usize,
},
Metrics {
view: ViewSpec,
summary: Box<MetricSummary>,
},
Files {
view: ViewSpec,
rows: Vec<FileRow>,
total: usize,
},
Summary(SummaryRow),
}
impl Section {
pub fn view(&self) -> ViewSpec {
match self {
Self::Extensions { .. } => ViewSpec::Extensions,
Self::Tree { view, .. } | Self::Metrics { view, .. } | Self::Files { view, .. } => {
*view
}
Self::Summary(_) => ViewSpec::Summary,
}
}
}
#[derive(Clone, Debug)]
pub struct Report {
pub age_reference_ns: Option<i64>,
pub format: crate::report_format::Format,
pub status: TreeStatus,
pub provenance: ReportProvenance,
pub scope: ScanScope,
pub requested_analysis: AnalysisSet,
pub requested_views: Vec<ViewSpec>,
pub omitted_views: Vec<ViewSpec>,
pub root: PathBuf,
pub notes: Vec<String>,
pub size: SizeMetric,
pub analysis: Option<ContentReportMetadata>,
pub ignored_entries: IgnoredEntries,
pub ignore_rules: ControlCoverage,
pub sections: Vec<Section>,
}
pub(crate) fn display_notes(query: &Query, ignore_rules: &ControlCoverage) -> Vec<String> {
let mut notes = Vec::new();
if !query.omitted_views.is_empty() {
let names: Vec<&str> = query.omitted_views.iter().map(|view| view.label()).collect();
notes.push(format!(
"note: omitted {} — requires content analysis: add {} lines, code, words, or all",
names.join(", "),
query.axes.analyze
));
}
notes.extend(refused_controls_note(ignore_rules, query.axes));
notes
}
const REFUSED_DIRECTORIES_NAMED: usize = 5;
fn refused_controls_note(ignore_rules: &ControlCoverage, axes: &AxisNames) -> Option<String> {
use crate::control::ControlRefusalReason::{Budget, LineLimit};
let ControlCoverage::Observed(observed) = ignore_rules else {
return None;
};
if observed.is_complete() {
return None;
}
let every_listed = observed.lists_every_refusal();
let listed =
|reason| observed.refusals.iter().filter(|refusal| refusal.reason == reason).count();
let fired: Vec<_> = [Budget, LineLimit]
.into_iter()
.filter(|reason| {
listed(*reason) > 0 || (!every_listed && observed.limits.limit_for(*reason).is_some())
})
.collect();
let size = |reason| {
observed.limits.limit_for(reason).map(|bytes| {
crate::report_format::human_bytes(u64::try_from(bytes).unwrap_or(u64::MAX))
})
};
let over = |reason| {
let (lead, noun) = match reason {
Budget => ("over", "ignore-rule budget"),
LineLimit => ("with a line over", "line limit"),
};
size(reason).map_or_else(
|| format!("{lead} the {noun}"),
|size| format!("{lead} the {size} {noun}"),
)
};
let why = if every_listed {
let parts: Vec<String> =
fired.iter().map(|reason| format!("{} {}", listed(*reason), over(*reason))).collect();
parts.join(", ")
} else {
let parts: Vec<String> = fired.iter().map(|reason| over(*reason)).collect();
parts.join(" or ")
};
let shown = observed.refusals.len().min(REFUSED_DIRECTORIES_NAMED);
let mut directories: Vec<String> = observed.refusals[..shown]
.iter()
.map(|refusal| match refusal.path.parent() {
Some(parent) if !parent.as_os_str().is_empty() => parent.display().to_string(),
_ => ".".to_string(),
})
.collect();
let unnamed = observed.refused.saturating_sub(u64::try_from(shown).unwrap_or(u64::MAX));
if unnamed > 0 {
directories.push(format!("{} more", crate::report_format::human_count(unnamed)));
}
let raises: Vec<String> = fired
.iter()
.filter_map(|reason| {
let knob = match reason {
Budget => axes.control_budget,
LineLimit => axes.control_line_limit,
};
size(*reason).map(|size| format!("{knob} above {size}"))
})
.collect();
let remedy = match raises.as_slice() {
[] => String::new(),
[raise] => format!(" To apply them, raise {raise}, or set it to all"),
raises => format!(" To apply them, raise {}, or set them to all", raises.join(" and ")),
};
let files = crate::report_format::human_count(observed.refused);
let noun = if observed.refused == 1 { "file" } else { "files" };
Some(format!(
"note: {files} .gitignore {noun} not applied ({why}), so ignored shares under {} are \
not exact; sizes are.{remedy}",
directories.join(", ")
))
}
pub fn report(
index: &Index,
request: &Request,
generated_at: std::time::SystemTime,
) -> crate::Result<Report> {
report_in(index, request, generated_at, NameIdentity::Native)
}
pub(crate) fn report_in(
index: &Index,
request: &Request,
generated_at: std::time::SystemTime,
identity: NameIdentity,
) -> crate::Result<Report> {
request.validate_read(&Basis::held_by(index)).map_err(crate::Error::InvalidRequest)?;
let query = &request.query;
let content = request.basis.content;
let walked = query.needs_selection_walk().then(|| walk(index, &query.selection, identity));
let row_consumers =
query.views.iter().copied().filter(|view| needs_unfiltered_entry_rows(*view)).count();
let unfiltered_rows = (walked.is_none() && row_consumers > 1).then(|| every_entry(index));
let mut sections: Vec<Section> = query
.views
.iter()
.map(|view| {
build_section(*view, index, query, content, walked.as_ref(), unfiltered_rows.as_deref())
})
.collect();
let age_reference_ns = crate::query::system_time_to_nanos(request.now);
for section in &mut sections {
if let Section::Files { rows, .. } = section {
for row in rows {
row.age_ns = match row.complete {
Some(false) => None,
Some(true) | None => {
age_reference_ns.map(|now| i128::from(now) - i128::from(row.mtime_ns))
}
};
}
}
}
let ignore_rules = index.control_coverage();
Ok(Report {
age_reference_ns,
format: query.format,
notes: display_notes(query, &ignore_rules),
status: TreeStatus::of(index, request),
provenance: ReportProvenance::of(index, content, generated_at),
scope: index.scope(),
requested_analysis: content,
requested_views: query.views.clone(),
omitted_views: query.omitted_views.clone(),
root: index.root_path().to_path_buf(),
size: query.selection.size,
analysis: index.content().and_then(|held| {
let wanted = index.content_identity(content);
let projected = held.admit(&wanted)?;
Some(ContentReportMetadata {
profile: projected.identity().analysis,
provenance: projected.identity().record_provenance(),
})
}),
ignored_entries: query.selection.ignored,
ignore_rules,
sections,
})
}
pub(crate) fn report_summary(
root: &Path,
scope: ScanScope,
request: &Request,
summary: SummaryRow,
status: TreeStatus,
provenance: ReportProvenance,
) -> Report {
Report {
age_reference_ns: crate::query::system_time_to_nanos(request.now),
format: request.query.format,
notes: Vec::new(),
status,
provenance,
scope,
requested_analysis: AnalysisSet::NONE,
requested_views: vec![ViewSpec::Summary],
omitted_views: Vec::new(),
root: root.to_path_buf(),
size: request.query.selection.size,
analysis: None,
ignored_entries: IgnoredEntries::Include,
ignore_rules: ControlCoverage::NotObserved,
sections: vec![Section::Summary(SummaryRow { ignored: None, ..summary })],
}
}
struct Walked {
observed: bool,
per_directory: BTreeMap<EntryId, SummaryRow>,
by_ext: BTreeMap<String, ExtTally>,
ignored_by_ext: BTreeMap<String, ExtTally>,
rows: Vec<FileRow>,
members: Vec<FileRow>,
visible: BTreeSet<EntryId>,
}
impl Walked {
fn summary_of(&self, id: EntryId) -> SummaryRow {
let mut row = self.per_directory.get(&id).copied().unwrap_or_default();
row.ignored = self.observed.then(|| row.ignored.unwrap_or_default());
row
}
}
fn unfiltered_summary(index: &Index, id: EntryId) -> SummaryRow {
let observed = index.observes_controls();
let Some((all, unignored)) = index.partition_scalars_of(id) else {
return SummaryRow {
ignored: observed.then(IgnoredTally::default),
..SummaryRow::default()
};
};
SummaryRow {
ignored: observed.then(|| IgnoredTally::between(all, unignored)),
..summary_from_scalars(all)
}
}
fn walk(index: &Index, selection: &Selection, identity: NameIdentity) -> Walked {
let observed = index.observes_controls();
let mut walked = Walked {
observed,
per_directory: BTreeMap::new(),
by_ext: BTreeMap::new(),
ignored_by_ext: BTreeMap::new(),
rows: Vec::new(),
members: Vec::new(),
visible: BTreeSet::new(),
};
debug_assert!(
observed || selection.ignored == IgnoredEntries::Include,
"a selection by ignored state over an unobserving index is refused before the walk"
);
let directories = (selection.kinds.is_empty() || selection.kinds.contains(&EntryKind::Dir))
.then(|| query_subtrees::measure(index, selection, identity));
let mut stack = vec![(EntryId::ROOT, PathBuf::new(), false, false)];
while let Some((id, path, expanded, covered)) = stack.pop() {
if expanded {
let mut total = walked.per_directory.remove(&id).unwrap_or_default();
if let Some(children) = index.children_of(id) {
for (_, child) in children {
if let Some(sub) = walked.per_directory.get(&child) {
let sub = *sub;
merge_summary(&mut total, &sub);
}
}
}
if total.files > 0 || total.dirs > 0 {
walked.visible.insert(id);
}
walked.per_directory.insert(id, total);
continue;
}
stack.push((id, path.clone(), true, covered));
let Some(children) = index.children_of(id) else {
continue;
};
let children: Vec<(PathBuf, EntryId)> =
children.map(|(name, child)| (path.join(name), child)).collect();
for (child_path, child) in children {
let (Some(kind), Some(attrs)) = (index.kind_of(child), index.attrs_of(child)) else {
continue;
};
let file_name = child_path.file_name().unwrap_or_default();
let ignored = index.ignored_bit_of(child).unwrap_or(false);
let mut measured = *attrs;
let subtree = directories.as_ref().and_then(|values| values.get(&child)).copied();
if let Some(subtree) = subtree {
measured.size = subtree.bytes;
measured.allocated = subtree.allocated;
measured.mtime_ns = subtree.mtime_ns;
}
let (pruned, matches) = query_subtrees::with_candidate(
&child_path,
kind,
measured,
ignored,
identity,
|candidate| {
(query_subtrees::pruned(selection, &candidate), selection.admits(&candidate))
},
);
if pruned {
continue;
}
let matches = matches
&& (selection.modified.is_unbounded()
|| subtree.is_none_or(|subtree| subtree.complete));
let row = FileRow {
path: child_path.clone(),
kind,
bytes: measured.size,
allocated: measured.allocated,
mtime_ns: measured.mtime_ns,
files: subtree.map(|subtree| subtree.files),
dirs: subtree.map(|subtree| subtree.dirs),
complete: subtree.map(|subtree| subtree.complete),
age_ns: None,
ignored: observed.then_some(ignored),
};
if matches {
walked.rows.push(row.clone());
}
if matches || (covered && selection.ignored.admits(ignored)) {
if kind == EntryKind::File {
walked.members.push(row);
} else if kind == EntryKind::Dir {
walked.visible.insert(child);
}
if kind == EntryKind::File {
let own = walked.per_directory.entry(id).or_default();
own.files += 1;
own.bytes += attrs.size;
own.allocated += attrs.allocated;
own.newest_mtime_ns = Some(
own.newest_mtime_ns.map_or(attrs.mtime_ns, |seen| seen.max(attrs.mtime_ns)),
);
let bucket = crate::classify::ext_bucket(file_name);
if ignored {
own.ignored.get_or_insert_with(IgnoredTally::default).add(IgnoredTally {
files: 1,
dirs: 0,
bytes: attrs.size,
allocated: attrs.allocated,
});
let tally = walked.ignored_by_ext.entry(bucket.clone()).or_default();
tally.files += 1;
tally.bytes += attrs.size;
tally.allocated += attrs.allocated;
}
let tally = walked.by_ext.entry(bucket).or_default();
tally.files += 1;
tally.bytes += attrs.size;
tally.allocated += attrs.allocated;
} else if kind == EntryKind::Dir {
let own = walked.per_directory.entry(id).or_default();
own.dirs += 1;
if ignored {
own.ignored.get_or_insert_with(IgnoredTally::default).dirs += 1;
}
}
}
if kind == EntryKind::Dir {
stack.push((child, child_path, false, covered || matches));
}
}
}
walked
}
fn merge_summary(into: &mut SummaryRow, from: &SummaryRow) {
into.files += from.files;
into.dirs += from.dirs;
into.bytes += from.bytes;
into.allocated += from.allocated;
into.newest_mtime_ns = match (into.newest_mtime_ns, from.newest_mtime_ns) {
(Some(left), Some(right)) => Some(left.max(right)),
(left, right) => left.or(right),
};
if let Some(share) = from.ignored {
into.ignored.get_or_insert_with(IgnoredTally::default).add(share);
}
}
fn needs_unfiltered_entry_rows(view: ViewSpec) -> bool {
matches!(
view,
ViewSpec::Types
| ViewSpec::Families
| ViewSpec::Languages
| ViewSpec::Documents
| ViewSpec::Files
| ViewSpec::Largest
| ViewSpec::Recent
)
}
fn entry_rows<'a>(
index: &Index,
walked: Option<&'a Walked>,
unfiltered_rows: Option<&'a [FileRow]>,
) -> Cow<'a, [FileRow]> {
match (walked, unfiltered_rows) {
(Some(walked), _) => Cow::Borrowed(&walked.rows),
(None, Some(rows)) => Cow::Borrowed(rows),
(None, None) => Cow::Owned(every_entry(index)),
}
}
fn build_section(
view: ViewSpec,
index: &Index,
query: &Query,
content: AnalysisSet,
walked: Option<&Walked>,
unfiltered_rows: Option<&[FileRow]>,
) -> Section {
if query.tree_for(view) {
return Section::Tree { view, root: tree_node(index, query, walked) };
}
match view {
ViewSpec::Summary => Section::Summary(match walked {
None => unfiltered_summary(index, EntryId::ROOT),
Some(walked) => walked.summary_of(EntryId::ROOT),
}),
ViewSpec::Extensions => {
let (rows, total) = extension_rows(index, query, walked);
Section::Extensions { rows, total }
}
ViewSpec::Types | ViewSpec::Families | ViewSpec::Languages | ViewSpec::Documents => {
Section::Metrics {
view,
summary: Box::new(metric_summary(
view,
index,
query,
content,
walked,
unfiltered_rows,
)),
}
}
ViewSpec::List
| ViewSpec::Tree
| ViewSpec::Files
| ViewSpec::Largest
| ViewSpec::Recent => {
let (rows, total) = file_rows(view, index, query, walked, unfiltered_rows);
Section::Files { view, rows, total }
}
}
}
fn summary_from_scalars(rollup: RollUpScalars) -> SummaryRow {
SummaryRow {
files: rollup.files,
dirs: rollup.dirs,
bytes: rollup.bytes,
allocated: rollup.allocated,
ignored: None,
newest_mtime_ns: (rollup.files > 0).then_some(rollup.newest_mtime_ns),
}
}
fn ignored_by_extension(
all: &BTreeMap<String, ExtTally>,
unignored: &BTreeMap<String, ExtTally>,
) -> BTreeMap<String, ExtTally> {
all.iter()
.filter_map(|(extension, tally)| {
let kept = unignored.get(extension).copied().unwrap_or_default();
let ignored = ExtTally {
files: tally.files.saturating_sub(kept.files),
bytes: tally.bytes.saturating_sub(kept.bytes),
allocated: tally.allocated.saturating_sub(kept.allocated),
};
(ignored.files > 0).then(|| (extension.clone(), ignored))
})
.collect()
}
fn extension_rows(index: &Index, query: &Query, walked: Option<&Walked>) -> (Vec<TypeRow>, usize) {
let observed = index.observes_controls();
let (tallies, ignored): (BTreeMap<String, ExtTally>, BTreeMap<String, ExtTally>) = match walked
{
None => match index.partition_total() {
Ok(partitions) => {
let ignored =
ignored_by_extension(&partitions.all.by_ext, &partitions.unignored.by_ext);
(partitions.all.by_ext, ignored)
}
Err(_not_observed) => (index.total().by_ext, BTreeMap::new()),
},
Some(walked) => (walked.by_ext.clone(), walked.ignored_by_ext.clone()),
};
let mut rows: Vec<TypeRow> = tallies
.into_iter()
.map(|(extension, tally)| {
let share = ignored.get(&extension).copied().unwrap_or_default();
TypeRow {
files: tally.files,
bytes: tally.bytes,
allocated: tally.allocated,
ignored: observed.then_some(IgnoredTally {
files: share.files,
dirs: 0,
bytes: share.bytes,
allocated: share.allocated,
}),
extension,
}
})
.collect();
sort_rows(
&mut rows,
query,
ViewSpec::Extensions,
|row, metric| match metric {
SizeMetric::Apparent => row.bytes,
SizeMetric::Allocated => row.allocated,
},
|row| row.files,
|_| None,
|row| row.extension.clone(),
);
let total = truncate(&mut rows, query.limit_for(ViewSpec::Extensions));
(rows, total)
}
fn metric_summary(
view: ViewSpec,
index: &Index,
query: &Query,
content: AnalysisSet,
walked: Option<&Walked>,
unfiltered_rows: Option<&[FileRow]>,
) -> MetricSummary {
let group = if view == ViewSpec::Families { MetricGroup::Family } else { MetricGroup::Type };
let files = walked.map_or_else(
|| entry_rows(index, None, unfiltered_rows),
|walked| Cow::Borrowed(walked.members.as_slice()),
);
let mut grouped = BTreeMap::<String, MetricRow>::new();
let wanted = index.content_identity(content);
let held = index.content().and_then(|held| held.admit(&wanted));
for file in files.iter().filter(|row| row.kind == EntryKind::File) {
let cached = held.and_then(|content| content.file(&file.path));
let classification = index.classify(&file.path);
let included = match view {
ViewSpec::Languages => classification.family == ContentFamily::Code,
ViewSpec::Documents => {
matches!(classification.family, ContentFamily::Prose | ContentFamily::Markup)
}
ViewSpec::Types | ViewSpec::Families => true,
ViewSpec::List
| ViewSpec::Tree
| ViewSpec::Extensions
| ViewSpec::Files
| ViewSpec::Largest
| ViewSpec::Recent
| ViewSpec::Summary => false,
};
if !included {
continue;
}
let id = match group {
MetricGroup::Type => classification.file_type.as_str().to_string(),
MetricGroup::Family => classification.family.as_str().to_string(),
};
let row = grouped.entry(id.clone()).or_insert_with(|| MetricRow {
analysis: content,
id,
family: classification.family,
files: 0,
bytes: 0,
allocated: 0,
analyzed_files: 0,
metrics: ReportMetricValues::for_analysis(content),
logical_word_stats: LogicalWordStats::default(),
visible_logical_word_stats: LogicalWordStats::default(),
document_raw_words: 0,
document_word_stats: LogicalWordStats::default(),
document_metric_files: 0,
coverage: BTreeMap::new(),
lines_coverage: BTreeMap::new(),
code_coverage: content.includes_code().then(BTreeMap::new),
words_coverage: content.includes_words().then(BTreeMap::new),
detection_sources: BTreeMap::new(),
detection_confidence: BTreeMap::new(),
generated_files: 0,
vendored_files: 0,
documentation_files: 0,
share: MetricShare::default(),
});
row.files = row.files.saturating_add(1);
row.bytes = row.bytes.saturating_add(file.bytes);
row.allocated = row.allocated.saturating_add(file.allocated);
let detection = cached.map_or(
(classification.source, classification.confidence, classification.flags),
|record| (record.detection.source, record.detection.confidence, record.detection.flags),
);
*row.detection_sources.entry(detection.0).or_default() += 1;
*row.detection_confidence.entry(detection.1).or_default() += 1;
row.generated_files = row.generated_files.saturating_add(u64::from(detection.2.generated));
row.vendored_files = row.vendored_files.saturating_add(u64::from(detection.2.vendored));
row.documentation_files =
row.documentation_files.saturating_add(u64::from(detection.2.documentation));
if let Some(record) = cached {
*row.lines_coverage.entry(record.lines.coverage()).or_default() += 1;
if let (Some(coverage), Some(outcome)) = (&mut row.code_coverage, record.code) {
*coverage.entry(outcome.coverage()).or_default() += 1;
}
if let (Some(coverage), Some(outcome)) = (&mut row.words_coverage, record.words) {
*coverage.entry(outcome.coverage()).or_default() += 1;
}
let selected = match view {
ViewSpec::Languages if content.includes_code() => {
record.code.map(|outcome| outcome.coverage())
}
ViewSpec::Documents if content.includes_words() => {
record.words.map(|outcome| outcome.coverage())
}
_ => None,
}
.unwrap_or(record.lines.coverage());
*row.coverage.entry(selected).or_default() += 1;
if selected == CoverageReason::Analyzed {
row.analyzed_files = row.analyzed_files.saturating_add(1);
}
if let Some(lines) = record.lines.value() {
add_optional(&mut row.metrics.physical_lines, Some(lines.physical_lines));
add_optional(&mut row.metrics.blank_lines, Some(lines.blank_lines));
add_optional(&mut row.metrics.nonblank_lines, Some(lines.nonblank_lines));
add_optional(&mut row.metrics.raw_words, Some(lines.raw_words));
row.document_raw_words = row.document_raw_words.saturating_add(lines.raw_words);
}
if let Some(code_metrics) = record.code.and_then(crate::content::AnalyzerOutcome::value)
{
add_optional(&mut row.metrics.code_lines, Some(code_metrics.code_lines));
add_optional(&mut row.metrics.comment_lines, Some(code_metrics.comment_lines));
add_optional(
&mut row.metrics.code_blank_lines,
Some(code_metrics.code_blank_lines),
);
}
if let Some(words) = record.words.and_then(crate::content::AnalyzerOutcome::value) {
add_optional(&mut row.metrics.paragraphs, Some(words.paragraphs));
add_optional(&mut row.metrics.visible_words, Some(words.visible_words));
row.logical_word_stats.add_assign(words.logical_word_stats);
row.visible_logical_word_stats.add_assign(words.visible_logical_word_stats);
row.document_metric_files = row.document_metric_files.saturating_add(1);
if classification.file_type.as_str() == "markdown" {
row.document_raw_words = row
.document_raw_words
.saturating_sub(record.lines.value().map_or(0, |lines| lines.raw_words));
row.document_raw_words =
row.document_raw_words.saturating_add(words.visible_words);
row.document_word_stats.add_assign(words.visible_logical_word_stats);
} else {
row.document_word_stats.add_assign(words.logical_word_stats);
}
}
}
}
for row in grouped.values_mut() {
row.finish_derived_metrics();
}
let mut total = MetricRow {
analysis: content,
id: "total".to_string(),
family: ContentFamily::Unknown,
files: 0,
bytes: 0,
allocated: 0,
analyzed_files: 0,
metrics: ReportMetricValues::for_analysis(content),
logical_word_stats: LogicalWordStats::default(),
visible_logical_word_stats: LogicalWordStats::default(),
document_raw_words: 0,
document_word_stats: LogicalWordStats::default(),
document_metric_files: 0,
coverage: BTreeMap::new(),
lines_coverage: BTreeMap::new(),
code_coverage: content.includes_code().then(BTreeMap::new),
words_coverage: content.includes_words().then(BTreeMap::new),
detection_sources: BTreeMap::new(),
detection_confidence: BTreeMap::new(),
generated_files: 0,
vendored_files: 0,
documentation_files: 0,
share: MetricShare::default(),
};
for row in grouped.values() {
total.files = total.files.saturating_add(row.files);
total.bytes = total.bytes.saturating_add(row.bytes);
total.allocated = total.allocated.saturating_add(row.allocated);
total.analyzed_files = total.analyzed_files.saturating_add(row.analyzed_files);
total.metrics.add_assign(&row.metrics);
total.logical_word_stats.add_assign(row.logical_word_stats);
total.visible_logical_word_stats.add_assign(row.visible_logical_word_stats);
total.document_raw_words = total.document_raw_words.saturating_add(row.document_raw_words);
total.document_word_stats.add_assign(row.document_word_stats);
total.document_metric_files =
total.document_metric_files.saturating_add(row.document_metric_files);
for (reason, count) in &row.coverage {
*total.coverage.entry(*reason).or_default() += count;
}
merge_coverage(&mut total.lines_coverage, &row.lines_coverage);
if let (Some(total), Some(row)) = (&mut total.code_coverage, &row.code_coverage) {
merge_coverage(total, row);
}
if let (Some(total), Some(row)) = (&mut total.words_coverage, &row.words_coverage) {
merge_coverage(total, row);
}
for (source, count) in &row.detection_sources {
*total.detection_sources.entry(*source).or_default() += count;
}
for (confidence, count) in &row.detection_confidence {
*total.detection_confidence.entry(*confidence).or_default() += count;
}
total.generated_files = total.generated_files.saturating_add(row.generated_files);
total.vendored_files = total.vendored_files.saturating_add(row.vendored_files);
total.documentation_files =
total.documentation_files.saturating_add(row.documentation_files);
}
total.finish_derived_metrics();
let byte_share_metric = match query.selection.size {
SizeMetric::Apparent => ShareMetric::ApparentBytes,
SizeMetric::Allocated => ShareMetric::AllocatedBytes,
};
let share_metric = match view {
ViewSpec::Languages if content.includes_code() => ShareMetric::CodeLines,
ViewSpec::Documents if content.includes_words() => ShareMetric::DocumentWords,
ViewSpec::Languages | ViewSpec::Documents if content.is_enabled() => ShareMetric::RawWords,
ViewSpec::Languages | ViewSpec::Documents | ViewSpec::Types | ViewSpec::Families => {
byte_share_metric
}
ViewSpec::List
| ViewSpec::Tree
| ViewSpec::Extensions
| ViewSpec::Files
| ViewSpec::Largest
| ViewSpec::Recent
| ViewSpec::Summary => {
unreachable!("only grouped views reach metric_summary")
}
};
let denominator = share_value(&total, share_metric);
total.share = MetricShare { numerator: denominator, denominator };
let mut rows = grouped.into_values().collect::<Vec<_>>();
for row in &mut rows {
row.share = MetricShare { numerator: share_value(row, share_metric), denominator };
}
sort_rows(
&mut rows,
query,
view,
|row, metric| match view {
ViewSpec::Languages | ViewSpec::Documents => share_value(row, share_metric),
_ => match metric {
SizeMetric::Apparent => row.bytes,
SizeMetric::Allocated => row.allocated,
},
},
|row| row.files,
|_| None,
|row| row.id.clone(),
);
let total_rows = truncate(&mut rows, query.limit_for(view));
MetricSummary {
group,
total,
rows,
total_rows,
share_metric,
words_per_page: query.words_per_page.max(1),
}
}
fn merge_coverage(total: &mut BTreeMap<CoverageReason, u64>, row: &BTreeMap<CoverageReason, u64>) {
for (reason, count) in row {
*total.entry(*reason).or_default() += count;
}
}
fn share_value(row: &MetricRow, metric: ShareMetric) -> u64 {
match metric {
ShareMetric::ApparentBytes => row.bytes,
ShareMetric::AllocatedBytes => row.allocated,
ShareMetric::CodeLines => row.metrics.code_lines.unwrap_or(0),
ShareMetric::DocumentWords => document_words(row).unwrap_or(0),
ShareMetric::RawWords => row.metrics.raw_words.unwrap_or(0),
}
}
pub fn document_words(row: &MetricRow) -> Option<u64> {
row.metrics.document_words
}
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub struct Pages {
pub words: u64,
pub words_per_page: u64,
}
pub fn pages(row: &MetricRow, words_per_page: u64) -> Option<Pages> {
document_words(row).map(|words| Pages { words, words_per_page: words_per_page.max(1) })
}
fn file_rows(
view: ViewSpec,
index: &Index,
query: &Query,
walked: Option<&Walked>,
unfiltered_rows: Option<&[FileRow]>,
) -> (Vec<FileRow>, usize) {
let mut rows = entry_rows(index, walked, unfiltered_rows).into_owned();
if view.files_only() {
rows.retain(|row| row.kind == EntryKind::File);
}
sort_rows(
&mut rows,
query,
view,
|row, metric| match metric {
SizeMetric::Apparent => row.bytes,
SizeMetric::Allocated => row.allocated,
},
|row| row.files.unwrap_or(1),
|row| Some(row.mtime_ns),
|row| row.path.to_string_lossy().into_owned(),
);
let total = truncate(&mut rows, query.limit_for(view));
(rows, total)
}
fn every_entry(index: &Index) -> Vec<FileRow> {
let observed = index.observes_controls();
let mut rows = Vec::new();
let mut stack: Vec<(EntryId, PathBuf)> = vec![(EntryId::ROOT, PathBuf::new())];
while let Some((id, path)) = stack.pop() {
let Some(children) = index.children_of(id) else {
continue;
};
let children: Vec<(PathBuf, EntryId)> =
children.map(|(name, child)| (path.join(name), child)).collect();
for (child_path, child) in children {
let (Some(kind), Some(attrs)) = (index.kind_of(child), index.attrs_of(child)) else {
continue;
};
rows.push(FileRow {
path: child_path.clone(),
kind,
bytes: attrs.size,
allocated: attrs.allocated,
mtime_ns: attrs.mtime_ns,
files: None,
dirs: None,
complete: None,
age_ns: None,
ignored: observed.then(|| index.ignored_bit_of(child).unwrap_or(false)),
});
if kind == EntryKind::Dir {
stack.push((child, child_path));
}
}
}
rows
}
fn tree_node(index: &Index, query: &Query, walked: Option<&Walked>) -> TreeNode {
let root_summary = match walked {
None => unfiltered_summary(index, EntryId::ROOT),
Some(walked) => walked.summary_of(EntryId::ROOT),
};
let mut root = TreeNode {
path: PathBuf::new(),
name: ".".to_string(),
kind: EntryKind::Dir,
bytes: root_summary.bytes,
allocated: root_summary.allocated,
files: root_summary.files,
dirs: root_summary.dirs,
ignored: root_summary.ignored,
newest_mtime_ns: root_summary.newest_mtime_ns,
children: Vec::new(),
truncated: false,
};
expand(index, query, walked, EntryId::ROOT, &PathBuf::new(), &mut root, 0);
root
}
fn expand(
index: &Index,
query: &Query,
walked: Option<&Walked>,
root_id: EntryId,
root_path: &Path,
node: &mut TreeNode,
start_depth: usize,
) {
struct Pending {
node: TreeNode,
id: EntryId,
depth: usize,
parent: Option<usize>,
}
let mut built = vec![Pending {
node: TreeNode {
path: root_path.to_path_buf(),
name: node.name.clone(),
kind: node.kind,
bytes: node.bytes,
allocated: node.allocated,
files: node.files,
dirs: node.dirs,
ignored: node.ignored,
newest_mtime_ns: node.newest_mtime_ns,
children: Vec::new(),
truncated: false,
},
id: root_id,
depth: start_depth,
parent: None,
}];
let mut cursor = 0;
while cursor < built.len() {
let (id, depth) = (built[cursor].id, built[cursor].depth);
let path = built[cursor].node.path.clone();
if !query.selection.depth.unwrap_or(ViewSpec::Tree.default_depth()).admits(depth) {
built[cursor].node.truncated = index.children_of(id).is_some_and(|mut children| {
children.any(|(_, child)| {
index.kind_of(child) == Some(EntryKind::Dir)
&& walked.is_none_or(|walked| walked.visible.contains(&child))
})
});
cursor += 1;
continue;
}
let mut rows = child_rows(index, query, walked, id, &path);
let kept = query
.selection
.limit
.unwrap_or(ViewSpec::Tree.default_limit())
.limit()
.unwrap_or(rows.len())
.min(rows.len());
built[cursor].node.truncated = kept < rows.len();
rows.truncate(kept);
for (child_node, child_id) in rows {
built.push(Pending {
node: child_node,
id: child_id,
depth: depth + 1,
parent: Some(cursor),
});
}
cursor += 1;
}
for position in (1..built.len()).rev() {
let child = built.remove(position);
let parent = child.parent.expect("only the root has no parent");
built[parent].node.children.insert(0, child.node);
}
let mut root = built.pop().expect("the root is always present");
node.children = std::mem::take(&mut root.node.children);
node.truncated = root.node.truncated;
}
fn child_rows(
index: &Index,
query: &Query,
walked: Option<&Walked>,
id: EntryId,
path: &Path,
) -> Vec<(TreeNode, EntryId)> {
let Some(children) = index.children_of(id) else {
return Vec::new();
};
let children: Vec<(PathBuf, EntryId)> =
children.map(|(name, child)| (path.join(name), child)).collect();
let mut rows: Vec<(TreeNode, EntryId)> = Vec::new();
for (child_path, child) in children {
let Some(kind) = index.kind_of(child) else {
continue;
};
if kind != EntryKind::Dir {
continue;
}
if walked.is_some_and(|walked| !walked.visible.contains(&child)) {
continue;
}
let summary = match walked {
None => unfiltered_summary(index, child),
Some(walked) => walked.summary_of(child),
};
let name = child_path
.file_name()
.map(|name| name.to_string_lossy().into_owned())
.unwrap_or_default();
rows.push((
TreeNode {
path: child_path,
name,
kind,
bytes: summary.bytes,
allocated: summary.allocated,
files: summary.files,
dirs: summary.dirs,
ignored: summary.ignored,
newest_mtime_ns: summary.newest_mtime_ns,
children: Vec::new(),
truncated: false,
},
child,
));
}
sort_rows_by(
&mut rows,
query,
ViewSpec::Tree,
|(row, _), metric| match metric {
SizeMetric::Apparent => row.bytes,
SizeMetric::Allocated => row.allocated,
},
|(row, _)| row.files,
|(row, _)| row.newest_mtime_ns,
|(row, _)| row.name.clone(),
);
rows
}
fn truncate<T>(rows: &mut Vec<T>, limit: Bound) -> usize {
let total = rows.len();
if let Some(limit) = limit.limit() {
rows.truncate(limit);
}
total
}
fn sort_rows<T>(
rows: &mut [T],
query: &Query,
view: ViewSpec,
size: impl Fn(&T, SizeMetric) -> u64,
count: impl Fn(&T) -> u64,
mtime: impl Fn(&T) -> Option<i64>,
name: impl Fn(&T) -> String,
) {
sort_rows_by(rows, query, view, size, count, mtime, name);
}
fn sort_rows_by<T>(
rows: &mut [T],
query: &Query,
view: ViewSpec,
size: impl Fn(&T, SizeMetric) -> u64,
count: impl Fn(&T) -> u64,
mtime: impl Fn(&T) -> Option<i64>,
name: impl Fn(&T) -> String,
) {
let key = query.selection.sort.unwrap_or_else(|| view.default_sort());
let metric = query.selection.size;
rows.sort_by(|left, right| {
let ordering = match key {
SortKey::Size => size(right, metric).cmp(&size(left, metric)),
SortKey::Count => count(right).cmp(&count(left)),
SortKey::Mtime => mtime(right).cmp(&mtime(left)),
SortKey::Name => name(left).cmp(&name(right)),
};
ordering.then_with(|| name(left).cmp(&name(right)))
});
if query.selection.reverse {
rows.reverse();
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::engine_contract::{Attrs, Observation, Op};
use crate::query::query_glob::Pattern;
use crate::query::query_selection::ModifiedWindow;
use std::fs;
use std::time::{Duration, UNIX_EPOCH};
fn attrs(size: u64, mtime_ns: i64) -> Attrs {
Attrs {
size,
allocated: size.div_ceil(512) * 512,
mtime_ns,
ctime_ns: mtime_ns,
inode: size.wrapping_mul(31).wrapping_add(mtime_ns.unsigned_abs()),
dev: 1,
}
}
fn upsert(path: &str, kind: EntryKind, attrs: Attrs) -> Op {
Op::Upsert { path: PathBuf::from(path), kind, attrs }
}
fn sample() -> Index {
let mut index = Index::new("/root");
index
.apply(&Observation::new(vec![
upsert("src", EntryKind::Dir, Attrs::default()),
upsert("src/main.rs", EntryKind::File, attrs(100, 10)),
upsert("src/lib.rs", EntryKind::File, attrs(200, 20)),
upsert("src/deep", EntryKind::Dir, Attrs::default()),
upsert("src/deep/nested.rs", EntryKind::File, attrs(50, 40)),
upsert("docs", EntryKind::Dir, Attrs::default()),
upsert("docs/guide.md", EntryKind::File, attrs(300, 30)),
upsert("notes.txt", EntryKind::File, attrs(7, 5)),
]))
.expect("apply");
index
}
#[test]
fn ages_use_one_signed_reference_and_unrepresentable_clocks_are_unknown() {
let mut index = sample();
index.apply_ok(&Observation::new(vec![
upsert("past", EntryKind::File, attrs(1, -10)),
upsert("future", EntryKind::File, attrs(1, i64::MAX)),
]));
let query = query(
&[ViewSpec::Files],
Selection { include: vec![pattern("past"), pattern("future")], ..Selection::default() },
);
let mut request = Request::new(Basis::held_by(&index), query, UNIX_EPOCH);
let answer = report(&index, &request, UNIX_EPOCH).expect("report");
let rows = files_of(&answer);
assert_eq!(answer.age_reference_ns, Some(0));
assert_eq!(
rows.iter().map(|r| r.age_ns).collect::<Vec<_>>(),
[Some(-i128::from(i64::MAX)), Some(10)]
);
request.now = UNIX_EPOCH + Duration::from_secs(10_000_000_000);
let answer = report(&index, &request, UNIX_EPOCH)
.expect("out-of-range reference is representable as unknown age");
assert_eq!(answer.age_reference_ns, None);
assert!(files_of(&answer).iter().all(|row| row.age_ns.is_none()));
}
#[test]
fn matching_a_directory_selects_its_subtree_once() {
let index = sample();
let selection = Selection {
include: vec![pattern("src"), pattern("deep")],
kinds: vec![EntryKind::Dir],
size: SizeMetric::Apparent,
min_size: Some(40),
..Selection::default()
};
let report = run(
&index,
&query(&[ViewSpec::Files, ViewSpec::Summary, ViewSpec::Extensions], selection),
);
let rows = files_of(&report);
assert_eq!(rows.len(), 2);
assert_eq!((rows[0].bytes, rows[0].mtime_ns), (350, 40));
let Section::Summary(summary) = &report.sections[1] else { panic!("summary") };
assert_eq!((summary.files, summary.dirs, summary.bytes), (3, 2, 350));
let Section::Extensions { rows, .. } = &report.sections[2] else { panic!("extensions") };
assert_eq!((rows[0].files, rows[0].bytes), (3, 350));
}
#[test]
fn subtree_predicates_include_directory_and_symlink_activity_but_only_file_bytes() {
let mut index = sample();
index
.apply(&Observation::new(vec![
upsert("src", EntryKind::Dir, attrs(9999, 45)),
upsert("src/empty", EntryKind::Dir, attrs(8888, 60)),
upsert("src/link", EntryKind::Symlink, attrs(7777, 70)),
upsert("empty", EntryKind::Dir, attrs(6666, -10)),
]))
.expect("apply");
let base = Selection {
include: vec![pattern("src"), pattern("empty")],
size: SizeMetric::Apparent,
..Selection::default()
};
let nested = PathBuf::from("src").join("empty").to_string_lossy().into_owned();
let rows = files_of(&run(&index, &query(&[ViewSpec::Files], base.clone())));
assert_eq!(
rows.iter()
.map(|r| (r.path.to_string_lossy().into_owned(), r.bytes, r.mtime_ns))
.collect::<Vec<_>>(),
[("empty".into(), 0, -10), ("src".into(), 350, 70), (nested.clone(), 0, 60)]
);
for (before, since, expected) in [
(70, 0, vec![nested.clone()]),
(71, 70, vec!["src".to_owned()]),
(0, -10, vec!["empty".to_owned()]),
] {
let selection = Selection {
modified: ModifiedWindow { before: Some(before), since: Some(since) },
..base.clone()
};
let rows = files_of(&run(&index, &query(&[ViewSpec::Files], selection)));
assert_eq!(
rows.iter().map(|r| r.path.to_string_lossy().into_owned()).collect::<Vec<_>>(),
expected
);
}
for (size, minimum, expected) in [
(SizeMetric::Apparent, 350, 1),
(SizeMetric::Apparent, 351, 0),
(SizeMetric::Allocated, 1536, 1),
(SizeMetric::Allocated, 1537, 0),
] {
let selection = Selection { size, min_size: Some(minimum), ..base.clone() };
assert_eq!(
files_of(&run(&index, &query(&[ViewSpec::Files], selection))).len(),
expected
);
}
}
#[test]
fn exclusions_apply_before_subtree_bounds_and_selected_ancestor_coverage() {
let index = classified_sample();
for (ignored, exclude, bytes, newest) in [
(IgnoredEntries::Include, vec![], 325, 70),
(IgnoredEntries::Exclude, vec![], 300, 20),
(IgnoredEntries::Include, vec![pattern("*.log")], 300, 20),
(IgnoredEntries::Include, vec![pattern("lib.rs")], 125, 70),
] {
let selection = Selection {
include: vec![pattern("src")],
kinds: vec![EntryKind::Dir],
ignored,
exclude,
size: SizeMetric::Apparent,
..Selection::default()
};
let report = run(
&index,
&query(&[ViewSpec::Files, ViewSpec::Summary, ViewSpec::Types], selection),
);
let row = &files_of(&report)[0];
assert_eq!((row.bytes, row.mtime_ns), (bytes, newest));
let Section::Summary(summary) = &report.sections[1] else { panic!("summary") };
assert_eq!(summary.bytes, bytes);
let Section::Metrics { summary, .. } = &report.sections[2] else { panic!("types") };
assert_eq!(summary.rows.iter().map(|r| r.bytes).sum::<u64>(), bytes);
}
let selection = Selection {
include: vec![pattern("build")],
exclude: vec![pattern("cache")],
kinds: vec![EntryKind::Dir],
..Selection::default()
};
let report = run(&index, &query(&[ViewSpec::Files, ViewSpec::Summary], selection));
assert_eq!(files_of(&report)[0].bytes, 0);
let Section::Summary(summary) = &report.sections[1] else { panic!("summary") };
assert_eq!((summary.files, summary.dirs, summary.bytes), (0, 1, 0));
let only = Selection {
include: vec![pattern("cache")],
kinds: vec![EntryKind::Dir],
ignored: IgnoredEntries::Only,
..Selection::default()
};
assert_eq!(files_of(&run(&index, &query(&[ViewSpec::Files], only)))[0].bytes, 1000);
}
#[test]
fn an_incomplete_subtree_reports_lower_bounds_and_matches_no_time_bound() {
let mut index = Index::new_with_scope(
"/root",
crate::ScanScope { max_depth: Some(2), ..crate::ScanScope::default() },
);
index.apply_ok(&Observation::new(vec![
upsert("env", EntryKind::Dir, attrs(0, 5)),
upsert("env/lib", EntryKind::Dir, attrs(0, 7)),
upsert("env/a.bin", EntryKind::File, attrs(100, 40)),
upsert("docs", EntryKind::Dir, attrs(0, 5)),
upsert("docs/guide.md", EntryKind::File, attrs(30, 50)),
]));
let directories = |selection: Selection| {
let selection =
Selection { kinds: vec![EntryKind::Dir], size: SizeMetric::Apparent, ..selection };
files_of(&run(&index, &flat(selection)))
.into_iter()
.map(|row| {
(row.path.to_string_lossy().into_owned(), row.complete, row.bytes, row.age_ns)
})
.collect::<Vec<_>>()
};
let env_lib = Path::new("env").join("lib").to_string_lossy().into_owned();
assert_eq!(
directories(Selection::default()),
vec![
("env".to_string(), Some(false), 100, None),
("docs".to_string(), Some(true), 30, Some(-50)),
(env_lib, Some(false), 0, None),
],
"sizes are lower bounds and the age is unknown below the boundary"
);
for modified in [
ModifiedWindow { since: None, before: Some(100) },
ModifiedWindow { since: Some(0), before: None },
] {
assert_eq!(
directories(Selection { modified, ..Selection::default() }),
vec![("docs".to_string(), Some(true), 30, Some(-50))],
"an unknown age satisfies no bound, not even one its lower bound would prove"
);
}
assert_eq!(
directories(Selection { min_size: Some(100), ..Selection::default() }),
vec![("env".to_string(), Some(false), 100, None)],
"a lower bound at or above the minimum proves the true size is too"
);
assert!(directories(Selection { min_size: Some(101), ..Selection::default() }).is_empty());
let files = files_of(&run(
&index,
&flat(Selection {
kinds: vec![EntryKind::File],
modified: ModifiedWindow { since: Some(45), before: None },
..Selection::default()
}),
));
assert_eq!(files.len(), 1);
assert_eq!((files[0].complete, files[0].age_ns), (None, Some(-50)));
}
#[test]
fn an_unscoped_partial_marker_marks_every_directory_row_incomplete() {
let directories = |index: &Index| {
files_of(&run(
index,
&flat(Selection { kinds: vec![EntryKind::Dir], ..Selection::default() }),
))
};
let mut partial = sample();
partial.set_initial_freshness(false);
let rows = directories(&partial);
assert_eq!(rows.len(), 3);
assert!(rows.iter().all(|row| row.complete == Some(false) && row.age_ns.is_none()));
let mut complete = sample();
complete.set_initial_freshness(true);
let rows = directories(&complete);
assert_eq!(rows.len(), 3);
assert!(rows.iter().all(|row| row.complete == Some(true) && row.age_ns.is_some()));
}
#[test]
fn an_opened_root_marks_a_directory_complete_only_once_discovery_listed_it() {
let handle = crate::index::IndexHandle::new(Index::new("/root"));
handle
.transition_discovery(crate::index::DiscoveryTransition::Begin)
.expect("begin discovery");
handle
.apply(&Observation::new(vec![
upsert("known", EntryKind::Dir, attrs(0, 5)),
upsert("pending", EntryKind::Dir, attrs(0, 5)),
]))
.expect("seed directories");
handle
.apply_discovery(
&Observation::new(Vec::new()),
crate::index::DiscoveryCommit {
directory_complete: Some(PathBuf::from("known")),
transition: None,
},
)
.expect("list one directory");
let completeness = handle
.read_with(|index| {
files_of(&run(
index,
&flat(Selection { kinds: vec![EntryKind::Dir], ..Selection::default() }),
))
.into_iter()
.map(|row| (row.path.to_string_lossy().into_owned(), row.complete))
.collect::<BTreeMap<_, _>>()
})
.expect("read");
assert_eq!(
completeness,
BTreeMap::from([
("known".to_string(), Some(true)),
("pending".to_string(), Some(false))
])
);
}
#[test]
fn a_filtered_tree_keeps_empty_matches_and_only_folds_visible_directories() {
let mut index = sample();
index
.apply(&Observation::new(vec![upsert("src/empty", EntryKind::Dir, Attrs::default())]))
.expect("apply");
let selection = Selection {
include: vec![pattern("empty")],
depth: Some(Bound::All),
..Selection::default()
};
let root = tree_of(&run(&index, &query(&[ViewSpec::Tree], selection)));
assert_eq!(root.children.len(), 1);
assert_eq!(root.children[0].name, "src");
assert_eq!(root.children[0].children[0].name, "empty");
let selection = Selection {
include: vec![pattern("notes.txt")],
depth: Some(Bound::Limit(0)),
..Selection::default()
};
let root = tree_of(&run(&index, &query(&[ViewSpec::Tree], selection)));
assert!(!root.truncated);
assert_eq!(root.bytes, 7);
}
#[test]
fn the_undocumented_docs_view_alias_is_rejected() {
assert_eq!(
ViewSpec::parse("documents").expect("the canonical view name parses"),
ViewSpec::Documents
);
assert_eq!(
ViewSpec::parse("docs").expect_err("an unreleased alias must not become a contract"),
format!("expected one of {}", ViewSpec::vocabulary())
);
}
fn generated_at() -> std::time::SystemTime {
UNIX_EPOCH + Duration::from_secs(1_001)
}
fn run(index: &Index, query: &Query) -> Report {
report(index, &crate::test_support::read_of(index, query.clone()), generated_at())
.expect("the query is answerable over this index")
}
fn query(views: &[ViewSpec], selection: Selection) -> Query {
Query { selection, views: views.to_vec(), ..Query::default() }
}
fn flat(selection: Selection) -> Query {
Query {
selection,
views: vec![ViewSpec::List],
format: crate::report_format::Format::Paths,
..Query::default()
}
}
fn pattern(source: &str) -> Pattern {
Pattern::parse(source).expect("pattern compiles")
}
fn summary_of(report: &Report) -> SummaryRow {
match report.sections.first().expect("a section") {
Section::Summary(row) => *row,
other => panic!("expected a summary, got {other:?}"),
}
}
fn files_of(report: &Report) -> Vec<FileRow> {
match report.sections.first().expect("a section") {
Section::Files { rows, .. } => rows.clone(),
other => panic!("expected files, got {other:?}"),
}
}
fn types_of(report: &Report) -> Vec<TypeRow> {
match report.sections.first().expect("a section") {
Section::Extensions { rows, .. } => rows.clone(),
other => panic!("expected types, got {other:?}"),
}
}
fn tree_of(report: &Report) -> TreeNode {
match report.sections.first().expect("a section") {
Section::Tree { root: node, .. } => node.clone(),
other => panic!("expected a tree, got {other:?}"),
}
}
#[test]
fn an_unfiltered_summary_matches_the_precomputed_rollup() {
let index = sample();
let row = summary_of(&run(&index, &query(&[ViewSpec::Summary], Selection::default())));
assert_eq!(row.files, 5);
assert_eq!(row.dirs, 3);
assert_eq!(row.bytes, 657);
assert_eq!(row.newest_mtime_ns, Some(40));
}
#[test]
fn the_two_tiers_agree_on_the_same_question() {
let index = sample();
let fast = summary_of(&run(&index, &query(&[ViewSpec::Summary], Selection::default())));
let admits_everything = Selection { min_size: Some(0), ..Selection::default() };
assert!(!admits_everything.is_unfiltered());
let slow = summary_of(&run(&index, &query(&[ViewSpec::Summary], admits_everything)));
assert_eq!(
(fast.files, fast.dirs, fast.bytes, fast.allocated, fast.newest_mtime_ns),
(slow.files, slow.dirs, slow.bytes, slow.allocated, slow.newest_mtime_ns)
);
}
#[test]
fn extension_rows_account_for_every_file_in_both_tiers() {
let mut index = sample();
index
.apply(&Observation::new(vec![
upsert("Makefile", EntryKind::File, attrs(28, 50)),
upsert(".gitignore", EntryKind::File, attrs(11, 51)),
]))
.expect("apply");
for selection in [
Selection::default(),
Selection { min_size: Some(0), ..Selection::default() },
Selection { kinds: vec![EntryKind::File], ..Selection::default() },
] {
let rows = types_of(&run(&index, &query(&[ViewSpec::Extensions], selection.clone())));
let summary = summary_of(&run(&index, &query(&[ViewSpec::Summary], selection.clone())));
assert_eq!(
rows.iter().map(|row| row.bytes).sum::<u64>(),
summary.bytes,
"bytes unaccounted for under {selection:?}: {rows:?}"
);
assert_eq!(
rows.iter().map(|row| row.files).sum::<u64>(),
summary.files,
"files unaccounted for under {selection:?}: {rows:?}"
);
}
}
#[test]
fn names_without_an_extension_share_one_bucket() {
let mut index = sample();
index
.apply(&Observation::new(vec![
upsert("Makefile", EntryKind::File, attrs(28, 50)),
upsert(".gitignore", EntryKind::File, attrs(11, 51)),
]))
.expect("apply");
let rows = types_of(&run(&index, &query(&[ViewSpec::Extensions], Selection::default())));
let bucket = rows
.iter()
.find(|row| row.extension == crate::classify::NO_EXTENSION)
.expect("a bucket for the extension-less names");
assert_eq!(bucket.files, 2);
assert_eq!(bucket.bytes, 39);
assert!(rows.iter().any(|row| row.extension == ".rs"), "{rows:?}");
}
#[test]
fn a_summary_counts_the_union_of_listed_entries_and_directory_contents() {
let index = sample();
for selection in [
Selection { kinds: vec![EntryKind::File], ..Selection::default() },
Selection { kinds: vec![EntryKind::Dir], ..Selection::default() },
Selection { include: vec![pattern("*.rs")], ..Selection::default() },
Selection { exclude: vec![pattern("docs")], ..Selection::default() },
Selection { min_size: Some(1_000_000), ..Selection::default() },
Selection { min_size: Some(0), ..Selection::default() },
] {
let summary = summary_of(&run(&index, &query(&[ViewSpec::Summary], selection.clone())));
let listed = files_of(&run(&index, &query(&[ViewSpec::Files], selection.clone())));
let dirs = listed.iter().filter(|row| row.kind == EntryKind::Dir).count() as u64;
let files = every_entry(&index)
.iter()
.filter(|entry| {
entry.kind == EntryKind::File
&& listed.iter().any(|row| {
entry.path == row.path
|| (row.kind == EntryKind::Dir && entry.path.starts_with(&row.path))
})
})
.count() as u64;
assert_eq!(summary.dirs, dirs, "directory counts disagree under {selection:?}");
assert_eq!(summary.files, files, "file counts disagree under {selection:?}");
}
}
#[test]
fn a_rejected_directory_is_still_descended_into() {
let index = sample();
let selection = Selection { kinds: vec![EntryKind::File], ..Selection::default() };
let row = summary_of(&run(&index, &query(&[ViewSpec::Summary], selection)));
assert_eq!(row.dirs, 0, "no directory was admitted");
assert_eq!(row.files, 5, "including src/deep/nested.rs, two levels down");
assert_eq!(row.bytes, 657, "and its bytes");
}
#[test]
fn nested_directory_counts_roll_up_through_every_level() {
let index = sample();
let selection = Selection { kinds: vec![EntryKind::Dir], ..Selection::default() };
let root = tree_of(&run(&index, &query(&[ViewSpec::Tree], selection)));
assert_eq!(root.dirs, 3, "src, src/deep, and docs");
assert_eq!(root.files, 4, "matching directories cover their regular files");
let src = root.children.iter().find(|node| node.name == "src").expect("src");
assert_eq!(src.dirs, 1, "src/deep, counted for src as well as for the root");
}
#[test]
fn selection_narrows_a_summary_to_what_it_admits() {
let index = sample();
let selection = Selection { include: vec![pattern("*.rs")], ..Selection::default() };
let row = summary_of(&run(&index, &query(&[ViewSpec::Summary], selection)));
assert_eq!(row.files, 3, "three .rs files");
assert_eq!(row.bytes, 350);
}
#[test]
fn a_files_view_lists_matching_entries_in_name_order_by_default() {
let index = sample();
let selection = Selection { include: vec![pattern("*.rs")], ..Selection::default() };
let rows = files_of(&run(&index, &query(&[ViewSpec::Files], selection)));
let paths: Vec<PathBuf> = rows.iter().map(|row| row.path.clone()).collect();
let expected: Vec<PathBuf> = [["src", "deep", "nested.rs"].iter().collect::<PathBuf>()]
.into_iter()
.chain([["src", "lib.rs"].iter().collect::<PathBuf>()])
.chain([["src", "main.rs"].iter().collect::<PathBuf>()])
.collect();
assert_eq!(paths, expected);
}
#[test]
fn sorting_and_limiting_compose_without_a_dedicated_view() {
let index = sample();
let selection = Selection {
kinds: vec![EntryKind::File],
sort: Some(SortKey::Size),
limit: Some(Bound::Limit(2)),
size: SizeMetric::Apparent,
..Selection::default()
};
let rows = files_of(&run(&index, &query(&[ViewSpec::Files], selection)));
assert_eq!(rows.len(), 2);
assert_eq!(rows[0].bytes, 300, "largest first");
assert_eq!(rows[1].bytes, 200);
}
#[test]
fn reverse_flips_whatever_order_is_in_effect() {
let index = sample();
let selection = Selection {
kinds: vec![EntryKind::File],
sort: Some(SortKey::Size),
reverse: true,
size: SizeMetric::Apparent,
..Selection::default()
};
let rows = files_of(&run(&index, &query(&[ViewSpec::Files], selection)));
assert_eq!(rows[0].bytes, 7, "smallest first once reversed");
}
#[test]
fn a_modified_window_selects_by_time() {
let index = sample();
let selection = Selection {
kinds: vec![EntryKind::File],
modified: ModifiedWindow { since: Some(20), before: Some(40) },
sort: Some(SortKey::Mtime),
..Selection::default()
};
let rows = files_of(&run(&index, &query(&[ViewSpec::Files], selection)));
let mut times: Vec<i64> = rows.iter().map(|row| row.mtime_ns).collect();
times.sort_unstable();
assert_eq!(times, vec![20, 30], "inclusive start, exclusive end");
}
#[test]
fn a_types_view_reports_both_size_metrics_per_extension() {
let index = sample();
let rows = types_of(&run(&index, &query(&[ViewSpec::Extensions], Selection::default())));
let rs = rows.iter().find(|row| row.extension == ".rs").expect(".rs present");
assert_eq!((rs.files, rs.bytes), (3, 350));
assert_eq!(rs.allocated, 1536, "three files, one 512-byte block each");
let order: Vec<&str> = rows.iter().map(|row| row.extension.as_str()).collect();
assert_eq!(order, vec![".rs", ".md", ".txt"]);
}
#[test]
fn a_tree_view_reports_directories_with_their_subtree_totals() {
let index = sample();
let tree = tree_of(&run(&index, &query(&[ViewSpec::Tree], Selection::default())));
assert_eq!(tree.name, ".");
assert_eq!(tree.bytes, 657);
let names: Vec<&str> = tree.children.iter().map(|child| child.name.as_str()).collect();
assert_eq!(names, vec!["src", "docs"]);
let src = &tree.children[0];
assert_eq!(src.bytes, 350);
let nested: Vec<&str> = src.children.iter().map(|child| child.name.as_str()).collect();
assert_eq!(nested, vec!["deep"]);
}
#[test]
fn depth_zero_keeps_dus_meaning_of_root_totals_only() {
let index = sample();
let selection = Selection { depth: Some(Bound::Limit(0)), ..Selection::default() };
let tree = tree_of(&run(&index, &query(&[ViewSpec::Tree], selection)));
assert_eq!(tree.bytes, 657, "totals still cover the whole tree");
assert!(tree.children.is_empty(), "but nothing below the root is listed");
assert!(tree.truncated, "and the report says so rather than implying emptiness");
}
#[test]
fn a_dropped_view_is_named_on_the_report_rather_than_by_one_surface() {
let (selected, omitted) = ViewSpec::resolve(Some("full"), AnalysisSet::NONE, "view")
.expect("full resolves without analyzers");
assert!(!omitted.is_empty(), "documents needs analysis and must be dropped");
let query = Query { views: selected, omitted_views: omitted, ..Query::default() };
let notes = display_notes(&query, &ControlCoverage::NotObserved);
assert_eq!(notes.len(), 1, "{notes:?}");
assert!(notes[0].contains("omitted documents"), "{notes:?}");
let (selected, omitted) = ViewSpec::resolve(Some("full"), AnalysisSet::ALL, "view")
.expect("full resolves with analyzers");
assert!(omitted.is_empty(), "every view is answerable with analysis enabled");
let query = Query { views: selected, omitted_views: omitted, ..Query::default() };
assert!(display_notes(&query, &ControlCoverage::NotObserved).is_empty());
}
#[test]
fn the_refused_controls_note_bounds_its_list_and_matches_its_remedy_to_the_reasons() {
use crate::control::{
ControlLimits, ControlObservation, ControlRefusalReason, RefusedControl,
};
let refused = |directory: &str, reason| RefusedControl {
path: Path::new(directory).join(".gitignore"),
reason,
};
let note = |limits, refusals: Vec<RefusedControl>, count: u64| {
let coverage = ControlCoverage::Observed(ControlObservation {
limits,
applied: 7,
refused: count,
refusals,
});
refused_controls_note(&coverage, &AxisNames::FLAGS).expect("a refusal is noted")
};
let defaults = ControlLimits::default();
let (budget, line_limit) = (ControlRefusalReason::Budget, ControlRefusalReason::LineLimit);
assert_eq!(
note(defaults, vec![refused("", budget), refused("pkg/a", budget)], 2),
"note: 2 .gitignore files not applied (2 over the 4.0 MiB ignore-rule budget), so \
ignored shares under ., pkg/a are not exact; sizes are. To apply them, raise \
--gitignore-budget above 4.0 MiB, or set it to all"
);
assert_eq!(
note(defaults, vec![refused("vendor", line_limit)], 1),
"note: 1 .gitignore file not applied (1 with a line over the 16 KiB line limit), so \
ignored shares under vendor are not exact; sizes are. To apply them, raise \
--gitignore-line-limit above 16 KiB, or set it to all"
);
assert_eq!(
note(defaults, vec![refused("a", budget), refused("b", line_limit)], 2),
"note: 2 .gitignore files not applied (1 over the 4.0 MiB ignore-rule budget, 1 with \
a line over the 16 KiB line limit), so ignored shares under a, b are not exact; \
sizes are. To apply them, raise --gitignore-budget above 4.0 MiB and \
--gitignore-line-limit above 16 KiB, or set them to all"
);
let listed: Vec<_> = (0..crate::MAX_RETAINED_ISSUES)
.map(|index| refused(&format!("d{index:02}"), budget))
.collect();
assert_eq!(
note(defaults, listed.clone(), 1_000),
"note: 1,000 .gitignore files not applied (over the 4.0 MiB ignore-rule budget or \
with a line over the 16 KiB line limit), so ignored shares under d00, d01, d02, \
d03, d04, 995 more are not exact; sizes are. To apply them, raise \
--gitignore-budget above 4.0 MiB and --gitignore-line-limit above 16 KiB, or set \
them to all"
);
assert_eq!(
note(ControlLimits { line_limit: None, ..defaults }, listed, 1_000),
"note: 1,000 .gitignore files not applied (over the 4.0 MiB ignore-rule budget), so \
ignored shares under d00, d01, d02, d03, d04, 995 more are not exact; sizes are. \
To apply them, raise --gitignore-budget above 4.0 MiB, or set it to all"
);
assert_eq!(
note(ControlLimits { budget: None, ..defaults }, vec![refused("a", budget)], 1),
"note: 1 .gitignore file not applied (1 over the ignore-rule budget), so ignored \
shares under a are not exact; sizes are."
);
let complete = ControlCoverage::Observed(ControlObservation {
limits: defaults,
applied: 3,
refused: 0,
refusals: Vec::new(),
});
assert_eq!(refused_controls_note(&complete, &AxisNames::FLAGS), None);
assert_eq!(refused_controls_note(&ControlCoverage::NotObserved, &AxisNames::FLAGS), None);
}
#[test]
fn a_diagnostic_names_the_axes_the_requesting_surface_uses() {
let (selected, omitted) = ViewSpec::resolve(Some("full"), AnalysisSet::NONE, "view")
.expect("full resolves without analyzers");
for (axes, mine, theirs) in [
(&AxisNames::FLAGS, "--analyze", "analyze"),
(&AxisNames::FIELDS, "analyze", "--analyze"),
] {
let query = Query {
views: selected.clone(),
omitted_views: omitted.clone(),
axes,
..Query::default()
};
let note = display_notes(&query, &ControlCoverage::NotObserved).remove(0);
assert!(note.contains(&format!("add {mine} ")), "{note} must name {mine}");
assert!(!note.contains(&format!("add {theirs} ")), "{note} must not name {theirs}");
}
for (axes, view, analyze) in
[(&AxisNames::FLAGS, "--view", "--analyze"), (&AxisNames::FIELDS, "view", "analyze")]
{
let error =
crate::query::RequestError::ViewNeedsContent(ViewSpec::Documents).message(axes);
assert!(error.starts_with(&format!("{view} documents")), "{error}");
assert!(error.contains(&format!("add {analyze} ")), "{error}");
let theirs = if analyze == "--analyze" { "analyze" } else { "--analyze" };
assert!(!error.contains(&format!("add {theirs} ")), "{error}");
}
}
#[test]
fn the_default_vocabulary_is_the_librarys_own() {
assert_eq!(*Query::default().axes, AxisNames::FIELDS);
}
#[test]
fn a_depth_bound_marks_only_hidden_directory_rows_as_truncated() {
let index = sample();
let selection = Selection { depth: Some(Bound::Limit(1)), ..Selection::default() };
let tree = tree_of(&run(&index, &query(&[ViewSpec::Tree], selection)));
let src = tree.children.iter().find(|child| child.name == "src").expect("src");
assert!(src.truncated, "src with a hidden directory child is truncated");
let docs = tree.children.iter().find(|child| child.name == "docs").expect("docs");
assert!(docs.children.is_empty());
assert!(
!docs.truncated,
"file children contribute to a directory row; they are not hidden tree rows"
);
}
#[test]
fn a_tree_limit_bounds_entries_per_directory_and_marks_truncation() {
let index = sample();
let selection = Selection { limit: Some(Bound::Limit(1)), ..Selection::default() };
let tree = tree_of(&run(&index, &query(&[ViewSpec::Tree], selection)));
assert_eq!(tree.children.len(), 1);
assert!(tree.truncated);
}
#[test]
fn requesting_more_views_never_changes_another_views_answer() {
let index = sample();
let alone = types_of(&run(&index, &query(&[ViewSpec::Extensions], Selection::default())));
let together = run(
&index,
&query(
&[ViewSpec::Extensions, ViewSpec::Tree, ViewSpec::Summary],
Selection::default(),
),
);
let with_others = match &together.sections[0] {
Section::Extensions { rows, .. } => rows.clone(),
other => panic!("expected types first, got {other:?}"),
};
assert_eq!(alone.len(), with_others.len());
for (left, right) in alone.iter().zip(with_others.iter()) {
assert_eq!(
(&left.extension, left.files, left.bytes),
(&right.extension, right.files, right.bytes)
);
}
assert_eq!(together.sections.len(), 3, "one section per view, in request order");
assert_eq!(together.sections[1].view(), ViewSpec::Tree);
assert_eq!(together.sections[2].view(), ViewSpec::Summary);
}
#[test]
fn analyzed_unfiltered_views_together_match_independent_answers_and_each_view_alone() {
const RUST: &str = "fn main() {\n println!(\"hi\");\n}\n";
const MARKDOWN: &str = "# Guide\n\nA small useful guide.\n";
const TEXT: &str = "plain notes here\n";
let root = tempfile::tempdir().expect("root");
fs::create_dir_all(root.path().join("src")).expect("src");
fs::create_dir_all(root.path().join("docs")).expect("docs");
fs::write(root.path().join("src/main.rs"), RUST).expect("rust");
fs::write(root.path().join("docs/guide.md"), MARKDOWN).expect("markdown");
fs::write(root.path().join("notes.txt"), TEXT).expect("text");
for (path, seconds) in [("src/main.rs", 10), ("notes.txt", 20), ("docs/guide.md", 30)] {
fs::File::options()
.write(true)
.open(root.path().join(path))
.expect("open for timestamp")
.set_times(
fs::FileTimes::new().set_modified(UNIX_EPOCH + Duration::from_secs(seconds)),
)
.expect("set timestamp");
}
let (mut index, _) = crate::scan::scan_into_index(
root.path(),
&crate::ScanConfig { read_controls: false, ..crate::ScanConfig::default() },
)
.expect("scan");
crate::content::analyze_index(
&mut index,
crate::content::AnalysisRequest {
profile: AnalysisSet::ALL,
..crate::content::AnalysisRequest::default()
},
);
let views = [
ViewSpec::Types,
ViewSpec::Families,
ViewSpec::Languages,
ViewSpec::Documents,
ViewSpec::Files,
ViewSpec::Largest,
ViewSpec::Recent,
ViewSpec::Summary,
ViewSpec::Tree,
ViewSpec::Extensions,
];
let selection = Selection { size: SizeMetric::Apparent, ..Selection::default() };
let together = run(&index, &query(&views, selection.clone()));
for (i, view) in views.iter().enumerate() {
let alone = run(&index, &query(&[*view], selection.clone()));
assert_eq!(
format!("{:?}", together.sections[i]),
format!("{:?}", alone.sections[0]),
"{view:?} changed when requested with the other views"
);
}
for (at, view, files) in [(0, ViewSpec::Types, 3), (1, ViewSpec::Families, 3)] {
let Section::Metrics { view: actual, summary } = &together.sections[at] else {
panic!("expected {view:?} metrics")
};
assert_eq!(*actual, view);
assert_eq!(summary.total.files, files);
assert_eq!(summary.total.analyzed_files, files);
}
let Section::Metrics { summary: languages, .. } = &together.sections[2] else {
panic!("languages")
};
assert_eq!(languages.total.files, 1);
assert_eq!(languages.total.metrics.code_lines, Some(3));
let Section::Metrics { summary: documents, .. } = &together.sections[3] else {
panic!("documents")
};
assert_eq!(documents.total.files, 2);
assert_eq!(documents.total.analyzed_files, 2);
assert_eq!(documents.total.document_metric_files, 2);
assert_eq!(documents.total.document_raw_words, 8);
assert_eq!(documents.total.document_word_stats.logical_words(), 8);
assert_eq!(documents.total.share, MetricShare { numerator: 8, denominator: 8 });
let Section::Files { rows: files, total, .. } = &together.sections[4] else {
panic!("files")
};
assert_eq!(*total, 5);
assert_eq!(
files.iter().map(|row| row.path.as_path()).collect::<Vec<_>>(),
["docs", "docs/guide.md", "notes.txt", "src", "src/main.rs"].map(Path::new).to_vec()
);
let Section::Files { rows: largest, total, .. } = &together.sections[5] else {
panic!("largest")
};
assert_eq!(*total, 3);
assert_eq!(
largest.iter().map(|row| row.path.as_path()).collect::<Vec<_>>(),
["src/main.rs", "docs/guide.md", "notes.txt"].map(Path::new).to_vec()
);
let Section::Files { rows: recent, total, .. } = &together.sections[6] else {
panic!("recent")
};
assert_eq!(*total, 3);
assert_eq!(
recent.iter().map(|row| row.path.as_path()).collect::<Vec<_>>(),
["docs/guide.md", "notes.txt", "src/main.rs"].map(Path::new).to_vec()
);
let Section::Summary(summary) = &together.sections[7] else { panic!("summary") };
assert_eq!((summary.files, summary.dirs), (3, 2));
assert_eq!(
summary.bytes,
u64::try_from(RUST.len() + MARKDOWN.len() + TEXT.len()).expect("fixture bytes")
);
let Section::Tree { root: tree, .. } = &together.sections[8] else { panic!("tree") };
assert_eq!((tree.files, tree.dirs), (3, 2));
let Section::Extensions { rows, total } = &together.sections[9] else {
panic!("extensions")
};
assert_eq!((*total, rows.len()), (3, 3));
}
#[test]
fn a_report_derives_provenance_from_its_index() {
let index = sample();
let report = run(&index, &query(&[ViewSpec::Summary], Selection::default()));
assert_eq!(report.provenance.source, ReportSource::ColdScan);
assert!(report.status.complete);
assert!(report.provenance.scan_started_at.is_some());
assert_eq!(report.provenance.generated_at, generated_at());
assert_eq!(report.root, Path::new("/root"));
}
#[test]
fn reporting_is_pure_and_repeatable() {
let index = sample();
let request = query(&[ViewSpec::Tree, ViewSpec::Extensions], Selection::default());
assert_eq!(format!("{:?}", run(&index, &request)), format!("{:?}", run(&index, &request)));
}
#[test]
fn metadata_grouping_views_use_the_generic_metric_projection() {
let index = sample();
let apparent = Selection { size: SizeMetric::Apparent, ..Selection::default() };
let report = run(
&index,
&query(&[ViewSpec::Types, ViewSpec::Families, ViewSpec::Languages], apparent),
);
let Section::Metrics { summary: types, .. } = &report.sections[0] else {
panic!("expected type metrics")
};
let rust = types.rows.iter().find(|row| row.id == "rust").expect("rust");
assert_eq!((rust.files, rust.bytes), (3, 350));
assert_eq!((rust.share.numerator, rust.share.denominator), (350, 657));
assert_eq!(types.share_metric, ShareMetric::ApparentBytes);
let Section::Metrics { summary: families, .. } = &report.sections[1] else {
panic!("expected family metrics")
};
assert!(families.rows.iter().any(|row| row.id == "code"));
assert!(families.rows.iter().any(|row| row.id == "prose"));
assert_eq!(families.share_metric, ShareMetric::ApparentBytes);
let Section::Metrics { summary: languages, .. } = &report.sections[2] else {
panic!("expected language metrics")
};
let rust = languages.rows.iter().find(|row| row.id == "rust").expect("rust");
assert_eq!((rust.files, rust.bytes), (3, 350));
assert_eq!((rust.share.numerator, rust.share.denominator), (350, 350));
assert_eq!(languages.share_metric, ShareMetric::ApparentBytes);
}
const CONTROL: &str = ".gitignore";
fn classified_sample() -> Index {
let mut index = Index::new_with_scope("/root", crate::test_support::observing_controls());
index
.apply(&Observation::new(vec![
Op::ControlUpsert {
path: PathBuf::from(CONTROL),
source: b"build/\n*.log\n".to_vec(),
},
upsert("src", EntryKind::Dir, Attrs::default()),
upsert("src/main.rs", EntryKind::File, attrs(100, 10)),
upsert("src/lib.rs", EntryKind::File, attrs(200, 20)),
upsert("src/debug.log", EntryKind::File, attrs(25, 70)),
upsert("docs", EntryKind::Dir, Attrs::default()),
upsert("docs/guide.md", EntryKind::File, attrs(300, 30)),
upsert("build", EntryKind::Dir, Attrs::default()),
upsert("build/cache", EntryKind::Dir, Attrs::default()),
upsert("build/cache/out.bin", EntryKind::File, attrs(1_000, 60)),
]))
.expect("apply");
index
}
fn ignored_of(row: &SummaryRow) -> IgnoredTally {
row.ignored.expect("an observing index reports an ignored share")
}
#[test]
fn the_two_tiers_agree_on_the_ignored_share() {
let index = classified_sample();
let expected = IgnoredTally { files: 2, dirs: 2, bytes: 1_025, allocated: 1_024 + 512 };
for selection in
[Selection::default(), Selection { min_size: Some(0), ..Selection::default() }]
{
let unfiltered = selection.is_unfiltered();
let summary = summary_of(&run(&index, &query(&[ViewSpec::Summary], selection.clone())));
assert_eq!(ignored_of(&summary), expected, "unfiltered: {unfiltered}");
let tree = tree_of(&run(&index, &query(&[ViewSpec::Tree], selection.clone())));
assert_eq!(tree.ignored, Some(expected), "unfiltered: {unfiltered}");
let child = |name: &str| {
tree.children.iter().find(|node| node.name == name).expect(name).ignored
};
assert_eq!(
child("build"),
Some(IgnoredTally { files: 1, dirs: 1, bytes: 1_000, allocated: 1_024 }),
"an ignored directory is wholly ignored below it, unfiltered: {unfiltered}"
);
assert_eq!(
child("src"),
Some(IgnoredTally { files: 1, dirs: 0, bytes: 25, allocated: 512 }),
"unfiltered: {unfiltered}"
);
assert_eq!(child("docs"), Some(IgnoredTally::default()), "observed, nothing ignored");
let rows = types_of(&run(&index, &query(&[ViewSpec::Extensions], selection.clone())));
let row = |extension: &str| {
rows.iter().find(|row| row.extension == extension).expect(extension).ignored
};
assert_eq!(
row(".log"),
Some(IgnoredTally { files: 1, dirs: 0, bytes: 25, allocated: 512 }),
"unfiltered: {unfiltered}"
);
assert_eq!(row(".rs"), Some(IgnoredTally::default()), "unfiltered: {unfiltered}");
let files = files_of(&run(&index, &query(&[ViewSpec::Files], selection)));
let flag =
|path: PathBuf| files.iter().find(|row| row.path == path).map(|row| row.ignored);
assert_eq!(flag(PathBuf::from("build")), Some(Some(true)));
assert_eq!(flag(PathBuf::from("src")), Some(Some(false)));
assert_eq!(flag(["src", "debug.log"].iter().collect()), Some(Some(true)));
assert_eq!(flag(["build", "cache", "out.bin"].iter().collect()), Some(Some(true)));
}
}
#[test]
fn ignored_entries_partition_the_tree_and_rank_by_what_they_select() {
let index = classified_sample();
let apparent = Selection { size: SizeMetric::Apparent, ..Selection::default() };
let with = |ignored| Selection { ignored, ..apparent.clone() };
let summary = |selection| summary_of(&run(&index, &query(&[ViewSpec::Summary], selection)));
let total = summary(apparent.clone());
let kept = summary(with(IgnoredEntries::Exclude));
let only = summary(with(IgnoredEntries::Only));
assert_eq!(
(kept.files + only.files, kept.dirs + only.dirs, kept.bytes + only.bytes),
(total.files, total.dirs, total.bytes)
);
assert_eq!(ignored_of(&kept), IgnoredTally::default());
let whole = ignored_of(&only);
assert_eq!((whole.files, whole.dirs, whole.bytes), (only.files, only.dirs, only.bytes));
let ranked = |selection| {
tree_of(&run(&index, &query(&[ViewSpec::Tree], selection)))
.children
.iter()
.map(|node| (node.name.clone(), node.bytes))
.collect::<Vec<_>>()
};
let row = |name: &str, bytes: u64| (name.to_string(), bytes);
assert_eq!(
ranked(apparent.clone()),
[row("build", 1_000), row("src", 325), row("docs", 300)]
);
assert_eq!(
ranked(with(IgnoredEntries::Exclude)),
[row("docs", 300), row("src", 300)],
"unignored sizes rank the rows, with the name breaking the tie"
);
assert_eq!(ranked(with(IgnoredEntries::Only)), [row("build", 1_000), row("src", 25)]);
}
#[test]
fn an_index_that_observed_no_control_state_has_no_ignored_share_to_select_by() {
let mut index =
Index::new_with_scope("/root", crate::test_support::not_observing_controls());
index
.apply(&Observation::new(vec![
upsert("build", EntryKind::Dir, Attrs::default()),
upsert("build/out.bin", EntryKind::File, attrs(1_000, 60)),
]))
.expect("apply");
let views = [ViewSpec::Summary, ViewSpec::Tree, ViewSpec::Extensions, ViewSpec::Files];
let report = run(&index, &query(&views, Selection::default()));
let Section::Summary(summary) = &report.sections[0] else { panic!("a summary") };
let Section::Tree { root: tree, .. } = &report.sections[1] else { panic!("a tree") };
let Section::Extensions { rows: extensions, .. } = &report.sections[2] else {
panic!("extensions")
};
let Section::Files { rows: files, .. } = &report.sections[3] else { panic!("files") };
assert_eq!(summary.ignored, None);
assert_eq!(tree.ignored, None);
assert!(tree.children.iter().all(|node| node.ignored.is_none()));
assert!(extensions.iter().all(|row| row.ignored.is_none()));
assert!(files.iter().all(|row| row.ignored.is_none()));
let exclude = Query {
selection: Selection { ignored: IgnoredEntries::Exclude, ..Selection::default() },
views: vec![ViewSpec::Summary],
..Query::default()
};
let only = Query {
selection: Selection { ignored: IgnoredEntries::Only, ..Selection::default() },
..exclude.clone()
};
for refused in [exclude, only] {
assert!(
matches!(
super::report(
&index,
&crate::test_support::read_of(&index, refused),
generated_at()
),
Err(crate::Error::InvalidRequest(
crate::query::RequestError::IgnoredWithoutObservation(_)
))
),
"a selection by ignored state over an unobserving index is refused"
);
}
assert_eq!(summary_of(&run(&index, &query(&views, Selection::default()))).files, 1);
}
}