Skip to main content

datui_lib/home/
discover.rs

1//! Dataset discovery for the home screen. Not a catalog: listings are computed from
2//! the filesystem when asked and forgotten at session end; between runs datui keeps
3//! only recent paths and measured shapes (`remembered`, valid while the files are
4//! unchanged). Every function scans one directory level, since data lives on slow
5//! mounts and huge partition trees.
6
7use std::path::{Path, PathBuf};
8
9/// Compression suffixes that may follow a data extension (`sales.csv.gz`).
10const COMPRESSION_EXTENSIONS: &[&str] = &["gz", "bz2", "xz", "zst", "zstd"];
11
12/// Upper bound on entries read from one directory, so a huge one cannot hang the UI.
13pub const MAX_ENTRIES_PER_DIR: usize = 5_000;
14
15/// What a home-screen row represents.
16#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
17#[serde(rename_all = "lowercase")]
18pub enum EntryKind {
19    /// A single data file.
20    File,
21    /// A file datui has no reader for: hidden on home until `Ctrl+A` shows it dimmed.
22    Other,
23    /// A directory of `key=value` partitions — one dataset, not a tree to walk.
24    Hive,
25    /// A directory of similarly-shaped data files, openable as one table.
26    MultiFile,
27    /// A Delta Lake table: `_delta_log/` beside the data files.
28    Delta,
29    /// An Apache Iceberg table: `metadata/` holding the snapshots, `data/` the files.
30    Iceberg,
31    /// An Apache Hudi table: `.hoodie/` holding the timeline.
32    Hudi,
33    /// An ordinary directory, to descend into.
34    Directory,
35    /// Somewhere remote not looked at yet: classifying would read it, which blocks when
36    /// the network is gone, so it is offered as openable, unlabeled. Also what an
37    /// unrecognized kind deserializes as, so an older build can still read a dataset
38    /// index written by a newer one (see `CLASSIFIER_VERSION`).
39    #[serde(other)]
40    Unknown,
41}
42
43/// Bump whenever classification changes. A cached kind is restored without
44/// re-deriving it (a remote row cannot be read cheaply), so it is restored only when
45/// written by a build with the same version; measurements in the record (rows,
46/// columns, cost) survive regardless.
47pub const CLASSIFIER_VERSION: u32 = 7;
48
49impl EntryKind {
50    /// Short label shown next to the entry name.
51    pub fn label(self) -> &'static str {
52        match self {
53            // A file with no reader says nothing: it is dimmed, and Enter shows its bytes.
54            EntryKind::File | EntryKind::Other => "",
55            EntryKind::Hive => "hive",
56            EntryKind::MultiFile => "multi",
57            EntryKind::Delta => "delta",
58            EntryKind::Iceberg => "iceberg",
59            EntryKind::Hudi => "hudi",
60            EntryKind::Directory => "dir",
61            EntryKind::Unknown => "",
62        }
63    }
64
65    /// Whether selecting this entry opens a dataset rather than navigating. An unexamined
66    /// remote path counts: datui can open a prefix or hive directory directly.
67    pub fn is_dataset(self) -> bool {
68        !matches!(self, EntryKind::Directory | EntryKind::Other) && !self.is_lake_table()
69    }
70
71    /// Whether this row is known to be a dataset, for counting: unlike
72    /// [`EntryKind::is_dataset`] ("may this be opened"), a row nothing has looked into
73    /// does not count.
74    pub fn is_known_dataset(self) -> bool {
75        self != EntryKind::Unknown && self.is_dataset()
76    }
77
78    /// A Delta, Iceberg or Hudi table root: a log beside the data files that says which
79    /// of them are live, which datui does not read yet.
80    pub fn is_lake_table(self) -> bool {
81        matches!(
82            self,
83            EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi
84        )
85    }
86
87    /// The format's name for prose; `label` is the row's lowercase chip.
88    pub fn lake_name(self) -> Option<&'static str> {
89        match self {
90            EntryKind::Delta => Some("Delta"),
91            EntryKind::Iceberg => Some("Iceberg"),
92            EntryKind::Hudi => Some("Hudi"),
93            _ => None,
94        }
95    }
96}
97
98/// What one listing of a directory found, counted rather than judged. The row's label
99/// comes from here (`12 parquet`), true whether or not the files are one table.
100#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
101pub struct Holds {
102    /// Data files by format, commonest first, as [`crate::FileFormat::name`] strings so
103    /// records survive across builds.
104    #[serde(default)]
105    pub formats: Vec<(String, usize)>,
106    /// Subdirectories, partitions among them (`folders` in 0.4.0 development builds).
107    #[serde(default, alias = "folders")]
108    pub directories: usize,
109    /// `key=value` subdirectories, which are also counted in `directories`.
110    #[serde(default)]
111    pub partitions: usize,
112    /// Files datui has no reader for (a README, a script, a notebook): neither data nor a
113    /// writer's own.
114    #[serde(default)]
115    pub not_read: usize,
116    /// Files with no extension, neither data nor `not_read` by name: Spark and GBIF part
117    /// files, which the open reads by their bytes.
118    #[serde(default)]
119    pub unnamed: usize,
120    /// Entries skipped as a writer's own, and the first few by name for the pane.
121    #[serde(default)]
122    pub skipped: usize,
123    #[serde(default)]
124    pub skipped_names: Vec<String>,
125    /// The listing stopped at [`MAX_ENTRIES_PER_DIR`], so every count is a floor (`5000+`).
126    #[serde(default)]
127    pub truncated: bool,
128    /// A Hugging Face DatasetDict saved with `save_to_disk` (`dataset_dict.json` beside
129    /// split directories), read as Arrow, one split at a time.
130    #[serde(default)]
131    pub dataset_dict: bool,
132}
133
134/// How many skipped names are kept for the pane. Enough to recognise the convention.
135pub(crate) const SKIPPED_NAMES_SHOWN: usize = 4;
136
137impl Holds {
138    /// Data files of every format.
139    pub fn data_files(&self) -> usize {
140        self.formats.iter().map(|(_, n)| n).sum()
141    }
142
143    /// The one format this directory holds, when it holds exactly one.
144    pub fn one_format(&self) -> Option<&str> {
145        match self.formats.as_slice() {
146            [(name, _)] => Some(name),
147            _ => None,
148        }
149    }
150
151    /// The weights' format and file count when this directory is a model (one weight
152    /// format plus only JSON). See `is_model_directory`.
153    pub fn model_weights(&self) -> Option<(&str, usize)> {
154        if !is_model_directory(counts_names(self)) {
155            return None;
156        }
157        self.formats
158            .iter()
159            .find(|(name, _)| is_weights(name))
160            .map(|(name, count)| (name.as_str(), *count))
161    }
162
163    /// The directory row's label when its kind does not name itself: `12 parquet`,
164    /// `mixed`, or `dir` when no data is directly inside.
165    pub fn label(&self) -> String {
166        let more = if self.truncated { "+" } else { "" };
167        // A model's weights beside config and tokenizer JSON: labeled as the model, not
168        // `mixed`.
169        if let Some((name, count)) = self.model_weights() {
170            return format!("{count}{more} {name}");
171        }
172        match self.formats.as_slice() {
173            // `dir` says there is no data file inside; a cut listing cannot claim that. A
174            // directory of directories counts them.
175            [] if self.directories == 1 => format!("1 dir{more}"),
176            [] if self.directories > 1 => format!("{}{more} dirs", self.directories),
177            [] => format!("dir{more}"),
178            // The `+` hedges the whole claim: past the cap another format may lurk.
179            [(name, count)] => format!("{count}{more} {name}"),
180            // No `+`: `mixed` is not a count, and more files cannot unmake it.
181            _ => "mixed".to_string(),
182        }
183    }
184
185    /// Whether nothing has been counted: a file, or a directory not looked into. (An
186    /// empty directory that was looked into reads `dir` too.)
187    pub fn is_empty(&self) -> bool {
188        self.formats.is_empty()
189            && self.directories == 0
190            && self.skipped == 0
191            && self.not_read == 0
192            && self.unnamed == 0
193            // Every field, even those implied by others today: this guards a placeholder from
194            // erasing a row's count and should not depend on that invariant.
195            && self.partitions == 0
196            && self.skipped_names.is_empty()
197            && !self.dataset_dict
198            // A cut listing that found nothing still says more exists (`dir+`).
199            && !self.truncated
200    }
201
202    /// The details pane's `contains` line: data files by format, directories and
203    /// partitions. Unreadable files and writer markers are left out (they would read as a
204    /// warning); a row inside the directory says what is hidden.
205    pub fn line(&self, with_partitions: bool) -> Option<String> {
206        let more = if self.truncated { "+" } else { "" };
207        let mut parts: Vec<String> = self
208            .formats
209            .iter()
210            .map(|(name, count)| format!("{count}{more} {name}"))
211            .collect();
212        // Partitions are counted in `directories` too; name only the rest.
213        let plain = self.directories.saturating_sub(self.partitions);
214        if plain > 0 {
215            let word = if plain == 1 {
216                "directory"
217            } else {
218                "directories"
219            };
220            parts.push(format!("{plain}{more} {word}"));
221        }
222        if with_partitions && self.partitions > 0 {
223            let word = if self.partitions == 1 {
224                "partition"
225            } else {
226                "partitions"
227            };
228            parts.push(format!("{}{more} {word}", self.partitions));
229        }
230        (!parts.is_empty()).then(|| parts.join(" · "))
231    }
232}
233
234impl Entry {
235    /// Whether Enter lists the tables inside: a file of several that is not itself a
236    /// table. A file opening one of its tables, or a spec's variants file, opens on
237    /// Enter; → lists them.
238    pub fn enter_lists_tables(&self) -> bool {
239        self.kind == EntryKind::File
240            && self.cost.tables.is_some_and(|n| n > 1)
241            && !self.cost.opens_one
242            && self.format_spec.is_none()
243    }
244
245    /// Whether home hides this row until Ctrl+A: an unreadable file or a database's
246    /// internal table.
247    pub fn hidden_by_default(&self) -> bool {
248        self.kind == EntryKind::Other || self.table.as_ref().is_some_and(|t| t.internal)
249    }
250
251    /// The short label beside a row's name, saying what it holds: a looked-into
252    /// directory's count (`12 parquet`, `mixed`, `dir`), a lake table's or hive root's
253    /// format, else the kind.
254    pub fn label(&self) -> std::borrow::Cow<'static, str> {
255        match self.kind {
256            // See `opens_whole_directory`.
257            _ if self.opens_whole_directory => "".into(),
258            EntryKind::Directory | EntryKind::MultiFile if !self.holds.is_empty() => {
259                self.holds.label().into()
260            }
261            EntryKind::File if self.format_spec.is_some() => {
262                self.format_spec.clone().unwrap_or_default().into()
263            }
264            EntryKind::File if self.cost.tables.is_some() => {
265                let n = self.cost.tables.unwrap_or_default();
266                format!("{n} {}", if n == 1 { "table" } else { "tables" }).into()
267            }
268            // A file named for its contents (as a collection names one): its format, which the
269            // name no longer says.
270            EntryKind::File
271                if crate::FileFormat::from_path(Path::new(&self.name)).is_none()
272                    && crate::FileFormat::from_path(&self.path).is_some() =>
273            {
274                crate::FileFormat::from_path(&self.path)
275                    .map(crate::FileFormat::name)
276                    .unwrap_or_default()
277                    .into()
278            }
279            kind => kind.label().into(),
280        }
281    }
282}
283
284/// One row on the home screen.
285#[derive(Debug, Clone)]
286pub struct Entry {
287    pub path: PathBuf,
288    pub kind: EntryKind,
289    /// Display name — the file or directory name, not the full path.
290    pub name: String,
291    /// Size in bytes; for multi-file datasets the sum of files inspected, a floor.
292    pub size: Option<u64>,
293    pub modified: Option<std::time::SystemTime>,
294    /// Row count, when it can be had without reading data (Parquet footers only).
295    pub rows: Option<usize>,
296    /// Column count, same caveat.
297    pub cols: Option<usize>,
298    /// Whether `cols` came from a spread of the directory (ends and middle, past the
299    /// footer budget), so it is a floor, shown as `6+`.
300    pub cols_sampled: bool,
301    /// Column names when free to get (a Parquet footer carries them with the row count).
302    pub columns: Vec<String>,
303    /// What opening this costs: where it lives, how it is stored and laid out, from bytes
304    /// already read.
305    pub cost: Cost,
306    /// What one listing found, for a directory; empty for a file or an unexamined
307    /// directory.
308    pub holds: Holds,
309    /// Whether this row is the door opening the browsed directory. It carries no label:
310    /// labels count what is directly inside, and this row reads all of it.
311    pub opens_whole_directory: bool,
312    /// The format spec that reads this file, by glob or by magic.
313    pub format_spec: Option<String>,
314    /// A table inside a file of tables (SQLite, a NumPy archive): its path is the file's
315    /// with the table's name appended.
316    pub table: Option<TableOf>,
317    /// Whether a measurement from [`crate::home::HomeState::enriched`] is folded into
318    /// this row, so the passes asking which rows to measure need not look it up.
319    pub measured: bool,
320}
321
322/// What a row inside a file of tables says about its table.
323#[derive(Debug, Clone, PartialEq, Eq)]
324pub struct TableOf {
325    /// The containing file's format; `None` for a spec's variant (see
326    /// [`Entry::format_spec`]).
327    pub format: Option<crate::FileFormat>,
328    /// What the file calls it: SQLite's `table`, `view`, `virtual` or `shadow`, or a NumPy
329    /// archive's `array`.
330    pub kind: String,
331    /// SQLite's own (schema, statistics, shadow tables): hidden until Ctrl+A, opened like
332    /// any other.
333    pub internal: bool,
334}
335
336/// What pressing Enter on a dataset will cost (200 MB of zstd Parquet is gigabytes in
337/// memory, and NFS is not tmpfs), all from what datui already reads: the mount table
338/// and the footer.
339#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
340pub struct Cost {
341    /// Filesystem or URL scheme: `nfs4`, `ext4`, `tmpfs`, `fuse.sshfs`, `s3`.
342    #[serde(default, skip_serializing_if = "Option::is_none")]
343    pub source: Option<String>,
344    /// Bytes once decompressed, as against on disk.
345    #[serde(default, skip_serializing_if = "Option::is_none")]
346    pub uncompressed: Option<u64>,
347    /// Compression codec, as the file itself names it.
348    #[serde(default, skip_serializing_if = "Option::is_none")]
349    pub codec: Option<String>,
350    /// Row groups: one huge group cannot be read in parallel or skipped; thousands of
351    /// tiny ones cost overhead.
352    #[serde(default, skip_serializing_if = "Option::is_none")]
353    pub row_groups: Option<usize>,
354    /// Partition layout, for a hive dataset.
355    #[serde(default, skip_serializing_if = "Option::is_none")]
356    pub partitions: Option<Partitions>,
357    /// Tables of its own, for a file of tables: one opens, several are listed.
358    #[serde(default, skip_serializing_if = "Option::is_none")]
359    pub tables: Option<usize>,
360    /// Whether a file of several tables opens one (a workbook's first sheet): Enter opens
361    /// it and → lists them.
362    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
363    pub opens_one: bool,
364    /// An Arrow IPC stream (converted before scanning) rather than an IPC file (scanned in
365    /// place), from its first bytes.
366    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
367    pub ipc_stream: bool,
368}
369
370/// How opening a file row will read it. See [`how_read`].
371#[derive(Debug, Clone, Copy, PartialEq, Eq)]
372pub struct HowRead {
373    pub mode: crate::ReadMode,
374    /// A remote file that is downloaded whole before it is read.
375    pub download: bool,
376}
377
378/// How opening `entry` reads it ([`crate::FileFormat::read_mode`] for its format and
379/// storage), and whether a remote one is downloaded first. `None` unless a file's name
380/// says its format.
381pub fn how_read(entry: &Entry) -> Option<HowRead> {
382    use crate::Stored;
383    if entry.kind != EntryKind::File {
384        return None;
385    }
386    let stored = if crate::CompressionFormat::from_extension(&entry.path).is_some() {
387        Stored::Compressed { in_memory: false }
388    } else if entry.cost.ipc_stream {
389        Stored::Stream
390    } else {
391        Stored::Plain
392    };
393    let choice = match &entry.format_spec {
394        Some(name) => crate::cli::FormatChoice::Spec(name.clone()),
395        // A table inside a file of tables, at its path inside it (`shop.db/orders`).
396        None if entry.table.is_some() => {
397            crate::cli::FormatChoice::Builtin(entry.table.as_ref().and_then(|t| t.format)?)
398        }
399        None => crate::cli::FormatChoice::Builtin(data_format(&entry.path)?),
400    };
401    let mode = choice.read_mode(stored)?;
402    let download = match crate::cloud::source::input_source(&entry.path) {
403        crate::cloud::source::InputSource::Local(_) => false,
404        crate::cloud::source::InputSource::Http(_) => {
405            choice.http_file() == crate::RemoteRead::Downloaded
406        }
407        // A remote Arrow file is not peeked at here, so it is taken for an IPC file.
408        _ => choice.bucket_object(stored) == crate::RemoteRead::Downloaded,
409    };
410    Some(HowRead { mode, download })
411}
412
413/// How a hive dataset is laid out, from directory names alone.
414#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
415pub struct Partitions {
416    /// Partition keys, outermost first: `["year", "month"]`.
417    pub keys: Vec<String>,
418    /// Distinct values seen for the outermost key, sorted; bounded, so not necessarily
419    /// all.
420    pub first_key_values: Vec<String>,
421    /// Directories counted at the outermost level.
422    pub count: usize,
423    /// The count stopped at a limit; there are more.
424    pub more: bool,
425}
426
427impl Entry {
428    /// A plain directory row — somewhere to step into, with nothing read from it.
429    pub fn directory(path: &Path) -> Self {
430        Self::new(path.to_path_buf(), EntryKind::Directory)
431    }
432
433    /// A file entry with a chosen display name, for tests.
434    pub fn for_test(path: &Path, name: &str) -> Self {
435        Self::new(path.to_path_buf(), EntryKind::File).with_name(name)
436    }
437
438    /// The row under a name other than its path's last part.
439    pub(crate) fn with_name(mut self, name: impl Into<String>) -> Self {
440        self.name = name.into();
441        self
442    }
443
444    pub(crate) fn new(path: PathBuf, kind: EntryKind) -> Self {
445        let name = path
446            .file_name()
447            .map(|n| n.to_string_lossy().into_owned())
448            .unwrap_or_else(|| path.to_string_lossy().into_owned());
449        Self {
450            path,
451            kind,
452            name,
453            size: None,
454            modified: None,
455            rows: None,
456            cols: None,
457            cols_sampled: false,
458            columns: Vec::new(),
459            cost: Cost::default(),
460            holds: Default::default(),
461            opens_whole_directory: false,
462            measured: false,
463            format_spec: None,
464            table: None,
465        }
466    }
467
468    /// Attach size and mtime from a directory entry's metadata.
469    pub(crate) fn with_fs_metadata(mut self, meta: &std::fs::Metadata) -> Self {
470        if meta.is_file() {
471            self.size = Some(meta.len());
472        }
473        self.modified = meta.modified().ok();
474        self
475    }
476}
477
478/// Whether a key or path is Parquet: named `.parquet`, or an extensionless part file
479/// in a `.parquet` directory (Spark, GBIF: `occurrence.parquet/000001`). Hidden and
480/// job files (`_SUCCESS`, `.crc`) are not.
481pub fn is_parquet_key(key: &str) -> bool {
482    let key = key.trim_end_matches('/');
483    let (directory, name) = match key.rsplit_once('/') {
484        Some((directory, name)) => (directory, name),
485        None => ("", key),
486    };
487    if is_bookkeeping(name) {
488        return false;
489    }
490    if name.to_ascii_lowercase().ends_with(".parquet") {
491        return true;
492    }
493    let directory_name = directory.rsplit('/').next().unwrap_or(directory);
494    !name.contains('.') && directory_name.to_ascii_lowercase().ends_with(".parquet")
495}
496
497#[cfg(test)]
498mod parquet_key_tests {
499    use super::is_parquet_key;
500
501    #[test]
502    fn parquet_without_an_extension_is_known_by_its_directory() {
503        assert!(is_parquet_key(
504            "occurrence/2026-09-01/occurrence.parquet/000001"
505        ));
506        assert!(is_parquet_key("data/part-0.parquet"));
507        assert!(is_parquet_key("DATA/PART-0.PARQUET"));
508        assert!(!is_parquet_key(
509            "occurrence/2026-09-01/occurrence.parquet/_SUCCESS"
510        ));
511        assert!(!is_parquet_key("occurrence.parquet/.part-0.crc"));
512        assert!(!is_parquet_key("occurrence/2026-09-01/citation.txt"));
513        assert!(!is_parquet_key("notes/000001"));
514    }
515}
516
517/// What a file with no usable extension is, from its first bytes (Parquet, Arrow,
518/// Avro and ORC carry signatures; CSV and JSON have none and are not guessed). Asked
519/// only of a directory being opened, never one being looked at; which signatures a
520/// listing trusts is each format's call (`crate::formats::readers::Trusted::listing`).
521pub fn sniff_format(path: &Path) -> Option<crate::FileFormat> {
522    crate::formats::readers::sniff_file(path, crate::formats::readers::Asked::Listing)
523}
524
525/// What a listing finds a file to be by its first bytes.
526#[derive(Debug, Clone)]
527pub enum Sniffed {
528    /// A format datui reads.
529    Format,
530    /// A format spec's, which reads it.
531    Spec(std::sync::Arc<crate::formats::Spec>),
532}
533
534/// [`sniff_format`], else the format spec an open would pick (by glob, else by magic
535/// and `match.where`), from one read of the file's head.
536pub fn sniff_listed(path: &Path, formats: &crate::formats::Registry) -> Option<Sniffed> {
537    use crate::formats::readers::{Asked, HEAD, head_of, sniff};
538    let head = head_of(path)?;
539    if sniff(&head, Some(path), Asked::Listing, |_| true).is_some() {
540        return Some(Sniffed::Format);
541    }
542    formats
543        .listed(path, &head, head.len() < HEAD)
544        .map(Sniffed::Spec)
545}
546
547/// Name `entry` a file of `spec`; a spec reading several record variants also makes
548/// it a place whose tables → lists.
549pub fn name_spec_file(entry: &mut Entry, spec: &crate::formats::Spec) {
550    entry.kind = EntryKind::File;
551    entry.format_spec = Some(spec.name.clone());
552    if spec.lists_variants() {
553        entry.cost.tables = Some(spec.records.variants.len());
554    }
555}
556
557/// Name an unlisted local file row (a recent) by the spec that reads it, as a listing
558/// would: by glob, else by magic when its name says nothing.
559pub fn name_unlisted_file(entry: &mut Entry, formats: &crate::formats::Registry) {
560    if formats.is_empty()
561        || entry.kind != EntryKind::File
562        || entry.table.is_some()
563        || is_data_file(&entry.path)
564    {
565        return;
566    }
567    let spec = match formats.by_glob(&entry.path, false).into_iter().next() {
568        Some(spec) => Some(spec),
569        None if worth_sniffing(&entry.path) && is_regular_file(&entry.path) => {
570            match sniff_listed(&entry.path, formats) {
571                Some(Sniffed::Spec(spec)) => Some(spec),
572                _ => None,
573            }
574        }
575        None => None,
576    };
577    if let Some(spec) = spec {
578        name_spec_file(entry, &spec);
579    }
580}
581
582/// How many extensionless files one listing sniffs; past this the rest are listed by
583/// name.
584pub(crate) const MAX_SNIFFS_PER_DIR: usize = 256;
585
586/// Whether a file's name has no extension at all: `part-00000`, `LICENSE`.
587pub fn has_no_extension(path: &Path) -> bool {
588    path.extension().is_none()
589}
590
591/// Whether a listing sniffs a file: no extension, an uninformative one (`.bin`), or
592/// only text (`.log`, which candump writes; `.txt`).
593pub fn worth_sniffing(path: &Path) -> bool {
594    path.extension().is_none_or(|e| {
595        e.eq_ignore_ascii_case("bin") || data_format(path).is_some_and(crate::FileFormat::is_lines)
596    })
597}
598
599/// Whether a local file is Parquet by its contents: `PAR1` at both ends.
600pub fn has_parquet_magic(path: &Path) -> bool {
601    use std::io::{Read, Seek, SeekFrom};
602    let Ok(mut file) = std::fs::File::open(path) else {
603        return false;
604    };
605    let mut head = [0u8; 4];
606    let mut tail = [0u8; 4];
607    file.read_exact(&mut head).is_ok()
608        && file.seek(SeekFrom::End(-4)).is_ok()
609        && file.read_exact(&mut tail).is_ok()
610        && &head == b"PAR1"
611        && &tail == b"PAR1"
612}
613
614/// Whether a path names a Parquet file, by extension or as an extensionless part file
615/// in a `.parquet` directory. Unlike [`is_parquet_key`], a writer's own file
616/// (`_manifest.parquet`) counts: it can be opened, even if it does not make its
617/// directory a dataset.
618pub fn is_parquet_path(path: &Path) -> bool {
619    path.extension()
620        .and_then(|e| e.to_str())
621        .is_some_and(|e| e.eq_ignore_ascii_case("parquet"))
622        || is_parquet_key(&directory_and_name(path))
623}
624
625/// Whether a path looks openable by name or place (a part file in a `.parquet`
626/// directory). Every route asks here so they agree.
627pub fn is_data_file(path: &Path) -> bool {
628    data_extension(path).is_some() || is_parquet_key(&directory_and_name(path))
629}
630
631/// The extension saying what a file is, past any compression suffix (`sales.csv.gz`
632/// is CSV); `None` when not something datui reads.
633pub fn data_extension(path: &Path) -> Option<String> {
634    let name = path.file_name().and_then(|n| n.to_str())?;
635    let lower = name.to_ascii_lowercase();
636    let mut parts: Vec<&str> = lower.rsplit('.').collect();
637    parts.reverse();
638    if parts.len() < 2 {
639        return None;
640    }
641    // Walk back past a compression suffix so `sales.csv.gz` still reads as CSV.
642    let mut idx = parts.len() - 1;
643    if COMPRESSION_EXTENSIONS.contains(&parts[idx]) && idx > 1 {
644        idx -= 1;
645    }
646    crate::FileFormat::from_extension(parts[idx]).map(|_| parts[idx].to_string())
647}
648
649/// What the home screen says of a file [`unreadable_by_name`] turns away.
650pub const NO_READER: &str = "datui has no reader for this file";
651
652/// Whether a name already rules the file out: an extension no reader takes, past any
653/// compression suffix. A bare `data.gz` or extensionless name is left to the open.
654pub fn unreadable_by_name(path: &Path) -> bool {
655    let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
656        return false;
657    };
658    let name = name.to_ascii_lowercase();
659    let parts: Vec<&str> = name.rsplit('.').collect();
660    let compressed = |last: &str| COMPRESSION_EXTENSIONS.contains(&last);
661    let ext = match parts[..] {
662        [last, inner, _, ..] if compressed(last) => inner,
663        [last, _] if compressed(last) => return false,
664        [last, _, ..] => last,
665        _ => return false,
666    };
667    crate::FileFormat::from_extension(ext).is_none()
668}
669
670/// The format a file's name says it holds, past any compression suffix. Extensions
671/// are not formats (`.ipc`, `.arrow`, `.arrows`, `.feather` are one); asking
672/// [`crate::FileFormat`] keeps home from listing what the reader cannot open.
673pub fn data_format(path: &Path) -> Option<crate::FileFormat> {
674    // A sharded checkpoint's index is the model, not a JSON table.
675    crate::FileFormat::from_name_ending(path)
676        .or_else(|| crate::FileFormat::from_extension(&data_extension(path)?))
677}
678
679/// The `dir/name` string `is_parquet_key` splits on `/`, so Windows backslashes do
680/// not make `occurrence.parquet\000001` one dotted name.
681pub(crate) fn directory_and_name(path: &Path) -> String {
682    let name = path.file_name().unwrap_or_default().to_string_lossy();
683    match path.parent().and_then(|p| p.file_name()) {
684        Some(directory) => format!("{}/{name}", directory.to_string_lossy()),
685        None => name.into_owned(),
686    }
687}
688
689/// Format rank: commonest first, Parquet winning ties, then by name, so the order
690/// never depends on `read_dir`. One order for the local label, the local reader
691/// choice and the cloud label, so a row's label and `Enter` agree.
692pub(crate) fn rank_formats(a: (&str, usize), b: (&str, usize)) -> std::cmp::Ordering {
693    // Text loses: a README among data files is not the table.
694    let text = crate::FileFormat::Text.name();
695    (a.0 == text)
696        .cmp(&(b.0 == text))
697        .then_with(|| b.1.cmp(&a.1))
698        .then_with(|| (a.0 != "parquet").cmp(&(b.0 != "parquet")))
699        .then_with(|| a.0.cmp(b.0))
700}
701
702/// Whether a file is Hugging Face `datasets` metadata beside Arrow shards, so its
703/// JSON does not outnumber a one-shard dataset and get read instead.
704pub(crate) fn is_hugging_face_metadata(name: &str) -> bool {
705    matches!(name, "dataset_info.json" | "state.json")
706}
707
708fn order_formats(counts: &mut [(crate::FileFormat, usize)]) {
709    counts.sort_by(|a, b| rank_formats((a.0.name(), a.1), (b.0.name(), b.1)));
710}
711
712/// Whether a listing entry is bookkeeping: a leading `_` or `.` (`_SUCCESS`,
713/// `_committed_*`, `_metadata.json`, `.crc`) or the `_$folder$` marker. One predicate
714/// for every listing and open, so they agree.
715pub fn is_bookkeeping(name: &str) -> bool {
716    // `_$folder$` first: the folder it stands for may be a partition (`year=2024_$folder$`
717    // beside `year=2024/`), which the partition test would call data.
718    if name.ends_with("_$folder$") {
719        return true;
720    }
721    // A `key=value` name is a partition, even with a leading `_` (Spark and Hive
722    // partition on internal columns like `_date=…`).
723    if is_partition_name(name) {
724        return false;
725    }
726    name.starts_with(['_', '.'])
727}
728
729/// Whether a format, by name, is model weights.
730fn is_weights(name: &str) -> bool {
731    name == crate::FileFormat::Safetensors.name() || name == crate::FileFormat::Gguf.name()
732}
733
734/// Whether formats side by side are a model: one weight format with only JSON beside
735/// (config, tokenizer), however many JSON files.
736pub(crate) fn is_model_directory<'a>(names: impl IntoIterator<Item = &'a str>) -> bool {
737    let mut weights = None;
738    for name in names {
739        if is_weights(name) {
740            if weights.is_some_and(|w| w != name) {
741                return false;
742            }
743            weights = Some(name);
744        } else if name != crate::FileFormat::Json.name() {
745            return false;
746        }
747    }
748    weights.is_some()
749}
750
751/// The format names `holds` counted.
752fn counts_names(holds: &Holds) -> impl Iterator<Item = &str> {
753    holds.formats.iter().map(|(name, _)| name.as_str())
754}
755
756/// Whether a name is a hive partition (`year=2024`): `key=value`, with a non-empty key.
757/// The value may be empty in practice.
758pub fn is_partition_name(name: &str) -> bool {
759    matches!(name.find('='), Some(i) if i > 0)
760}
761
762/// What the data files sitting directly in a directory say about how to read it.
763///
764/// A directory opened as one dataset has to be read by *something*, and the only honest
765/// source for that is the files in it. Before this existed the answer was assumed:
766/// any directory was scanned as Parquet, so a directory of `.json.gz` was opened by
767/// seeking to the end of each file for a `PAR1` that was never going to be there.
768#[derive(Debug, Clone, PartialEq, Eq)]
769pub enum DirectoryFormat {
770    /// Every data file directly in the directory reads as this one format, and these are
771    /// the files. Sorted, because a concatenation's row order is its file order.
772    One(crate::FileFormat, Vec<PathBuf>),
773    /// The files name more than one format. The commonest of them is the table — a
774    /// directory of a thousand CSVs and one stray JSON is a directory of CSVs — and the
775    /// rest are counted by format so the read can say what it passed over. Parquet wins a
776    /// tie, because it is the format a directory of data files is most likely to be about
777    /// and the one every other route here reads in place.
778    Mixed {
779        format: crate::FileFormat,
780        files: Vec<PathBuf>,
781        passed_over: Vec<(crate::FileFormat, usize)>,
782    },
783    /// The directory settles nothing by itself: it holds no readable data file directly,
784    /// or it has subdirectories and so may hold its data below. A hive dataset looks like
785    /// this — its files are a level down, under `key=value`.
786    Deeper,
787}
788
789/// What [`DirectoryFormat`] the data files directly in `dir` amount to.
790///
791/// One level only, and no file is opened: this reads names, exactly as the rest of
792/// this module does. A directory of two hundred thousand files costs one listing
793/// and nothing per file beyond it — the entry's own type comes back with the name, so
794/// there is no `stat` to bound.
795///
796/// Deliberately *not* bounded by [`MAX_ENTRIES_PER_DIR`], unlike every listing in this
797/// module. The files this returns are not a menu to show, they are the table to read:
798/// stopping at five thousand of a directory's six thousand CSVs would open it with a row
799/// count, a schema union and every aggregate quietly computed over a subset, and the
800/// `take` running before the sort would drop whichever file the directory read
801/// happened to return last. The Parquet route this mirrors hands the directory to a scan
802/// that enumerates it, with no cap either.
803pub fn directory_format(dir: &Path) -> DirectoryFormat {
804    let Ok(iter) = std::fs::read_dir(dir) else {
805        return DirectoryFormat::Deeper;
806    };
807
808    // Every format the directory names, with the files of each. A directory of one format
809    // takes the only entry; a directory of several takes the commonest and reports the
810    // rest, which is what stops one stray file deciding a directory cannot be read.
811    let mut by_format: Vec<(crate::FileFormat, Vec<PathBuf>)> = Vec::new();
812    // Files whose names say nothing, kept in case their bytes do. See below.
813    let mut nameless: Vec<PathBuf> = Vec::new();
814    let mut partitioned = false;
815    for entry in iter.flatten() {
816        let path = entry.path();
817        let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
818            continue;
819        };
820        // A table format's own files are not the table's, and a dotfile is nobody's.
821        // The same test every other route makes, so they agree on what is data.
822        if is_bookkeeping(name) {
823            continue;
824        }
825        // The type the directory read already returned, rather than a `stat` per
826        // entry: on a share that is a round trip per entry, and the question is only
827        // whether this is a file. A symlink still gets the stat, because `d_type`
828        // cannot say what is on the other end of one.
829        let is_file = match entry.file_type() {
830            Ok(kind) if kind.is_symlink() => is_regular_file(&path),
831            Ok(kind) => kind.is_file(),
832            Err(_) => is_regular_file(&path),
833        };
834        if !is_file {
835            // Only a partition, not any subdirectory: a directory of CSVs with some
836            // unrelated directory beside them is still a directory of CSVs, and handing
837            // that to a scanner that walks trees is the very thing this function exists
838            // to stop.
839            partitioned |= is_partition_dir(&path);
840            continue;
841        }
842        let Some(found) = data_format(&path) else {
843            // A name that says nothing. Kept rather than dropped, because the bytes may
844            // still say what it is — see below, where they are asked.
845            if path.extension().is_none() {
846                nameless.push(path);
847            }
848            continue;
849        };
850        match by_format.iter_mut().find(|(f, _)| *f == found) {
851            Some((_, of_that_format)) => of_that_format.push(path),
852            None => by_format.push((found, vec![path])),
853        }
854    }
855
856    // Extensionless files (Spark and GBIF part files) in a directory whose names settled
857    // nothing; skipped when names already answer (a `LICENSE` beside Parquet). A
858    // [`spread`] is sniffed, not every file: the ends must agree, and then all are taken
859    // as that format (the scan reports any odd one out).
860    if by_format.is_empty() && !nameless.is_empty() {
861        nameless.sort();
862        let picks = spread(nameless.len());
863        let sniffed: Vec<crate::FileFormat> = picks
864            .iter()
865            .filter_map(|i| sniff_format(&nameless[*i]))
866            .collect();
867        if sniffed.len() == picks.len()
868            && let Some(found) = sniffed.first().copied()
869            && sniffed.iter().all(|f| *f == found)
870        {
871            by_format.push((found, nameless));
872        }
873    }
874
875    // Text is data only where nothing else is: a README beside Parquet is not a
876    // candidate.
877    if by_format.iter().any(|(f, _)| !f.is_lines()) {
878        by_format.retain(|(f, _)| !f.is_lines());
879    }
880
881    // A Hugging Face dataset's own JSON files, beside its shards.
882    if by_format
883        .iter()
884        .any(|(f, _)| *f == crate::FileFormat::Arrow)
885    {
886        for (format, files) in &mut by_format {
887            if *format == crate::FileFormat::Json {
888                files.retain(|f| {
889                    !f.file_name()
890                        .and_then(|n| n.to_str())
891                        .is_some_and(is_hugging_face_metadata)
892                });
893            }
894        }
895        by_format.retain(|(_, files)| !files.is_empty());
896    }
897
898    // Decided after the whole listing, not at the first deciding entry, so read order
899    // does not matter. One `key=value` below and the data is down there.
900    if partitioned {
901        return DirectoryFormat::Deeper;
902    }
903
904    // The shared format rank, so the reader picked is the format the label names.
905    by_format.sort_by(|a, b| rank_formats((a.0.name(), a.1.len()), (b.0.name(), b.1.len())));
906    // A directory of model weights is the model, however much config and tokenizer
907    // JSON sits beside it; the JSON is passed over.
908    if is_model_directory(by_format.iter().map(|(f, _)| f.name()))
909        && let Some(at) = by_format.iter().position(|(f, _)| is_weights(f.name()))
910    {
911        let weights = by_format.remove(at);
912        by_format.insert(0, weights);
913    }
914    let mut by_format = by_format.into_iter();
915    let Some((format, mut files)) = by_format.next() else {
916        return DirectoryFormat::Deeper;
917    };
918    files.sort();
919    let passed_over: Vec<(crate::FileFormat, usize)> =
920        by_format.map(|(f, of_that)| (f, of_that.len())).collect();
921    if passed_over.is_empty() {
922        DirectoryFormat::One(format, files)
923    } else {
924        DirectoryFormat::Mixed {
925            format,
926            files,
927            passed_over,
928        }
929    }
930}
931
932/// How far down a hive root is followed for its files (year/month/day/hour is four);
933/// deeper is not a hive dataset, and each level costs a read on a share.
934const MAX_HIVE_DEPTH: usize = 16;
935
936/// What the files under a hive root's partitions are. [`directory_format`] can only
937/// say `Deeper` of a root, so this follows one spine (the first partition at each
938/// level, as a hive scan reads its schema) to the bottom. A sample: disagreeing
939/// partitions report as the first one.
940pub fn hive_leaf_format(dir: &Path) -> DirectoryFormat {
941    let mut at = dir.to_path_buf();
942    for _ in 0..MAX_HIVE_DEPTH {
943        match directory_format(&at) {
944            // Nothing here settles it. Follow the partitions down, if there are any.
945            DirectoryFormat::Deeper => match first_partition(&at) {
946                Some(next) => at = next,
947                None => return DirectoryFormat::Deeper,
948            },
949            settled => return settled,
950        }
951    }
952    DirectoryFormat::Deeper
953}
954
955/// The first `key=value` subdirectory of `dir` by name, so runs agree.
956fn first_partition(dir: &Path) -> Option<PathBuf> {
957    let iter = std::fs::read_dir(dir).ok()?;
958    iter.flatten()
959        .take(MAX_ENTRIES_PER_DIR)
960        .map(|entry| entry.path())
961        .filter(|path| is_partition_dir(path) && path.is_dir())
962        .min()
963}
964
965/// Whether a directory name is a hive partition (`year=2024`).
966fn is_partition_dir(path: &Path) -> bool {
967    path.file_name()
968        .and_then(|n| n.to_str())
969        .is_some_and(is_partition_name)
970}
971
972/// An empty extensionless object: a tool's folder marker (`yellow/year=2032`), whether
973/// or not the folder still exists. Nothing datui opens is both empty and nameless.
974pub fn is_empty_marker(name: &str, size: u64) -> bool {
975    size == 0 && !name.contains('.')
976}
977
978/// Classify a directory without walking it: one listing bounded by
979/// [`MAX_ENTRIES_PER_DIR`] (not a probe of the first few entries, whose answer would
980/// depend on read order). Includes rows on network mounts: one `getdents` walk,
981/// stat'ing only symlinks. `home_open_selected` calls it on the key thread; others
982/// on workers.
983pub fn classify_directory(path: &Path) -> EntryKind {
984    look_at_directory(path).0
985}
986
987/// The kind (what `Enter` does) and what the listing found (what the label says),
988/// from one read.
989pub fn look_at_directory(path: &Path) -> (EntryKind, Holds) {
990    // Before the listing: three `join` tests answer it, where counting would walk every
991    // table of a warehouse each pass. A bucket prefix finds markers in its listing.
992    if let Some(lake) = lake_table(path) {
993        return (lake, Holds::default());
994    }
995    let Ok(iter) = std::fs::read_dir(path) else {
996        return (EntryKind::Directory, Holds::default());
997    };
998    let directory = path.file_name().unwrap_or_default().to_string_lossy();
999    let rules = Rules {
1000        directory: &directory,
1001        // As the listing of the rows inside does, so the tally agrees with them.
1002        sniff: (!crate::home::is_remote_path(path)).then_some(path),
1003        in_bucket: false,
1004    };
1005    // One past the cap, so "there is more" is known; bounded where entries come from,
1006    // since `.crc` files beside each data file would double the walk. Past the cap the
1007    // first entries decide.
1008    let mut truncated = false;
1009    let seen = iter
1010        .flatten()
1011        .take(MAX_ENTRIES_PER_DIR + 1)
1012        .enumerate()
1013        .map_while(|(at, entry)| {
1014            truncated = at == MAX_ENTRIES_PER_DIR;
1015            (!truncated).then(|| seen_on_disk(&entry))
1016        });
1017    let (kind, mut holds) = classify(seen, &rules);
1018    holds.truncated = truncated;
1019    (kind, holds)
1020}
1021
1022/// A local entry as [`classify`] sees it, from the read's file type (no stat per
1023/// entry; symlinks still need one). Regular files only: a FIFO named `a.csv` blocks
1024/// its opener, a broken symlink opens as nothing.
1025fn seen_on_disk(entry: &std::fs::DirEntry) -> Seen {
1026    let (is_dir, is_file) = match entry.file_type() {
1027        Ok(kind) if !kind.is_symlink() => (kind.is_dir(), kind.is_file()),
1028        _ => std::fs::metadata(entry.path()).map_or((false, false), |m| (m.is_dir(), m.is_file())),
1029    };
1030    Seen {
1031        name: entry.file_name().to_string_lossy().into_owned(),
1032        is_dir,
1033        is_file,
1034        size: None,
1035    }
1036}
1037
1038/// One entry of a listing, on disk or in a bucket, as [`classify`] asks about it.
1039#[derive(Debug, Clone)]
1040pub struct Seen {
1041    pub name: String,
1042    pub is_dir: bool,
1043    /// A regular file an open can read (not a FIFO, socket or broken symlink).
1044    pub is_file: bool,
1045    /// Bytes, where listed; an empty extensionless file is a folder marker, not data.
1046    pub size: Option<u64>,
1047}
1048
1049/// Where the local and the bucket listings deliberately differ.
1050pub struct Rules<'a> {
1051    /// The listed directory's name: an extensionless part file in `occurrence.parquet/` is
1052    /// Parquet by where it sits.
1053    pub directory: &'a str,
1054    /// Where to look inside nameless files, a few per listing; `None` where each open is
1055    /// a round trip that may not return.
1056    pub sniff: Option<&'a Path>,
1057    /// A bucket prefix: lake tables are found by names in the listing (Iceberg by layout
1058    /// alone), a DatasetDict by `dataset_dict.json`, and only Parquet (read in place) is
1059    /// offered as many files that are one table.
1060    pub in_bucket: bool,
1061}
1062
1063/// A directory's kind and holdings from one listing level: the one rule for
1064/// [`look_at_directory`] and bucket prefixes, so a directory and its mirror agree;
1065/// [`Rules`] holds their differences. The caller sets truncation.
1066pub fn classify(seen: impl Iterator<Item = Seen>, rules: &Rules) -> (EntryKind, Holds) {
1067    use crate::FileFormat;
1068    let mut holds = Holds::default();
1069    let mut counts: Vec<(FileFormat, usize)> = Vec::new();
1070    let mut skipped: Vec<String> = Vec::new();
1071    // Counted as data until the listing is done; see `is_hugging_face_metadata`.
1072    let mut hugging_face: Vec<String> = Vec::new();
1073    let mut dict_file = false;
1074    let mut lake: Vec<&'static str> = Vec::new();
1075    // `present` is everything but a writer's own: what the majority is measured against.
1076    let (mut present, mut data_files, mut parquet) = (0usize, 0usize, 0usize);
1077    let mut sniffs_left = rules.sniff.map_or(0, |_| MAX_SNIFFS_PER_DIR);
1078    for s in seen {
1079        // Lake markers are spec, not strays: looked for before bookkeeping skips
1080        // `_delta_log`.
1081        if s.is_dir
1082            && rules.in_bucket
1083            && let Some(marker) = ["_delta_log", ".hoodie", "metadata", "data"]
1084                .into_iter()
1085                .find(|m| *m == s.name)
1086        {
1087            lake.push(marker);
1088        }
1089        if is_bookkeeping(&s.name) {
1090            skipped.push(s.name);
1091            continue;
1092        }
1093        present += 1;
1094        if s.is_dir {
1095            holds.directories += 1;
1096            holds.partitions += usize::from(is_partition_name(&s.name));
1097            continue;
1098        }
1099        if s.size.is_some_and(|size| is_empty_marker(&s.name, size)) {
1100            holds.not_read += 1;
1101            continue;
1102        }
1103        let name = Path::new(&s.name);
1104        let key = format!("{}/{}", rules.directory, s.name);
1105        let named = data_format(name);
1106        let found = named
1107            // Text by its name, unless its bytes say more: below.
1108            .filter(|f| !f.is_lines())
1109            // A sharded checkpoint's index counts as JSON, so the label counts shards; the read
1110            // still uses it.
1111            .map(
1112                |f| match crate::formats::model_files::is_safetensors_index(name) {
1113                    true => FileFormat::Json,
1114                    false => f,
1115                },
1116            )
1117            // Data by where it sits rather than by its name: see [`is_data_file`].
1118            .or_else(|| is_parquet_key(&key).then_some(FileFormat::Parquet))
1119            .or_else(|| {
1120                let dir = rules.sniff?;
1121                spend_sniff(&mut sniffs_left, s.is_file, name)
1122                    .then(|| sniff_format(&dir.join(name)))?
1123            })
1124            .or(named)
1125            .filter(|_| s.is_file);
1126        let Some(found) = found else {
1127            // No reader and no name; extensionless files are counted apart (Spark part files).
1128            match s.is_file && has_no_extension(name) {
1129                true => holds.unnamed += 1,
1130                false => holds.not_read += 1,
1131            }
1132            continue;
1133        };
1134        if found == FileFormat::Json && is_hugging_face_metadata(&s.name) {
1135            hugging_face.push(s.name.clone());
1136        }
1137        dict_file |= found == FileFormat::Json && s.name == crate::formats::hf_splits::DATASET_DICT;
1138        data_files += 1;
1139        parquet += usize::from(is_parquet_key(&key));
1140        match counts.iter_mut().find(|(f, _)| *f == found) {
1141            Some((_, n)) => *n += 1,
1142            None => counts.push((found, 1)),
1143        }
1144    }
1145
1146    // Hugging Face's JSON is its writer's own, like `_SUCCESS`, as is a DatasetDict's
1147    // `dataset_dict.json`.
1148    if !counts.iter().any(|(f, _)| *f == FileFormat::Arrow) {
1149        hugging_face.clear();
1150    }
1151    holds.dataset_dict = rules.in_bucket && dict_file && holds.directories > 0;
1152    if holds.dataset_dict {
1153        hugging_face.push(crate::formats::hf_splits::DATASET_DICT.to_string());
1154    }
1155    if let Some((_, n)) = counts.iter_mut().find(|(f, _)| *f == FileFormat::Json) {
1156        *n -= hugging_face.len();
1157        data_files -= hugging_face.len();
1158        present -= hugging_face.len();
1159        skipped.append(&mut hugging_face);
1160    }
1161    // Text is data only where nothing else is.
1162    if counts.iter().any(|(f, _)| !f.is_lines()) {
1163        for (_, n) in counts.iter_mut().filter(|(f, _)| f.is_lines()) {
1164            data_files -= *n;
1165            holds.not_read += std::mem::take(n);
1166        }
1167    }
1168    counts.retain(|(_, n)| *n > 0);
1169    order_formats(&mut counts);
1170    let one_readable = matches!(counts.as_slice(), [(f, _)] if f.reads_many_files());
1171    holds.formats = counts
1172        .into_iter()
1173        .map(|(f, n)| (f.name().to_string(), n))
1174        .collect();
1175    // The first few by name, so runs agree; an object and prefix of one name are one
1176    // thing.
1177    skipped.sort();
1178    skipped.dedup();
1179    holds.skipped = skipped.len();
1180    skipped.truncate(SKIPPED_NAMES_SHOWN);
1181    holds.skipped_names = skipped;
1182
1183    // Lake tables' data files agree on a schema, so the rules below would wrongly say
1184    // "one table". Iceberg needs its whole shape (data under `data/`).
1185    let marked = |m| lake.contains(&m);
1186    if marked("_delta_log") {
1187        return (EntryKind::Delta, holds);
1188    }
1189    if marked(".hoodie") {
1190        return (EntryKind::Hudi, holds);
1191    }
1192    if marked("metadata") && marked("data") && parquet == 0 {
1193        return (EntryKind::Iceberg, holds);
1194    }
1195    // No majority rule: one stray `notes=old/` reads `hive`, but majorities refused real
1196    // hive roots with a README or `scripts/` beside them, the worse mistake.
1197    if holds.partitions > 0 && holds.partitions >= data_files {
1198        return (EntryKind::Hive, holds);
1199    }
1200    // One multi-file-readable format that dominates: a couple of stray CSVs make a
1201    // place, not a dataset. A model opens as the model despite its JSON.
1202    let one_table = if rules.in_bucket {
1203        parquet > 1 && parquet == data_files && parquet * 2 >= present
1204    } else {
1205        (data_files > 1 && one_readable && data_files * 2 >= present)
1206            || (holds.partitions == 0 && is_model_directory(counts_names(&holds)))
1207    };
1208    let kind = match one_table {
1209        true => EntryKind::MultiFile,
1210        false => EntryKind::Directory,
1211    };
1212    (kind, holds)
1213}
1214
1215/// Spend one of a listing's sniffs on `name`: a regular file whose name says nothing
1216/// ([`worth_sniffing`]), budget permitting.
1217fn spend_sniff(left: &mut usize, is_file: bool, name: &Path) -> bool {
1218    let spend = is_file && *left > 0 && worth_sniffing(name);
1219    *left -= usize::from(spend);
1220    spend
1221}
1222
1223/// Entries under `metadata/` checked for Iceberg: `v1.metadata.json` is written on
1224/// the first commit and never removed, but read order is arbitrary.
1225const ICEBERG_METADATA_PROBE: usize = 64;
1226
1227/// Whether `path` is a lake table root, and which. Marker names are part of these
1228/// formats' specs (unlike filename conventions); three `join` tests answer it
1229/// without walking a table's files.
1230fn lake_table(path: &Path) -> Option<EntryKind> {
1231    if path.join("_delta_log").is_dir() {
1232        return Some(EntryKind::Delta);
1233    }
1234    if path.join(".hoodie").is_dir() {
1235        return Some(EntryKind::Hudi);
1236    }
1237    // Iceberg's marker is a plain name, so it needs the whole shape: `metadata/` with a
1238    // metadata file, beside `data/`.
1239    let metadata = path.join("metadata");
1240    if path.join("data").is_dir()
1241        && metadata.is_dir()
1242        && std::fs::read_dir(&metadata).is_ok_and(|entries| {
1243            entries
1244                .flatten()
1245                .take(ICEBERG_METADATA_PROBE)
1246                .any(|e| e.file_name().to_string_lossy().ends_with(".metadata.json"))
1247        })
1248    {
1249        return Some(EntryKind::Iceberg);
1250    }
1251    None
1252}
1253
1254/// What one directory listing produced, and whether it saw all of it.
1255#[derive(Debug, Clone, Default)]
1256pub struct Scan {
1257    pub entries: Vec<Entry>,
1258    /// The directory held more than `MAX_ENTRIES_PER_DIR`; `entries` is a prefix, which
1259    /// the UI must say.
1260    pub truncated: bool,
1261}
1262
1263/// One local directory level as home lists it, naming sniffed files by `formats`'
1264/// magic. Names and kinds only, from the directory itself: no stat per entry, so a
1265/// directory of fifty thousand files costs its reads, not fifty thousand more. Size
1266/// and mtime come with [`stat_row`] for the rows shown, or every row when a sort
1267/// needs them.
1268pub fn scan_dir_specs(dir: &Path, formats: &crate::formats::Registry) -> Scan {
1269    scan_dir_with(dir, formats, false, |_| {})
1270}
1271
1272/// Fill in a row's size and mtime, which a local listing leaves to the rows shown:
1273/// one stat, not following a link (a listing leaves links out). Whether it is there.
1274pub fn stat_row(entry: &mut Entry) -> bool {
1275    match std::fs::symlink_metadata(&entry.path) {
1276        Ok(meta) => {
1277            if meta.is_file() {
1278                entry.size = Some(meta.len());
1279            }
1280            entry.modified = meta.modified().ok();
1281            true
1282        }
1283        Err(_) => false,
1284    }
1285}
1286
1287/// Whether a row on disk still lacks the size and mtime its listing left out: no URL,
1288/// no table inside a file.
1289pub fn unstated(entry: &Entry) -> bool {
1290    entry.modified.is_none() && on_disk(entry)
1291}
1292
1293/// Whether a row is a path on disk a stat answers for: no URL, no table inside a file.
1294pub fn on_disk(entry: &Entry) -> bool {
1295    entry.table.is_none()
1296        && !crate::home::is_cloud_place(&entry.path)
1297        && matches!(
1298            crate::cloud::source::input_source(&entry.path),
1299            crate::cloud::source::InputSource::Local(_)
1300        )
1301}
1302
1303/// How often a listing still being read shows what it has so far.
1304const LISTING_PROGRESS_EVERY: std::time::Duration = std::time::Duration::from_millis(250);
1305
1306/// List one directory with bounded work: one `read_dir` and a stat per entry, up to
1307/// [`MAX_ENTRIES_PER_DIR`]; errors give an empty listing. Nothing is classified:
1308/// subdirectories return [`EntryKind::Unknown`] and are looked into later from the
1309/// viewport, so a label is a fact about the row, not its position. `progress` gets
1310/// the rows read since the last call, every `LISTING_PROGRESS_EVERY`, so a slow share
1311/// shows rows as they arrive.
1312pub fn scan_dir_progressive(dir: &Path, progress: impl FnMut(&[Entry])) -> Scan {
1313    scan_dir_with(dir, &crate::formats::Registry::default(), true, progress)
1314}
1315
1316/// List `dir`; with `stat`, each entry's size and mtime too (a share's probe, whose
1317/// rows nothing on the UI's side may stat later).
1318fn scan_dir_with(
1319    dir: &Path,
1320    formats: &crate::formats::Registry,
1321    stat: bool,
1322    mut progress: impl FnMut(&[Entry]),
1323) -> Scan {
1324    let Ok(iter) = std::fs::read_dir(dir) else {
1325        return Scan::default();
1326    };
1327    let mut shown = std::time::Instant::now();
1328
1329    let mut entries = Vec::new();
1330    // How many of `entries` `progress` has been handed.
1331    let mut sent = 0usize;
1332    let mut seen = 0usize;
1333    let mut truncated = false;
1334    // Extensionless files are sniffed (a few bytes) so part files list as data; never on
1335    // a share, where each open is a round trip.
1336    let mut sniffs_left = if crate::home::is_remote_path(dir) {
1337        0
1338    } else {
1339        MAX_SNIFFS_PER_DIR
1340    };
1341
1342    // One past the cap: enough to know more exists without paying to process it.
1343    for dir_entry in iter.flatten().take(MAX_ENTRIES_PER_DIR + 1) {
1344        seen += 1;
1345        if seen > MAX_ENTRIES_PER_DIR {
1346            truncated = true;
1347            break;
1348        }
1349
1350        let path = dir_entry.path();
1351        let name = dir_entry.file_name();
1352        if name.to_string_lossy().starts_with('.') {
1353            continue;
1354        }
1355
1356        let meta = if stat {
1357            match dir_entry.metadata() {
1358                Ok(meta) => Some(meta),
1359                Err(_) => continue,
1360            }
1361        } else {
1362            None
1363        };
1364        // The directory's own word for the type (`d_type`), unless already stat'ed; a
1365        // filesystem that does not say is stat'ed by `file_type` itself. Links are not
1366        // followed, so they are left out as before.
1367        let file_type = match &meta {
1368            Some(meta) => meta.file_type(),
1369            None => match dir_entry.file_type() {
1370                Ok(file_type) => file_type,
1371                Err(_) => continue,
1372            },
1373        };
1374
1375        let mut spec = None;
1376        let kind = if file_type.is_dir() {
1377            EntryKind::Unknown
1378        } else if file_type.is_file() && is_data_file(&path) {
1379            EntryKind::File
1380        } else if spend_sniff(&mut sniffs_left, file_type.is_file(), &path) {
1381            match sniff_listed(&path, formats) {
1382                Some(Sniffed::Format) => EntryKind::File,
1383                Some(Sniffed::Spec(found)) => {
1384                    spec = Some(found);
1385                    EntryKind::File
1386                }
1387                None => EntryKind::Other,
1388            }
1389        } else if file_type.is_file() {
1390            EntryKind::Other
1391        } else {
1392            // Not a directory or regular file: a FIFO named `x.parquet` must never be offered.
1393            continue;
1394        };
1395
1396        let mut entry = Entry::new(path, kind);
1397        if let Some(meta) = &meta {
1398            entry = entry.with_fs_metadata(meta);
1399        }
1400        if let Some(spec) = spec {
1401            name_spec_file(&mut entry, &spec);
1402        }
1403        entries.push(entry);
1404        if shown.elapsed() >= LISTING_PROGRESS_EVERY {
1405            progress(&entries[sent..]);
1406            sent = entries.len();
1407            shown = std::time::Instant::now();
1408        }
1409    }
1410
1411    sort_entries(&mut entries);
1412    Scan { entries, truncated }
1413}
1414
1415/// Datasets first, then directories, each alphabetical (a listing is scanned by
1416/// name). Unexamined rows sort with directories, where they land if plain, so a kind
1417/// arriving later never moves a row under the cursor.
1418pub(crate) fn sort_entries(entries: &mut [Entry]) {
1419    entries.sort_by(|a, b| {
1420        // Data, then directories, then what datui cannot read.
1421        let group = |k: EntryKind| match k {
1422            k if k.is_known_dataset() => 0,
1423            EntryKind::Other => 2,
1424            _ => 1,
1425        };
1426        group(a.kind).cmp(&group(b.kind)).then_with(|| {
1427            a.name
1428                .to_ascii_lowercase()
1429                .cmp(&b.name.to_ascii_lowercase())
1430        })
1431    });
1432}
1433
1434/// How far below a directory the footer walk goes (year/month/day/hour is four).
1435const MAX_WALK_DEPTH: u8 = 4;
1436
1437/// Upper bound on footers read to size a multi-file or hive dataset; past it the count
1438/// is left blank rather than partial.
1439const MAX_FOOTERS_PER_DATASET: usize = 64;
1440
1441/// Fill in row and column counts from Parquet footers: one file, or a bounded sum for
1442/// hive and multi-file datasets. Non-Parquet keeps `None`, a blank in the UI.
1443pub fn enrich(entry: &mut Entry) {
1444    enrich_as(entry, &crate::formats::schema_union::ReadAs::default())
1445}
1446
1447/// As [`enrich`], reading files as the following open will: where the header is
1448/// decides what the names are, so judging with other settings would misjudge (e.g.
1449/// `--no-header`).
1450pub fn enrich_as(entry: &mut Entry, as_read: &crate::formats::schema_union::ReadAs) {
1451    enrich_with(entry, as_read, None)
1452}
1453
1454/// As [`enrich_as`], using the shape an open kept in `remembered` while the files are
1455/// unchanged, instead of sampling footers.
1456pub fn enrich_with(
1457    entry: &mut Entry,
1458    as_read: &crate::formats::schema_union::ReadAs,
1459    remembered: Option<&crate::cache::CacheManager>,
1460) {
1461    match entry.kind {
1462        EntryKind::File => {
1463            enrich_parquet(entry);
1464            enrich_tables(entry);
1465            enrich_arrow(entry);
1466        }
1467        EntryKind::Hive | EntryKind::MultiFile => enrich_dataset(entry, as_read, remembered),
1468        // Nothing to read for plain or unexamined directories, nor lake tables (summing their
1469        // footers counts tombstoned and rewritten rows).
1470        EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => {}
1471        EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => {}
1472    }
1473}
1474
1475/// Sum footers across a bounded set of Parquet files under `entry`.
1476fn enrich_dataset(
1477    entry: &mut Entry,
1478    as_read: &crate::formats::schema_union::ReadAs,
1479    remembered: Option<&crate::cache::CacheManager>,
1480) {
1481    // A directory whose own files are a format this cannot count is not described by
1482    // Parquet beneath it (`6 json` reporting the subdirectories' columns). Parquet with
1483    // Parquet beneath keeps the walk: `Enter` reads the subtree, so counts must too; the
1484    // `holds` line explains the difference.
1485
1486    // The partition layout comes from directory names, so it is known even when the rows
1487    // are too many to count.
1488    if entry.kind == EntryKind::Hive {
1489        entry.cost.partitions = partition_layout(&entry.path);
1490    }
1491
1492    // Whether the footers below describe this directory (exact for its own format). Not
1493    // asked of a hive root: its own files are strays (a `schema.json`), and its data's
1494    // format can only be sampled.
1495    let reads_as_parquet = entry.kind == EntryKind::Hive
1496        || match entry.holds.one_format() {
1497            // No single format: unreachable today (mixed formats are `Directory`), so default to
1498            // the safe side.
1499            None => true,
1500            // An unreadable name is not Parquet: missing counts are undone by opening; another
1501            // format's counts are not.
1502            Some(name) => crate::FileFormat::from_name(name) == Some(crate::FileFormat::Parquet),
1503        };
1504    if !reads_as_parquet {
1505        entry.size = None;
1506        judge_by_names(entry, as_read);
1507        return;
1508    }
1509
1510    // A directory's stat size is its inode, not its contents: dropped here, restored only
1511    // if the files are totalled.
1512    entry.size = None;
1513
1514    let files = parquet_files_under(&entry.path);
1515    // Past the budget, footers an open kept are used if the listing still matches.
1516    if files.len() > MAX_FOOTERS_PER_DATASET
1517        && let Some((listed, footers)) = remembered
1518            .and_then(|cache| crate::formats::dataset_files::remembered_footers(&entry.path, cache))
1519    {
1520        measure_from_footers(entry, &listed, &footers);
1521        return;
1522    }
1523    if files.is_empty() || files.len() > MAX_FOOTERS_PER_DATASET {
1524        // Whether these are one table needs only three spread footers, and matters most for
1525        // directories past the counting limit.
1526        let sampled = sample_footers(&files);
1527        let names: Vec<Vec<String>> = sampled.iter().map(column_names).collect();
1528        let tops: Vec<Vec<String>> = names
1529            .iter()
1530            .map(|n| crate::formats::schema_union::top_level_columns(n))
1531            .collect();
1532        if entry.kind == EntryKind::MultiFile && one_table_from(&tops) == Some(false) {
1533            // One-table is asked of the subtree (what opening unions), but holdings count only
1534            // the direct files (the label's), since a downgraded row never opens as one table.
1535            let own_files = direct_children(&files, &entry.path);
1536            let own = sample_footers(&own_files);
1537            // The names are every sampled one under it: home's search index should find a
1538            // directory by any column looking inside reaches.
1539            entry.columns = union_of(&names);
1540            // A floor only if one of its own footers went unread.
1541            entry.cols_sampled = own.len() < own_files.len();
1542            // The columns a reader sees, from the schema rather than splitting leaf paths on dots
1543            // (`user.id` vs struct `user` with `id`).
1544            let top = union_of(&own.iter().map(top_level_names).collect::<Vec<_>>());
1545            downgrade_to_directory(entry, (!top.is_empty()).then_some(top.len()));
1546            return;
1547        }
1548        // Still worth knowing the shape, even when the row count is out of reach.
1549        if let Some(meta) = sampled.first() {
1550            // Three files, not the first: a growing dataset's newest columns are in its last
1551            // file. Still a sample; the count is already `?`.
1552            entry.columns = union_of(&names);
1553            let top = union_of(&sampled.iter().map(top_level_names).collect::<Vec<_>>());
1554            entry.cols = Some(top.len() + partition_columns_beyond(entry, &top));
1555            entry.cols_sampled = true;
1556            // From one file: codec and row-group sizing are the writer's, uniform in practice.
1557            physical_facts(meta, &mut entry.cost);
1558            entry.cost.uncompressed = None;
1559        }
1560        return;
1561    }
1562
1563    let mut rows = 0usize;
1564    let mut bytes = 0u64;
1565    // Every column any file has, in first-seen order, so columns added over time are
1566    // found.
1567    let mut columns: Vec<String> = Vec::new();
1568    let mut seen_columns = std::collections::HashSet::new();
1569    // Top-level columns unioned the same way, kept beside the leaves (see
1570    // [`top_level_names`]).
1571    let mut top_level: Vec<String> = Vec::new();
1572    let mut seen_top_level = std::collections::HashSet::new();
1573    let mut per_file: Vec<Vec<String>> = Vec::with_capacity(files.len());
1574    // Width and size from the directory's own files only (a downgraded row is never one
1575    // table); column names stay the subtree's, for search.
1576    let mut own_bytes = 0u64;
1577    let mut own_top_level: Vec<String> = Vec::new();
1578    let mut own_seen_top = std::collections::HashSet::new();
1579    let mut cost = Cost::default();
1580    let mut uncompressed = 0u64;
1581    let mut row_groups = 0usize;
1582    for file in &files {
1583        let Some(meta) = crate::formats::parquet_footer::read_parquet_metadata(file) else {
1584            return; // A file we cannot read makes the total a guess; report nothing.
1585        };
1586        rows += meta.num_rows;
1587        let names = column_names(&meta);
1588        for name in &names {
1589            if seen_columns.insert(name.clone()) {
1590                columns.push(name.clone());
1591            }
1592        }
1593        for name in top_level_names(&meta) {
1594            if seen_top_level.insert(name.clone()) {
1595                top_level.push(name);
1596            }
1597        }
1598        // Top-level columns, not footer leaves: see
1599        // [`crate::formats::schema_union::top_level_columns`].
1600        per_file.push(crate::formats::schema_union::top_level_columns(&names));
1601        // Again for its own files (`2 parquet` must mean those two), from one stat feeding
1602        // both totals.
1603        let file_bytes = std::fs::metadata(file).map(|m| m.len()).unwrap_or(0);
1604        bytes += file_bytes;
1605        if file.parent() == Some(entry.path.as_path()) {
1606            own_bytes += file_bytes;
1607            for name in top_level_names(&meta) {
1608                if own_seen_top.insert(name.clone()) {
1609                    own_top_level.push(name);
1610                }
1611            }
1612        }
1613        let mut per_file = Cost::default();
1614        physical_facts(&meta, &mut per_file);
1615        uncompressed += per_file.uncompressed.unwrap_or(0);
1616        row_groups += per_file.row_groups.unwrap_or(0);
1617        if cost.codec.is_none() {
1618            cost.codec = per_file.codec;
1619        }
1620    }
1621    // With the footers read, one-table is known. Separate tables make a place to look
1622    // inside (summed rows mean nothing). Only `multi` is reconsidered: hive files hold
1623    // one table by construction.
1624    if entry.kind == EntryKind::MultiFile && !crate::formats::schema_union::is_nested(&per_file) {
1625        // Its own files' bytes, matching what the label and columns count.
1626        entry.size = Some(own_bytes);
1627        // Not one table, but column search should still find it: the count is its own
1628        // files', the names every one under it.
1629        entry.columns = columns;
1630        entry.cols_sampled = false;
1631        downgrade_to_directory(
1632            entry,
1633            (!own_top_level.is_empty()).then_some(own_top_level.len()),
1634        );
1635        return;
1636    }
1637
1638    entry.rows = Some(rows);
1639    entry.cols = Some(top_level.len() + partition_columns_beyond(entry, &top_level));
1640    entry.size = Some(bytes);
1641    entry.columns = columns;
1642    cost.uncompressed = (uncompressed > 0).then_some(uncompressed);
1643    cost.row_groups = (row_groups > 0).then_some(row_groups);
1644    cost.partitions = entry.cost.partitions.take();
1645    entry.cost = cost;
1646}
1647
1648/// Partition keys the files lack, hoisted in as columns by the open, so the width
1649/// counts them (`12 × 4`, not `12 × 2`).
1650fn partition_columns_beyond(entry: &Entry, top_level: &[String]) -> usize {
1651    entry.cost.partitions.as_ref().map_or(0, |layout| {
1652        layout
1653            .keys
1654            .iter()
1655            .filter(|key| !top_level.contains(key))
1656            .count()
1657    })
1658}
1659
1660/// The files of `dir` itself, out of a walk that went below it.
1661fn direct_children(files: &[PathBuf], dir: &Path) -> Vec<PathBuf> {
1662    files
1663        .iter()
1664        .filter(|f| f.parent() == Some(dir))
1665        .cloned()
1666        .collect()
1667}
1668
1669/// Which of `files` to read for a sample: the ends and the middle. Sorted names
1670/// cluster one table's files at the start, and the last is the newest. Three reads.
1671pub(crate) fn spread(files: usize) -> Vec<usize> {
1672    let mut picks = match files {
1673        0 => Vec::new(),
1674        n => vec![0, n / 2, n - 1],
1675    };
1676    picks.dedup();
1677    picks
1678}
1679
1680/// Whether a few files' top-level columns are one table. `None` from fewer than two:
1681/// the directory keeps its name-based kind.
1682pub(crate) fn one_table_from(footers: &[Vec<String>]) -> Option<bool> {
1683    (footers.len() >= 2).then(|| crate::formats::schema_union::is_nested(footers))
1684}
1685
1686/// The footers of the [`spread`] of a directory too large to read every one of.
1687fn sample_footers(files: &[PathBuf]) -> Vec<crate::formats::parquet_footer::Footer> {
1688    spread(files.len())
1689        .into_iter()
1690        .filter_map(|i| crate::formats::parquet_footer::read_parquet_metadata(&files[i]))
1691        .collect()
1692}
1693
1694/// Ask a footerless directory whether its files are one table by their header names:
1695/// [`one_table_from`]'s rule for CSV and NDJSON, so `Enter` does not promise a table
1696/// the read then refuses. Silence (too few files, costly schema, a parse failure)
1697/// keeps the name-based kind, safe because the read unions by name and widens types
1698/// (`crate::formats::readers::polars::union_of_files`).
1699fn judge_by_names(entry: &mut Entry, as_read: &crate::formats::schema_union::ReadAs) {
1700    if entry.kind != EntryKind::MultiFile {
1701        return;
1702    }
1703    let Some(format) = entry
1704        .holds
1705        .one_format()
1706        .and_then(crate::FileFormat::from_name)
1707    else {
1708        return;
1709    };
1710    // The directory's own files, which the label counts and the open reads; `MultiFile`
1711    // is flat by construction and has one format, so only `One` arrives.
1712    let DirectoryFormat::One(_, files) = directory_format(&entry.path) else {
1713        return;
1714    };
1715    // Read as the following open will: header placement decides the names.
1716    let sampled = crate::formats::schema_union::sample_files(&files, format, as_read);
1717    if sampled.nests == Some(false) {
1718        // The sample's columns, for column search, as the Parquet path keeps when
1719        // downgrading; from the spread, so cost does not grow with the directory.
1720        let cols = (!sampled.columns.is_empty()).then_some(sampled.columns.len());
1721        // A floor only when files went unopened.
1722        entry.cols_sampled = sampled.read < files.len();
1723        entry.columns = sampled.columns;
1724        downgrade_to_directory(entry, cols);
1725    }
1726}
1727
1728/// A directory of separate tables becomes a place to look inside: no row count (a sum
1729/// of unrelated tables), but the union's column count (`15 parquet · 72 columns`).
1730/// `cols` is passed in because leaf paths cannot be split back into top-level
1731/// columns (see [`top_level_names`]).
1732fn downgrade_to_directory(entry: &mut Entry, cols: Option<usize>) {
1733    entry.kind = EntryKind::Directory;
1734    entry.rows = None;
1735    entry.cols = cols;
1736    entry.cost = Cost {
1737        partitions: entry.cost.partitions.take(),
1738        ..Cost::default()
1739    };
1740}
1741
1742/// The columns a reader sees: the schema's own top-level fields.
1743///
1744/// Not the leaves a footer names, and not those leaves split on a dot either. Leaves
1745/// counted directly double for a directory whose writer changed — the same nested column
1746/// written by parquet-mr and by Arrow gives `inputs.list.element.address` in one file and
1747/// `inputs.bag.array_element.address` in the other, and a union of leaf paths holds both.
1748/// Splitting the dotted path fixes that and breaks something else: a column literally
1749/// named `user.id` is one column, and so is a struct `user` with a field `id`, and the
1750/// string cannot tell them apart.
1751///
1752/// The schema knows. `fields()` is the root's own children, which is what a struct counts
1753/// as here, what `schema_preview` lists in the details pane, and what the table shows.
1754fn top_level_names(meta: &crate::formats::parquet_footer::Footer) -> Vec<String> {
1755    meta.schema_descr
1756        .fields()
1757        .iter()
1758        .map(|field| field.name().to_string())
1759        .collect()
1760}
1761
1762/// Every column name any of the files has, in the order they first appear.
1763fn union_of(per_file: &[Vec<String>]) -> Vec<String> {
1764    let mut seen = std::collections::HashSet::new();
1765    per_file
1766        .iter()
1767        .flatten()
1768        .filter(|name| seen.insert(name.as_str()))
1769        .cloned()
1770        .collect()
1771}
1772
1773/// The Parquet files under `dir`, sorted, as deep as a dataset goes, stopping one past
1774/// the budget (which says there are too many to count). The open's own walk.
1775fn parquet_files_under(dir: &Path) -> Vec<PathBuf> {
1776    let mut files = crate::formats::dataset_files::LocalFiles::new(dir)
1777        .first_files(MAX_WALK_DEPTH as usize + 1, MAX_FOOTERS_PER_DATASET);
1778    // The walk takes a directory entry's own type; a file this reads must be one.
1779    files.retain(|p| is_regular_file(p));
1780    files
1781}
1782
1783/// Measure a dataset from footers an open kept: rows, width, size, row groups, and
1784/// whether its files are one table; nothing is read.
1785fn measure_from_footers(
1786    entry: &mut Entry,
1787    files: &[crate::formats::dataset_files::DatasetFile],
1788    footers: &[Option<crate::formats::schema_union::FileFooter>],
1789) {
1790    let per_file: Vec<Vec<String>> = footers
1791        .iter()
1792        .flatten()
1793        .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
1794        .collect();
1795    let columns = union_of(&per_file);
1796    if entry.kind == EntryKind::MultiFile && !crate::formats::schema_union::is_nested(&per_file) {
1797        // As a full footer read judges it: label, width and size count the directory's own
1798        // files.
1799        let own: Vec<usize> = files
1800            .iter()
1801            .enumerate()
1802            .filter(|(_, f)| Path::new(&f.key).parent() == Some(entry.path.as_path()))
1803            .map(|(i, _)| i)
1804            .collect();
1805        entry.size = Some(own.iter().map(|&i| files[i].size).sum());
1806        let own_columns = union_of(
1807            &own.iter()
1808                .filter_map(|&i| footers[i].as_ref())
1809                .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
1810                .collect::<Vec<_>>(),
1811        );
1812        entry.columns = columns;
1813        entry.cols_sampled = false;
1814        downgrade_to_directory(
1815            entry,
1816            (!own_columns.is_empty()).then_some(own_columns.len()),
1817        );
1818        return;
1819    }
1820    let footers: Vec<&crate::formats::schema_union::FileFooter> =
1821        footers.iter().flatten().collect();
1822    let uncompressed: u64 = footers
1823        .iter()
1824        .flat_map(|f| &f.column_bytes)
1825        .map(|(_, bytes)| *bytes as u64)
1826        .sum();
1827    let row_groups: usize = footers.iter().map(|f| f.row_group_rows.len()).sum();
1828    entry.rows = Some(footers.iter().map(|f| f.rows()).sum());
1829    entry.cols = Some(columns.len() + partition_columns_beyond(entry, &columns));
1830    entry.size = Some(files.iter().map(|f| f.size).sum());
1831    entry.columns = columns;
1832    entry.cols_sampled = false;
1833    entry.cost = Cost {
1834        uncompressed: (uncompressed > 0).then_some(uncompressed),
1835        row_groups: (row_groups > 0).then_some(row_groups),
1836        partitions: entry.cost.partitions.take(),
1837        ..Cost::default()
1838    };
1839}
1840
1841/// Fill in a Parquet file's counts from its footer, reading no column data. Other
1842/// formats keep `None` (a CSV's rows need a scan).
1843pub fn enrich_parquet(entry: &mut Entry) {
1844    if entry.kind != EntryKind::File {
1845        return;
1846    }
1847    if !is_parquet_path(&entry.path) {
1848        return;
1849    }
1850    if !is_regular_file(&entry.path) {
1851        return;
1852    }
1853    if let Some(meta) = crate::formats::parquet_footer::read_parquet_metadata(&entry.path) {
1854        entry.rows = Some(meta.num_rows);
1855        entry.columns = column_names(&meta);
1856        // Top-level columns, as a directory's row reports them: `schema_descr` names leaves
1857        // (one struct of three fields would count four). See [`top_level_names`].
1858        entry.cols = Some(top_level_names(&meta).len());
1859        physical_facts(&meta, &mut entry.cost);
1860    }
1861}
1862
1863/// A file of tables' tables (SQLite schema, NumPy archive directory): how many, whether
1864/// Enter opens one, and that one's columns. A file whose bytes contradict its name (a
1865/// `.db` that is not SQLite) is unopenable.
1866pub fn enrich_tables(entry: &mut Entry) {
1867    if entry.kind != EntryKind::File || entry.table.is_some() {
1868        return;
1869    }
1870    let named = data_format(&entry.path);
1871    // A name that says a format holding no tables settles it, past no compression
1872    // suffix (an archive may hold anything): no file is opened to ask.
1873    if named.is_some_and(|f| !f.holds_tables())
1874        && crate::CompressionFormat::from_extension(&entry.path).is_none()
1875    {
1876        return;
1877    }
1878    if !is_regular_file(&entry.path) {
1879        return;
1880    }
1881    let Some(format) = crate::formats::members::holder(&entry.path) else {
1882        if named.is_some_and(|f| f.holds_tables() && crate::formats::readers::of(f).bytes_decide) {
1883            entry.kind = EntryKind::Other;
1884        }
1885        return;
1886    };
1887    let Ok(tables) = crate::formats::members::tables(&entry.path, format) else {
1888        return;
1889    };
1890    let own: Vec<&crate::formats::sqlite::Table> = tables.iter().filter(|t| !t.internal).collect();
1891    entry.cost.tables = Some(own.len());
1892    entry.cost.opens_one = format.opens_one_table();
1893    if let [one] = own.as_slice()
1894        && !one.columns.is_empty()
1895    {
1896        entry.columns = one.columns.iter().map(|(name, _)| name.clone()).collect();
1897        entry.cols = Some(entry.columns.len());
1898    }
1899}
1900
1901/// A file of tables listed as rows (a database's tables and views, an archive's arrays),
1902/// each at its path inside the file; SQLite's own marked hidden.
1903pub fn database_rows(file: &Path) -> Vec<Entry> {
1904    let Some(format) = crate::formats::members::holder(file) else {
1905        return Vec::new();
1906    };
1907    let Ok(mut tables) = crate::formats::members::tables(file, format) else {
1908        return Vec::new();
1909    };
1910    // A database's tables by name; an archive's arrays in the order they were saved.
1911    if format
1912        .descriptor()
1913        .tables
1914        .as_ref()
1915        .is_some_and(|t| t.by_name)
1916    {
1917        tables.sort_by_cached_key(|t| t.name.to_lowercase());
1918    }
1919    let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
1920    tables
1921        .into_iter()
1922        .map(|table| table_entry(file, format, table, modified))
1923        .collect()
1924}
1925
1926/// A Hugging Face cache directory's splits as rows at their paths inside it
1927/// (`cache/test`), opened with `--table`. Empty otherwise, or for one split (its door
1928/// opens it).
1929pub fn split_rows(dir: &Path) -> Vec<Entry> {
1930    let splits = crate::formats::hf_splits::cache_splits(dir);
1931    if splits.len() < 2 {
1932        return Vec::new();
1933    }
1934    splits
1935        .into_iter()
1936        .map(|split| split_entry(dir, split))
1937        .collect()
1938}
1939
1940/// The row of a split named by its path inside its cache directory (a recent); `None`
1941/// if none.
1942pub fn split_row(path: &Path) -> Option<Entry> {
1943    let (dir, split) = crate::formats::hf_splits::split_place(path)?;
1944    Some(split_entry(&dir, split))
1945}
1946
1947fn split_entry(dir: &Path, split: String) -> Entry {
1948    let mut entry = Entry::new(dir.join(&split), EntryKind::File).with_name(split);
1949    entry.table = Some(TableOf {
1950        format: Some(crate::FileFormat::Arrow),
1951        kind: "split".to_string(),
1952        internal: false,
1953    });
1954    entry
1955}
1956
1957/// A spec-read file's variants as rows at their paths inside it (`day.itch/add`),
1958/// opened with `--table`. Empty for other files.
1959pub fn variant_rows(file: &Path, formats: &crate::formats::Registry) -> Vec<Entry> {
1960    let Some((spec, tables)) = crate::formats::members::variants(file, formats) else {
1961        return Vec::new();
1962    };
1963    let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
1964    tables
1965        .into_iter()
1966        .map(|table| variant_entry(file, &spec, table, modified))
1967        .collect()
1968}
1969
1970/// The row of a variant named by its path inside its file (a recent); `None` if none.
1971pub fn variant_row(path: &Path, formats: &crate::formats::Registry) -> Option<Entry> {
1972    let (file, name) = crate::formats::members::split_variant(path, formats)?;
1973    let (spec, tables) = crate::formats::members::variants(&file, formats)?;
1974    let table = tables.into_iter().find(|t| t.name == name)?;
1975    let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
1976    let mut entry = variant_entry(&file, &spec, table, modified);
1977    entry.path = path.to_path_buf();
1978    Some(entry)
1979}
1980
1981fn variant_entry(
1982    file: &Path,
1983    spec: &str,
1984    table: crate::formats::sqlite::Table,
1985    modified: Option<std::time::SystemTime>,
1986) -> Entry {
1987    let mut entry = Entry::new(
1988        crate::formats::members::place(file, &table.name),
1989        EntryKind::File,
1990    )
1991    .with_name(table.name);
1992    entry.modified = modified;
1993    entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
1994    entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
1995    entry.format_spec = Some(spec.to_string());
1996    entry.table = Some(TableOf {
1997        format: None,
1998        kind: table.kind,
1999        internal: false,
2000    });
2001    entry
2002}
2003
2004/// The row of a table named by its path inside its file (`app.db/users`, a recent);
2005/// `None` if none.
2006pub fn table_row(path: &Path) -> Option<Entry> {
2007    let (file, name) = crate::formats::members::split(path)?;
2008    let format = crate::formats::members::holder(&file)?;
2009    let table = crate::formats::members::tables(&file, format)
2010        .ok()?
2011        .into_iter()
2012        .find(|t| t.name == name)?;
2013    let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
2014    let mut entry = table_entry(&file, format, table, modified);
2015    entry.path = path.to_path_buf();
2016    Some(entry)
2017}
2018
2019fn table_entry(
2020    file: &Path,
2021    format: crate::FileFormat,
2022    table: crate::formats::sqlite::Table,
2023    modified: Option<std::time::SystemTime>,
2024) -> Entry {
2025    let mut entry = Entry::new(
2026        crate::formats::members::place(file, &table.name),
2027        EntryKind::File,
2028    )
2029    .with_name(table.name);
2030    entry.modified = modified;
2031    entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
2032    entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
2033    entry.table = Some(TableOf {
2034        format: Some(format),
2035        kind: table.kind,
2036        internal: table.internal,
2037    });
2038    entry
2039}
2040
2041/// The first bytes of `path`, as many as fit in `buf`. A short read is the whole file.
2042fn read_head<'a>(path: &Path, buf: &'a mut [u8]) -> Option<&'a [u8]> {
2043    use std::io::Read;
2044    let mut file = std::fs::File::open(path).ok()?;
2045    let mut filled = 0;
2046    loop {
2047        match file.read(&mut buf[filled..]) {
2048            Ok(0) => break,
2049            Ok(n) => filled += n,
2050            Err(_) => return None,
2051        }
2052        if filled == buf.len() {
2053            break;
2054        }
2055    }
2056    Some(&buf[..filled])
2057}
2058
2059/// Whether an Arrow file is an IPC stream (an IPC file starts `ARROW1`): eight bytes,
2060/// so the listing can say which will be converted.
2061fn enrich_arrow(entry: &mut Entry) {
2062    if entry.kind != EntryKind::File
2063        || data_format(&entry.path) != Some(crate::FileFormat::Arrow)
2064        || crate::CompressionFormat::from_extension(&entry.path).is_some()
2065        || !is_regular_file(&entry.path)
2066    {
2067        return;
2068    }
2069    let mut head = [0u8; 8];
2070    if let Some(head) = read_head(&entry.path, &mut head) {
2071        entry.cost.ipc_stream = !head.starts_with(b"ARROW1");
2072    }
2073}
2074
2075/// Pull layout and compression from an already-read footer: what reading the file
2076/// will do, beyond its size.
2077pub fn physical_facts(meta: &crate::formats::parquet_footer::Footer, cost: &mut Cost) {
2078    if meta.row_groups.is_empty() {
2079        return;
2080    }
2081    cost.row_groups = Some(meta.row_groups.len());
2082
2083    let mut uncompressed: u64 = 0;
2084    let mut codecs: Vec<String> = Vec::new();
2085    for rg in &meta.row_groups {
2086        uncompressed = uncompressed.saturating_add(rg.total_byte_size() as u64);
2087        for cc in rg.parquet_columns() {
2088            let codec = format!("{:?}", cc.compression()).to_lowercase();
2089            if !codecs.contains(&codec) {
2090                codecs.push(codec);
2091            }
2092        }
2093    }
2094    if uncompressed > 0 {
2095        cost.uncompressed = Some(uncompressed);
2096    }
2097    // Usually one codec; when not, say so rather than imply uniformity.
2098    cost.codec = match codecs.len() {
2099        0 => None,
2100        1 => Some(codecs.remove(0)),
2101        n => Some(format!("mixed ({n})")),
2102    };
2103}
2104
2105/// Outermost directories examined to describe a hive partitioning: enough to name the
2106/// keys and show the shape; a decade of daily partitions is not worth counting on a
2107/// share.
2108const MAX_PARTITION_DIRS: usize = 512;
2109
2110/// Describe how a hive dataset is partitioned, from directory names alone.
2111pub fn partition_layout(dir: &Path) -> Option<Partitions> {
2112    let iter = std::fs::read_dir(dir).ok()?;
2113    let mut values: Vec<String> = Vec::new();
2114    let mut keys: Vec<String> = Vec::new();
2115    let mut count = 0usize;
2116    let mut more = false;
2117
2118    for entry in iter.flatten() {
2119        if count >= MAX_PARTITION_DIRS {
2120            more = true;
2121            break;
2122        }
2123        let name = entry.file_name().to_string_lossy().into_owned();
2124        let Some((key, value)) = name.split_once('=') else {
2125            continue;
2126        };
2127        if !entry.path().is_dir() {
2128            continue;
2129        }
2130        if keys.is_empty() {
2131            keys.push(key.to_string());
2132            // Only the first partition is descended for nested keys: one is representative.
2133            keys.extend(nested_keys(&entry.path()));
2134        }
2135        values.push(value.to_string());
2136        count += 1;
2137    }
2138
2139    if keys.is_empty() {
2140        return None;
2141    }
2142    values.sort();
2143    values.dedup();
2144    Some(Partitions {
2145        keys,
2146        first_key_values: values,
2147        count,
2148        more,
2149    })
2150}
2151
2152/// Partition keys below `dir`, following the first child at each level.
2153fn nested_keys(dir: &Path) -> Vec<String> {
2154    let mut keys = Vec::new();
2155    let mut current = dir.to_path_buf();
2156    // Bounded: each level is a directory read.
2157    for _ in 0..6 {
2158        let Ok(iter) = std::fs::read_dir(&current) else {
2159            break;
2160        };
2161        let Some(child) = iter
2162            .flatten()
2163            .find(|e| e.file_name().to_string_lossy().contains('=') && e.path().is_dir())
2164        else {
2165            break;
2166        };
2167        let name = child.file_name().to_string_lossy().into_owned();
2168        let Some((key, _)) = name.split_once('=') else {
2169            break;
2170        };
2171        keys.push(key.to_string());
2172        current = child.path();
2173    }
2174    keys
2175}
2176
2177/// Render a row count compactly (`2.4M`).
2178pub fn format_rows(rows: usize) -> String {
2179    let r = rows as f64;
2180    if rows >= 1_000_000_000 {
2181        format!("{:.1}B", r / 1e9)
2182    } else if rows >= 1_000_000 {
2183        format!("{:.1}M", r / 1e6)
2184    } else if rows >= 10_000 {
2185        format!("{:.0}k", r / 1e3)
2186    } else if rows >= 1_000 {
2187        // Below ten thousand show the exact count: 3,653 days is ten years; "4k" is not.
2188        let mut out = String::new();
2189        let digits = rows.to_string();
2190        for (i, c) in digits.chars().enumerate() {
2191            if i > 0 && (digits.len() - i).is_multiple_of(3) {
2192                out.push(',');
2193            }
2194            out.push(c);
2195        }
2196        out
2197    } else {
2198        rows.to_string()
2199    }
2200}
2201
2202/// Render "how long ago" compactly (`2d`, `3h`).
2203pub fn format_age(t: std::time::SystemTime) -> String {
2204    let Ok(elapsed) = t.elapsed() else {
2205        return String::new();
2206    };
2207    let secs = elapsed.as_secs();
2208    if secs < 60 {
2209        "now".to_string()
2210    } else if secs < 3600 {
2211        format!("{}m", secs / 60)
2212    } else if secs < 86_400 {
2213        format!("{}h", secs / 3600)
2214    } else if secs < 86_400 * 365 {
2215        format!("{}d", secs / 86_400)
2216    } else {
2217        format!("{}y", secs / (86_400 * 365))
2218    }
2219}
2220
2221/// Column name and type, for the home screen's preview pane.
2222pub type SchemaPreview = Vec<(String, polars::prelude::DataType)>;
2223
2224/// The preview of a table inside a file of tables, or of such a file (or NumPy array
2225/// file) as it opens. `None` for other entries, `Some(None)` when there is nothing to
2226/// show.
2227fn table_preview(entry: &Entry) -> Option<Option<SchemaPreview>> {
2228    let (file, format, name) = match &entry.table {
2229        Some(table) => match (crate::formats::members::split(&entry.path), table.format) {
2230            (Some((file, _)), Some(format)) => (file, format, Some(entry.name.as_str())),
2231            _ => return Some(None),
2232        },
2233        None if is_regular_file(&entry.path) => {
2234            let format = crate::formats::members::holder(&entry.path).or_else(|| {
2235                data_format(&entry.path).filter(|f| {
2236                    f.holds_tables() && crate::formats::readers::of(*f).table_schema.is_some()
2237                })
2238            })?;
2239            (entry.path.clone(), format, None)
2240        }
2241        None => return None,
2242    };
2243    Some(
2244        crate::formats::readers::of(format)
2245            .table_schema
2246            .and_then(|schema| schema(&file, name)),
2247    )
2248}
2249
2250/// The first Parquet file at or under `dir`, bounded in breadth and depth so thousands
2251/// of partitions cost what three do.
2252fn first_parquet_under(dir: &Path, depth: u8) -> Option<PathBuf> {
2253    if depth > MAX_WALK_DEPTH {
2254        return None;
2255    }
2256    let mut subdirs = Vec::new();
2257    for entry in std::fs::read_dir(dir).ok()?.flatten().take(64) {
2258        let path = entry.path();
2259        if path.is_dir() {
2260            subdirs.push(path);
2261        } else if is_parquet_path(&path) && is_regular_file(&path) {
2262            return Some(path);
2263        }
2264    }
2265    subdirs.sort();
2266    subdirs
2267        .into_iter()
2268        .take(4)
2269        .find_map(|d| first_parquet_under(&d, depth + 1))
2270}
2271
2272/// Column names from a Parquet footer.
2273pub fn column_names(meta: &crate::formats::parquet_footer::Footer) -> Vec<String> {
2274    meta.schema_descr
2275        .columns()
2276        .iter()
2277        .map(|c| c.path_in_schema.join("."))
2278        .collect()
2279}
2280
2281/// Whether `path` is a regular file safe to open: a FIFO, device or socket named
2282/// `.parquet` would block or worse, so reads are gated on kind (stat never blocks
2283/// like an open can).
2284fn is_regular_file(path: &Path) -> bool {
2285    std::fs::metadata(path)
2286        .map(|m| m.file_type().is_file())
2287        .unwrap_or(false)
2288}
2289
2290/// A dataset's column names and types without reading data; Parquet only, `None`
2291/// otherwise (the UI says so).
2292pub fn schema_preview(entry: &Entry) -> Option<SchemaPreview> {
2293    // `SerReader` is what brings `ParquetReader::new` into scope.
2294    use polars::prelude::{ParquetReader, Schema, SchemaExt, SerReader};
2295
2296    let file_path = match entry.kind {
2297        EntryKind::File => {
2298            if let Some(preview) = table_preview(entry) {
2299                return preview;
2300            }
2301            if !is_parquet_path(&entry.path) {
2302                return None;
2303            }
2304            entry.path.clone()
2305        }
2306        EntryKind::Hive | EntryKind::MultiFile => first_parquet_under(&entry.path, 0)?,
2307        EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => return None,
2308        // One data file's schema is not a lake table's (Iceberg field IDs and Delta column
2309        // mapping rename columns).
2310        EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => return None,
2311    };
2312
2313    if !is_regular_file(&file_path) {
2314        return None;
2315    }
2316    let file = std::fs::File::open(&file_path).ok()?;
2317    let mut reader = ParquetReader::new(file);
2318    let arrow_schema = reader.schema().ok()?;
2319    let schema = Schema::from_arrow_schema(arrow_schema.as_ref());
2320    let mut preview: SchemaPreview = Vec::new();
2321    // A hive table opens with partition keys hoisted first, so the pane lists them there,
2322    // typed from the path as the scan infers them.
2323    if entry.kind == EntryKind::Hive
2324        && let Ok(below) = file_path.strip_prefix(&entry.path)
2325    {
2326        for part in below.parent().into_iter().flat_map(Path::components) {
2327            let part = part.as_os_str().to_string_lossy();
2328            if let Some((key, value)) = part.split_once('=')
2329                && !key.is_empty()
2330                && schema.get(key).is_none()
2331            {
2332                // The scan's own inference, dates and booleans included.
2333                let dtype = if value.is_empty() || value == "__HIVE_DEFAULT_PARTITION__" {
2334                    polars::prelude::DataType::String
2335                } else {
2336                    polars::io::csv::read::schema_inference::infer_field_schema(value, true, false)
2337                };
2338                preview.push((key.to_string(), dtype));
2339            }
2340        }
2341    }
2342    preview.extend(
2343        schema
2344            .iter()
2345            .map(|(name, dtype)| (name.to_string(), dtype.clone())),
2346    );
2347    Some(preview)
2348}
2349
2350#[cfg(test)]
2351mod classification_tests;