Skip to main content

datui_lib/home/
discover.rs

1//! Dataset discovery for the home screen. Not a catalog: listings are computed from
2//! the filesystem when asked and forgotten at session end; between runs datui keeps
3//! only recent paths and measured shapes (`remembered`, valid while the files are
4//! unchanged). Every function scans one directory level, since data lives on slow
5//! mounts and huge partition trees.
6
7use std::path::{Path, PathBuf};
8
9/// Compression suffixes that may follow a data extension (`sales.csv.gz`).
10const COMPRESSION_EXTENSIONS: &[&str] = &["gz", "bz2", "xz", "zst", "zstd"];
11
12/// Upper bound on entries read from one directory, so a huge one cannot hang the UI.
13pub const MAX_ENTRIES_PER_DIR: usize = 5_000;
14
15/// What a home-screen row represents.
16#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
17#[serde(rename_all = "lowercase")]
18pub enum EntryKind {
19    /// A single data file.
20    File,
21    /// A file datui has no reader for: hidden on home until `Ctrl+A` shows it dimmed.
22    Other,
23    /// A directory of `key=value` partitions โ€” one dataset, not a tree to walk.
24    Hive,
25    /// A directory of similarly-shaped data files, openable as one table.
26    MultiFile,
27    /// A Delta Lake table: `_delta_log/` beside the data files.
28    Delta,
29    /// An Apache Iceberg table: `metadata/` holding the snapshots, `data/` the files.
30    Iceberg,
31    /// An Apache Hudi table: `.hoodie/` holding the timeline.
32    Hudi,
33    /// An ordinary directory, to descend into.
34    Directory,
35    /// Somewhere remote not looked at yet: classifying would read it, which blocks when
36    /// the network is gone, so it is offered as openable, unlabeled. Also what an
37    /// unrecognized kind deserializes as, so an older build can still read a dataset
38    /// index written by a newer one (see `CLASSIFIER_VERSION`).
39    #[serde(other)]
40    Unknown,
41}
42
43/// Bump whenever classification changes. A cached kind is restored without
44/// re-deriving it (a remote row cannot be read cheaply), so it is restored only when
45/// written by a build with the same version; measurements in the record (rows,
46/// columns, cost) survive regardless.
47pub const CLASSIFIER_VERSION: u32 = 7;
48
49impl EntryKind {
50    /// Short label shown next to the entry name.
51    pub fn label(self) -> &'static str {
52        match self {
53            // A file with no reader says nothing: it is dimmed, and Enter shows its bytes.
54            EntryKind::File | EntryKind::Other => "",
55            EntryKind::Hive => "hive",
56            EntryKind::MultiFile => "multi",
57            EntryKind::Delta => "delta",
58            EntryKind::Iceberg => "iceberg",
59            EntryKind::Hudi => "hudi",
60            EntryKind::Directory => "dir",
61            EntryKind::Unknown => "",
62        }
63    }
64
65    /// Whether selecting this entry opens a dataset rather than navigating. An unexamined
66    /// remote path counts: datui can open a prefix or hive directory directly.
67    pub fn is_dataset(self) -> bool {
68        !matches!(self, EntryKind::Directory | EntryKind::Other) && !self.is_lake_table()
69    }
70
71    /// Whether this row is known to be a dataset, for counting: unlike
72    /// [`EntryKind::is_dataset`] ("may this be opened"), a row nothing has looked into
73    /// does not count.
74    pub fn is_known_dataset(self) -> bool {
75        self != EntryKind::Unknown && self.is_dataset()
76    }
77
78    /// A Delta, Iceberg or Hudi table root: a log beside the data files that says which
79    /// of them are live, which datui does not read yet.
80    pub fn is_lake_table(self) -> bool {
81        matches!(
82            self,
83            EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi
84        )
85    }
86
87    /// The format's name for prose; `label` is the row's lowercase chip.
88    pub fn lake_name(self) -> Option<&'static str> {
89        match self {
90            EntryKind::Delta => Some("Delta"),
91            EntryKind::Iceberg => Some("Iceberg"),
92            EntryKind::Hudi => Some("Hudi"),
93            _ => None,
94        }
95    }
96}
97
98/// What one listing of a directory found, counted rather than judged. The row's label
99/// comes from here (`12 parquet`), true whether or not the files are one table.
100#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
101pub struct Holds {
102    /// Data files by format, commonest first, as [`crate::FileFormat::name`] strings so
103    /// records survive across builds.
104    #[serde(default)]
105    pub formats: Vec<(String, usize)>,
106    /// Subdirectories, partitions among them (`folders` in 0.4.0 development builds).
107    #[serde(default, alias = "folders")]
108    pub directories: usize,
109    /// `key=value` subdirectories, which are also counted in `directories`.
110    #[serde(default)]
111    pub partitions: usize,
112    /// Files datui has no reader for (a README, a script, a notebook): neither data nor a
113    /// writer's own.
114    #[serde(default)]
115    pub not_read: usize,
116    /// Files with no extension, neither data nor `not_read` by name: Spark and GBIF part
117    /// files, which the open reads by their bytes.
118    #[serde(default)]
119    pub unnamed: usize,
120    /// Entries skipped as a writer's own, and the first few by name for the pane.
121    #[serde(default)]
122    pub skipped: usize,
123    #[serde(default)]
124    pub skipped_names: Vec<String>,
125    /// The listing stopped at [`MAX_ENTRIES_PER_DIR`], so every count is a floor (`5000+`).
126    #[serde(default)]
127    pub truncated: bool,
128    /// A Hugging Face DatasetDict saved with `save_to_disk` (`dataset_dict.json` beside
129    /// split directories), read as Arrow, one split at a time.
130    #[serde(default)]
131    pub dataset_dict: bool,
132}
133
134/// How many skipped names are kept for the pane. Enough to recognise the convention.
135pub(crate) const SKIPPED_NAMES_SHOWN: usize = 4;
136
137impl Holds {
138    /// Data files of every format.
139    pub fn data_files(&self) -> usize {
140        self.formats.iter().map(|(_, n)| n).sum()
141    }
142
143    /// The one format this directory holds, when it holds exactly one.
144    pub fn one_format(&self) -> Option<&str> {
145        match self.formats.as_slice() {
146            [(name, _)] => Some(name),
147            _ => None,
148        }
149    }
150
151    /// The weights' format and file count when this directory is a model (one weight
152    /// format plus only JSON). See `is_model_directory`.
153    pub fn model_weights(&self) -> Option<(&str, usize)> {
154        if !is_model_directory(counts_names(self)) {
155            return None;
156        }
157        self.formats
158            .iter()
159            .find(|(name, _)| is_weights(name))
160            .map(|(name, count)| (name.as_str(), *count))
161    }
162
163    /// The directory row's label when its kind does not name itself: `12 parquet`,
164    /// `mixed`, or `dir` when no data is directly inside.
165    pub fn label(&self) -> String {
166        let more = if self.truncated { "+" } else { "" };
167        // A model's weights beside config and tokenizer JSON: labeled as the model, not
168        // `mixed`.
169        if let Some((name, count)) = self.model_weights() {
170            return format!("{count}{more} {name}");
171        }
172        match self.formats.as_slice() {
173            // `dir` says there is no data file inside; a cut listing cannot claim that. A
174            // directory of directories counts them.
175            [] if self.directories == 1 => format!("1 dir{more}"),
176            [] if self.directories > 1 => format!("{}{more} dirs", self.directories),
177            [] => format!("dir{more}"),
178            // The `+` hedges the whole claim: past the cap another format may lurk.
179            [(name, count)] => format!("{count}{more} {name}"),
180            // No `+`: `mixed` is not a count, and more files cannot unmake it.
181            _ => "mixed".to_string(),
182        }
183    }
184
185    /// Whether nothing has been counted: a file, or a directory not looked into. (An
186    /// empty directory that was looked into reads `dir` too.)
187    pub fn is_empty(&self) -> bool {
188        self.formats.is_empty()
189            && self.directories == 0
190            && self.skipped == 0
191            && self.not_read == 0
192            && self.unnamed == 0
193            // Every field, even those implied by others today: this guards a placeholder from
194            // erasing a row's count and should not depend on that invariant.
195            && self.partitions == 0
196            && self.skipped_names.is_empty()
197            && !self.dataset_dict
198            // A cut listing that found nothing still says more exists (`dir+`).
199            && !self.truncated
200    }
201
202    /// The details pane's `contains` line: data files by format, directories and
203    /// partitions. Unreadable files and writer markers are left out (they would read as a
204    /// warning); a row inside the directory says what is hidden.
205    pub fn line(&self, with_partitions: bool) -> Option<String> {
206        let more = if self.truncated { "+" } else { "" };
207        let mut parts: Vec<String> = self
208            .formats
209            .iter()
210            .map(|(name, count)| format!("{count}{more} {name}"))
211            .collect();
212        // Partitions are counted in `directories` too; name only the rest.
213        let plain = self.directories.saturating_sub(self.partitions);
214        if plain > 0 {
215            let word = if plain == 1 {
216                "directory"
217            } else {
218                "directories"
219            };
220            parts.push(format!("{plain}{more} {word}"));
221        }
222        if with_partitions && self.partitions > 0 {
223            let word = if self.partitions == 1 {
224                "partition"
225            } else {
226                "partitions"
227            };
228            parts.push(format!("{}{more} {word}", self.partitions));
229        }
230        (!parts.is_empty()).then(|| parts.join(" ยท "))
231    }
232}
233
234impl Entry {
235    /// Whether Enter lists the tables inside: a file of several that is not itself a
236    /// table. A file opening one of its tables, or a spec's variants file, opens on
237    /// Enter; โ†’ lists them.
238    pub fn enter_lists_tables(&self) -> bool {
239        self.kind == EntryKind::File
240            && self.cost.tables.is_some_and(|n| n > 1)
241            && !self.cost.opens_one
242            && self.format_spec.is_none()
243    }
244
245    /// Whether home hides this row until Ctrl+A: an unreadable file or a database's
246    /// internal table.
247    pub fn hidden_by_default(&self) -> bool {
248        self.kind == EntryKind::Other || self.table.as_ref().is_some_and(|t| t.internal)
249    }
250
251    /// The short label beside a row's name, saying what it holds: a looked-into
252    /// directory's count (`12 parquet`, `mixed`, `dir`), a lake table's or hive root's
253    /// format, else the kind.
254    pub fn label(&self) -> std::borrow::Cow<'static, str> {
255        match self.kind {
256            // See `opens_whole_directory`.
257            _ if self.opens_whole_directory => "".into(),
258            EntryKind::Directory | EntryKind::MultiFile if !self.holds.is_empty() => {
259                self.holds.label().into()
260            }
261            EntryKind::File if self.format_spec.is_some() => {
262                self.format_spec.clone().unwrap_or_default().into()
263            }
264            EntryKind::File if self.cost.tables.is_some() => {
265                let n = self.cost.tables.unwrap_or_default();
266                format!("{n} {}", if n == 1 { "table" } else { "tables" }).into()
267            }
268            // A file named for its contents (as a collection names one): its format, which the
269            // name no longer says.
270            EntryKind::File
271                if crate::FileFormat::from_path(Path::new(&self.name)).is_none()
272                    && crate::FileFormat::from_path(&self.path).is_some() =>
273            {
274                crate::FileFormat::from_path(&self.path)
275                    .map(crate::FileFormat::name)
276                    .unwrap_or_default()
277                    .into()
278            }
279            kind => kind.label().into(),
280        }
281    }
282}
283
284/// One row on the home screen.
285#[derive(Debug, Clone)]
286pub struct Entry {
287    pub path: PathBuf,
288    pub kind: EntryKind,
289    /// Display name โ€” the file or directory name, not the full path.
290    pub name: String,
291    /// Size in bytes; for multi-file datasets the sum of files inspected, a floor.
292    pub size: Option<u64>,
293    pub modified: Option<std::time::SystemTime>,
294    /// Row count, when it can be had without reading data (Parquet footers only).
295    pub rows: Option<usize>,
296    /// Column count, same caveat.
297    pub cols: Option<usize>,
298    /// Whether `cols` came from a spread of the directory (ends and middle, past the
299    /// footer budget), so it is a floor, shown as `6+`.
300    pub cols_sampled: bool,
301    /// Column names when free to get (a Parquet footer carries them with the row count).
302    pub columns: Vec<String>,
303    /// What opening this costs: where it lives, how it is stored and laid out, from bytes
304    /// already read.
305    pub cost: Cost,
306    /// What one listing found, for a directory; empty for a file or an unexamined
307    /// directory.
308    pub holds: Holds,
309    /// Whether this row is the door opening the browsed directory. It carries no label:
310    /// labels count what is directly inside, and this row reads all of it.
311    pub opens_whole_directory: bool,
312    /// The format spec that reads this file, by glob or by magic.
313    pub format_spec: Option<String>,
314    /// A table inside a file of tables (SQLite, a NumPy archive): its path is the file's
315    /// with the table's name appended.
316    pub table: Option<TableOf>,
317    /// Whether a measurement from [`crate::home::HomeState::enriched`] is folded into
318    /// this row, so the passes asking which rows to measure need not look it up.
319    pub measured: bool,
320}
321
322/// What a row inside a file of tables says about its table.
323#[derive(Debug, Clone, PartialEq, Eq)]
324pub struct TableOf {
325    /// The containing file's format; `None` for a spec's variant (see
326    /// [`Entry::format_spec`]).
327    pub format: Option<crate::FileFormat>,
328    /// What the file calls it: SQLite's `table`, `view`, `virtual` or `shadow`, or a NumPy
329    /// archive's `array`.
330    pub kind: String,
331    /// SQLite's own (schema, statistics, shadow tables): hidden until Ctrl+A, opened like
332    /// any other.
333    pub internal: bool,
334}
335
336/// What pressing Enter on a dataset will cost (200 MB of zstd Parquet is gigabytes in
337/// memory, and NFS is not tmpfs), all from what datui already reads: the mount table
338/// and the footer.
339#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
340pub struct Cost {
341    /// Filesystem or URL scheme: `nfs4`, `ext4`, `tmpfs`, `fuse.sshfs`, `s3`.
342    #[serde(default, skip_serializing_if = "Option::is_none")]
343    pub source: Option<String>,
344    /// Bytes once decompressed, as against on disk.
345    #[serde(default, skip_serializing_if = "Option::is_none")]
346    pub uncompressed: Option<u64>,
347    /// Compression codec, as the file itself names it.
348    #[serde(default, skip_serializing_if = "Option::is_none")]
349    pub codec: Option<String>,
350    /// Row groups: one huge group cannot be read in parallel or skipped; thousands of
351    /// tiny ones cost overhead.
352    #[serde(default, skip_serializing_if = "Option::is_none")]
353    pub row_groups: Option<usize>,
354    /// Partition layout, for a hive dataset.
355    #[serde(default, skip_serializing_if = "Option::is_none")]
356    pub partitions: Option<Partitions>,
357    /// Tables of its own, for a file of tables: one opens, several are listed.
358    #[serde(default, skip_serializing_if = "Option::is_none")]
359    pub tables: Option<usize>,
360    /// Whether a file of several tables opens one (a workbook's first sheet): Enter opens
361    /// it and โ†’ lists them.
362    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
363    pub opens_one: bool,
364    /// An Arrow IPC stream (converted before scanning) rather than an IPC file (scanned in
365    /// place), from its first bytes.
366    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
367    pub ipc_stream: bool,
368}
369
370/// How opening a file row will read it. See [`how_read`].
371#[derive(Debug, Clone, Copy, PartialEq, Eq)]
372pub struct HowRead {
373    pub mode: crate::ReadMode,
374    /// A remote file that is downloaded whole before it is read.
375    pub download: bool,
376}
377
378/// How opening `entry` reads it ([`crate::FileFormat::read_mode`] for its format and
379/// storage), and whether a remote one is downloaded first. `None` unless a file's name
380/// says its format.
381pub fn how_read(entry: &Entry) -> Option<HowRead> {
382    use crate::Stored;
383    if entry.kind != EntryKind::File {
384        return None;
385    }
386    let stored = if crate::CompressionFormat::from_extension(&entry.path).is_some() {
387        Stored::Compressed { in_memory: false }
388    } else if entry.cost.ipc_stream {
389        Stored::Stream
390    } else {
391        Stored::Plain
392    };
393    let choice = match &entry.format_spec {
394        Some(name) => crate::cli::FormatChoice::Spec(name.clone()),
395        // A table inside a file of tables, at its path inside it (`shop.db/orders`).
396        None if entry.table.is_some() => {
397            crate::cli::FormatChoice::Builtin(entry.table.as_ref().and_then(|t| t.format)?)
398        }
399        None => crate::cli::FormatChoice::Builtin(data_format(&entry.path)?),
400    };
401    let mode = choice.read_mode(stored)?;
402    let download = match crate::cloud::source::input_source(&entry.path) {
403        crate::cloud::source::InputSource::Local(_) => false,
404        crate::cloud::source::InputSource::Http(_) => {
405            choice.http_file() == crate::RemoteRead::Downloaded
406        }
407        // A remote Arrow file is not peeked at here, so it is taken for an IPC file.
408        _ => choice.bucket_object(stored) == crate::RemoteRead::Downloaded,
409    };
410    Some(HowRead { mode, download })
411}
412
413/// How a hive dataset is laid out, from directory names alone.
414#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
415pub struct Partitions {
416    /// Partition keys, outermost first: `["year", "month"]`.
417    pub keys: Vec<String>,
418    /// Distinct values seen for the outermost key, sorted; bounded, so not necessarily
419    /// all.
420    pub first_key_values: Vec<String>,
421    /// Directories counted at the outermost level.
422    pub count: usize,
423    /// The count stopped at a limit; there are more.
424    pub more: bool,
425}
426
427impl Entry {
428    /// A plain directory row โ€” somewhere to step into, with nothing read from it.
429    pub fn directory(path: &Path) -> Self {
430        Self::new(path.to_path_buf(), EntryKind::Directory)
431    }
432
433    /// A file entry with a chosen display name, for tests.
434    pub fn for_test(path: &Path, name: &str) -> Self {
435        Self::new(path.to_path_buf(), EntryKind::File).with_name(name)
436    }
437
438    /// The row under a name other than its path's last part.
439    pub(crate) fn with_name(mut self, name: impl Into<String>) -> Self {
440        self.name = name.into();
441        self
442    }
443
444    pub(crate) fn new(path: PathBuf, kind: EntryKind) -> Self {
445        let name = path
446            .file_name()
447            .map(|n| n.to_string_lossy().into_owned())
448            .unwrap_or_else(|| path.to_string_lossy().into_owned());
449        Self {
450            path,
451            kind,
452            name,
453            size: None,
454            modified: None,
455            rows: None,
456            cols: None,
457            cols_sampled: false,
458            columns: Vec::new(),
459            cost: Cost::default(),
460            holds: Default::default(),
461            opens_whole_directory: false,
462            measured: false,
463            format_spec: None,
464            table: None,
465        }
466    }
467
468    /// Attach size and mtime from a directory entry's metadata.
469    pub(crate) fn with_fs_metadata(mut self, meta: &std::fs::Metadata) -> Self {
470        if meta.is_file() {
471            self.size = Some(meta.len());
472        }
473        self.modified = meta.modified().ok();
474        self
475    }
476}
477
478/// Whether a key or path is Parquet: named `.parquet`, or an extensionless part file
479/// in a `.parquet` directory (Spark, GBIF: `occurrence.parquet/000001`). Hidden and
480/// job files (`_SUCCESS`, `.crc`) are not.
481pub fn is_parquet_key(key: &str) -> bool {
482    let key = key.trim_end_matches('/');
483    let (directory, name) = match key.rsplit_once('/') {
484        Some((directory, name)) => (directory, name),
485        None => ("", key),
486    };
487    if is_bookkeeping(name) {
488        return false;
489    }
490    if name.to_ascii_lowercase().ends_with(".parquet") {
491        return true;
492    }
493    let directory_name = directory.rsplit('/').next().unwrap_or(directory);
494    !name.contains('.') && directory_name.to_ascii_lowercase().ends_with(".parquet")
495}
496
497#[cfg(test)]
498mod parquet_key_tests {
499    use super::is_parquet_key;
500
501    #[test]
502    fn parquet_without_an_extension_is_known_by_its_directory() {
503        assert!(is_parquet_key(
504            "occurrence/2026-09-01/occurrence.parquet/000001"
505        ));
506        assert!(is_parquet_key("data/part-0.parquet"));
507        assert!(is_parquet_key("DATA/PART-0.PARQUET"));
508        assert!(!is_parquet_key(
509            "occurrence/2026-09-01/occurrence.parquet/_SUCCESS"
510        ));
511        assert!(!is_parquet_key("occurrence.parquet/.part-0.crc"));
512        assert!(!is_parquet_key("occurrence/2026-09-01/citation.txt"));
513        assert!(!is_parquet_key("notes/000001"));
514    }
515}
516
517/// What a file with no usable extension is, from its first bytes (Parquet, Arrow,
518/// Avro and ORC carry signatures; CSV and JSON have none and are not guessed). Asked
519/// only of a directory being opened, never one being looked at; which signatures a
520/// listing trusts is each format's call (`crate::formats::readers::Trusted::listing`).
521pub fn sniff_format(path: &Path) -> Option<crate::FileFormat> {
522    crate::formats::readers::sniff_file(path, crate::formats::readers::Asked::Listing)
523}
524
525/// What a listing finds a file to be by its first bytes.
526#[derive(Debug, Clone)]
527pub enum Sniffed {
528    /// A format datui reads.
529    Format,
530    /// A format spec's, which reads it.
531    Spec(std::sync::Arc<crate::formats::Spec>),
532}
533
534/// [`sniff_format`], else the format spec an open would pick (by glob, else by magic
535/// and `match.where`), from one read of the file's head.
536pub fn sniff_listed(path: &Path, formats: &crate::formats::Registry) -> Option<Sniffed> {
537    use crate::formats::readers::{Asked, HEAD, head_of, sniff};
538    let head = head_of(path)?;
539    if sniff(&head, Some(path), Asked::Listing, |_| true).is_some() {
540        return Some(Sniffed::Format);
541    }
542    formats
543        .listed(path, &head, head.len() < HEAD)
544        .map(Sniffed::Spec)
545}
546
547/// Name `entry` a file of `spec`; a spec reading several record variants also makes
548/// it a place whose tables โ†’ lists.
549pub fn name_spec_file(entry: &mut Entry, spec: &crate::formats::Spec) {
550    entry.kind = EntryKind::File;
551    entry.format_spec = Some(spec.name.clone());
552    if spec.lists_variants() {
553        entry.cost.tables = Some(spec.records.variants.len());
554    }
555}
556
557/// Name an unlisted local file row (a recent) by the spec that reads it, as a listing
558/// would: by glob, else by magic when its name says nothing.
559pub fn name_unlisted_file(entry: &mut Entry, formats: &crate::formats::Registry) {
560    if formats.is_empty()
561        || entry.kind != EntryKind::File
562        || entry.table.is_some()
563        || is_data_file(&entry.path)
564    {
565        return;
566    }
567    let spec = match formats.by_glob(&entry.path, false).into_iter().next() {
568        Some(spec) => Some(spec),
569        None if worth_sniffing(&entry.path) && is_regular_file(&entry.path) => {
570            match sniff_listed(&entry.path, formats) {
571                Some(Sniffed::Spec(spec)) => Some(spec),
572                _ => None,
573            }
574        }
575        None => None,
576    };
577    if let Some(spec) = spec {
578        name_spec_file(entry, &spec);
579    }
580}
581
582/// How many extensionless files one listing sniffs; past this the rest are listed by
583/// name.
584pub(crate) const MAX_SNIFFS_PER_DIR: usize = 256;
585
586/// Whether a file's name has no extension at all: `part-00000`, `LICENSE`.
587pub fn has_no_extension(path: &Path) -> bool {
588    path.extension().is_none()
589}
590
591/// Whether a listing sniffs a file: no extension, an uninformative one (`.bin`), or
592/// only text (`.log`, which candump writes; `.txt`).
593pub fn worth_sniffing(path: &Path) -> bool {
594    path.extension().is_none_or(|e| {
595        e.eq_ignore_ascii_case("bin") || data_format(path).is_some_and(crate::FileFormat::is_lines)
596    })
597}
598
599/// Whether a local file is Parquet by its contents: `PAR1` at both ends.
600pub fn has_parquet_magic(path: &Path) -> bool {
601    use std::io::{Read, Seek, SeekFrom};
602    let Ok(mut file) = std::fs::File::open(path) else {
603        return false;
604    };
605    let mut head = [0u8; 4];
606    let mut tail = [0u8; 4];
607    file.read_exact(&mut head).is_ok()
608        && file.seek(SeekFrom::End(-4)).is_ok()
609        && file.read_exact(&mut tail).is_ok()
610        && &head == b"PAR1"
611        && &tail == b"PAR1"
612}
613
614/// Whether a path names a Parquet file, by extension or as an extensionless part file
615/// in a `.parquet` directory. Unlike [`is_parquet_key`], a writer's own file
616/// (`_manifest.parquet`) counts: it can be opened, even if it does not make its
617/// directory a dataset.
618pub fn is_parquet_path(path: &Path) -> bool {
619    path.extension()
620        .and_then(|e| e.to_str())
621        .is_some_and(|e| e.eq_ignore_ascii_case("parquet"))
622        || is_parquet_key(&directory_and_name(path))
623}
624
625/// Whether a path looks openable by name or place (a part file in a `.parquet`
626/// directory). Every route asks here so they agree.
627pub fn is_data_file(path: &Path) -> bool {
628    data_extension(path).is_some() || is_parquet_key(&directory_and_name(path))
629}
630
631/// The extension saying what a file is, past any compression suffix (`sales.csv.gz`
632/// is CSV); `None` when not something datui reads.
633pub fn data_extension(path: &Path) -> Option<String> {
634    let name = path.file_name().and_then(|n| n.to_str())?;
635    let lower = name.to_ascii_lowercase();
636    let mut parts: Vec<&str> = lower.rsplit('.').collect();
637    parts.reverse();
638    if parts.len() < 2 {
639        return None;
640    }
641    // Walk back past a compression suffix so `sales.csv.gz` still reads as CSV.
642    let mut idx = parts.len() - 1;
643    if COMPRESSION_EXTENSIONS.contains(&parts[idx]) && idx > 1 {
644        idx -= 1;
645    }
646    crate::FileFormat::from_extension(parts[idx]).map(|_| parts[idx].to_string())
647}
648
649/// What the home screen says of a file [`unreadable_by_name`] turns away.
650pub const NO_READER: &str = "datui has no reader for this file";
651
652/// Whether a name already rules the file out: an extension no reader takes, past any
653/// compression suffix. A bare `data.gz` or extensionless name is left to the open.
654pub fn unreadable_by_name(path: &Path) -> bool {
655    let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
656        return false;
657    };
658    let name = name.to_ascii_lowercase();
659    let parts: Vec<&str> = name.rsplit('.').collect();
660    let compressed = |last: &str| COMPRESSION_EXTENSIONS.contains(&last);
661    let ext = match parts[..] {
662        [last, inner, _, ..] if compressed(last) => inner,
663        [last, _] if compressed(last) => return false,
664        [last, _, ..] => last,
665        _ => return false,
666    };
667    crate::FileFormat::from_extension(ext).is_none()
668}
669
670/// The format a file's name says it holds, past any compression suffix. Extensions
671/// are not formats (`.ipc`, `.arrow`, `.arrows`, `.feather` are one); asking
672/// [`crate::FileFormat`] keeps home from listing what the reader cannot open.
673pub fn data_format(path: &Path) -> Option<crate::FileFormat> {
674    // A sharded checkpoint's index is the model, not a JSON table.
675    crate::FileFormat::from_name_ending(path)
676        .or_else(|| crate::FileFormat::from_extension(&data_extension(path)?))
677}
678
679/// The `dir/name` string `is_parquet_key` splits on `/`, so Windows backslashes do
680/// not make `occurrence.parquet\000001` one dotted name.
681pub(crate) fn directory_and_name(path: &Path) -> String {
682    let name = path.file_name().unwrap_or_default().to_string_lossy();
683    match path.parent().and_then(|p| p.file_name()) {
684        Some(directory) => format!("{}/{name}", directory.to_string_lossy()),
685        None => name.into_owned(),
686    }
687}
688
689/// Format rank: commonest first, Parquet winning ties, then by name, so the order
690/// never depends on `read_dir`. One order for the local label, the local reader
691/// choice and the cloud label, so a row's label and `Enter` agree.
692pub(crate) fn rank_formats(a: (&str, usize), b: (&str, usize)) -> std::cmp::Ordering {
693    // Text loses: a README among data files is not the table.
694    let text = crate::FileFormat::Text.name();
695    (a.0 == text)
696        .cmp(&(b.0 == text))
697        .then_with(|| b.1.cmp(&a.1))
698        .then_with(|| (a.0 != "parquet").cmp(&(b.0 != "parquet")))
699        .then_with(|| a.0.cmp(b.0))
700}
701
702/// Whether a file is Hugging Face `datasets` metadata beside Arrow shards, so its
703/// JSON does not outnumber a one-shard dataset and get read instead.
704pub(crate) fn is_hugging_face_metadata(name: &str) -> bool {
705    matches!(name, "dataset_info.json" | "state.json")
706}
707
708fn order_formats(counts: &mut [(crate::FileFormat, usize)]) {
709    counts.sort_by(|a, b| rank_formats((a.0.name(), a.1), (b.0.name(), b.1)));
710}
711
712/// Whether a listing entry is bookkeeping: a leading `_` or `.` (`_SUCCESS`,
713/// `_committed_*`, `_metadata.json`, `.crc`) or the `_$folder$` marker. One predicate
714/// for every listing and open, so they agree.
715pub fn is_bookkeeping(name: &str) -> bool {
716    // `_$folder$` first: the folder it stands for may be a partition (`year=2024_$folder$`
717    // beside `year=2024/`), which the partition test would call data.
718    if name.ends_with("_$folder$") {
719        return true;
720    }
721    // A `key=value` name is a partition, even with a leading `_` (Spark and Hive
722    // partition on internal columns like `_date=โ€ฆ`).
723    if is_partition_name(name) {
724        return false;
725    }
726    name.starts_with(['_', '.'])
727}
728
729/// Whether a format, by name, is model weights.
730fn is_weights(name: &str) -> bool {
731    name == crate::FileFormat::Safetensors.name() || name == crate::FileFormat::Gguf.name()
732}
733
734/// Whether formats side by side are a model: one weight format with only JSON beside
735/// (config, tokenizer), however many JSON files.
736pub(crate) fn is_model_directory<'a>(names: impl IntoIterator<Item = &'a str>) -> bool {
737    let mut weights = None;
738    for name in names {
739        if is_weights(name) {
740            if weights.is_some_and(|w| w != name) {
741                return false;
742            }
743            weights = Some(name);
744        } else if name != crate::FileFormat::Json.name() {
745            return false;
746        }
747    }
748    weights.is_some()
749}
750
751/// The format names `holds` counted.
752fn counts_names(holds: &Holds) -> impl Iterator<Item = &str> {
753    holds.formats.iter().map(|(name, _)| name.as_str())
754}
755
756/// Whether a name is a hive partition (`year=2024`): `key=value`, with a non-empty key.
757/// The value may be empty in practice.
758pub fn is_partition_name(name: &str) -> bool {
759    matches!(name.find('='), Some(i) if i > 0)
760}
761
762/// What the data files sitting directly in a directory say about how to read it.
763///
764/// A directory opened as one dataset has to be read by *something*, and the only honest
765/// source for that is the files in it. Before this existed the answer was assumed:
766/// any directory was scanned as Parquet, so a directory of `.json.gz` was opened by
767/// seeking to the end of each file for a `PAR1` that was never going to be there.
768#[derive(Debug, Clone, PartialEq, Eq)]
769pub enum DirectoryFormat {
770    /// Every data file directly in the directory reads as this one format, and these are
771    /// the files. Sorted, because a concatenation's row order is its file order.
772    One(crate::FileFormat, Vec<PathBuf>),
773    /// The files name more than one format. The commonest of them is the table โ€” a
774    /// directory of a thousand CSVs and one stray JSON is a directory of CSVs โ€” and the
775    /// rest are counted by format so the read can say what it passed over. Parquet wins a
776    /// tie, because it is the format a directory of data files is most likely to be about
777    /// and the one every other route here reads in place.
778    Mixed {
779        format: crate::FileFormat,
780        files: Vec<PathBuf>,
781        passed_over: Vec<(crate::FileFormat, usize)>,
782    },
783    /// The directory settles nothing by itself: it holds no readable data file directly,
784    /// or it has subdirectories and so may hold its data below. A hive dataset looks like
785    /// this โ€” its files are a level down, under `key=value`.
786    Deeper,
787}
788
789/// What [`DirectoryFormat`] the data files directly in `dir` amount to.
790///
791/// One level only, and no file is opened: this reads names, exactly as the rest of
792/// this module does. A directory of two hundred thousand files costs one listing
793/// and nothing per file beyond it โ€” the entry's own type comes back with the name, so
794/// there is no `stat` to bound.
795///
796/// Deliberately *not* bounded by [`MAX_ENTRIES_PER_DIR`], unlike every listing in this
797/// module. The files this returns are not a menu to show, they are the table to read:
798/// stopping at five thousand of a directory's six thousand CSVs would open it with a row
799/// count, a schema union and every aggregate quietly computed over a subset, and the
800/// `take` running before the sort would drop whichever file the directory read
801/// happened to return last. The Parquet route this mirrors hands the directory to a scan
802/// that enumerates it, with no cap either.
803pub fn directory_format(dir: &Path) -> DirectoryFormat {
804    let Ok(iter) = std::fs::read_dir(dir) else {
805        return DirectoryFormat::Deeper;
806    };
807
808    // Every format the directory names, with the files of each. A directory of one format
809    // takes the only entry; a directory of several takes the commonest and reports the
810    // rest, which is what stops one stray file deciding a directory cannot be read.
811    let mut by_format: Vec<(crate::FileFormat, Vec<PathBuf>)> = Vec::new();
812    // Files whose names say nothing, kept in case their bytes do. See below.
813    let mut nameless: Vec<PathBuf> = Vec::new();
814    let mut partitioned = false;
815    for entry in iter.flatten() {
816        let path = entry.path();
817        let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
818            continue;
819        };
820        // A table format's own files are not the table's, and a dotfile is nobody's.
821        // The same test every other route makes, so they agree on what is data.
822        if is_bookkeeping(name) {
823            continue;
824        }
825        // The type the directory read already returned, rather than a `stat` per
826        // entry: on a share that is a round trip per entry, and the question is only
827        // whether this is a file. A symlink still gets the stat, because `d_type`
828        // cannot say what is on the other end of one.
829        let is_file = match entry.file_type() {
830            Ok(kind) if kind.is_symlink() => is_regular_file(&path),
831            Ok(kind) => kind.is_file(),
832            Err(_) => is_regular_file(&path),
833        };
834        if !is_file {
835            // Only a partition, not any subdirectory: a directory of CSVs with some
836            // unrelated directory beside them is still a directory of CSVs, and handing
837            // that to a scanner that walks trees is the very thing this function exists
838            // to stop.
839            partitioned |= is_partition_dir(&path);
840            continue;
841        }
842        let Some(found) = data_format(&path) else {
843            // A name that says nothing. Kept rather than dropped, because the bytes may
844            // still say what it is โ€” see below, where they are asked.
845            if path.extension().is_none() {
846                nameless.push(path);
847            }
848            continue;
849        };
850        match by_format.iter_mut().find(|(f, _)| *f == found) {
851            Some((_, of_that_format)) => of_that_format.push(path),
852            None => by_format.push((found, vec![path])),
853        }
854    }
855
856    // Extensionless files (Spark and GBIF part files) in a directory whose names settled
857    // nothing; skipped when names already answer (a `LICENSE` beside Parquet). A
858    // [`spread`] is sniffed, not every file: the ends must agree, and then all are taken
859    // as that format (the scan reports any odd one out).
860    if by_format.is_empty() && !nameless.is_empty() {
861        nameless.sort();
862        let picks = spread(nameless.len());
863        let sniffed: Vec<crate::FileFormat> = picks
864            .iter()
865            .filter_map(|i| sniff_format(&nameless[*i]))
866            .collect();
867        if sniffed.len() == picks.len()
868            && let Some(found) = sniffed.first().copied()
869            && sniffed.iter().all(|f| *f == found)
870        {
871            by_format.push((found, nameless));
872        }
873    }
874
875    // Text is data only where nothing else is: a README beside Parquet is not a
876    // candidate.
877    if by_format.iter().any(|(f, _)| !f.is_lines()) {
878        by_format.retain(|(f, _)| !f.is_lines());
879    }
880
881    // A Hugging Face dataset's own JSON files, beside its shards.
882    if by_format
883        .iter()
884        .any(|(f, _)| *f == crate::FileFormat::Arrow)
885    {
886        for (format, files) in &mut by_format {
887            if *format == crate::FileFormat::Json {
888                files.retain(|f| {
889                    !f.file_name()
890                        .and_then(|n| n.to_str())
891                        .is_some_and(is_hugging_face_metadata)
892                });
893            }
894        }
895        by_format.retain(|(_, files)| !files.is_empty());
896    }
897
898    // Decided after the whole listing, not at the first deciding entry, so read order
899    // does not matter. One `key=value` below and the data is down there.
900    if partitioned {
901        return DirectoryFormat::Deeper;
902    }
903
904    // The shared format rank, so the reader picked is the format the label names.
905    by_format.sort_by(|a, b| rank_formats((a.0.name(), a.1.len()), (b.0.name(), b.1.len())));
906    // A directory of model weights is the model, however much config and tokenizer
907    // JSON sits beside it; the JSON is passed over.
908    if is_model_directory(by_format.iter().map(|(f, _)| f.name()))
909        && let Some(at) = by_format.iter().position(|(f, _)| is_weights(f.name()))
910    {
911        let weights = by_format.remove(at);
912        by_format.insert(0, weights);
913    }
914    let mut by_format = by_format.into_iter();
915    let Some((format, mut files)) = by_format.next() else {
916        return DirectoryFormat::Deeper;
917    };
918    files.sort();
919    let passed_over: Vec<(crate::FileFormat, usize)> =
920        by_format.map(|(f, of_that)| (f, of_that.len())).collect();
921    if passed_over.is_empty() {
922        DirectoryFormat::One(format, files)
923    } else {
924        DirectoryFormat::Mixed {
925            format,
926            files,
927            passed_over,
928        }
929    }
930}
931
932/// How far down a hive root is followed for its files (year/month/day/hour is four);
933/// deeper is not a hive dataset, and each level costs a read on a share.
934const MAX_HIVE_DEPTH: usize = 16;
935
936/// What the files under a hive root's partitions are. [`directory_format`] can only
937/// say `Deeper` of a root, so this follows one spine (the first partition at each
938/// level, as a hive scan reads its schema) to the bottom. A sample: disagreeing
939/// partitions report as the first one.
940pub fn hive_leaf_format(dir: &Path) -> DirectoryFormat {
941    let mut at = dir.to_path_buf();
942    for _ in 0..MAX_HIVE_DEPTH {
943        match directory_format(&at) {
944            // Nothing here settles it. Follow the partitions down, if there are any.
945            DirectoryFormat::Deeper => match first_partition(&at) {
946                Some(next) => at = next,
947                None => return DirectoryFormat::Deeper,
948            },
949            settled => return settled,
950        }
951    }
952    DirectoryFormat::Deeper
953}
954
955/// The first `key=value` subdirectory of `dir` by name, so runs agree.
956fn first_partition(dir: &Path) -> Option<PathBuf> {
957    let iter = std::fs::read_dir(dir).ok()?;
958    iter.flatten()
959        .take(MAX_ENTRIES_PER_DIR)
960        .map(|entry| entry.path())
961        .filter(|path| is_partition_dir(path) && path.is_dir())
962        .min()
963}
964
965/// Whether a directory name is a hive partition (`year=2024`).
966fn is_partition_dir(path: &Path) -> bool {
967    path.file_name()
968        .and_then(|n| n.to_str())
969        .is_some_and(is_partition_name)
970}
971
972/// An empty extensionless object: a tool's folder marker (`yellow/year=2032`), whether
973/// or not the folder still exists. Nothing datui opens is both empty and nameless.
974pub fn is_empty_marker(name: &str, size: u64) -> bool {
975    size == 0 && !name.contains('.')
976}
977
978/// Classify a directory without walking it: one listing bounded by
979/// [`MAX_ENTRIES_PER_DIR`] (not a probe of the first few entries, whose answer would
980/// depend on read order). Includes rows on network mounts: one `getdents` walk,
981/// stat'ing only symlinks. `home_open_selected` calls it on the key thread; others
982/// on workers.
983pub fn classify_directory(path: &Path) -> EntryKind {
984    look_at_directory(path).0
985}
986
987/// The kind (what `Enter` does) and what the listing found (what the label says),
988/// from one read.
989pub fn look_at_directory(path: &Path) -> (EntryKind, Holds) {
990    // Before the listing: three `join` tests answer it, where counting would walk every
991    // table of a warehouse each pass. A bucket prefix finds markers in its listing.
992    if let Some(lake) = lake_table(path) {
993        return (lake, Holds::default());
994    }
995    let Ok(iter) = std::fs::read_dir(path) else {
996        return (EntryKind::Directory, Holds::default());
997    };
998    let directory = path.file_name().unwrap_or_default().to_string_lossy();
999    let rules = Rules {
1000        directory: &directory,
1001        // As the listing of the rows inside does, so the tally agrees with them.
1002        sniff: (!crate::home::is_remote_path(path)).then_some(path),
1003        in_bucket: false,
1004    };
1005    // One past the cap, so "there is more" is known; bounded where entries come from,
1006    // since `.crc` files beside each data file would double the walk. Past the cap the
1007    // first entries decide.
1008    let mut truncated = false;
1009    let seen = iter
1010        .flatten()
1011        .take(MAX_ENTRIES_PER_DIR + 1)
1012        .enumerate()
1013        .map_while(|(at, entry)| {
1014            truncated = at == MAX_ENTRIES_PER_DIR;
1015            (!truncated).then(|| seen_on_disk(&entry))
1016        });
1017    let (kind, mut holds) = classify(seen, &rules);
1018    holds.truncated = truncated;
1019    (kind, holds)
1020}
1021
1022/// A local entry as [`classify`] sees it, from the read's file type (no stat per
1023/// entry; symlinks still need one). Regular files only: a FIFO named `a.csv` blocks
1024/// its opener, a broken symlink opens as nothing.
1025fn seen_on_disk(entry: &std::fs::DirEntry) -> Seen {
1026    let (is_dir, is_file) = match entry.file_type() {
1027        Ok(kind) if !kind.is_symlink() => (kind.is_dir(), kind.is_file()),
1028        _ => std::fs::metadata(entry.path()).map_or((false, false), |m| (m.is_dir(), m.is_file())),
1029    };
1030    Seen {
1031        name: entry.file_name().to_string_lossy().into_owned(),
1032        is_dir,
1033        is_file,
1034        size: None,
1035    }
1036}
1037
1038/// One entry of a listing, on disk or in a bucket, as [`classify`] asks about it.
1039#[derive(Debug, Clone)]
1040pub struct Seen {
1041    pub name: String,
1042    pub is_dir: bool,
1043    /// A regular file an open can read (not a FIFO, socket or broken symlink).
1044    pub is_file: bool,
1045    /// Bytes, where listed; an empty extensionless file is a folder marker, not data.
1046    pub size: Option<u64>,
1047}
1048
1049/// Where the local and the bucket listings deliberately differ.
1050pub struct Rules<'a> {
1051    /// The listed directory's name: an extensionless part file in `occurrence.parquet/` is
1052    /// Parquet by where it sits.
1053    pub directory: &'a str,
1054    /// Where to look inside nameless files, a few per listing; `None` where each open is
1055    /// a round trip that may not return.
1056    pub sniff: Option<&'a Path>,
1057    /// A bucket prefix: lake tables are found by names in the listing (Iceberg by layout
1058    /// alone), a DatasetDict by `dataset_dict.json`, and only Parquet (read in place) is
1059    /// offered as many files that are one table.
1060    pub in_bucket: bool,
1061}
1062
1063/// A directory's kind and holdings from one listing level: the one rule for
1064/// [`look_at_directory`] and bucket prefixes, so a directory and its mirror agree;
1065/// [`Rules`] holds their differences. The caller sets truncation.
1066pub fn classify(seen: impl Iterator<Item = Seen>, rules: &Rules) -> (EntryKind, Holds) {
1067    use crate::FileFormat;
1068    let mut holds = Holds::default();
1069    let mut counts: Vec<(FileFormat, usize)> = Vec::new();
1070    let mut skipped: Vec<String> = Vec::new();
1071    // Counted as data until the listing is done; see `is_hugging_face_metadata`.
1072    let mut hugging_face: Vec<String> = Vec::new();
1073    let mut dict_file = false;
1074    let mut lake: Vec<&'static str> = Vec::new();
1075    // `present` is everything but a writer's own: what the majority is measured against.
1076    let (mut present, mut data_files, mut parquet) = (0usize, 0usize, 0usize);
1077    let mut sniffs_left = rules.sniff.map_or(0, |_| MAX_SNIFFS_PER_DIR);
1078    for s in seen {
1079        // Lake markers are spec, not strays: looked for before bookkeeping skips
1080        // `_delta_log`.
1081        if s.is_dir
1082            && rules.in_bucket
1083            && let Some(marker) = ["_delta_log", ".hoodie", "metadata", "data"]
1084                .into_iter()
1085                .find(|m| *m == s.name)
1086        {
1087            lake.push(marker);
1088        }
1089        if is_bookkeeping(&s.name) {
1090            skipped.push(s.name);
1091            continue;
1092        }
1093        present += 1;
1094        if s.is_dir {
1095            holds.directories += 1;
1096            holds.partitions += usize::from(is_partition_name(&s.name));
1097            continue;
1098        }
1099        if s.size.is_some_and(|size| is_empty_marker(&s.name, size)) {
1100            holds.not_read += 1;
1101            continue;
1102        }
1103        let name = Path::new(&s.name);
1104        let key = format!("{}/{}", rules.directory, s.name);
1105        let named = data_format(name);
1106        let found = named
1107            // Text by its name, unless its bytes say more: below.
1108            .filter(|f| !f.is_lines())
1109            // A sharded checkpoint's index counts as JSON, so the label counts shards; the read
1110            // still uses it.
1111            .map(
1112                |f| match crate::formats::model_files::is_safetensors_index(name) {
1113                    true => FileFormat::Json,
1114                    false => f,
1115                },
1116            )
1117            // Data by where it sits rather than by its name: see [`is_data_file`].
1118            .or_else(|| is_parquet_key(&key).then_some(FileFormat::Parquet))
1119            .or_else(|| {
1120                let dir = rules.sniff?;
1121                spend_sniff(&mut sniffs_left, s.is_file, name)
1122                    .then(|| sniff_format(&dir.join(name)))?
1123            })
1124            .or(named)
1125            .filter(|_| s.is_file);
1126        let Some(found) = found else {
1127            // No reader and no name; extensionless files are counted apart (Spark part files).
1128            match s.is_file && has_no_extension(name) {
1129                true => holds.unnamed += 1,
1130                false => holds.not_read += 1,
1131            }
1132            continue;
1133        };
1134        if found == FileFormat::Json && is_hugging_face_metadata(&s.name) {
1135            hugging_face.push(s.name.clone());
1136        }
1137        dict_file |= found == FileFormat::Json && s.name == crate::formats::hf_splits::DATASET_DICT;
1138        data_files += 1;
1139        parquet += usize::from(is_parquet_key(&key));
1140        match counts.iter_mut().find(|(f, _)| *f == found) {
1141            Some((_, n)) => *n += 1,
1142            None => counts.push((found, 1)),
1143        }
1144    }
1145
1146    // Hugging Face's JSON is its writer's own, like `_SUCCESS`, as is a DatasetDict's
1147    // `dataset_dict.json`.
1148    if !counts.iter().any(|(f, _)| *f == FileFormat::Arrow) {
1149        hugging_face.clear();
1150    }
1151    holds.dataset_dict = rules.in_bucket && dict_file && holds.directories > 0;
1152    if holds.dataset_dict {
1153        hugging_face.push(crate::formats::hf_splits::DATASET_DICT.to_string());
1154    }
1155    if let Some((_, n)) = counts.iter_mut().find(|(f, _)| *f == FileFormat::Json) {
1156        *n -= hugging_face.len();
1157        data_files -= hugging_face.len();
1158        present -= hugging_face.len();
1159        skipped.append(&mut hugging_face);
1160    }
1161    // Text is data only where nothing else is.
1162    if counts.iter().any(|(f, _)| !f.is_lines()) {
1163        for (_, n) in counts.iter_mut().filter(|(f, _)| f.is_lines()) {
1164            data_files -= *n;
1165            holds.not_read += std::mem::take(n);
1166        }
1167    }
1168    counts.retain(|(_, n)| *n > 0);
1169    order_formats(&mut counts);
1170    let one_readable = matches!(counts.as_slice(), [(f, _)] if f.reads_many_files());
1171    holds.formats = counts
1172        .into_iter()
1173        .map(|(f, n)| (f.name().to_string(), n))
1174        .collect();
1175    // The first few by name, so runs agree; an object and prefix of one name are one
1176    // thing.
1177    skipped.sort();
1178    skipped.dedup();
1179    holds.skipped = skipped.len();
1180    skipped.truncate(SKIPPED_NAMES_SHOWN);
1181    holds.skipped_names = skipped;
1182
1183    // Lake tables' data files agree on a schema, so the rules below would wrongly say
1184    // "one table". Iceberg needs its whole shape (data under `data/`).
1185    let marked = |m| lake.contains(&m);
1186    if marked("_delta_log") {
1187        return (EntryKind::Delta, holds);
1188    }
1189    if marked(".hoodie") {
1190        return (EntryKind::Hudi, holds);
1191    }
1192    if marked("metadata") && marked("data") && parquet == 0 {
1193        return (EntryKind::Iceberg, holds);
1194    }
1195    // No majority rule: one stray `notes=old/` reads `hive`, but majorities refused real
1196    // hive roots with a README or `scripts/` beside them, the worse mistake.
1197    if holds.partitions > 0 && holds.partitions >= data_files {
1198        return (EntryKind::Hive, holds);
1199    }
1200    // One multi-file-readable format that dominates: a couple of stray CSVs make a
1201    // place, not a dataset. A model opens as the model despite its JSON.
1202    let one_table = if rules.in_bucket {
1203        parquet > 1 && parquet == data_files && parquet * 2 >= present
1204    } else {
1205        (data_files > 1 && one_readable && data_files * 2 >= present)
1206            || (holds.partitions == 0 && is_model_directory(counts_names(&holds)))
1207    };
1208    let kind = match one_table {
1209        true => EntryKind::MultiFile,
1210        false => EntryKind::Directory,
1211    };
1212    (kind, holds)
1213}
1214
1215/// Spend one of a listing's sniffs on `name`: a regular file whose name says nothing
1216/// ([`worth_sniffing`]), budget permitting.
1217fn spend_sniff(left: &mut usize, is_file: bool, name: &Path) -> bool {
1218    let spend = is_file && *left > 0 && worth_sniffing(name);
1219    *left -= usize::from(spend);
1220    spend
1221}
1222
1223/// Entries under `metadata/` checked for Iceberg: `v1.metadata.json` is written on
1224/// the first commit and never removed, but read order is arbitrary.
1225const ICEBERG_METADATA_PROBE: usize = 64;
1226
1227/// Whether `path` is a lake table root, and which. Marker names are part of these
1228/// formats' specs (unlike filename conventions); three `join` tests answer it
1229/// without walking a table's files.
1230fn lake_table(path: &Path) -> Option<EntryKind> {
1231    if path.join("_delta_log").is_dir() {
1232        return Some(EntryKind::Delta);
1233    }
1234    if path.join(".hoodie").is_dir() {
1235        return Some(EntryKind::Hudi);
1236    }
1237    // Iceberg's marker is a plain name, so it needs the whole shape: `metadata/` with a
1238    // metadata file, beside `data/`.
1239    let metadata = path.join("metadata");
1240    if path.join("data").is_dir()
1241        && metadata.is_dir()
1242        && std::fs::read_dir(&metadata).is_ok_and(|entries| {
1243            entries
1244                .flatten()
1245                .take(ICEBERG_METADATA_PROBE)
1246                .any(|e| e.file_name().to_string_lossy().ends_with(".metadata.json"))
1247        })
1248    {
1249        return Some(EntryKind::Iceberg);
1250    }
1251    None
1252}
1253
1254/// What one directory listing produced, and whether it saw all of it.
1255#[derive(Debug, Clone, Default)]
1256pub struct Scan {
1257    pub entries: Vec<Entry>,
1258    /// The directory held more than `MAX_ENTRIES_PER_DIR`; `entries` is a prefix, which
1259    /// the UI must say.
1260    pub truncated: bool,
1261}
1262
1263/// [`scan_dir_progressive`], also naming sniffed files by `formats`' magic.
1264pub fn scan_dir_specs(dir: &Path, formats: &crate::formats::Registry) -> Scan {
1265    scan_dir_with(dir, formats, |_| {})
1266}
1267
1268/// How often a listing still being read shows what it has so far.
1269const LISTING_PROGRESS_EVERY: std::time::Duration = std::time::Duration::from_millis(250);
1270
1271/// List one directory with bounded work: one `read_dir` and a stat per entry, up to
1272/// [`MAX_ENTRIES_PER_DIR`]; errors give an empty listing. Nothing is classified:
1273/// subdirectories return [`EntryKind::Unknown`] and are looked into later from the
1274/// viewport, so a label is a fact about the row, not its position. `progress` gets
1275/// the rows read since the last call, every `LISTING_PROGRESS_EVERY`, so a slow share
1276/// shows rows as they arrive.
1277pub fn scan_dir_progressive(dir: &Path, progress: impl FnMut(&[Entry])) -> Scan {
1278    scan_dir_with(dir, &crate::formats::Registry::default(), progress)
1279}
1280
1281fn scan_dir_with(
1282    dir: &Path,
1283    formats: &crate::formats::Registry,
1284    mut progress: impl FnMut(&[Entry]),
1285) -> Scan {
1286    let Ok(iter) = std::fs::read_dir(dir) else {
1287        return Scan::default();
1288    };
1289    let mut shown = std::time::Instant::now();
1290
1291    let mut entries = Vec::new();
1292    // How many of `entries` `progress` has been handed.
1293    let mut sent = 0usize;
1294    let mut seen = 0usize;
1295    let mut truncated = false;
1296    // Extensionless files are sniffed (a few bytes) so part files list as data; never on
1297    // a share, where each open is a round trip.
1298    let mut sniffs_left = if crate::home::is_remote_path(dir) {
1299        0
1300    } else {
1301        MAX_SNIFFS_PER_DIR
1302    };
1303
1304    // One past the cap: enough to know more exists without paying to process it.
1305    for dir_entry in iter.flatten().take(MAX_ENTRIES_PER_DIR + 1) {
1306        seen += 1;
1307        if seen > MAX_ENTRIES_PER_DIR {
1308            truncated = true;
1309            break;
1310        }
1311
1312        let path = dir_entry.path();
1313        let name = dir_entry.file_name();
1314        if name.to_string_lossy().starts_with('.') {
1315            continue;
1316        }
1317
1318        let Ok(meta) = dir_entry.metadata() else {
1319            continue;
1320        };
1321
1322        let mut spec = None;
1323        let kind = if meta.is_dir() {
1324            EntryKind::Unknown
1325        } else if meta.is_file() && is_data_file(&path) {
1326            EntryKind::File
1327        } else if spend_sniff(&mut sniffs_left, meta.is_file(), &path) {
1328            match sniff_listed(&path, formats) {
1329                Some(Sniffed::Format) => EntryKind::File,
1330                Some(Sniffed::Spec(found)) => {
1331                    spec = Some(found);
1332                    EntryKind::File
1333                }
1334                None => EntryKind::Other,
1335            }
1336        } else if meta.is_file() {
1337            EntryKind::Other
1338        } else {
1339            // Not a directory or regular file: a FIFO named `x.parquet` must never be offered.
1340            continue;
1341        };
1342
1343        let mut entry = Entry::new(path, kind).with_fs_metadata(&meta);
1344        if let Some(spec) = spec {
1345            name_spec_file(&mut entry, &spec);
1346        }
1347        entries.push(entry);
1348        if shown.elapsed() >= LISTING_PROGRESS_EVERY {
1349            progress(&entries[sent..]);
1350            sent = entries.len();
1351            shown = std::time::Instant::now();
1352        }
1353    }
1354
1355    sort_entries(&mut entries);
1356    Scan { entries, truncated }
1357}
1358
1359/// Datasets first, then directories, each alphabetical (a listing is scanned by
1360/// name). Unexamined rows sort with directories, where they land if plain, so a kind
1361/// arriving later never moves a row under the cursor.
1362pub(crate) fn sort_entries(entries: &mut [Entry]) {
1363    entries.sort_by(|a, b| {
1364        // Data, then directories, then what datui cannot read.
1365        let group = |k: EntryKind| match k {
1366            k if k.is_known_dataset() => 0,
1367            EntryKind::Other => 2,
1368            _ => 1,
1369        };
1370        group(a.kind).cmp(&group(b.kind)).then_with(|| {
1371            a.name
1372                .to_ascii_lowercase()
1373                .cmp(&b.name.to_ascii_lowercase())
1374        })
1375    });
1376}
1377
1378/// How far below a directory the footer walk goes (year/month/day/hour is four).
1379const MAX_WALK_DEPTH: u8 = 4;
1380
1381/// Upper bound on footers read to size a multi-file or hive dataset; past it the count
1382/// is left blank rather than partial.
1383const MAX_FOOTERS_PER_DATASET: usize = 64;
1384
1385/// Fill in row and column counts from Parquet footers: one file, or a bounded sum for
1386/// hive and multi-file datasets. Non-Parquet keeps `None`, a blank in the UI.
1387pub fn enrich(entry: &mut Entry) {
1388    enrich_as(entry, &crate::formats::schema_union::ReadAs::default())
1389}
1390
1391/// As [`enrich`], reading files as the following open will: where the header is
1392/// decides what the names are, so judging with other settings would misjudge (e.g.
1393/// `--no-header`).
1394pub fn enrich_as(entry: &mut Entry, as_read: &crate::formats::schema_union::ReadAs) {
1395    enrich_with(entry, as_read, None)
1396}
1397
1398/// As [`enrich_as`], using the shape an open kept in `remembered` while the files are
1399/// unchanged, instead of sampling footers.
1400pub fn enrich_with(
1401    entry: &mut Entry,
1402    as_read: &crate::formats::schema_union::ReadAs,
1403    remembered: Option<&crate::cache::CacheManager>,
1404) {
1405    match entry.kind {
1406        EntryKind::File => {
1407            enrich_parquet(entry);
1408            enrich_tables(entry);
1409            enrich_arrow(entry);
1410        }
1411        EntryKind::Hive | EntryKind::MultiFile => enrich_dataset(entry, as_read, remembered),
1412        // Nothing to read for plain or unexamined directories, nor lake tables (summing their
1413        // footers counts tombstoned and rewritten rows).
1414        EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => {}
1415        EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => {}
1416    }
1417}
1418
1419/// Sum footers across a bounded set of Parquet files under `entry`.
1420fn enrich_dataset(
1421    entry: &mut Entry,
1422    as_read: &crate::formats::schema_union::ReadAs,
1423    remembered: Option<&crate::cache::CacheManager>,
1424) {
1425    // A directory whose own files are a format this cannot count is not described by
1426    // Parquet beneath it (`6 json` reporting the subdirectories' columns). Parquet with
1427    // Parquet beneath keeps the walk: `Enter` reads the subtree, so counts must too; the
1428    // `holds` line explains the difference.
1429
1430    // The partition layout comes from directory names, so it is known even when the rows
1431    // are too many to count.
1432    if entry.kind == EntryKind::Hive {
1433        entry.cost.partitions = partition_layout(&entry.path);
1434    }
1435
1436    // Whether the footers below describe this directory (exact for its own format). Not
1437    // asked of a hive root: its own files are strays (a `schema.json`), and its data's
1438    // format can only be sampled.
1439    let reads_as_parquet = entry.kind == EntryKind::Hive
1440        || match entry.holds.one_format() {
1441            // No single format: unreachable today (mixed formats are `Directory`), so default to
1442            // the safe side.
1443            None => true,
1444            // An unreadable name is not Parquet: missing counts are undone by opening; another
1445            // format's counts are not.
1446            Some(name) => crate::FileFormat::from_name(name) == Some(crate::FileFormat::Parquet),
1447        };
1448    if !reads_as_parquet {
1449        entry.size = None;
1450        judge_by_names(entry, as_read);
1451        return;
1452    }
1453
1454    // A directory's stat size is its inode, not its contents: dropped here, restored only
1455    // if the files are totalled.
1456    entry.size = None;
1457
1458    let files = parquet_files_under(&entry.path);
1459    // Past the budget, footers an open kept are used if the listing still matches.
1460    if files.len() > MAX_FOOTERS_PER_DATASET
1461        && let Some((listed, footers)) = remembered
1462            .and_then(|cache| crate::formats::dataset_files::remembered_footers(&entry.path, cache))
1463    {
1464        measure_from_footers(entry, &listed, &footers);
1465        return;
1466    }
1467    if files.is_empty() || files.len() > MAX_FOOTERS_PER_DATASET {
1468        // Whether these are one table needs only three spread footers, and matters most for
1469        // directories past the counting limit.
1470        let sampled = sample_footers(&files);
1471        let names: Vec<Vec<String>> = sampled.iter().map(column_names).collect();
1472        let tops: Vec<Vec<String>> = names
1473            .iter()
1474            .map(|n| crate::formats::schema_union::top_level_columns(n))
1475            .collect();
1476        if entry.kind == EntryKind::MultiFile && one_table_from(&tops) == Some(false) {
1477            // One-table is asked of the subtree (what opening unions), but holdings count only
1478            // the direct files (the label's), since a downgraded row never opens as one table.
1479            let own_files = direct_children(&files, &entry.path);
1480            let own = sample_footers(&own_files);
1481            // The names are every sampled one under it: home's search index should find a
1482            // directory by any column looking inside reaches.
1483            entry.columns = union_of(&names);
1484            // A floor only if one of its own footers went unread.
1485            entry.cols_sampled = own.len() < own_files.len();
1486            // The columns a reader sees, from the schema rather than splitting leaf paths on dots
1487            // (`user.id` vs struct `user` with `id`).
1488            let top = union_of(&own.iter().map(top_level_names).collect::<Vec<_>>());
1489            downgrade_to_directory(entry, (!top.is_empty()).then_some(top.len()));
1490            return;
1491        }
1492        // Still worth knowing the shape, even when the row count is out of reach.
1493        if let Some(meta) = sampled.first() {
1494            // Three files, not the first: a growing dataset's newest columns are in its last
1495            // file. Still a sample; the count is already `?`.
1496            entry.columns = union_of(&names);
1497            let top = union_of(&sampled.iter().map(top_level_names).collect::<Vec<_>>());
1498            entry.cols = Some(top.len() + partition_columns_beyond(entry, &top));
1499            entry.cols_sampled = true;
1500            // From one file: codec and row-group sizing are the writer's, uniform in practice.
1501            physical_facts(meta, &mut entry.cost);
1502            entry.cost.uncompressed = None;
1503        }
1504        return;
1505    }
1506
1507    let mut rows = 0usize;
1508    let mut bytes = 0u64;
1509    // Every column any file has, in first-seen order, so columns added over time are
1510    // found.
1511    let mut columns: Vec<String> = Vec::new();
1512    let mut seen_columns = std::collections::HashSet::new();
1513    // Top-level columns unioned the same way, kept beside the leaves (see
1514    // [`top_level_names`]).
1515    let mut top_level: Vec<String> = Vec::new();
1516    let mut seen_top_level = std::collections::HashSet::new();
1517    let mut per_file: Vec<Vec<String>> = Vec::with_capacity(files.len());
1518    // Width and size from the directory's own files only (a downgraded row is never one
1519    // table); column names stay the subtree's, for search.
1520    let mut own_bytes = 0u64;
1521    let mut own_top_level: Vec<String> = Vec::new();
1522    let mut own_seen_top = std::collections::HashSet::new();
1523    let mut cost = Cost::default();
1524    let mut uncompressed = 0u64;
1525    let mut row_groups = 0usize;
1526    for file in &files {
1527        let Some(meta) = crate::formats::parquet_footer::read_parquet_metadata(file) else {
1528            return; // A file we cannot read makes the total a guess; report nothing.
1529        };
1530        rows += meta.num_rows;
1531        let names = column_names(&meta);
1532        for name in &names {
1533            if seen_columns.insert(name.clone()) {
1534                columns.push(name.clone());
1535            }
1536        }
1537        for name in top_level_names(&meta) {
1538            if seen_top_level.insert(name.clone()) {
1539                top_level.push(name);
1540            }
1541        }
1542        // Top-level columns, not footer leaves: see
1543        // [`crate::formats::schema_union::top_level_columns`].
1544        per_file.push(crate::formats::schema_union::top_level_columns(&names));
1545        // Again for its own files (`2 parquet` must mean those two), from one stat feeding
1546        // both totals.
1547        let file_bytes = std::fs::metadata(file).map(|m| m.len()).unwrap_or(0);
1548        bytes += file_bytes;
1549        if file.parent() == Some(entry.path.as_path()) {
1550            own_bytes += file_bytes;
1551            for name in top_level_names(&meta) {
1552                if own_seen_top.insert(name.clone()) {
1553                    own_top_level.push(name);
1554                }
1555            }
1556        }
1557        let mut per_file = Cost::default();
1558        physical_facts(&meta, &mut per_file);
1559        uncompressed += per_file.uncompressed.unwrap_or(0);
1560        row_groups += per_file.row_groups.unwrap_or(0);
1561        if cost.codec.is_none() {
1562            cost.codec = per_file.codec;
1563        }
1564    }
1565    // With the footers read, one-table is known. Separate tables make a place to look
1566    // inside (summed rows mean nothing). Only `multi` is reconsidered: hive files hold
1567    // one table by construction.
1568    if entry.kind == EntryKind::MultiFile && !crate::formats::schema_union::is_nested(&per_file) {
1569        // Its own files' bytes, matching what the label and columns count.
1570        entry.size = Some(own_bytes);
1571        // Not one table, but column search should still find it: the count is its own
1572        // files', the names every one under it.
1573        entry.columns = columns;
1574        entry.cols_sampled = false;
1575        downgrade_to_directory(
1576            entry,
1577            (!own_top_level.is_empty()).then_some(own_top_level.len()),
1578        );
1579        return;
1580    }
1581
1582    entry.rows = Some(rows);
1583    entry.cols = Some(top_level.len() + partition_columns_beyond(entry, &top_level));
1584    entry.size = Some(bytes);
1585    entry.columns = columns;
1586    cost.uncompressed = (uncompressed > 0).then_some(uncompressed);
1587    cost.row_groups = (row_groups > 0).then_some(row_groups);
1588    cost.partitions = entry.cost.partitions.take();
1589    entry.cost = cost;
1590}
1591
1592/// Partition keys the files lack, hoisted in as columns by the open, so the width
1593/// counts them (`12 ร— 4`, not `12 ร— 2`).
1594fn partition_columns_beyond(entry: &Entry, top_level: &[String]) -> usize {
1595    entry.cost.partitions.as_ref().map_or(0, |layout| {
1596        layout
1597            .keys
1598            .iter()
1599            .filter(|key| !top_level.contains(key))
1600            .count()
1601    })
1602}
1603
1604/// The files of `dir` itself, out of a walk that went below it.
1605fn direct_children(files: &[PathBuf], dir: &Path) -> Vec<PathBuf> {
1606    files
1607        .iter()
1608        .filter(|f| f.parent() == Some(dir))
1609        .cloned()
1610        .collect()
1611}
1612
1613/// Which of `files` to read for a sample: the ends and the middle. Sorted names
1614/// cluster one table's files at the start, and the last is the newest. Three reads.
1615pub(crate) fn spread(files: usize) -> Vec<usize> {
1616    let mut picks = match files {
1617        0 => Vec::new(),
1618        n => vec![0, n / 2, n - 1],
1619    };
1620    picks.dedup();
1621    picks
1622}
1623
1624/// Whether a few files' top-level columns are one table. `None` from fewer than two:
1625/// the directory keeps its name-based kind.
1626pub(crate) fn one_table_from(footers: &[Vec<String>]) -> Option<bool> {
1627    (footers.len() >= 2).then(|| crate::formats::schema_union::is_nested(footers))
1628}
1629
1630/// The footers of the [`spread`] of a directory too large to read every one of.
1631fn sample_footers(files: &[PathBuf]) -> Vec<crate::formats::parquet_footer::Footer> {
1632    spread(files.len())
1633        .into_iter()
1634        .filter_map(|i| crate::formats::parquet_footer::read_parquet_metadata(&files[i]))
1635        .collect()
1636}
1637
1638/// Ask a footerless directory whether its files are one table by their header names:
1639/// [`one_table_from`]'s rule for CSV and NDJSON, so `Enter` does not promise a table
1640/// the read then refuses. Silence (too few files, costly schema, a parse failure)
1641/// keeps the name-based kind, safe because the read unions by name and widens types
1642/// (`crate::formats::readers::polars::union_of_files`).
1643fn judge_by_names(entry: &mut Entry, as_read: &crate::formats::schema_union::ReadAs) {
1644    if entry.kind != EntryKind::MultiFile {
1645        return;
1646    }
1647    let Some(format) = entry
1648        .holds
1649        .one_format()
1650        .and_then(crate::FileFormat::from_name)
1651    else {
1652        return;
1653    };
1654    // The directory's own files, which the label counts and the open reads; `MultiFile`
1655    // is flat by construction and has one format, so only `One` arrives.
1656    let DirectoryFormat::One(_, files) = directory_format(&entry.path) else {
1657        return;
1658    };
1659    // Read as the following open will: header placement decides the names.
1660    let sampled = crate::formats::schema_union::sample_files(&files, format, as_read);
1661    if sampled.nests == Some(false) {
1662        // The sample's columns, for column search, as the Parquet path keeps when
1663        // downgrading; from the spread, so cost does not grow with the directory.
1664        let cols = (!sampled.columns.is_empty()).then_some(sampled.columns.len());
1665        // A floor only when files went unopened.
1666        entry.cols_sampled = sampled.read < files.len();
1667        entry.columns = sampled.columns;
1668        downgrade_to_directory(entry, cols);
1669    }
1670}
1671
1672/// A directory of separate tables becomes a place to look inside: no row count (a sum
1673/// of unrelated tables), but the union's column count (`15 parquet ยท 72 columns`).
1674/// `cols` is passed in because leaf paths cannot be split back into top-level
1675/// columns (see [`top_level_names`]).
1676fn downgrade_to_directory(entry: &mut Entry, cols: Option<usize>) {
1677    entry.kind = EntryKind::Directory;
1678    entry.rows = None;
1679    entry.cols = cols;
1680    entry.cost = Cost {
1681        partitions: entry.cost.partitions.take(),
1682        ..Cost::default()
1683    };
1684}
1685
1686/// The columns a reader sees: the schema's own top-level fields.
1687///
1688/// Not the leaves a footer names, and not those leaves split on a dot either. Leaves
1689/// counted directly double for a directory whose writer changed โ€” the same nested column
1690/// written by parquet-mr and by Arrow gives `inputs.list.element.address` in one file and
1691/// `inputs.bag.array_element.address` in the other, and a union of leaf paths holds both.
1692/// Splitting the dotted path fixes that and breaks something else: a column literally
1693/// named `user.id` is one column, and so is a struct `user` with a field `id`, and the
1694/// string cannot tell them apart.
1695///
1696/// The schema knows. `fields()` is the root's own children, which is what a struct counts
1697/// as here, what `schema_preview` lists in the details pane, and what the table shows.
1698fn top_level_names(meta: &crate::formats::parquet_footer::Footer) -> Vec<String> {
1699    meta.schema_descr
1700        .fields()
1701        .iter()
1702        .map(|field| field.name().to_string())
1703        .collect()
1704}
1705
1706/// Every column name any of the files has, in the order they first appear.
1707fn union_of(per_file: &[Vec<String>]) -> Vec<String> {
1708    let mut seen = std::collections::HashSet::new();
1709    per_file
1710        .iter()
1711        .flatten()
1712        .filter(|name| seen.insert(name.as_str()))
1713        .cloned()
1714        .collect()
1715}
1716
1717/// The Parquet files under `dir`, sorted, as deep as a dataset goes, stopping one past
1718/// the budget (which says there are too many to count). The open's own walk.
1719fn parquet_files_under(dir: &Path) -> Vec<PathBuf> {
1720    let mut files = crate::formats::dataset_files::LocalFiles::new(dir)
1721        .first_files(MAX_WALK_DEPTH as usize + 1, MAX_FOOTERS_PER_DATASET);
1722    // The walk takes a directory entry's own type; a file this reads must be one.
1723    files.retain(|p| is_regular_file(p));
1724    files
1725}
1726
1727/// Measure a dataset from footers an open kept: rows, width, size, row groups, and
1728/// whether its files are one table; nothing is read.
1729fn measure_from_footers(
1730    entry: &mut Entry,
1731    files: &[crate::formats::dataset_files::DatasetFile],
1732    footers: &[Option<crate::formats::schema_union::FileFooter>],
1733) {
1734    let per_file: Vec<Vec<String>> = footers
1735        .iter()
1736        .flatten()
1737        .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
1738        .collect();
1739    let columns = union_of(&per_file);
1740    if entry.kind == EntryKind::MultiFile && !crate::formats::schema_union::is_nested(&per_file) {
1741        // As a full footer read judges it: label, width and size count the directory's own
1742        // files.
1743        let own: Vec<usize> = files
1744            .iter()
1745            .enumerate()
1746            .filter(|(_, f)| Path::new(&f.key).parent() == Some(entry.path.as_path()))
1747            .map(|(i, _)| i)
1748            .collect();
1749        entry.size = Some(own.iter().map(|&i| files[i].size).sum());
1750        let own_columns = union_of(
1751            &own.iter()
1752                .filter_map(|&i| footers[i].as_ref())
1753                .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
1754                .collect::<Vec<_>>(),
1755        );
1756        entry.columns = columns;
1757        entry.cols_sampled = false;
1758        downgrade_to_directory(
1759            entry,
1760            (!own_columns.is_empty()).then_some(own_columns.len()),
1761        );
1762        return;
1763    }
1764    let footers: Vec<&crate::formats::schema_union::FileFooter> =
1765        footers.iter().flatten().collect();
1766    let uncompressed: u64 = footers
1767        .iter()
1768        .flat_map(|f| &f.column_bytes)
1769        .map(|(_, bytes)| *bytes as u64)
1770        .sum();
1771    let row_groups: usize = footers.iter().map(|f| f.row_group_rows.len()).sum();
1772    entry.rows = Some(footers.iter().map(|f| f.rows()).sum());
1773    entry.cols = Some(columns.len() + partition_columns_beyond(entry, &columns));
1774    entry.size = Some(files.iter().map(|f| f.size).sum());
1775    entry.columns = columns;
1776    entry.cols_sampled = false;
1777    entry.cost = Cost {
1778        uncompressed: (uncompressed > 0).then_some(uncompressed),
1779        row_groups: (row_groups > 0).then_some(row_groups),
1780        partitions: entry.cost.partitions.take(),
1781        ..Cost::default()
1782    };
1783}
1784
1785/// Fill in a Parquet file's counts from its footer, reading no column data. Other
1786/// formats keep `None` (a CSV's rows need a scan).
1787pub fn enrich_parquet(entry: &mut Entry) {
1788    if entry.kind != EntryKind::File {
1789        return;
1790    }
1791    if !is_parquet_path(&entry.path) {
1792        return;
1793    }
1794    if !is_regular_file(&entry.path) {
1795        return;
1796    }
1797    if let Some(meta) = crate::formats::parquet_footer::read_parquet_metadata(&entry.path) {
1798        entry.rows = Some(meta.num_rows);
1799        entry.columns = column_names(&meta);
1800        // Top-level columns, as a directory's row reports them: `schema_descr` names leaves
1801        // (one struct of three fields would count four). See [`top_level_names`].
1802        entry.cols = Some(top_level_names(&meta).len());
1803        physical_facts(&meta, &mut entry.cost);
1804    }
1805}
1806
1807/// A file of tables' tables (SQLite schema, NumPy archive directory): how many, whether
1808/// Enter opens one, and that one's columns. A file whose bytes contradict its name (a
1809/// `.db` that is not SQLite) is unopenable.
1810pub fn enrich_tables(entry: &mut Entry) {
1811    if entry.kind != EntryKind::File || entry.table.is_some() {
1812        return;
1813    }
1814    let named = data_format(&entry.path);
1815    if !is_regular_file(&entry.path) {
1816        return;
1817    }
1818    let Some(format) = crate::formats::members::holder(&entry.path) else {
1819        if named.is_some_and(|f| f.holds_tables() && crate::formats::readers::of(f).bytes_decide) {
1820            entry.kind = EntryKind::Other;
1821        }
1822        return;
1823    };
1824    let Ok(tables) = crate::formats::members::tables(&entry.path, format) else {
1825        return;
1826    };
1827    let own: Vec<&crate::formats::sqlite::Table> = tables.iter().filter(|t| !t.internal).collect();
1828    entry.cost.tables = Some(own.len());
1829    entry.cost.opens_one = format.opens_one_table();
1830    if let [one] = own.as_slice()
1831        && !one.columns.is_empty()
1832    {
1833        entry.columns = one.columns.iter().map(|(name, _)| name.clone()).collect();
1834        entry.cols = Some(entry.columns.len());
1835    }
1836}
1837
1838/// A file of tables listed as rows (a database's tables and views, an archive's arrays),
1839/// each at its path inside the file; SQLite's own marked hidden.
1840pub fn database_rows(file: &Path) -> Vec<Entry> {
1841    let Some(format) = crate::formats::members::holder(file) else {
1842        return Vec::new();
1843    };
1844    let Ok(mut tables) = crate::formats::members::tables(file, format) else {
1845        return Vec::new();
1846    };
1847    // A database's tables by name; an archive's arrays in the order they were saved.
1848    if format
1849        .descriptor()
1850        .tables
1851        .as_ref()
1852        .is_some_and(|t| t.by_name)
1853    {
1854        tables.sort_by_cached_key(|t| t.name.to_lowercase());
1855    }
1856    let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
1857    tables
1858        .into_iter()
1859        .map(|table| table_entry(file, format, table, modified))
1860        .collect()
1861}
1862
1863/// A Hugging Face cache directory's splits as rows at their paths inside it
1864/// (`cache/test`), opened with `--table`. Empty otherwise, or for one split (its door
1865/// opens it).
1866pub fn split_rows(dir: &Path) -> Vec<Entry> {
1867    let splits = crate::formats::hf_splits::cache_splits(dir);
1868    if splits.len() < 2 {
1869        return Vec::new();
1870    }
1871    splits
1872        .into_iter()
1873        .map(|split| split_entry(dir, split))
1874        .collect()
1875}
1876
1877/// The row of a split named by its path inside its cache directory (a recent); `None`
1878/// if none.
1879pub fn split_row(path: &Path) -> Option<Entry> {
1880    let (dir, split) = crate::formats::hf_splits::split_place(path)?;
1881    Some(split_entry(&dir, split))
1882}
1883
1884fn split_entry(dir: &Path, split: String) -> Entry {
1885    let mut entry = Entry::new(dir.join(&split), EntryKind::File).with_name(split);
1886    entry.table = Some(TableOf {
1887        format: Some(crate::FileFormat::Arrow),
1888        kind: "split".to_string(),
1889        internal: false,
1890    });
1891    entry
1892}
1893
1894/// A spec-read file's variants as rows at their paths inside it (`day.itch/add`),
1895/// opened with `--table`. Empty for other files.
1896pub fn variant_rows(file: &Path, formats: &crate::formats::Registry) -> Vec<Entry> {
1897    let Some((spec, tables)) = crate::formats::members::variants(file, formats) else {
1898        return Vec::new();
1899    };
1900    let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
1901    tables
1902        .into_iter()
1903        .map(|table| variant_entry(file, &spec, table, modified))
1904        .collect()
1905}
1906
1907/// The row of a variant named by its path inside its file (a recent); `None` if none.
1908pub fn variant_row(path: &Path, formats: &crate::formats::Registry) -> Option<Entry> {
1909    let (file, name) = crate::formats::members::split_variant(path, formats)?;
1910    let (spec, tables) = crate::formats::members::variants(&file, formats)?;
1911    let table = tables.into_iter().find(|t| t.name == name)?;
1912    let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
1913    let mut entry = variant_entry(&file, &spec, table, modified);
1914    entry.path = path.to_path_buf();
1915    Some(entry)
1916}
1917
1918fn variant_entry(
1919    file: &Path,
1920    spec: &str,
1921    table: crate::formats::sqlite::Table,
1922    modified: Option<std::time::SystemTime>,
1923) -> Entry {
1924    let mut entry = Entry::new(
1925        crate::formats::members::place(file, &table.name),
1926        EntryKind::File,
1927    )
1928    .with_name(table.name);
1929    entry.modified = modified;
1930    entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
1931    entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
1932    entry.format_spec = Some(spec.to_string());
1933    entry.table = Some(TableOf {
1934        format: None,
1935        kind: table.kind,
1936        internal: false,
1937    });
1938    entry
1939}
1940
1941/// The row of a table named by its path inside its file (`app.db/users`, a recent);
1942/// `None` if none.
1943pub fn table_row(path: &Path) -> Option<Entry> {
1944    let (file, name) = crate::formats::members::split(path)?;
1945    let format = crate::formats::members::holder(&file)?;
1946    let table = crate::formats::members::tables(&file, format)
1947        .ok()?
1948        .into_iter()
1949        .find(|t| t.name == name)?;
1950    let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
1951    let mut entry = table_entry(&file, format, table, modified);
1952    entry.path = path.to_path_buf();
1953    Some(entry)
1954}
1955
1956fn table_entry(
1957    file: &Path,
1958    format: crate::FileFormat,
1959    table: crate::formats::sqlite::Table,
1960    modified: Option<std::time::SystemTime>,
1961) -> Entry {
1962    let mut entry = Entry::new(
1963        crate::formats::members::place(file, &table.name),
1964        EntryKind::File,
1965    )
1966    .with_name(table.name);
1967    entry.modified = modified;
1968    entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
1969    entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
1970    entry.table = Some(TableOf {
1971        format: Some(format),
1972        kind: table.kind,
1973        internal: table.internal,
1974    });
1975    entry
1976}
1977
1978/// The first bytes of `path`, as many as fit in `buf`. A short read is the whole file.
1979fn read_head<'a>(path: &Path, buf: &'a mut [u8]) -> Option<&'a [u8]> {
1980    use std::io::Read;
1981    let mut file = std::fs::File::open(path).ok()?;
1982    let mut filled = 0;
1983    loop {
1984        match file.read(&mut buf[filled..]) {
1985            Ok(0) => break,
1986            Ok(n) => filled += n,
1987            Err(_) => return None,
1988        }
1989        if filled == buf.len() {
1990            break;
1991        }
1992    }
1993    Some(&buf[..filled])
1994}
1995
1996/// Whether an Arrow file is an IPC stream (an IPC file starts `ARROW1`): eight bytes,
1997/// so the listing can say which will be converted.
1998fn enrich_arrow(entry: &mut Entry) {
1999    if entry.kind != EntryKind::File
2000        || data_format(&entry.path) != Some(crate::FileFormat::Arrow)
2001        || crate::CompressionFormat::from_extension(&entry.path).is_some()
2002        || !is_regular_file(&entry.path)
2003    {
2004        return;
2005    }
2006    let mut head = [0u8; 8];
2007    if let Some(head) = read_head(&entry.path, &mut head) {
2008        entry.cost.ipc_stream = !head.starts_with(b"ARROW1");
2009    }
2010}
2011
2012/// Pull layout and compression from an already-read footer: what reading the file
2013/// will do, beyond its size.
2014pub fn physical_facts(meta: &crate::formats::parquet_footer::Footer, cost: &mut Cost) {
2015    if meta.row_groups.is_empty() {
2016        return;
2017    }
2018    cost.row_groups = Some(meta.row_groups.len());
2019
2020    let mut uncompressed: u64 = 0;
2021    let mut codecs: Vec<String> = Vec::new();
2022    for rg in &meta.row_groups {
2023        uncompressed = uncompressed.saturating_add(rg.total_byte_size() as u64);
2024        for cc in rg.parquet_columns() {
2025            let codec = format!("{:?}", cc.compression()).to_lowercase();
2026            if !codecs.contains(&codec) {
2027                codecs.push(codec);
2028            }
2029        }
2030    }
2031    if uncompressed > 0 {
2032        cost.uncompressed = Some(uncompressed);
2033    }
2034    // Usually one codec; when not, say so rather than imply uniformity.
2035    cost.codec = match codecs.len() {
2036        0 => None,
2037        1 => Some(codecs.remove(0)),
2038        n => Some(format!("mixed ({n})")),
2039    };
2040}
2041
2042/// Outermost directories examined to describe a hive partitioning: enough to name the
2043/// keys and show the shape; a decade of daily partitions is not worth counting on a
2044/// share.
2045const MAX_PARTITION_DIRS: usize = 512;
2046
2047/// Describe how a hive dataset is partitioned, from directory names alone.
2048pub fn partition_layout(dir: &Path) -> Option<Partitions> {
2049    let iter = std::fs::read_dir(dir).ok()?;
2050    let mut values: Vec<String> = Vec::new();
2051    let mut keys: Vec<String> = Vec::new();
2052    let mut count = 0usize;
2053    let mut more = false;
2054
2055    for entry in iter.flatten() {
2056        if count >= MAX_PARTITION_DIRS {
2057            more = true;
2058            break;
2059        }
2060        let name = entry.file_name().to_string_lossy().into_owned();
2061        let Some((key, value)) = name.split_once('=') else {
2062            continue;
2063        };
2064        if !entry.path().is_dir() {
2065            continue;
2066        }
2067        if keys.is_empty() {
2068            keys.push(key.to_string());
2069            // Only the first partition is descended for nested keys: one is representative.
2070            keys.extend(nested_keys(&entry.path()));
2071        }
2072        values.push(value.to_string());
2073        count += 1;
2074    }
2075
2076    if keys.is_empty() {
2077        return None;
2078    }
2079    values.sort();
2080    values.dedup();
2081    Some(Partitions {
2082        keys,
2083        first_key_values: values,
2084        count,
2085        more,
2086    })
2087}
2088
2089/// Partition keys below `dir`, following the first child at each level.
2090fn nested_keys(dir: &Path) -> Vec<String> {
2091    let mut keys = Vec::new();
2092    let mut current = dir.to_path_buf();
2093    // Bounded: each level is a directory read.
2094    for _ in 0..6 {
2095        let Ok(iter) = std::fs::read_dir(&current) else {
2096            break;
2097        };
2098        let Some(child) = iter
2099            .flatten()
2100            .find(|e| e.file_name().to_string_lossy().contains('=') && e.path().is_dir())
2101        else {
2102            break;
2103        };
2104        let name = child.file_name().to_string_lossy().into_owned();
2105        let Some((key, _)) = name.split_once('=') else {
2106            break;
2107        };
2108        keys.push(key.to_string());
2109        current = child.path();
2110    }
2111    keys
2112}
2113
2114/// Render a row count compactly (`2.4M`).
2115pub fn format_rows(rows: usize) -> String {
2116    let r = rows as f64;
2117    if rows >= 1_000_000_000 {
2118        format!("{:.1}B", r / 1e9)
2119    } else if rows >= 1_000_000 {
2120        format!("{:.1}M", r / 1e6)
2121    } else if rows >= 10_000 {
2122        format!("{:.0}k", r / 1e3)
2123    } else if rows >= 1_000 {
2124        // Below ten thousand show the exact count: 3,653 days is ten years; "4k" is not.
2125        let mut out = String::new();
2126        let digits = rows.to_string();
2127        for (i, c) in digits.chars().enumerate() {
2128            if i > 0 && (digits.len() - i).is_multiple_of(3) {
2129                out.push(',');
2130            }
2131            out.push(c);
2132        }
2133        out
2134    } else {
2135        rows.to_string()
2136    }
2137}
2138
2139/// Render "how long ago" compactly (`2d`, `3h`).
2140pub fn format_age(t: std::time::SystemTime) -> String {
2141    let Ok(elapsed) = t.elapsed() else {
2142        return String::new();
2143    };
2144    let secs = elapsed.as_secs();
2145    if secs < 60 {
2146        "now".to_string()
2147    } else if secs < 3600 {
2148        format!("{}m", secs / 60)
2149    } else if secs < 86_400 {
2150        format!("{}h", secs / 3600)
2151    } else if secs < 86_400 * 365 {
2152        format!("{}d", secs / 86_400)
2153    } else {
2154        format!("{}y", secs / (86_400 * 365))
2155    }
2156}
2157
2158/// Column name and type, for the home screen's preview pane.
2159pub type SchemaPreview = Vec<(String, polars::prelude::DataType)>;
2160
2161/// The preview of a table inside a file of tables, or of such a file (or NumPy array
2162/// file) as it opens. `None` for other entries, `Some(None)` when there is nothing to
2163/// show.
2164fn table_preview(entry: &Entry) -> Option<Option<SchemaPreview>> {
2165    let (file, format, name) = match &entry.table {
2166        Some(table) => match (crate::formats::members::split(&entry.path), table.format) {
2167            (Some((file, _)), Some(format)) => (file, format, Some(entry.name.as_str())),
2168            _ => return Some(None),
2169        },
2170        None if is_regular_file(&entry.path) => {
2171            let format = crate::formats::members::holder(&entry.path).or_else(|| {
2172                data_format(&entry.path).filter(|f| {
2173                    f.holds_tables() && crate::formats::readers::of(*f).table_schema.is_some()
2174                })
2175            })?;
2176            (entry.path.clone(), format, None)
2177        }
2178        None => return None,
2179    };
2180    Some(
2181        crate::formats::readers::of(format)
2182            .table_schema
2183            .and_then(|schema| schema(&file, name)),
2184    )
2185}
2186
2187/// The first Parquet file at or under `dir`, bounded in breadth and depth so thousands
2188/// of partitions cost what three do.
2189fn first_parquet_under(dir: &Path, depth: u8) -> Option<PathBuf> {
2190    if depth > MAX_WALK_DEPTH {
2191        return None;
2192    }
2193    let mut subdirs = Vec::new();
2194    for entry in std::fs::read_dir(dir).ok()?.flatten().take(64) {
2195        let path = entry.path();
2196        if path.is_dir() {
2197            subdirs.push(path);
2198        } else if is_parquet_path(&path) && is_regular_file(&path) {
2199            return Some(path);
2200        }
2201    }
2202    subdirs.sort();
2203    subdirs
2204        .into_iter()
2205        .take(4)
2206        .find_map(|d| first_parquet_under(&d, depth + 1))
2207}
2208
2209/// Column names from a Parquet footer.
2210pub fn column_names(meta: &crate::formats::parquet_footer::Footer) -> Vec<String> {
2211    meta.schema_descr
2212        .columns()
2213        .iter()
2214        .map(|c| c.path_in_schema.join("."))
2215        .collect()
2216}
2217
2218/// Whether `path` is a regular file safe to open: a FIFO, device or socket named
2219/// `.parquet` would block or worse, so reads are gated on kind (stat never blocks
2220/// like an open can).
2221fn is_regular_file(path: &Path) -> bool {
2222    std::fs::metadata(path)
2223        .map(|m| m.file_type().is_file())
2224        .unwrap_or(false)
2225}
2226
2227/// A dataset's column names and types without reading data; Parquet only, `None`
2228/// otherwise (the UI says so).
2229pub fn schema_preview(entry: &Entry) -> Option<SchemaPreview> {
2230    // `SerReader` is what brings `ParquetReader::new` into scope.
2231    use polars::prelude::{ParquetReader, Schema, SchemaExt, SerReader};
2232
2233    let file_path = match entry.kind {
2234        EntryKind::File => {
2235            if let Some(preview) = table_preview(entry) {
2236                return preview;
2237            }
2238            if !is_parquet_path(&entry.path) {
2239                return None;
2240            }
2241            entry.path.clone()
2242        }
2243        EntryKind::Hive | EntryKind::MultiFile => first_parquet_under(&entry.path, 0)?,
2244        EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => return None,
2245        // One data file's schema is not a lake table's (Iceberg field IDs and Delta column
2246        // mapping rename columns).
2247        EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => return None,
2248    };
2249
2250    if !is_regular_file(&file_path) {
2251        return None;
2252    }
2253    let file = std::fs::File::open(&file_path).ok()?;
2254    let mut reader = ParquetReader::new(file);
2255    let arrow_schema = reader.schema().ok()?;
2256    let schema = Schema::from_arrow_schema(arrow_schema.as_ref());
2257    let mut preview: SchemaPreview = Vec::new();
2258    // A hive table opens with partition keys hoisted first, so the pane lists them there,
2259    // typed from the path as the scan infers them.
2260    if entry.kind == EntryKind::Hive
2261        && let Ok(below) = file_path.strip_prefix(&entry.path)
2262    {
2263        for part in below.parent().into_iter().flat_map(Path::components) {
2264            let part = part.as_os_str().to_string_lossy();
2265            if let Some((key, value)) = part.split_once('=')
2266                && !key.is_empty()
2267                && schema.get(key).is_none()
2268            {
2269                // The scan's own inference, dates and booleans included.
2270                let dtype = if value.is_empty() || value == "__HIVE_DEFAULT_PARTITION__" {
2271                    polars::prelude::DataType::String
2272                } else {
2273                    polars::io::csv::read::schema_inference::infer_field_schema(value, true, false)
2274                };
2275                preview.push((key.to_string(), dtype));
2276            }
2277        }
2278    }
2279    preview.extend(
2280        schema
2281            .iter()
2282            .map(|(name, dtype)| (name.to_string(), dtype.clone())),
2283    );
2284    Some(preview)
2285}
2286
2287#[cfg(test)]
2288mod classification_tests;