Skip to main content

datui_lib/home/
discover.rs

1//! Dataset discovery for the home screen. Not a catalog: listings are computed from
2//! the filesystem when asked and forgotten at session end; between runs datui keeps
3//! only recent paths and measured shapes (`remembered`, valid while the files are
4//! unchanged). Every function scans one directory level, since data lives on slow
5//! mounts and huge partition trees.
6
7use std::path::{Path, PathBuf};
8
9/// Compression suffixes that may follow a data extension (`sales.csv.gz`).
10const COMPRESSION_EXTENSIONS: &[&str] = &["gz", "bz2", "xz", "zst", "zstd"];
11
12/// Upper bound on entries read from one directory, so a huge one cannot hang the UI.
13pub const MAX_ENTRIES_PER_DIR: usize = 5_000;
14
15/// What a home-screen row represents.
16#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
17#[serde(rename_all = "lowercase")]
18pub enum EntryKind {
19    /// A single data file.
20    File,
21    /// A file datui has no reader for: hidden on home until `Ctrl+A` shows it dimmed.
22    Other,
23    /// A directory of `key=value` partitions โ€” one dataset, not a tree to walk.
24    Hive,
25    /// A directory of similarly-shaped data files, openable as one table.
26    MultiFile,
27    /// A Delta Lake table: `_delta_log/` beside the data files.
28    Delta,
29    /// An Apache Iceberg table: `metadata/` holding the snapshots, `data/` the files.
30    Iceberg,
31    /// An Apache Hudi table: `.hoodie/` holding the timeline.
32    Hudi,
33    /// An ordinary directory, to descend into.
34    Directory,
35    /// Somewhere remote not looked at yet: classifying would read it, which blocks when
36    /// the network is gone, so it is offered as openable, unlabeled. Also what an
37    /// unrecognized kind deserializes as, so an older build can still read a dataset
38    /// index written by a newer one (see `CLASSIFIER_VERSION`).
39    #[serde(other)]
40    Unknown,
41}
42
43/// Bump whenever classification changes. A cached kind is restored without
44/// re-deriving it (a remote row cannot be read cheaply), so it is restored only when
45/// written by a build with the same version; measurements in the record (rows,
46/// columns, cost) survive regardless.
47pub const CLASSIFIER_VERSION: u32 = 7;
48
49impl EntryKind {
50    /// Short label shown next to the entry name.
51    pub fn label(self) -> &'static str {
52        match self {
53            // A file with no reader says nothing: it is dimmed, and Enter shows its bytes.
54            EntryKind::File | EntryKind::Other => "",
55            EntryKind::Hive => "hive",
56            EntryKind::MultiFile => "multi",
57            EntryKind::Delta => "delta",
58            EntryKind::Iceberg => "iceberg",
59            EntryKind::Hudi => "hudi",
60            EntryKind::Directory => "dir",
61            EntryKind::Unknown => "",
62        }
63    }
64
65    /// Whether selecting this entry opens a dataset rather than navigating. An unexamined
66    /// remote path counts: datui can open a prefix or hive directory directly.
67    pub fn is_dataset(self) -> bool {
68        !matches!(self, EntryKind::Directory | EntryKind::Other) && !self.is_lake_table()
69    }
70
71    /// Whether this row is known to be a dataset, for counting: unlike
72    /// [`EntryKind::is_dataset`] ("may this be opened"), a row nothing has looked into
73    /// does not count.
74    pub fn is_known_dataset(self) -> bool {
75        self != EntryKind::Unknown && self.is_dataset()
76    }
77
78    /// A Delta, Iceberg or Hudi table root: a log beside the data files that says which
79    /// of them are live, which datui does not read yet.
80    pub fn is_lake_table(self) -> bool {
81        matches!(
82            self,
83            EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi
84        )
85    }
86
87    /// The format's name for prose; `label` is the row's lowercase chip.
88    pub fn lake_name(self) -> Option<&'static str> {
89        match self {
90            EntryKind::Delta => Some("Delta"),
91            EntryKind::Iceberg => Some("Iceberg"),
92            EntryKind::Hudi => Some("Hudi"),
93            _ => None,
94        }
95    }
96}
97
98/// What one listing of a directory found, counted rather than judged. The row's label
99/// comes from here (`12 parquet`), true whether or not the files are one table.
100#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
101pub struct Holds {
102    /// Data files by format, commonest first, as [`crate::FileFormat::name`] strings so
103    /// records survive across builds.
104    #[serde(default)]
105    pub formats: Vec<(String, usize)>,
106    /// Subdirectories, partitions among them (`folders` in 0.4.0 development builds).
107    #[serde(default, alias = "folders")]
108    pub directories: usize,
109    /// `key=value` subdirectories, which are also counted in `directories`.
110    #[serde(default)]
111    pub partitions: usize,
112    /// Files datui has no reader for (a README, a script, a notebook): neither data nor a
113    /// writer's own.
114    #[serde(default)]
115    pub not_read: usize,
116    /// Files with no extension, neither data nor `not_read` by name: Spark and GBIF part
117    /// files, which the open reads by their bytes.
118    #[serde(default)]
119    pub unnamed: usize,
120    /// Entries skipped as a writer's own, and the first few by name for the pane.
121    #[serde(default)]
122    pub skipped: usize,
123    #[serde(default)]
124    pub skipped_names: Vec<String>,
125    /// The listing stopped at [`MAX_ENTRIES_PER_DIR`], so every count is a floor (`5000+`).
126    #[serde(default)]
127    pub truncated: bool,
128    /// A Hugging Face DatasetDict saved with `save_to_disk` (`dataset_dict.json` beside
129    /// split directories), read as Arrow, one split at a time.
130    #[serde(default)]
131    pub dataset_dict: bool,
132}
133
134/// How many skipped names are kept for the pane. Enough to recognise the convention.
135pub(crate) const SKIPPED_NAMES_SHOWN: usize = 4;
136
137impl Holds {
138    /// Data files of every format.
139    pub fn data_files(&self) -> usize {
140        self.formats.iter().map(|(_, n)| n).sum()
141    }
142
143    /// The one format this directory holds, when it holds exactly one.
144    pub fn one_format(&self) -> Option<&str> {
145        match self.formats.as_slice() {
146            [(name, _)] => Some(name),
147            _ => None,
148        }
149    }
150
151    /// The weights' format and file count when this directory is a model (one weight
152    /// format plus only JSON). See `is_model_directory`.
153    pub fn model_weights(&self) -> Option<(&str, usize)> {
154        if !is_model_directory(counts_names(self)) {
155            return None;
156        }
157        self.formats
158            .iter()
159            .find(|(name, _)| is_weights(name))
160            .map(|(name, count)| (name.as_str(), *count))
161    }
162
163    /// The directory row's label when its kind does not name itself: `12 parquet`,
164    /// `mixed`, or `dir` when no data is directly inside.
165    pub fn label(&self) -> String {
166        let more = if self.truncated { "+" } else { "" };
167        // A model's weights beside config and tokenizer JSON: labeled as the model, not
168        // `mixed`.
169        if let Some((name, count)) = self.model_weights() {
170            return format!("{count}{more} {name}");
171        }
172        match self.formats.as_slice() {
173            // `dir` says there is no data file inside; a cut listing cannot claim that. A
174            // directory of directories counts them.
175            [] if self.directories == 1 => format!("1 dir{more}"),
176            [] if self.directories > 1 => format!("{}{more} dirs", self.directories),
177            [] => format!("dir{more}"),
178            // The `+` hedges the whole claim: past the cap another format may lurk.
179            [(name, count)] => format!("{count}{more} {name}"),
180            // No `+`: `mixed` is not a count, and more files cannot unmake it.
181            _ => "mixed".to_string(),
182        }
183    }
184
185    /// Whether nothing has been counted: a file, or a directory not looked into. (An
186    /// empty directory that was looked into reads `dir` too.)
187    pub fn is_empty(&self) -> bool {
188        self.formats.is_empty()
189            && self.directories == 0
190            && self.skipped == 0
191            && self.not_read == 0
192            && self.unnamed == 0
193            // Every field, even those implied by others today: this guards a placeholder from
194            // erasing a row's count and should not depend on that invariant.
195            && self.partitions == 0
196            && self.skipped_names.is_empty()
197            && !self.dataset_dict
198            // A cut listing that found nothing still says more exists (`dir+`).
199            && !self.truncated
200    }
201
202    /// The details pane's `contains` line: data files by format, directories and
203    /// partitions. Unreadable files and writer markers are left out (they would read as a
204    /// warning); a row inside the directory says what is hidden.
205    pub fn line(&self, with_partitions: bool) -> Option<String> {
206        let more = if self.truncated { "+" } else { "" };
207        let mut parts: Vec<String> = self
208            .formats
209            .iter()
210            .map(|(name, count)| format!("{count}{more} {name}"))
211            .collect();
212        // Partitions are counted in `directories` too; name only the rest.
213        let plain = self.directories.saturating_sub(self.partitions);
214        if plain > 0 {
215            let word = if plain == 1 {
216                "directory"
217            } else {
218                "directories"
219            };
220            parts.push(format!("{plain}{more} {word}"));
221        }
222        if with_partitions && self.partitions > 0 {
223            let word = if self.partitions == 1 {
224                "partition"
225            } else {
226                "partitions"
227            };
228            parts.push(format!("{}{more} {word}", self.partitions));
229        }
230        (!parts.is_empty()).then(|| parts.join(" ยท "))
231    }
232}
233
234impl Entry {
235    /// Whether Enter lists the tables inside: a file of several that is not itself a
236    /// table. A file opening one of its tables, or a spec's variants file, opens on
237    /// Enter; โ†’ lists them.
238    pub fn enter_lists_tables(&self) -> bool {
239        self.kind == EntryKind::File
240            && self.cost.tables.is_some_and(|n| n > 1)
241            && !self.cost.opens_one
242            && self.format_spec.is_none()
243    }
244
245    /// Whether home hides this row until Ctrl+A: an unreadable file or a database's
246    /// internal table.
247    pub fn hidden_by_default(&self) -> bool {
248        self.kind == EntryKind::Other || self.table.as_ref().is_some_and(|t| t.internal)
249    }
250
251    /// The short label beside a row's name, saying what it holds: a looked-into
252    /// directory's count (`12 parquet`, `mixed`, `dir`), a lake table's or hive root's
253    /// format, else the kind.
254    pub fn label(&self) -> std::borrow::Cow<'static, str> {
255        match self.kind {
256            // See `opens_whole_directory`.
257            _ if self.opens_whole_directory => "".into(),
258            EntryKind::Directory | EntryKind::MultiFile if !self.holds.is_empty() => {
259                self.holds.label().into()
260            }
261            EntryKind::File if self.format_spec.is_some() => {
262                self.format_spec.clone().unwrap_or_default().into()
263            }
264            EntryKind::File if self.cost.tables.is_some() => {
265                let n = self.cost.tables.unwrap_or_default();
266                format!("{n} {}", if n == 1 { "table" } else { "tables" }).into()
267            }
268            // A file named for its contents (as a collection names one): its format, which the
269            // name no longer says.
270            EntryKind::File
271                if crate::FileFormat::from_path(Path::new(&self.name)).is_none()
272                    && crate::FileFormat::from_path(&self.path).is_some() =>
273            {
274                crate::FileFormat::from_path(&self.path)
275                    .map(crate::FileFormat::name)
276                    .unwrap_or_default()
277                    .into()
278            }
279            kind => kind.label().into(),
280        }
281    }
282}
283
284/// One row on the home screen.
285#[derive(Debug, Clone)]
286pub struct Entry {
287    pub path: PathBuf,
288    pub kind: EntryKind,
289    /// Display name โ€” the file or directory name, not the full path.
290    pub name: String,
291    /// Size in bytes; for multi-file datasets the sum of files inspected, a floor.
292    pub size: Option<u64>,
293    pub modified: Option<std::time::SystemTime>,
294    /// Row count, when it can be had without reading data (Parquet footers only).
295    pub rows: Option<usize>,
296    /// Column count, same caveat.
297    pub cols: Option<usize>,
298    /// Whether `cols` came from a spread of the directory (ends and middle, past the
299    /// footer budget), so it is a floor, shown as `6+`.
300    pub cols_sampled: bool,
301    /// Column names when free to get (a Parquet footer carries them with the row count).
302    pub columns: Vec<String>,
303    /// What opening this costs: where it lives, how it is stored and laid out, from bytes
304    /// already read.
305    pub cost: Cost,
306    /// What one listing found, for a directory; empty for a file or an unexamined
307    /// directory.
308    pub holds: Holds,
309    /// Whether this row is the door opening the browsed directory. It carries no label:
310    /// labels count what is directly inside, and this row reads all of it.
311    pub opens_whole_directory: bool,
312    /// The format spec that reads this file, by glob or by magic.
313    pub format_spec: Option<String>,
314    /// A table inside a file of tables (SQLite, a NumPy archive): its path is the file's
315    /// with the table's name appended.
316    pub table: Option<TableOf>,
317}
318
319/// What a row inside a file of tables says about its table.
320#[derive(Debug, Clone, PartialEq, Eq)]
321pub struct TableOf {
322    /// The containing file's format; `None` for a spec's variant (see
323    /// [`Entry::format_spec`]).
324    pub format: Option<crate::FileFormat>,
325    /// What the file calls it: SQLite's `table`, `view`, `virtual` or `shadow`, or a NumPy
326    /// archive's `array`.
327    pub kind: String,
328    /// SQLite's own (schema, statistics, shadow tables): hidden until Ctrl+A, opened like
329    /// any other.
330    pub internal: bool,
331}
332
333/// What pressing Enter on a dataset will cost (200 MB of zstd Parquet is gigabytes in
334/// memory, and NFS is not tmpfs), all from what datui already reads: the mount table
335/// and the footer.
336#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
337pub struct Cost {
338    /// Filesystem or URL scheme: `nfs4`, `ext4`, `tmpfs`, `fuse.sshfs`, `s3`.
339    #[serde(default, skip_serializing_if = "Option::is_none")]
340    pub source: Option<String>,
341    /// Bytes once decompressed, as against on disk.
342    #[serde(default, skip_serializing_if = "Option::is_none")]
343    pub uncompressed: Option<u64>,
344    /// Compression codec, as the file itself names it.
345    #[serde(default, skip_serializing_if = "Option::is_none")]
346    pub codec: Option<String>,
347    /// Row groups: one huge group cannot be read in parallel or skipped; thousands of
348    /// tiny ones cost overhead.
349    #[serde(default, skip_serializing_if = "Option::is_none")]
350    pub row_groups: Option<usize>,
351    /// Partition layout, for a hive dataset.
352    #[serde(default, skip_serializing_if = "Option::is_none")]
353    pub partitions: Option<Partitions>,
354    /// Tables of its own, for a file of tables: one opens, several are listed.
355    #[serde(default, skip_serializing_if = "Option::is_none")]
356    pub tables: Option<usize>,
357    /// Whether a file of several tables opens one (a workbook's first sheet): Enter opens
358    /// it and โ†’ lists them.
359    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
360    pub opens_one: bool,
361    /// An Arrow IPC stream (converted before scanning) rather than an IPC file (scanned in
362    /// place), from its first bytes.
363    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
364    pub ipc_stream: bool,
365}
366
367/// How opening a file row will read it. See [`how_read`].
368#[derive(Debug, Clone, Copy, PartialEq, Eq)]
369pub struct HowRead {
370    pub mode: crate::ReadMode,
371    /// A remote file that is downloaded whole before it is read.
372    pub download: bool,
373}
374
375/// How opening `entry` reads it ([`crate::FileFormat::read_mode`] for its format and
376/// storage), and whether a remote one is downloaded first. `None` unless a file's name
377/// says its format.
378pub fn how_read(entry: &Entry) -> Option<HowRead> {
379    use crate::Stored;
380    if entry.kind != EntryKind::File {
381        return None;
382    }
383    let stored = if crate::CompressionFormat::from_extension(&entry.path).is_some() {
384        Stored::Compressed { in_memory: false }
385    } else if entry.cost.ipc_stream {
386        Stored::Stream
387    } else {
388        Stored::Plain
389    };
390    let choice = match &entry.format_spec {
391        Some(name) => crate::cli::FormatChoice::Spec(name.clone()),
392        // A table inside a file of tables, at its path inside it (`shop.db/orders`).
393        None if entry.table.is_some() => {
394            crate::cli::FormatChoice::Builtin(entry.table.as_ref().and_then(|t| t.format)?)
395        }
396        None => crate::cli::FormatChoice::Builtin(data_format(&entry.path)?),
397    };
398    let mode = choice.read_mode(stored)?;
399    let download = match crate::cloud::source::input_source(&entry.path) {
400        crate::cloud::source::InputSource::Local(_) => false,
401        crate::cloud::source::InputSource::Http(_) => {
402            choice.http_file() == crate::RemoteRead::Downloaded
403        }
404        // A remote Arrow file is not peeked at here, so it is taken for an IPC file.
405        _ => choice.bucket_object(stored) == crate::RemoteRead::Downloaded,
406    };
407    Some(HowRead { mode, download })
408}
409
410/// How a hive dataset is laid out, from directory names alone.
411#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
412pub struct Partitions {
413    /// Partition keys, outermost first: `["year", "month"]`.
414    pub keys: Vec<String>,
415    /// Distinct values seen for the outermost key, sorted; bounded, so not necessarily
416    /// all.
417    pub first_key_values: Vec<String>,
418    /// Directories counted at the outermost level.
419    pub count: usize,
420    /// The count stopped at a limit; there are more.
421    pub more: bool,
422}
423
424impl Entry {
425    /// A plain directory row โ€” somewhere to step into, with nothing read from it.
426    pub fn directory(path: &Path) -> Self {
427        Self::new(path.to_path_buf(), EntryKind::Directory)
428    }
429
430    /// A file entry with a chosen display name, for tests.
431    pub fn for_test(path: &Path, name: &str) -> Self {
432        Self::new(path.to_path_buf(), EntryKind::File).with_name(name)
433    }
434
435    /// The row under a name other than its path's last part.
436    pub(crate) fn with_name(mut self, name: impl Into<String>) -> Self {
437        self.name = name.into();
438        self
439    }
440
441    pub(crate) fn new(path: PathBuf, kind: EntryKind) -> Self {
442        let name = path
443            .file_name()
444            .map(|n| n.to_string_lossy().into_owned())
445            .unwrap_or_else(|| path.to_string_lossy().into_owned());
446        Self {
447            path,
448            kind,
449            name,
450            size: None,
451            modified: None,
452            rows: None,
453            cols: None,
454            cols_sampled: false,
455            columns: Vec::new(),
456            cost: Cost::default(),
457            holds: Default::default(),
458            opens_whole_directory: false,
459            format_spec: None,
460            table: None,
461        }
462    }
463
464    /// Attach size and mtime from a directory entry's metadata.
465    pub(crate) fn with_fs_metadata(mut self, meta: &std::fs::Metadata) -> Self {
466        if meta.is_file() {
467            self.size = Some(meta.len());
468        }
469        self.modified = meta.modified().ok();
470        self
471    }
472}
473
474/// Whether a key or path is Parquet: named `.parquet`, or an extensionless part file
475/// in a `.parquet` directory (Spark, GBIF: `occurrence.parquet/000001`). Hidden and
476/// job files (`_SUCCESS`, `.crc`) are not.
477pub fn is_parquet_key(key: &str) -> bool {
478    let key = key.trim_end_matches('/');
479    let (directory, name) = match key.rsplit_once('/') {
480        Some((directory, name)) => (directory, name),
481        None => ("", key),
482    };
483    if is_bookkeeping(name) {
484        return false;
485    }
486    if name.to_ascii_lowercase().ends_with(".parquet") {
487        return true;
488    }
489    let directory_name = directory.rsplit('/').next().unwrap_or(directory);
490    !name.contains('.') && directory_name.to_ascii_lowercase().ends_with(".parquet")
491}
492
493#[cfg(test)]
494mod parquet_key_tests {
495    use super::is_parquet_key;
496
497    #[test]
498    fn parquet_without_an_extension_is_known_by_its_directory() {
499        assert!(is_parquet_key(
500            "occurrence/2026-09-01/occurrence.parquet/000001"
501        ));
502        assert!(is_parquet_key("data/part-0.parquet"));
503        assert!(is_parquet_key("DATA/PART-0.PARQUET"));
504        assert!(!is_parquet_key(
505            "occurrence/2026-09-01/occurrence.parquet/_SUCCESS"
506        ));
507        assert!(!is_parquet_key("occurrence.parquet/.part-0.crc"));
508        assert!(!is_parquet_key("occurrence/2026-09-01/citation.txt"));
509        assert!(!is_parquet_key("notes/000001"));
510    }
511}
512
513/// What a file with no usable extension is, from its first bytes (Parquet, Arrow,
514/// Avro and ORC carry signatures; CSV and JSON have none and are not guessed). Asked
515/// only of a directory being opened, never one being looked at; which signatures a
516/// listing trusts is each format's call (`crate::formats::readers::Trusted::listing`).
517pub fn sniff_format(path: &Path) -> Option<crate::FileFormat> {
518    crate::formats::readers::sniff_file(path, crate::formats::readers::Asked::Listing)
519}
520
521/// What a listing finds a file to be by its first bytes.
522#[derive(Debug, Clone)]
523pub enum Sniffed {
524    /// A format datui reads.
525    Format,
526    /// A format spec's, which reads it.
527    Spec(std::sync::Arc<crate::formats::Spec>),
528}
529
530/// [`sniff_format`], else the format spec an open would pick (by glob, else by magic
531/// and `match.where`), from one read of the file's head.
532pub fn sniff_listed(path: &Path, formats: &crate::formats::Registry) -> Option<Sniffed> {
533    use crate::formats::readers::{Asked, HEAD, head_of, sniff};
534    let head = head_of(path)?;
535    if sniff(&head, Some(path), Asked::Listing, |_| true).is_some() {
536        return Some(Sniffed::Format);
537    }
538    formats
539        .listed(path, &head, head.len() < HEAD)
540        .map(Sniffed::Spec)
541}
542
543/// Name `entry` a file of `spec`; a spec reading several record variants also makes
544/// it a place whose tables โ†’ lists.
545pub fn name_spec_file(entry: &mut Entry, spec: &crate::formats::Spec) {
546    entry.kind = EntryKind::File;
547    entry.format_spec = Some(spec.name.clone());
548    if spec.lists_variants() {
549        entry.cost.tables = Some(spec.records.variants.len());
550    }
551}
552
553/// Name an unlisted local file row (a recent) by the spec that reads it, as a listing
554/// would: by glob, else by magic when its name says nothing.
555pub fn name_unlisted_file(entry: &mut Entry, formats: &crate::formats::Registry) {
556    if formats.is_empty()
557        || entry.kind != EntryKind::File
558        || entry.table.is_some()
559        || is_data_file(&entry.path)
560    {
561        return;
562    }
563    let spec = match formats.by_glob(&entry.path, false).into_iter().next() {
564        Some(spec) => Some(spec),
565        None if worth_sniffing(&entry.path) && is_regular_file(&entry.path) => {
566            match sniff_listed(&entry.path, formats) {
567                Some(Sniffed::Spec(spec)) => Some(spec),
568                _ => None,
569            }
570        }
571        None => None,
572    };
573    if let Some(spec) = spec {
574        name_spec_file(entry, &spec);
575    }
576}
577
578/// How many extensionless files one listing sniffs; past this the rest are listed by
579/// name.
580pub(crate) const MAX_SNIFFS_PER_DIR: usize = 256;
581
582/// Whether a file's name has no extension at all: `part-00000`, `LICENSE`.
583pub fn has_no_extension(path: &Path) -> bool {
584    path.extension().is_none()
585}
586
587/// Whether a listing sniffs a file: no extension, an uninformative one (`.bin`), or
588/// only text (`.log`, which candump writes; `.txt`).
589pub fn worth_sniffing(path: &Path) -> bool {
590    path.extension().is_none_or(|e| {
591        e.eq_ignore_ascii_case("bin") || data_format(path).is_some_and(crate::FileFormat::is_lines)
592    })
593}
594
595/// Whether a local file is Parquet by its contents: `PAR1` at both ends.
596pub fn has_parquet_magic(path: &Path) -> bool {
597    use std::io::{Read, Seek, SeekFrom};
598    let Ok(mut file) = std::fs::File::open(path) else {
599        return false;
600    };
601    let mut head = [0u8; 4];
602    let mut tail = [0u8; 4];
603    file.read_exact(&mut head).is_ok()
604        && file.seek(SeekFrom::End(-4)).is_ok()
605        && file.read_exact(&mut tail).is_ok()
606        && &head == b"PAR1"
607        && &tail == b"PAR1"
608}
609
610/// Whether a path names a Parquet file, by extension or as an extensionless part file
611/// in a `.parquet` directory. Unlike [`is_parquet_key`], a writer's own file
612/// (`_manifest.parquet`) counts: it can be opened, even if it does not make its
613/// directory a dataset.
614pub fn is_parquet_path(path: &Path) -> bool {
615    path.extension()
616        .and_then(|e| e.to_str())
617        .is_some_and(|e| e.eq_ignore_ascii_case("parquet"))
618        || is_parquet_key(&directory_and_name(path))
619}
620
621/// Whether a path looks openable by name or place (a part file in a `.parquet`
622/// directory). Every route asks here so they agree.
623pub fn is_data_file(path: &Path) -> bool {
624    data_extension(path).is_some() || is_parquet_key(&directory_and_name(path))
625}
626
627/// The extension saying what a file is, past any compression suffix (`sales.csv.gz`
628/// is CSV); `None` when not something datui reads.
629pub fn data_extension(path: &Path) -> Option<String> {
630    let name = path.file_name().and_then(|n| n.to_str())?;
631    let lower = name.to_ascii_lowercase();
632    let mut parts: Vec<&str> = lower.rsplit('.').collect();
633    parts.reverse();
634    if parts.len() < 2 {
635        return None;
636    }
637    // Walk back past a compression suffix so `sales.csv.gz` still reads as CSV.
638    let mut idx = parts.len() - 1;
639    if COMPRESSION_EXTENSIONS.contains(&parts[idx]) && idx > 1 {
640        idx -= 1;
641    }
642    crate::FileFormat::from_extension(parts[idx]).map(|_| parts[idx].to_string())
643}
644
645/// What the home screen says of a file [`unreadable_by_name`] turns away.
646pub const NO_READER: &str = "datui has no reader for this file";
647
648/// Whether a name already rules the file out: an extension no reader takes, past any
649/// compression suffix. A bare `data.gz` or extensionless name is left to the open.
650pub fn unreadable_by_name(path: &Path) -> bool {
651    let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
652        return false;
653    };
654    let name = name.to_ascii_lowercase();
655    let parts: Vec<&str> = name.rsplit('.').collect();
656    let compressed = |last: &str| COMPRESSION_EXTENSIONS.contains(&last);
657    let ext = match parts[..] {
658        [last, inner, _, ..] if compressed(last) => inner,
659        [last, _] if compressed(last) => return false,
660        [last, _, ..] => last,
661        _ => return false,
662    };
663    crate::FileFormat::from_extension(ext).is_none()
664}
665
666/// The format a file's name says it holds, past any compression suffix. Extensions
667/// are not formats (`.ipc`, `.arrow`, `.arrows`, `.feather` are one); asking
668/// [`crate::FileFormat`] keeps home from listing what the reader cannot open.
669pub fn data_format(path: &Path) -> Option<crate::FileFormat> {
670    // A sharded checkpoint's index is the model, not a JSON table.
671    crate::FileFormat::from_name_ending(path)
672        .or_else(|| crate::FileFormat::from_extension(&data_extension(path)?))
673}
674
675/// The `dir/name` string `is_parquet_key` splits on `/`, so Windows backslashes do
676/// not make `occurrence.parquet\000001` one dotted name.
677pub(crate) fn directory_and_name(path: &Path) -> String {
678    let name = path.file_name().unwrap_or_default().to_string_lossy();
679    match path.parent().and_then(|p| p.file_name()) {
680        Some(directory) => format!("{}/{name}", directory.to_string_lossy()),
681        None => name.into_owned(),
682    }
683}
684
685/// Format rank: commonest first, Parquet winning ties, then by name, so the order
686/// never depends on `read_dir`. One order for the local label, the local reader
687/// choice and the cloud label, so a row's label and `Enter` agree.
688pub(crate) fn rank_formats(a: (&str, usize), b: (&str, usize)) -> std::cmp::Ordering {
689    // Text loses: a README among data files is not the table.
690    let text = crate::FileFormat::Text.name();
691    (a.0 == text)
692        .cmp(&(b.0 == text))
693        .then_with(|| b.1.cmp(&a.1))
694        .then_with(|| (a.0 != "parquet").cmp(&(b.0 != "parquet")))
695        .then_with(|| a.0.cmp(b.0))
696}
697
698/// Whether a file is Hugging Face `datasets` metadata beside Arrow shards, so its
699/// JSON does not outnumber a one-shard dataset and get read instead.
700pub(crate) fn is_hugging_face_metadata(name: &str) -> bool {
701    matches!(name, "dataset_info.json" | "state.json")
702}
703
704fn order_formats(counts: &mut [(crate::FileFormat, usize)]) {
705    counts.sort_by(|a, b| rank_formats((a.0.name(), a.1), (b.0.name(), b.1)));
706}
707
708/// Whether a listing entry is bookkeeping: a leading `_` or `.` (`_SUCCESS`,
709/// `_committed_*`, `_metadata.json`, `.crc`) or the `_$folder$` marker. One predicate
710/// for every listing and open, so they agree.
711pub fn is_bookkeeping(name: &str) -> bool {
712    // `_$folder$` first: the folder it stands for may be a partition (`year=2024_$folder$`
713    // beside `year=2024/`), which the partition test would call data.
714    if name.ends_with("_$folder$") {
715        return true;
716    }
717    // A `key=value` name is a partition, even with a leading `_` (Spark and Hive
718    // partition on internal columns like `_date=โ€ฆ`).
719    if is_partition_name(name) {
720        return false;
721    }
722    name.starts_with(['_', '.'])
723}
724
725/// Whether a format, by name, is model weights.
726fn is_weights(name: &str) -> bool {
727    name == crate::FileFormat::Safetensors.name() || name == crate::FileFormat::Gguf.name()
728}
729
730/// Whether formats side by side are a model: one weight format with only JSON beside
731/// (config, tokenizer), however many JSON files.
732pub(crate) fn is_model_directory<'a>(names: impl IntoIterator<Item = &'a str>) -> bool {
733    let mut weights = None;
734    for name in names {
735        if is_weights(name) {
736            if weights.is_some_and(|w| w != name) {
737                return false;
738            }
739            weights = Some(name);
740        } else if name != crate::FileFormat::Json.name() {
741            return false;
742        }
743    }
744    weights.is_some()
745}
746
747/// The format names `holds` counted.
748fn counts_names(holds: &Holds) -> impl Iterator<Item = &str> {
749    holds.formats.iter().map(|(name, _)| name.as_str())
750}
751
752/// Whether a name is a hive partition (`year=2024`): `key=value`, with a non-empty key.
753/// The value may be empty in practice.
754pub fn is_partition_name(name: &str) -> bool {
755    matches!(name.find('='), Some(i) if i > 0)
756}
757
758/// What the data files sitting directly in a directory say about how to read it.
759///
760/// A directory opened as one dataset has to be read by *something*, and the only honest
761/// source for that is the files in it. Before this existed the answer was assumed:
762/// any directory was scanned as Parquet, so a directory of `.json.gz` was opened by
763/// seeking to the end of each file for a `PAR1` that was never going to be there.
764#[derive(Debug, Clone, PartialEq, Eq)]
765pub enum DirectoryFormat {
766    /// Every data file directly in the directory reads as this one format, and these are
767    /// the files. Sorted, because a concatenation's row order is its file order.
768    One(crate::FileFormat, Vec<PathBuf>),
769    /// The files name more than one format. The commonest of them is the table โ€” a
770    /// directory of a thousand CSVs and one stray JSON is a directory of CSVs โ€” and the
771    /// rest are counted by format so the read can say what it passed over. Parquet wins a
772    /// tie, because it is the format a directory of data files is most likely to be about
773    /// and the one every other route here reads in place.
774    Mixed {
775        format: crate::FileFormat,
776        files: Vec<PathBuf>,
777        passed_over: Vec<(crate::FileFormat, usize)>,
778    },
779    /// The directory settles nothing by itself: it holds no readable data file directly,
780    /// or it has subdirectories and so may hold its data below. A hive dataset looks like
781    /// this โ€” its files are a level down, under `key=value`.
782    Deeper,
783}
784
785/// What [`DirectoryFormat`] the data files directly in `dir` amount to.
786///
787/// One level only, and no file is opened: this reads names, exactly as the rest of
788/// this module does. A directory of two hundred thousand files costs one listing
789/// and nothing per file beyond it โ€” the entry's own type comes back with the name, so
790/// there is no `stat` to bound.
791///
792/// Deliberately *not* bounded by [`MAX_ENTRIES_PER_DIR`], unlike every listing in this
793/// module. The files this returns are not a menu to show, they are the table to read:
794/// stopping at five thousand of a directory's six thousand CSVs would open it with a row
795/// count, a schema union and every aggregate quietly computed over a subset, and the
796/// `take` running before the sort would drop whichever file the directory read
797/// happened to return last. The Parquet route this mirrors hands the directory to a scan
798/// that enumerates it, with no cap either.
799pub fn directory_format(dir: &Path) -> DirectoryFormat {
800    let Ok(iter) = std::fs::read_dir(dir) else {
801        return DirectoryFormat::Deeper;
802    };
803
804    // Every format the directory names, with the files of each. A directory of one format
805    // takes the only entry; a directory of several takes the commonest and reports the
806    // rest, which is what stops one stray file deciding a directory cannot be read.
807    let mut by_format: Vec<(crate::FileFormat, Vec<PathBuf>)> = Vec::new();
808    // Files whose names say nothing, kept in case their bytes do. See below.
809    let mut nameless: Vec<PathBuf> = Vec::new();
810    let mut partitioned = false;
811    for entry in iter.flatten() {
812        let path = entry.path();
813        let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
814            continue;
815        };
816        // A table format's own files are not the table's, and a dotfile is nobody's.
817        // The same test every other route makes, so they agree on what is data.
818        if is_bookkeeping(name) {
819            continue;
820        }
821        // The type the directory read already returned, rather than a `stat` per
822        // entry: on a share that is a round trip per entry, and the question is only
823        // whether this is a file. A symlink still gets the stat, because `d_type`
824        // cannot say what is on the other end of one.
825        let is_file = match entry.file_type() {
826            Ok(kind) if kind.is_symlink() => is_regular_file(&path),
827            Ok(kind) => kind.is_file(),
828            Err(_) => is_regular_file(&path),
829        };
830        if !is_file {
831            // Only a partition, not any subdirectory: a directory of CSVs with some
832            // unrelated directory beside them is still a directory of CSVs, and handing
833            // that to a scanner that walks trees is the very thing this function exists
834            // to stop.
835            partitioned |= is_partition_dir(&path);
836            continue;
837        }
838        let Some(found) = data_format(&path) else {
839            // A name that says nothing. Kept rather than dropped, because the bytes may
840            // still say what it is โ€” see below, where they are asked.
841            if path.extension().is_none() {
842                nameless.push(path);
843            }
844            continue;
845        };
846        match by_format.iter_mut().find(|(f, _)| *f == found) {
847            Some((_, of_that_format)) => of_that_format.push(path),
848            None => by_format.push((found, vec![path])),
849        }
850    }
851
852    // Extensionless files (Spark and GBIF part files) in a directory whose names settled
853    // nothing; skipped when names already answer (a `LICENSE` beside Parquet). A
854    // [`spread`] is sniffed, not every file: the ends must agree, and then all are taken
855    // as that format (the scan reports any odd one out).
856    if by_format.is_empty() && !nameless.is_empty() {
857        nameless.sort();
858        let picks = spread(nameless.len());
859        let sniffed: Vec<crate::FileFormat> = picks
860            .iter()
861            .filter_map(|i| sniff_format(&nameless[*i]))
862            .collect();
863        if sniffed.len() == picks.len()
864            && let Some(found) = sniffed.first().copied()
865            && sniffed.iter().all(|f| *f == found)
866        {
867            by_format.push((found, nameless));
868        }
869    }
870
871    // Text is data only where nothing else is: a README beside Parquet is not a
872    // candidate.
873    if by_format.iter().any(|(f, _)| !f.is_lines()) {
874        by_format.retain(|(f, _)| !f.is_lines());
875    }
876
877    // A Hugging Face dataset's own JSON files, beside its shards.
878    if by_format
879        .iter()
880        .any(|(f, _)| *f == crate::FileFormat::Arrow)
881    {
882        for (format, files) in &mut by_format {
883            if *format == crate::FileFormat::Json {
884                files.retain(|f| {
885                    !f.file_name()
886                        .and_then(|n| n.to_str())
887                        .is_some_and(is_hugging_face_metadata)
888                });
889            }
890        }
891        by_format.retain(|(_, files)| !files.is_empty());
892    }
893
894    // Decided after the whole listing, not at the first deciding entry, so read order
895    // does not matter. One `key=value` below and the data is down there.
896    if partitioned {
897        return DirectoryFormat::Deeper;
898    }
899
900    // The shared format rank, so the reader picked is the format the label names.
901    by_format.sort_by(|a, b| rank_formats((a.0.name(), a.1.len()), (b.0.name(), b.1.len())));
902    // A directory of model weights is the model, however much config and tokenizer
903    // JSON sits beside it; the JSON is passed over.
904    if is_model_directory(by_format.iter().map(|(f, _)| f.name()))
905        && let Some(at) = by_format.iter().position(|(f, _)| is_weights(f.name()))
906    {
907        let weights = by_format.remove(at);
908        by_format.insert(0, weights);
909    }
910    let mut by_format = by_format.into_iter();
911    let Some((format, mut files)) = by_format.next() else {
912        return DirectoryFormat::Deeper;
913    };
914    files.sort();
915    let passed_over: Vec<(crate::FileFormat, usize)> =
916        by_format.map(|(f, of_that)| (f, of_that.len())).collect();
917    if passed_over.is_empty() {
918        DirectoryFormat::One(format, files)
919    } else {
920        DirectoryFormat::Mixed {
921            format,
922            files,
923            passed_over,
924        }
925    }
926}
927
928/// How far down a hive root is followed for its files (year/month/day/hour is four);
929/// deeper is not a hive dataset, and each level costs a read on a share.
930const MAX_HIVE_DEPTH: usize = 16;
931
932/// What the files under a hive root's partitions are. [`directory_format`] can only
933/// say `Deeper` of a root, so this follows one spine (the first partition at each
934/// level, as a hive scan reads its schema) to the bottom. A sample: disagreeing
935/// partitions report as the first one.
936pub fn hive_leaf_format(dir: &Path) -> DirectoryFormat {
937    let mut at = dir.to_path_buf();
938    for _ in 0..MAX_HIVE_DEPTH {
939        match directory_format(&at) {
940            // Nothing here settles it. Follow the partitions down, if there are any.
941            DirectoryFormat::Deeper => match first_partition(&at) {
942                Some(next) => at = next,
943                None => return DirectoryFormat::Deeper,
944            },
945            settled => return settled,
946        }
947    }
948    DirectoryFormat::Deeper
949}
950
951/// The first `key=value` subdirectory of `dir` by name, so runs agree.
952fn first_partition(dir: &Path) -> Option<PathBuf> {
953    let iter = std::fs::read_dir(dir).ok()?;
954    iter.flatten()
955        .take(MAX_ENTRIES_PER_DIR)
956        .map(|entry| entry.path())
957        .filter(|path| is_partition_dir(path) && path.is_dir())
958        .min()
959}
960
961/// Whether a directory name is a hive partition (`year=2024`).
962fn is_partition_dir(path: &Path) -> bool {
963    path.file_name()
964        .and_then(|n| n.to_str())
965        .is_some_and(is_partition_name)
966}
967
968/// An empty extensionless object: a tool's folder marker (`yellow/year=2032`), whether
969/// or not the folder still exists. Nothing datui opens is both empty and nameless.
970pub fn is_empty_marker(name: &str, size: u64) -> bool {
971    size == 0 && !name.contains('.')
972}
973
974/// Classify a directory without walking it: one listing bounded by
975/// [`MAX_ENTRIES_PER_DIR`] (not a probe of the first few entries, whose answer would
976/// depend on read order). Includes rows on network mounts: one `getdents` walk,
977/// stat'ing only symlinks. `home_open_selected` calls it on the key thread; others
978/// on workers.
979pub fn classify_directory(path: &Path) -> EntryKind {
980    look_at_directory(path).0
981}
982
983/// The kind (what `Enter` does) and what the listing found (what the label says),
984/// from one read.
985pub fn look_at_directory(path: &Path) -> (EntryKind, Holds) {
986    // Before the listing: three `join` tests answer it, where counting would walk every
987    // table of a warehouse each pass. A bucket prefix finds markers in its listing.
988    if let Some(lake) = lake_table(path) {
989        return (lake, Holds::default());
990    }
991    let Ok(iter) = std::fs::read_dir(path) else {
992        return (EntryKind::Directory, Holds::default());
993    };
994    let directory = path.file_name().unwrap_or_default().to_string_lossy();
995    let rules = Rules {
996        directory: &directory,
997        // As the listing of the rows inside does, so the tally agrees with them.
998        sniff: (!crate::home::is_remote_path(path)).then_some(path),
999        in_bucket: false,
1000    };
1001    // One past the cap, so "there is more" is known; bounded where entries come from,
1002    // since `.crc` files beside each data file would double the walk. Past the cap the
1003    // first entries decide.
1004    let mut truncated = false;
1005    let seen = iter
1006        .flatten()
1007        .take(MAX_ENTRIES_PER_DIR + 1)
1008        .enumerate()
1009        .map_while(|(at, entry)| {
1010            truncated = at == MAX_ENTRIES_PER_DIR;
1011            (!truncated).then(|| seen_on_disk(&entry))
1012        });
1013    let (kind, mut holds) = classify(seen, &rules);
1014    holds.truncated = truncated;
1015    (kind, holds)
1016}
1017
1018/// A local entry as [`classify`] sees it, from the read's file type (no stat per
1019/// entry; symlinks still need one). Regular files only: a FIFO named `a.csv` blocks
1020/// its opener, a broken symlink opens as nothing.
1021fn seen_on_disk(entry: &std::fs::DirEntry) -> Seen {
1022    let (is_dir, is_file) = match entry.file_type() {
1023        Ok(kind) if !kind.is_symlink() => (kind.is_dir(), kind.is_file()),
1024        _ => std::fs::metadata(entry.path()).map_or((false, false), |m| (m.is_dir(), m.is_file())),
1025    };
1026    Seen {
1027        name: entry.file_name().to_string_lossy().into_owned(),
1028        is_dir,
1029        is_file,
1030        size: None,
1031    }
1032}
1033
1034/// One entry of a listing, on disk or in a bucket, as [`classify`] asks about it.
1035#[derive(Debug, Clone)]
1036pub struct Seen {
1037    pub name: String,
1038    pub is_dir: bool,
1039    /// A regular file an open can read (not a FIFO, socket or broken symlink).
1040    pub is_file: bool,
1041    /// Bytes, where listed; an empty extensionless file is a folder marker, not data.
1042    pub size: Option<u64>,
1043}
1044
1045/// Where the local and the bucket listings deliberately differ.
1046pub struct Rules<'a> {
1047    /// The listed directory's name: an extensionless part file in `occurrence.parquet/` is
1048    /// Parquet by where it sits.
1049    pub directory: &'a str,
1050    /// Where to look inside nameless files, a few per listing; `None` where each open is
1051    /// a round trip that may not return.
1052    pub sniff: Option<&'a Path>,
1053    /// A bucket prefix: lake tables are found by names in the listing (Iceberg by layout
1054    /// alone), a DatasetDict by `dataset_dict.json`, and only Parquet (read in place) is
1055    /// offered as many files that are one table.
1056    pub in_bucket: bool,
1057}
1058
1059/// A directory's kind and holdings from one listing level: the one rule for
1060/// [`look_at_directory`] and bucket prefixes, so a directory and its mirror agree;
1061/// [`Rules`] holds their differences. The caller sets truncation.
1062pub fn classify(seen: impl Iterator<Item = Seen>, rules: &Rules) -> (EntryKind, Holds) {
1063    use crate::FileFormat;
1064    let mut holds = Holds::default();
1065    let mut counts: Vec<(FileFormat, usize)> = Vec::new();
1066    let mut skipped: Vec<String> = Vec::new();
1067    // Counted as data until the listing is done; see `is_hugging_face_metadata`.
1068    let mut hugging_face: Vec<String> = Vec::new();
1069    let mut dict_file = false;
1070    let mut lake: Vec<&'static str> = Vec::new();
1071    // `present` is everything but a writer's own: what the majority is measured against.
1072    let (mut present, mut data_files, mut parquet) = (0usize, 0usize, 0usize);
1073    let mut sniffs_left = rules.sniff.map_or(0, |_| MAX_SNIFFS_PER_DIR);
1074    for s in seen {
1075        // Lake markers are spec, not strays: looked for before bookkeeping skips
1076        // `_delta_log`.
1077        if s.is_dir
1078            && rules.in_bucket
1079            && let Some(marker) = ["_delta_log", ".hoodie", "metadata", "data"]
1080                .into_iter()
1081                .find(|m| *m == s.name)
1082        {
1083            lake.push(marker);
1084        }
1085        if is_bookkeeping(&s.name) {
1086            skipped.push(s.name);
1087            continue;
1088        }
1089        present += 1;
1090        if s.is_dir {
1091            holds.directories += 1;
1092            holds.partitions += usize::from(is_partition_name(&s.name));
1093            continue;
1094        }
1095        if s.size.is_some_and(|size| is_empty_marker(&s.name, size)) {
1096            holds.not_read += 1;
1097            continue;
1098        }
1099        let name = Path::new(&s.name);
1100        let key = format!("{}/{}", rules.directory, s.name);
1101        let named = data_format(name);
1102        let found = named
1103            // Text by its name, unless its bytes say more: below.
1104            .filter(|f| !f.is_lines())
1105            // A sharded checkpoint's index counts as JSON, so the label counts shards; the read
1106            // still uses it.
1107            .map(
1108                |f| match crate::formats::model_files::is_safetensors_index(name) {
1109                    true => FileFormat::Json,
1110                    false => f,
1111                },
1112            )
1113            // Data by where it sits rather than by its name: see [`is_data_file`].
1114            .or_else(|| is_parquet_key(&key).then_some(FileFormat::Parquet))
1115            .or_else(|| {
1116                let dir = rules.sniff?;
1117                spend_sniff(&mut sniffs_left, s.is_file, name)
1118                    .then(|| sniff_format(&dir.join(name)))?
1119            })
1120            .or(named)
1121            .filter(|_| s.is_file);
1122        let Some(found) = found else {
1123            // No reader and no name; extensionless files are counted apart (Spark part files).
1124            match s.is_file && has_no_extension(name) {
1125                true => holds.unnamed += 1,
1126                false => holds.not_read += 1,
1127            }
1128            continue;
1129        };
1130        if found == FileFormat::Json && is_hugging_face_metadata(&s.name) {
1131            hugging_face.push(s.name.clone());
1132        }
1133        dict_file |= found == FileFormat::Json && s.name == crate::formats::hf_splits::DATASET_DICT;
1134        data_files += 1;
1135        parquet += usize::from(is_parquet_key(&key));
1136        match counts.iter_mut().find(|(f, _)| *f == found) {
1137            Some((_, n)) => *n += 1,
1138            None => counts.push((found, 1)),
1139        }
1140    }
1141
1142    // Hugging Face's JSON is its writer's own, like `_SUCCESS`, as is a DatasetDict's
1143    // `dataset_dict.json`.
1144    if !counts.iter().any(|(f, _)| *f == FileFormat::Arrow) {
1145        hugging_face.clear();
1146    }
1147    holds.dataset_dict = rules.in_bucket && dict_file && holds.directories > 0;
1148    if holds.dataset_dict {
1149        hugging_face.push(crate::formats::hf_splits::DATASET_DICT.to_string());
1150    }
1151    if let Some((_, n)) = counts.iter_mut().find(|(f, _)| *f == FileFormat::Json) {
1152        *n -= hugging_face.len();
1153        data_files -= hugging_face.len();
1154        present -= hugging_face.len();
1155        skipped.append(&mut hugging_face);
1156    }
1157    // Text is data only where nothing else is.
1158    if counts.iter().any(|(f, _)| !f.is_lines()) {
1159        for (_, n) in counts.iter_mut().filter(|(f, _)| f.is_lines()) {
1160            data_files -= *n;
1161            holds.not_read += std::mem::take(n);
1162        }
1163    }
1164    counts.retain(|(_, n)| *n > 0);
1165    order_formats(&mut counts);
1166    let one_readable = matches!(counts.as_slice(), [(f, _)] if f.reads_many_files());
1167    holds.formats = counts
1168        .into_iter()
1169        .map(|(f, n)| (f.name().to_string(), n))
1170        .collect();
1171    // The first few by name, so runs agree; an object and prefix of one name are one
1172    // thing.
1173    skipped.sort();
1174    skipped.dedup();
1175    holds.skipped = skipped.len();
1176    skipped.truncate(SKIPPED_NAMES_SHOWN);
1177    holds.skipped_names = skipped;
1178
1179    // Lake tables' data files agree on a schema, so the rules below would wrongly say
1180    // "one table". Iceberg needs its whole shape (data under `data/`).
1181    let marked = |m| lake.contains(&m);
1182    if marked("_delta_log") {
1183        return (EntryKind::Delta, holds);
1184    }
1185    if marked(".hoodie") {
1186        return (EntryKind::Hudi, holds);
1187    }
1188    if marked("metadata") && marked("data") && parquet == 0 {
1189        return (EntryKind::Iceberg, holds);
1190    }
1191    // No majority rule: one stray `notes=old/` reads `hive`, but majorities refused real
1192    // hive roots with a README or `scripts/` beside them, the worse mistake.
1193    if holds.partitions > 0 && holds.partitions >= data_files {
1194        return (EntryKind::Hive, holds);
1195    }
1196    // One multi-file-readable format that dominates: a couple of stray CSVs make a
1197    // place, not a dataset. A model opens as the model despite its JSON.
1198    let one_table = if rules.in_bucket {
1199        parquet > 1 && parquet == data_files && parquet * 2 >= present
1200    } else {
1201        (data_files > 1 && one_readable && data_files * 2 >= present)
1202            || (holds.partitions == 0 && is_model_directory(counts_names(&holds)))
1203    };
1204    let kind = match one_table {
1205        true => EntryKind::MultiFile,
1206        false => EntryKind::Directory,
1207    };
1208    (kind, holds)
1209}
1210
1211/// Spend one of a listing's sniffs on `name`: a regular file whose name says nothing
1212/// ([`worth_sniffing`]), budget permitting.
1213fn spend_sniff(left: &mut usize, is_file: bool, name: &Path) -> bool {
1214    let spend = is_file && *left > 0 && worth_sniffing(name);
1215    *left -= usize::from(spend);
1216    spend
1217}
1218
1219/// Entries under `metadata/` checked for Iceberg: `v1.metadata.json` is written on
1220/// the first commit and never removed, but read order is arbitrary.
1221const ICEBERG_METADATA_PROBE: usize = 64;
1222
1223/// Whether `path` is a lake table root, and which. Marker names are part of these
1224/// formats' specs (unlike filename conventions); three `join` tests answer it
1225/// without walking a table's files.
1226fn lake_table(path: &Path) -> Option<EntryKind> {
1227    if path.join("_delta_log").is_dir() {
1228        return Some(EntryKind::Delta);
1229    }
1230    if path.join(".hoodie").is_dir() {
1231        return Some(EntryKind::Hudi);
1232    }
1233    // Iceberg's marker is a plain name, so it needs the whole shape: `metadata/` with a
1234    // metadata file, beside `data/`.
1235    let metadata = path.join("metadata");
1236    if path.join("data").is_dir()
1237        && metadata.is_dir()
1238        && std::fs::read_dir(&metadata).is_ok_and(|entries| {
1239            entries
1240                .flatten()
1241                .take(ICEBERG_METADATA_PROBE)
1242                .any(|e| e.file_name().to_string_lossy().ends_with(".metadata.json"))
1243        })
1244    {
1245        return Some(EntryKind::Iceberg);
1246    }
1247    None
1248}
1249
1250/// What one directory listing produced, and whether it saw all of it.
1251#[derive(Debug, Clone, Default)]
1252pub struct Scan {
1253    pub entries: Vec<Entry>,
1254    /// The directory held more than `MAX_ENTRIES_PER_DIR`; `entries` is a prefix, which
1255    /// the UI must say.
1256    pub truncated: bool,
1257}
1258
1259/// [`scan_dir_progressive`], also naming sniffed files by `formats`' magic.
1260pub fn scan_dir_specs(dir: &Path, formats: &crate::formats::Registry) -> Scan {
1261    scan_dir_with(dir, formats, |_| {})
1262}
1263
1264/// How often a listing still being read shows what it has so far.
1265const LISTING_PROGRESS_EVERY: std::time::Duration = std::time::Duration::from_millis(250);
1266
1267/// List one directory with bounded work: one `read_dir` and a stat per entry, up to
1268/// [`MAX_ENTRIES_PER_DIR`]; errors give an empty listing. Nothing is classified:
1269/// subdirectories return [`EntryKind::Unknown`] and are looked into later from the
1270/// viewport, so a label is a fact about the row, not its position. `progress` gets
1271/// the rows read since the last call, every `LISTING_PROGRESS_EVERY`, so a slow share
1272/// shows rows as they arrive.
1273pub fn scan_dir_progressive(dir: &Path, progress: impl FnMut(&[Entry])) -> Scan {
1274    scan_dir_with(dir, &crate::formats::Registry::default(), progress)
1275}
1276
1277fn scan_dir_with(
1278    dir: &Path,
1279    formats: &crate::formats::Registry,
1280    mut progress: impl FnMut(&[Entry]),
1281) -> Scan {
1282    let Ok(iter) = std::fs::read_dir(dir) else {
1283        return Scan::default();
1284    };
1285    let mut shown = std::time::Instant::now();
1286
1287    let mut entries = Vec::new();
1288    // How many of `entries` `progress` has been handed.
1289    let mut sent = 0usize;
1290    let mut seen = 0usize;
1291    let mut truncated = false;
1292    // Extensionless files are sniffed (a few bytes) so part files list as data; never on
1293    // a share, where each open is a round trip.
1294    let mut sniffs_left = if crate::home::is_remote_path(dir) {
1295        0
1296    } else {
1297        MAX_SNIFFS_PER_DIR
1298    };
1299
1300    // One past the cap: enough to know more exists without paying to process it.
1301    for dir_entry in iter.flatten().take(MAX_ENTRIES_PER_DIR + 1) {
1302        seen += 1;
1303        if seen > MAX_ENTRIES_PER_DIR {
1304            truncated = true;
1305            break;
1306        }
1307
1308        let path = dir_entry.path();
1309        let name = dir_entry.file_name();
1310        if name.to_string_lossy().starts_with('.') {
1311            continue;
1312        }
1313
1314        let Ok(meta) = dir_entry.metadata() else {
1315            continue;
1316        };
1317
1318        let mut spec = None;
1319        let kind = if meta.is_dir() {
1320            EntryKind::Unknown
1321        } else if meta.is_file() && is_data_file(&path) {
1322            EntryKind::File
1323        } else if spend_sniff(&mut sniffs_left, meta.is_file(), &path) {
1324            match sniff_listed(&path, formats) {
1325                Some(Sniffed::Format) => EntryKind::File,
1326                Some(Sniffed::Spec(found)) => {
1327                    spec = Some(found);
1328                    EntryKind::File
1329                }
1330                None => EntryKind::Other,
1331            }
1332        } else if meta.is_file() {
1333            EntryKind::Other
1334        } else {
1335            // Not a directory or regular file: a FIFO named `x.parquet` must never be offered.
1336            continue;
1337        };
1338
1339        let mut entry = Entry::new(path, kind).with_fs_metadata(&meta);
1340        if let Some(spec) = spec {
1341            name_spec_file(&mut entry, &spec);
1342        }
1343        entries.push(entry);
1344        if shown.elapsed() >= LISTING_PROGRESS_EVERY {
1345            progress(&entries[sent..]);
1346            sent = entries.len();
1347            shown = std::time::Instant::now();
1348        }
1349    }
1350
1351    sort_entries(&mut entries);
1352    Scan { entries, truncated }
1353}
1354
1355/// Datasets first, then directories, each alphabetical (a listing is scanned by
1356/// name). Unexamined rows sort with directories, where they land if plain, so a kind
1357/// arriving later never moves a row under the cursor.
1358pub(crate) fn sort_entries(entries: &mut [Entry]) {
1359    entries.sort_by(|a, b| {
1360        // Data, then directories, then what datui cannot read.
1361        let group = |k: EntryKind| match k {
1362            k if k.is_known_dataset() => 0,
1363            EntryKind::Other => 2,
1364            _ => 1,
1365        };
1366        group(a.kind).cmp(&group(b.kind)).then_with(|| {
1367            a.name
1368                .to_ascii_lowercase()
1369                .cmp(&b.name.to_ascii_lowercase())
1370        })
1371    });
1372}
1373
1374/// How far below a directory the footer walk goes (year/month/day/hour is four).
1375const MAX_WALK_DEPTH: u8 = 4;
1376
1377/// Upper bound on footers read to size a multi-file or hive dataset; past it the count
1378/// is left blank rather than partial.
1379const MAX_FOOTERS_PER_DATASET: usize = 64;
1380
1381/// Fill in row and column counts from Parquet footers: one file, or a bounded sum for
1382/// hive and multi-file datasets. Non-Parquet keeps `None`, a blank in the UI.
1383pub fn enrich(entry: &mut Entry) {
1384    enrich_as(entry, &crate::formats::schema_union::ReadAs::default())
1385}
1386
1387/// As [`enrich`], reading files as the following open will: where the header is
1388/// decides what the names are, so judging with other settings would misjudge (e.g.
1389/// `--no-header`).
1390pub fn enrich_as(entry: &mut Entry, as_read: &crate::formats::schema_union::ReadAs) {
1391    enrich_with(entry, as_read, None)
1392}
1393
1394/// As [`enrich_as`], using the shape an open kept in `remembered` while the files are
1395/// unchanged, instead of sampling footers.
1396pub fn enrich_with(
1397    entry: &mut Entry,
1398    as_read: &crate::formats::schema_union::ReadAs,
1399    remembered: Option<&crate::cache::CacheManager>,
1400) {
1401    match entry.kind {
1402        EntryKind::File => {
1403            enrich_parquet(entry);
1404            enrich_tables(entry);
1405            enrich_arrow(entry);
1406        }
1407        EntryKind::Hive | EntryKind::MultiFile => enrich_dataset(entry, as_read, remembered),
1408        // Nothing to read for plain or unexamined directories, nor lake tables (summing their
1409        // footers counts tombstoned and rewritten rows).
1410        EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => {}
1411        EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => {}
1412    }
1413}
1414
1415/// Sum footers across a bounded set of Parquet files under `entry`.
1416fn enrich_dataset(
1417    entry: &mut Entry,
1418    as_read: &crate::formats::schema_union::ReadAs,
1419    remembered: Option<&crate::cache::CacheManager>,
1420) {
1421    // A directory whose own files are a format this cannot count is not described by
1422    // Parquet beneath it (`6 json` reporting the subdirectories' columns). Parquet with
1423    // Parquet beneath keeps the walk: `Enter` reads the subtree, so counts must too; the
1424    // `holds` line explains the difference.
1425
1426    // The partition layout comes from directory names, so it is known even when the rows
1427    // are too many to count.
1428    if entry.kind == EntryKind::Hive {
1429        entry.cost.partitions = partition_layout(&entry.path);
1430    }
1431
1432    // Whether the footers below describe this directory (exact for its own format). Not
1433    // asked of a hive root: its own files are strays (a `schema.json`), and its data's
1434    // format can only be sampled.
1435    let reads_as_parquet = entry.kind == EntryKind::Hive
1436        || match entry.holds.one_format() {
1437            // No single format: unreachable today (mixed formats are `Directory`), so default to
1438            // the safe side.
1439            None => true,
1440            // An unreadable name is not Parquet: missing counts are undone by opening; another
1441            // format's counts are not.
1442            Some(name) => crate::FileFormat::from_name(name) == Some(crate::FileFormat::Parquet),
1443        };
1444    if !reads_as_parquet {
1445        entry.size = None;
1446        judge_by_names(entry, as_read);
1447        return;
1448    }
1449
1450    // A directory's stat size is its inode, not its contents: dropped here, restored only
1451    // if the files are totalled.
1452    entry.size = None;
1453
1454    let files = parquet_files_under(&entry.path);
1455    // Past the budget, footers an open kept are used if the listing still matches.
1456    if files.len() > MAX_FOOTERS_PER_DATASET
1457        && let Some((listed, footers)) = remembered
1458            .and_then(|cache| crate::formats::dataset_files::remembered_footers(&entry.path, cache))
1459    {
1460        measure_from_footers(entry, &listed, &footers);
1461        return;
1462    }
1463    if files.is_empty() || files.len() > MAX_FOOTERS_PER_DATASET {
1464        // Whether these are one table needs only three spread footers, and matters most for
1465        // directories past the counting limit.
1466        let sampled = sample_footers(&files);
1467        let names: Vec<Vec<String>> = sampled.iter().map(column_names).collect();
1468        let tops: Vec<Vec<String>> = names
1469            .iter()
1470            .map(|n| crate::formats::schema_union::top_level_columns(n))
1471            .collect();
1472        if entry.kind == EntryKind::MultiFile && one_table_from(&tops) == Some(false) {
1473            // One-table is asked of the subtree (what opening unions), but holdings count only
1474            // the direct files (the label's), since a downgraded row never opens as one table.
1475            let own_files = direct_children(&files, &entry.path);
1476            let own = sample_footers(&own_files);
1477            // The names are every sampled one under it: home's search index should find a
1478            // directory by any column looking inside reaches.
1479            entry.columns = union_of(&names);
1480            // A floor only if one of its own footers went unread.
1481            entry.cols_sampled = own.len() < own_files.len();
1482            // The columns a reader sees, from the schema rather than splitting leaf paths on dots
1483            // (`user.id` vs struct `user` with `id`).
1484            let top = union_of(&own.iter().map(top_level_names).collect::<Vec<_>>());
1485            downgrade_to_directory(entry, (!top.is_empty()).then_some(top.len()));
1486            return;
1487        }
1488        // Still worth knowing the shape, even when the row count is out of reach.
1489        if let Some(meta) = sampled.first() {
1490            // Three files, not the first: a growing dataset's newest columns are in its last
1491            // file. Still a sample; the count is already `?`.
1492            entry.columns = union_of(&names);
1493            let top = union_of(&sampled.iter().map(top_level_names).collect::<Vec<_>>());
1494            entry.cols = Some(top.len() + partition_columns_beyond(entry, &top));
1495            entry.cols_sampled = true;
1496            // From one file: codec and row-group sizing are the writer's, uniform in practice.
1497            physical_facts(meta, &mut entry.cost);
1498            entry.cost.uncompressed = None;
1499        }
1500        return;
1501    }
1502
1503    let mut rows = 0usize;
1504    let mut bytes = 0u64;
1505    // Every column any file has, in first-seen order, so columns added over time are
1506    // found.
1507    let mut columns: Vec<String> = Vec::new();
1508    let mut seen_columns = std::collections::HashSet::new();
1509    // Top-level columns unioned the same way, kept beside the leaves (see
1510    // [`top_level_names`]).
1511    let mut top_level: Vec<String> = Vec::new();
1512    let mut seen_top_level = std::collections::HashSet::new();
1513    let mut per_file: Vec<Vec<String>> = Vec::with_capacity(files.len());
1514    // Width and size from the directory's own files only (a downgraded row is never one
1515    // table); column names stay the subtree's, for search.
1516    let mut own_bytes = 0u64;
1517    let mut own_top_level: Vec<String> = Vec::new();
1518    let mut own_seen_top = std::collections::HashSet::new();
1519    let mut cost = Cost::default();
1520    let mut uncompressed = 0u64;
1521    let mut row_groups = 0usize;
1522    for file in &files {
1523        let Some(meta) = crate::formats::parquet_footer::read_parquet_metadata(file) else {
1524            return; // A file we cannot read makes the total a guess; report nothing.
1525        };
1526        rows += meta.num_rows;
1527        let names = column_names(&meta);
1528        for name in &names {
1529            if seen_columns.insert(name.clone()) {
1530                columns.push(name.clone());
1531            }
1532        }
1533        for name in top_level_names(&meta) {
1534            if seen_top_level.insert(name.clone()) {
1535                top_level.push(name);
1536            }
1537        }
1538        // Top-level columns, not footer leaves: see
1539        // [`crate::formats::schema_union::top_level_columns`].
1540        per_file.push(crate::formats::schema_union::top_level_columns(&names));
1541        // Again for its own files (`2 parquet` must mean those two), from one stat feeding
1542        // both totals.
1543        let file_bytes = std::fs::metadata(file).map(|m| m.len()).unwrap_or(0);
1544        bytes += file_bytes;
1545        if file.parent() == Some(entry.path.as_path()) {
1546            own_bytes += file_bytes;
1547            for name in top_level_names(&meta) {
1548                if own_seen_top.insert(name.clone()) {
1549                    own_top_level.push(name);
1550                }
1551            }
1552        }
1553        let mut per_file = Cost::default();
1554        physical_facts(&meta, &mut per_file);
1555        uncompressed += per_file.uncompressed.unwrap_or(0);
1556        row_groups += per_file.row_groups.unwrap_or(0);
1557        if cost.codec.is_none() {
1558            cost.codec = per_file.codec;
1559        }
1560    }
1561    // With the footers read, one-table is known. Separate tables make a place to look
1562    // inside (summed rows mean nothing). Only `multi` is reconsidered: hive files hold
1563    // one table by construction.
1564    if entry.kind == EntryKind::MultiFile && !crate::formats::schema_union::is_nested(&per_file) {
1565        // Its own files' bytes, matching what the label and columns count.
1566        entry.size = Some(own_bytes);
1567        // Not one table, but column search should still find it: the count is its own
1568        // files', the names every one under it.
1569        entry.columns = columns;
1570        entry.cols_sampled = false;
1571        downgrade_to_directory(
1572            entry,
1573            (!own_top_level.is_empty()).then_some(own_top_level.len()),
1574        );
1575        return;
1576    }
1577
1578    entry.rows = Some(rows);
1579    entry.cols = Some(top_level.len() + partition_columns_beyond(entry, &top_level));
1580    entry.size = Some(bytes);
1581    entry.columns = columns;
1582    cost.uncompressed = (uncompressed > 0).then_some(uncompressed);
1583    cost.row_groups = (row_groups > 0).then_some(row_groups);
1584    cost.partitions = entry.cost.partitions.take();
1585    entry.cost = cost;
1586}
1587
1588/// Partition keys the files lack, hoisted in as columns by the open, so the width
1589/// counts them (`12 ร— 4`, not `12 ร— 2`).
1590fn partition_columns_beyond(entry: &Entry, top_level: &[String]) -> usize {
1591    entry.cost.partitions.as_ref().map_or(0, |layout| {
1592        layout
1593            .keys
1594            .iter()
1595            .filter(|key| !top_level.contains(key))
1596            .count()
1597    })
1598}
1599
1600/// The files of `dir` itself, out of a walk that went below it.
1601fn direct_children(files: &[PathBuf], dir: &Path) -> Vec<PathBuf> {
1602    files
1603        .iter()
1604        .filter(|f| f.parent() == Some(dir))
1605        .cloned()
1606        .collect()
1607}
1608
1609/// Which of `files` to read for a sample: the ends and the middle. Sorted names
1610/// cluster one table's files at the start, and the last is the newest. Three reads.
1611pub(crate) fn spread(files: usize) -> Vec<usize> {
1612    let mut picks = match files {
1613        0 => Vec::new(),
1614        n => vec![0, n / 2, n - 1],
1615    };
1616    picks.dedup();
1617    picks
1618}
1619
1620/// Whether a few files' top-level columns are one table. `None` from fewer than two:
1621/// the directory keeps its name-based kind.
1622pub(crate) fn one_table_from(footers: &[Vec<String>]) -> Option<bool> {
1623    (footers.len() >= 2).then(|| crate::formats::schema_union::is_nested(footers))
1624}
1625
1626/// The footers of the [`spread`] of a directory too large to read every one of.
1627fn sample_footers(files: &[PathBuf]) -> Vec<crate::formats::parquet_footer::Footer> {
1628    spread(files.len())
1629        .into_iter()
1630        .filter_map(|i| crate::formats::parquet_footer::read_parquet_metadata(&files[i]))
1631        .collect()
1632}
1633
1634/// Ask a footerless directory whether its files are one table by their header names:
1635/// [`one_table_from`]'s rule for CSV and NDJSON, so `Enter` does not promise a table
1636/// the read then refuses. Silence (too few files, costly schema, a parse failure)
1637/// keeps the name-based kind, safe because the read unions by name and widens types
1638/// (`crate::formats::readers::polars::union_of_files`).
1639fn judge_by_names(entry: &mut Entry, as_read: &crate::formats::schema_union::ReadAs) {
1640    if entry.kind != EntryKind::MultiFile {
1641        return;
1642    }
1643    let Some(format) = entry
1644        .holds
1645        .one_format()
1646        .and_then(crate::FileFormat::from_name)
1647    else {
1648        return;
1649    };
1650    // The directory's own files, which the label counts and the open reads; `MultiFile`
1651    // is flat by construction and has one format, so only `One` arrives.
1652    let DirectoryFormat::One(_, files) = directory_format(&entry.path) else {
1653        return;
1654    };
1655    // Read as the following open will: header placement decides the names.
1656    let sampled = crate::formats::schema_union::sample_files(&files, format, as_read);
1657    if sampled.nests == Some(false) {
1658        // The sample's columns, for column search, as the Parquet path keeps when
1659        // downgrading; from the spread, so cost does not grow with the directory.
1660        let cols = (!sampled.columns.is_empty()).then_some(sampled.columns.len());
1661        // A floor only when files went unopened.
1662        entry.cols_sampled = sampled.read < files.len();
1663        entry.columns = sampled.columns;
1664        downgrade_to_directory(entry, cols);
1665    }
1666}
1667
1668/// A directory of separate tables becomes a place to look inside: no row count (a sum
1669/// of unrelated tables), but the union's column count (`15 parquet ยท 72 columns`).
1670/// `cols` is passed in because leaf paths cannot be split back into top-level
1671/// columns (see [`top_level_names`]).
1672fn downgrade_to_directory(entry: &mut Entry, cols: Option<usize>) {
1673    entry.kind = EntryKind::Directory;
1674    entry.rows = None;
1675    entry.cols = cols;
1676    entry.cost = Cost {
1677        partitions: entry.cost.partitions.take(),
1678        ..Cost::default()
1679    };
1680}
1681
1682/// The columns a reader sees: the schema's own top-level fields.
1683///
1684/// Not the leaves a footer names, and not those leaves split on a dot either. Leaves
1685/// counted directly double for a directory whose writer changed โ€” the same nested column
1686/// written by parquet-mr and by Arrow gives `inputs.list.element.address` in one file and
1687/// `inputs.bag.array_element.address` in the other, and a union of leaf paths holds both.
1688/// Splitting the dotted path fixes that and breaks something else: a column literally
1689/// named `user.id` is one column, and so is a struct `user` with a field `id`, and the
1690/// string cannot tell them apart.
1691///
1692/// The schema knows. `fields()` is the root's own children, which is what a struct counts
1693/// as here, what `schema_preview` lists in the details pane, and what the table shows.
1694fn top_level_names(meta: &crate::formats::parquet_footer::Footer) -> Vec<String> {
1695    meta.schema_descr
1696        .fields()
1697        .iter()
1698        .map(|field| field.name().to_string())
1699        .collect()
1700}
1701
1702/// Every column name any of the files has, in the order they first appear.
1703fn union_of(per_file: &[Vec<String>]) -> Vec<String> {
1704    let mut seen = std::collections::HashSet::new();
1705    per_file
1706        .iter()
1707        .flatten()
1708        .filter(|name| seen.insert(name.as_str()))
1709        .cloned()
1710        .collect()
1711}
1712
1713/// The Parquet files under `dir`, sorted, as deep as a dataset goes, stopping one past
1714/// the budget (which says there are too many to count). The open's own walk.
1715fn parquet_files_under(dir: &Path) -> Vec<PathBuf> {
1716    let mut files = crate::formats::dataset_files::LocalFiles::new(dir)
1717        .first_files(MAX_WALK_DEPTH as usize + 1, MAX_FOOTERS_PER_DATASET);
1718    // The walk takes a directory entry's own type; a file this reads must be one.
1719    files.retain(|p| is_regular_file(p));
1720    files
1721}
1722
1723/// Measure a dataset from footers an open kept: rows, width, size, row groups, and
1724/// whether its files are one table; nothing is read.
1725fn measure_from_footers(
1726    entry: &mut Entry,
1727    files: &[crate::formats::dataset_files::DatasetFile],
1728    footers: &[Option<crate::formats::schema_union::FileFooter>],
1729) {
1730    let per_file: Vec<Vec<String>> = footers
1731        .iter()
1732        .flatten()
1733        .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
1734        .collect();
1735    let columns = union_of(&per_file);
1736    if entry.kind == EntryKind::MultiFile && !crate::formats::schema_union::is_nested(&per_file) {
1737        // As a full footer read judges it: label, width and size count the directory's own
1738        // files.
1739        let own: Vec<usize> = files
1740            .iter()
1741            .enumerate()
1742            .filter(|(_, f)| Path::new(&f.key).parent() == Some(entry.path.as_path()))
1743            .map(|(i, _)| i)
1744            .collect();
1745        entry.size = Some(own.iter().map(|&i| files[i].size).sum());
1746        let own_columns = union_of(
1747            &own.iter()
1748                .filter_map(|&i| footers[i].as_ref())
1749                .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
1750                .collect::<Vec<_>>(),
1751        );
1752        entry.columns = columns;
1753        entry.cols_sampled = false;
1754        downgrade_to_directory(
1755            entry,
1756            (!own_columns.is_empty()).then_some(own_columns.len()),
1757        );
1758        return;
1759    }
1760    let footers: Vec<&crate::formats::schema_union::FileFooter> =
1761        footers.iter().flatten().collect();
1762    let uncompressed: u64 = footers
1763        .iter()
1764        .flat_map(|f| &f.column_bytes)
1765        .map(|(_, bytes)| *bytes as u64)
1766        .sum();
1767    let row_groups: usize = footers.iter().map(|f| f.row_group_rows.len()).sum();
1768    entry.rows = Some(footers.iter().map(|f| f.rows()).sum());
1769    entry.cols = Some(columns.len() + partition_columns_beyond(entry, &columns));
1770    entry.size = Some(files.iter().map(|f| f.size).sum());
1771    entry.columns = columns;
1772    entry.cols_sampled = false;
1773    entry.cost = Cost {
1774        uncompressed: (uncompressed > 0).then_some(uncompressed),
1775        row_groups: (row_groups > 0).then_some(row_groups),
1776        partitions: entry.cost.partitions.take(),
1777        ..Cost::default()
1778    };
1779}
1780
1781/// Fill in a Parquet file's counts from its footer, reading no column data. Other
1782/// formats keep `None` (a CSV's rows need a scan).
1783pub fn enrich_parquet(entry: &mut Entry) {
1784    if entry.kind != EntryKind::File {
1785        return;
1786    }
1787    if !is_parquet_path(&entry.path) {
1788        return;
1789    }
1790    if !is_regular_file(&entry.path) {
1791        return;
1792    }
1793    if let Some(meta) = crate::formats::parquet_footer::read_parquet_metadata(&entry.path) {
1794        entry.rows = Some(meta.num_rows);
1795        entry.columns = column_names(&meta);
1796        // Top-level columns, as a directory's row reports them: `schema_descr` names leaves
1797        // (one struct of three fields would count four). See [`top_level_names`].
1798        entry.cols = Some(top_level_names(&meta).len());
1799        physical_facts(&meta, &mut entry.cost);
1800    }
1801}
1802
1803/// A file of tables' tables (SQLite schema, NumPy archive directory): how many, whether
1804/// Enter opens one, and that one's columns. A file whose bytes contradict its name (a
1805/// `.db` that is not SQLite) is unopenable.
1806pub fn enrich_tables(entry: &mut Entry) {
1807    if entry.kind != EntryKind::File || entry.table.is_some() {
1808        return;
1809    }
1810    let named = data_format(&entry.path);
1811    if !is_regular_file(&entry.path) {
1812        return;
1813    }
1814    let Some(format) = crate::formats::members::holder(&entry.path) else {
1815        if named.is_some_and(|f| f.holds_tables() && crate::formats::readers::of(f).bytes_decide) {
1816            entry.kind = EntryKind::Other;
1817        }
1818        return;
1819    };
1820    let Ok(tables) = crate::formats::members::tables(&entry.path, format) else {
1821        return;
1822    };
1823    let own: Vec<&crate::formats::sqlite::Table> = tables.iter().filter(|t| !t.internal).collect();
1824    entry.cost.tables = Some(own.len());
1825    entry.cost.opens_one = format.opens_one_table();
1826    if let [one] = own.as_slice()
1827        && !one.columns.is_empty()
1828    {
1829        entry.columns = one.columns.iter().map(|(name, _)| name.clone()).collect();
1830        entry.cols = Some(entry.columns.len());
1831    }
1832}
1833
1834/// A file of tables listed as rows (a database's tables and views, an archive's arrays),
1835/// each at its path inside the file; SQLite's own marked hidden.
1836pub fn database_rows(file: &Path) -> Vec<Entry> {
1837    let Some(format) = crate::formats::members::holder(file) else {
1838        return Vec::new();
1839    };
1840    let Ok(mut tables) = crate::formats::members::tables(file, format) else {
1841        return Vec::new();
1842    };
1843    // A database's tables by name; an archive's arrays in the order they were saved.
1844    if format
1845        .descriptor()
1846        .tables
1847        .as_ref()
1848        .is_some_and(|t| t.by_name)
1849    {
1850        tables.sort_by_cached_key(|t| t.name.to_lowercase());
1851    }
1852    let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
1853    tables
1854        .into_iter()
1855        .map(|table| table_entry(file, format, table, modified))
1856        .collect()
1857}
1858
1859/// A Hugging Face cache directory's splits as rows at their paths inside it
1860/// (`cache/test`), opened with `--table`. Empty otherwise, or for one split (its door
1861/// opens it).
1862pub fn split_rows(dir: &Path) -> Vec<Entry> {
1863    let splits = crate::formats::hf_splits::cache_splits(dir);
1864    if splits.len() < 2 {
1865        return Vec::new();
1866    }
1867    splits
1868        .into_iter()
1869        .map(|split| split_entry(dir, split))
1870        .collect()
1871}
1872
1873/// The row of a split named by its path inside its cache directory (a recent); `None`
1874/// if none.
1875pub fn split_row(path: &Path) -> Option<Entry> {
1876    let (dir, split) = crate::formats::hf_splits::split_place(path)?;
1877    Some(split_entry(&dir, split))
1878}
1879
1880fn split_entry(dir: &Path, split: String) -> Entry {
1881    let mut entry = Entry::new(dir.join(&split), EntryKind::File).with_name(split);
1882    entry.table = Some(TableOf {
1883        format: Some(crate::FileFormat::Arrow),
1884        kind: "split".to_string(),
1885        internal: false,
1886    });
1887    entry
1888}
1889
1890/// A spec-read file's variants as rows at their paths inside it (`day.itch/add`),
1891/// opened with `--table`. Empty for other files.
1892pub fn variant_rows(file: &Path, formats: &crate::formats::Registry) -> Vec<Entry> {
1893    let Some((spec, tables)) = crate::formats::members::variants(file, formats) else {
1894        return Vec::new();
1895    };
1896    let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
1897    tables
1898        .into_iter()
1899        .map(|table| variant_entry(file, &spec, table, modified))
1900        .collect()
1901}
1902
1903/// The row of a variant named by its path inside its file (a recent); `None` if none.
1904pub fn variant_row(path: &Path, formats: &crate::formats::Registry) -> Option<Entry> {
1905    let (file, name) = crate::formats::members::split_variant(path, formats)?;
1906    let (spec, tables) = crate::formats::members::variants(&file, formats)?;
1907    let table = tables.into_iter().find(|t| t.name == name)?;
1908    let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
1909    let mut entry = variant_entry(&file, &spec, table, modified);
1910    entry.path = path.to_path_buf();
1911    Some(entry)
1912}
1913
1914fn variant_entry(
1915    file: &Path,
1916    spec: &str,
1917    table: crate::formats::sqlite::Table,
1918    modified: Option<std::time::SystemTime>,
1919) -> Entry {
1920    let mut entry = Entry::new(
1921        crate::formats::members::place(file, &table.name),
1922        EntryKind::File,
1923    )
1924    .with_name(table.name);
1925    entry.modified = modified;
1926    entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
1927    entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
1928    entry.format_spec = Some(spec.to_string());
1929    entry.table = Some(TableOf {
1930        format: None,
1931        kind: table.kind,
1932        internal: false,
1933    });
1934    entry
1935}
1936
1937/// The row of a table named by its path inside its file (`app.db/users`, a recent);
1938/// `None` if none.
1939pub fn table_row(path: &Path) -> Option<Entry> {
1940    let (file, name) = crate::formats::members::split(path)?;
1941    let format = crate::formats::members::holder(&file)?;
1942    let table = crate::formats::members::tables(&file, format)
1943        .ok()?
1944        .into_iter()
1945        .find(|t| t.name == name)?;
1946    let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
1947    let mut entry = table_entry(&file, format, table, modified);
1948    entry.path = path.to_path_buf();
1949    Some(entry)
1950}
1951
1952fn table_entry(
1953    file: &Path,
1954    format: crate::FileFormat,
1955    table: crate::formats::sqlite::Table,
1956    modified: Option<std::time::SystemTime>,
1957) -> Entry {
1958    let mut entry = Entry::new(
1959        crate::formats::members::place(file, &table.name),
1960        EntryKind::File,
1961    )
1962    .with_name(table.name);
1963    entry.modified = modified;
1964    entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
1965    entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
1966    entry.table = Some(TableOf {
1967        format: Some(format),
1968        kind: table.kind,
1969        internal: table.internal,
1970    });
1971    entry
1972}
1973
1974/// The first bytes of `path`, as many as fit in `buf`. A short read is the whole file.
1975fn read_head<'a>(path: &Path, buf: &'a mut [u8]) -> Option<&'a [u8]> {
1976    use std::io::Read;
1977    let mut file = std::fs::File::open(path).ok()?;
1978    let mut filled = 0;
1979    loop {
1980        match file.read(&mut buf[filled..]) {
1981            Ok(0) => break,
1982            Ok(n) => filled += n,
1983            Err(_) => return None,
1984        }
1985        if filled == buf.len() {
1986            break;
1987        }
1988    }
1989    Some(&buf[..filled])
1990}
1991
1992/// Whether an Arrow file is an IPC stream (an IPC file starts `ARROW1`): eight bytes,
1993/// so the listing can say which will be converted.
1994fn enrich_arrow(entry: &mut Entry) {
1995    if entry.kind != EntryKind::File
1996        || data_format(&entry.path) != Some(crate::FileFormat::Arrow)
1997        || crate::CompressionFormat::from_extension(&entry.path).is_some()
1998        || !is_regular_file(&entry.path)
1999    {
2000        return;
2001    }
2002    let mut head = [0u8; 8];
2003    if let Some(head) = read_head(&entry.path, &mut head) {
2004        entry.cost.ipc_stream = !head.starts_with(b"ARROW1");
2005    }
2006}
2007
2008/// Pull layout and compression from an already-read footer: what reading the file
2009/// will do, beyond its size.
2010pub fn physical_facts(meta: &crate::formats::parquet_footer::Footer, cost: &mut Cost) {
2011    if meta.row_groups.is_empty() {
2012        return;
2013    }
2014    cost.row_groups = Some(meta.row_groups.len());
2015
2016    let mut uncompressed: u64 = 0;
2017    let mut codecs: Vec<String> = Vec::new();
2018    for rg in &meta.row_groups {
2019        uncompressed = uncompressed.saturating_add(rg.total_byte_size() as u64);
2020        for cc in rg.parquet_columns() {
2021            let codec = format!("{:?}", cc.compression()).to_lowercase();
2022            if !codecs.contains(&codec) {
2023                codecs.push(codec);
2024            }
2025        }
2026    }
2027    if uncompressed > 0 {
2028        cost.uncompressed = Some(uncompressed);
2029    }
2030    // Usually one codec; when not, say so rather than imply uniformity.
2031    cost.codec = match codecs.len() {
2032        0 => None,
2033        1 => Some(codecs.remove(0)),
2034        n => Some(format!("mixed ({n})")),
2035    };
2036}
2037
2038/// Outermost directories examined to describe a hive partitioning: enough to name the
2039/// keys and show the shape; a decade of daily partitions is not worth counting on a
2040/// share.
2041const MAX_PARTITION_DIRS: usize = 512;
2042
2043/// Describe how a hive dataset is partitioned, from directory names alone.
2044pub fn partition_layout(dir: &Path) -> Option<Partitions> {
2045    let iter = std::fs::read_dir(dir).ok()?;
2046    let mut values: Vec<String> = Vec::new();
2047    let mut keys: Vec<String> = Vec::new();
2048    let mut count = 0usize;
2049    let mut more = false;
2050
2051    for entry in iter.flatten() {
2052        if count >= MAX_PARTITION_DIRS {
2053            more = true;
2054            break;
2055        }
2056        let name = entry.file_name().to_string_lossy().into_owned();
2057        let Some((key, value)) = name.split_once('=') else {
2058            continue;
2059        };
2060        if !entry.path().is_dir() {
2061            continue;
2062        }
2063        if keys.is_empty() {
2064            keys.push(key.to_string());
2065            // Only the first partition is descended for nested keys: one is representative.
2066            keys.extend(nested_keys(&entry.path()));
2067        }
2068        values.push(value.to_string());
2069        count += 1;
2070    }
2071
2072    if keys.is_empty() {
2073        return None;
2074    }
2075    values.sort();
2076    values.dedup();
2077    Some(Partitions {
2078        keys,
2079        first_key_values: values,
2080        count,
2081        more,
2082    })
2083}
2084
2085/// Partition keys below `dir`, following the first child at each level.
2086fn nested_keys(dir: &Path) -> Vec<String> {
2087    let mut keys = Vec::new();
2088    let mut current = dir.to_path_buf();
2089    // Bounded: each level is a directory read.
2090    for _ in 0..6 {
2091        let Ok(iter) = std::fs::read_dir(&current) else {
2092            break;
2093        };
2094        let Some(child) = iter
2095            .flatten()
2096            .find(|e| e.file_name().to_string_lossy().contains('=') && e.path().is_dir())
2097        else {
2098            break;
2099        };
2100        let name = child.file_name().to_string_lossy().into_owned();
2101        let Some((key, _)) = name.split_once('=') else {
2102            break;
2103        };
2104        keys.push(key.to_string());
2105        current = child.path();
2106    }
2107    keys
2108}
2109
2110/// Render a row count compactly (`2.4M`).
2111pub fn format_rows(rows: usize) -> String {
2112    let r = rows as f64;
2113    if rows >= 1_000_000_000 {
2114        format!("{:.1}B", r / 1e9)
2115    } else if rows >= 1_000_000 {
2116        format!("{:.1}M", r / 1e6)
2117    } else if rows >= 10_000 {
2118        format!("{:.0}k", r / 1e3)
2119    } else if rows >= 1_000 {
2120        // Below ten thousand show the exact count: 3,653 days is ten years; "4k" is not.
2121        let mut out = String::new();
2122        let digits = rows.to_string();
2123        for (i, c) in digits.chars().enumerate() {
2124            if i > 0 && (digits.len() - i).is_multiple_of(3) {
2125                out.push(',');
2126            }
2127            out.push(c);
2128        }
2129        out
2130    } else {
2131        rows.to_string()
2132    }
2133}
2134
2135/// Render "how long ago" compactly (`2d`, `3h`).
2136pub fn format_age(t: std::time::SystemTime) -> String {
2137    let Ok(elapsed) = t.elapsed() else {
2138        return String::new();
2139    };
2140    let secs = elapsed.as_secs();
2141    if secs < 60 {
2142        "now".to_string()
2143    } else if secs < 3600 {
2144        format!("{}m", secs / 60)
2145    } else if secs < 86_400 {
2146        format!("{}h", secs / 3600)
2147    } else if secs < 86_400 * 365 {
2148        format!("{}d", secs / 86_400)
2149    } else {
2150        format!("{}y", secs / (86_400 * 365))
2151    }
2152}
2153
2154/// Column name and type, for the home screen's preview pane.
2155pub type SchemaPreview = Vec<(String, polars::prelude::DataType)>;
2156
2157/// The preview of a table inside a file of tables, or of such a file (or NumPy array
2158/// file) as it opens. `None` for other entries, `Some(None)` when there is nothing to
2159/// show.
2160fn table_preview(entry: &Entry) -> Option<Option<SchemaPreview>> {
2161    let (file, format, name) = match &entry.table {
2162        Some(table) => match (crate::formats::members::split(&entry.path), table.format) {
2163            (Some((file, _)), Some(format)) => (file, format, Some(entry.name.as_str())),
2164            _ => return Some(None),
2165        },
2166        None if is_regular_file(&entry.path) => {
2167            let format = crate::formats::members::holder(&entry.path).or_else(|| {
2168                data_format(&entry.path).filter(|f| {
2169                    f.holds_tables() && crate::formats::readers::of(*f).table_schema.is_some()
2170                })
2171            })?;
2172            (entry.path.clone(), format, None)
2173        }
2174        None => return None,
2175    };
2176    Some(
2177        crate::formats::readers::of(format)
2178            .table_schema
2179            .and_then(|schema| schema(&file, name)),
2180    )
2181}
2182
2183/// The first Parquet file at or under `dir`, bounded in breadth and depth so thousands
2184/// of partitions cost what three do.
2185fn first_parquet_under(dir: &Path, depth: u8) -> Option<PathBuf> {
2186    if depth > MAX_WALK_DEPTH {
2187        return None;
2188    }
2189    let mut subdirs = Vec::new();
2190    for entry in std::fs::read_dir(dir).ok()?.flatten().take(64) {
2191        let path = entry.path();
2192        if path.is_dir() {
2193            subdirs.push(path);
2194        } else if is_parquet_path(&path) && is_regular_file(&path) {
2195            return Some(path);
2196        }
2197    }
2198    subdirs.sort();
2199    subdirs
2200        .into_iter()
2201        .take(4)
2202        .find_map(|d| first_parquet_under(&d, depth + 1))
2203}
2204
2205/// Column names from a Parquet footer.
2206pub fn column_names(meta: &crate::formats::parquet_footer::Footer) -> Vec<String> {
2207    meta.schema_descr
2208        .columns()
2209        .iter()
2210        .map(|c| c.path_in_schema.join("."))
2211        .collect()
2212}
2213
2214/// Whether `path` is a regular file safe to open: a FIFO, device or socket named
2215/// `.parquet` would block or worse, so reads are gated on kind (stat never blocks
2216/// like an open can).
2217fn is_regular_file(path: &Path) -> bool {
2218    std::fs::metadata(path)
2219        .map(|m| m.file_type().is_file())
2220        .unwrap_or(false)
2221}
2222
2223/// A dataset's column names and types without reading data; Parquet only, `None`
2224/// otherwise (the UI says so).
2225pub fn schema_preview(entry: &Entry) -> Option<SchemaPreview> {
2226    // `SerReader` is what brings `ParquetReader::new` into scope.
2227    use polars::prelude::{ParquetReader, Schema, SchemaExt, SerReader};
2228
2229    let file_path = match entry.kind {
2230        EntryKind::File => {
2231            if let Some(preview) = table_preview(entry) {
2232                return preview;
2233            }
2234            if !is_parquet_path(&entry.path) {
2235                return None;
2236            }
2237            entry.path.clone()
2238        }
2239        EntryKind::Hive | EntryKind::MultiFile => first_parquet_under(&entry.path, 0)?,
2240        EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => return None,
2241        // One data file's schema is not a lake table's (Iceberg field IDs and Delta column
2242        // mapping rename columns).
2243        EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => return None,
2244    };
2245
2246    if !is_regular_file(&file_path) {
2247        return None;
2248    }
2249    let file = std::fs::File::open(&file_path).ok()?;
2250    let mut reader = ParquetReader::new(file);
2251    let arrow_schema = reader.schema().ok()?;
2252    let schema = Schema::from_arrow_schema(arrow_schema.as_ref());
2253    let mut preview: SchemaPreview = Vec::new();
2254    // A hive table opens with partition keys hoisted first, so the pane lists them there,
2255    // typed from the path as the scan infers them.
2256    if entry.kind == EntryKind::Hive
2257        && let Ok(below) = file_path.strip_prefix(&entry.path)
2258    {
2259        for part in below.parent().into_iter().flat_map(Path::components) {
2260            let part = part.as_os_str().to_string_lossy();
2261            if let Some((key, value)) = part.split_once('=')
2262                && !key.is_empty()
2263                && schema.get(key).is_none()
2264            {
2265                // The scan's own inference, dates and booleans included.
2266                let dtype = if value.is_empty() || value == "__HIVE_DEFAULT_PARTITION__" {
2267                    polars::prelude::DataType::String
2268                } else {
2269                    polars::io::csv::read::schema_inference::infer_field_schema(value, true, false)
2270                };
2271                preview.push((key.to_string(), dtype));
2272            }
2273        }
2274    }
2275    preview.extend(
2276        schema
2277            .iter()
2278            .map(|(name, dtype)| (name.to_string(), dtype.clone())),
2279    );
2280    Some(preview)
2281}
2282
2283#[cfg(test)]
2284mod classification_tests;