Skip to main content

datui_lib/
discover.rs

1//! Dataset discovery for the home screen.
2//!
3//! This is deliberately *not* a catalogue. Nothing here is persisted: every listing
4//! is computed from the filesystem when asked for, and forgotten when the session
5//! ends. The only state datui keeps between runs is a list of recently opened paths.
6//!
7//! Discovery is also deliberately shallow. Interesting datasets tend to live on
8//! mounts — network filesystems, spinning disks, hive trees with a hundred thousand
9//! partition files — so a recursive walk would make the home screen slowest exactly
10//! where the data is most interesting. Every function here scans one directory level
11//! and stops.
12
13use std::path::{Path, PathBuf};
14
15/// Compression suffixes that may follow a data extension (`sales.csv.gz`).
16const COMPRESSION_EXTENSIONS: &[&str] = &["gz", "bz2", "xz", "zst", "zstd"];
17
18/// Upper bound on entries read from a single directory, so a pathological directory
19/// cannot hang the UI.
20pub const MAX_ENTRIES_PER_DIR: usize = 5_000;
21
22/// What a home-screen row represents.
23#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
24#[serde(rename_all = "lowercase")]
25pub enum EntryKind {
26    /// A single data file.
27    File,
28    /// A file datui has no reader for. Hidden on the home screen until `Ctrl+A` shows
29    /// it, dimmed, so a directory can be seen as it is.
30    Other,
31    /// A directory of `key=value` partitions — one dataset, not a tree to walk.
32    Hive,
33    /// A directory of similarly-shaped data files, openable as one table.
34    MultiFile,
35    /// A Delta Lake table: `_delta_log/` beside the data files.
36    Delta,
37    /// An Apache Iceberg table: `metadata/` holding the snapshots, `data/` the files.
38    Iceberg,
39    /// An Apache Hudi table: `.hoodie/` holding the timeline.
40    Hudi,
41    /// An ordinary directory, to descend into.
42    Directory,
43    /// Somewhere remote that has not been looked at yet. Classifying it would mean
44    /// reading it, which is the call that blocks when the network is gone — so it is
45    /// offered as openable and left unlabelled rather than guessed at.
46    ///
47    /// Also what a kind this build does not recognize reads back as. The dataset index
48    /// is one JSON map, and a value an older datui cannot parse would otherwise fail the
49    /// whole map and discard every dataset fact it had — see `CLASSIFIER_VERSION`, which
50    /// is why a new kind can appear in a file an older build reads.
51    #[serde(other)]
52    Unknown,
53}
54
55/// Bumped whenever a build starts classifying something differently.
56///
57/// A cached kind is the only thing a remote row has to go on — it was never stat'ed, and
58/// classifying it means reading it — so it is restored rather than re-derived. That makes
59/// it a way for an answer this build would not give to come back: a Delta root measured
60/// before lake tables were recognized was recorded as `multifile`, and restoring that
61/// opens it as one table, which is the whole of #237 read back off disk.
62///
63/// So the kind is restored only when the build that wrote it classified the way this one
64/// does. Everything else in the record — rows, columns, cost — is a measurement rather
65/// than a judgement, and survives.
66///
67/// 7: a file with no extension is a SQLite database when its first bytes say so.
68///
69/// 6: on a local disk, a file with no extension is data when its first bytes carry a
70/// Parquet, Arrow, Avro or ORC signature, so a directory of Spark part files a 5 called
71/// `dir` is one dataset. Unidentified ones are `unnamed` rather than `not_read`.
72///
73/// 5: a directory of CSV or NDJSON is judged by the names at the front of its files, the
74/// way a directory of Parquet is judged by its footers — so one a 4 called `multi` on its
75/// filenames alone may be a place to look inside. A cached kind is restored without
76/// looking again, so a record written by 4 would keep the answer this build exists to
77/// correct (#275 follow-up).
78///
79/// 4: a directory's row carries what one listing of it found, beside its kind, and the
80/// two are restored together — a record written by 3 carries the kind and not the count,
81/// and a row given a kind from the cache is never looked into again (#275, phase 2).
82///
83/// 3: one listing instead of a probe of the first eight entries, formats instead of
84/// extension strings, and one bookkeeping predicate. A directory of `.arrow` beside
85/// `.ipc` was `dir` and is now one dataset; a directory whose ninth entry decided it was
86/// answered by whatever the filesystem returned first (#275, phase 1).
87pub const CLASSIFIER_VERSION: u32 = 7;
88
89impl EntryKind {
90    /// Short label shown next to the entry name.
91    pub fn label(self) -> &'static str {
92        match self {
93            // A file with no reader says nothing: it is dimmed, and Enter shows its
94            // bytes. `binary` would be wrong for the README or log it often is.
95            EntryKind::File | EntryKind::Other => "",
96            EntryKind::Hive => "hive",
97            EntryKind::MultiFile => "multi",
98            EntryKind::Delta => "delta",
99            EntryKind::Iceberg => "iceberg",
100            EntryKind::Hudi => "hudi",
101            EntryKind::Directory => "dir",
102            EntryKind::Unknown => "",
103        }
104    }
105
106    /// Whether selecting this entry opens a dataset rather than navigating.
107    ///
108    /// An unexamined remote path counts: datui can open an object-store prefix or a
109    /// hive directory directly, and descending into one is not possible anyway
110    /// without the listing this deliberately has not fetched.
111    pub fn is_dataset(self) -> bool {
112        !matches!(self, EntryKind::Directory | EntryKind::Other) && !self.is_lake_table()
113    }
114
115    /// Whether this row is *known* to be a dataset.
116    ///
117    /// [`EntryKind::is_dataset`] answers "may this be opened", and a row nothing has
118    /// looked into answers yes: it is offered, and looked into before it is acted on.
119    /// This one answers "is this a dataset", which such a row cannot answer at all —
120    /// and that is the question counting asks. A directory of two hundred subdirectories
121    /// nobody has looked into is not two hundred datasets.
122    pub fn is_known_dataset(self) -> bool {
123        self != EntryKind::Unknown && self.is_dataset()
124    }
125
126    /// A Delta, Iceberg or Hudi table root: a log beside the data files that says which
127    /// of them are live, which datui does not read yet.
128    pub fn is_lake_table(self) -> bool {
129        matches!(
130            self,
131            EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi
132        )
133    }
134
135    /// The format's name for prose. `label` is the row's chip, and is lowercase like
136    /// `hive` and `multi` beside it.
137    pub fn lake_name(self) -> Option<&'static str> {
138        match self {
139            EntryKind::Delta => Some("Delta"),
140            EntryKind::Iceberg => Some("Iceberg"),
141            EntryKind::Hudi => Some("Hudi"),
142            _ => None,
143        }
144    }
145}
146
147/// What one listing of a directory found in it, counted rather than judged.
148///
149/// The label a directory's row carries comes from here, so it says what is inside rather
150/// than what `Enter` will do with it. A count that is wrong then costs a reader nothing:
151/// `12 parquet` is true of a directory whether or not its files are one table.
152#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
153pub struct Holds {
154    /// Data files by format, commonest first. The name is [`crate::FileFormat::name`],
155    /// kept as a string so a record written by one build reads in the next.
156    #[serde(default)]
157    pub formats: Vec<(String, usize)>,
158    /// Subdirectories, partitions among them. Development builds of 0.4.0 wrote it as
159    /// `folders`; the alias keeps a cache from one of those readable.
160    #[serde(default, alias = "folders")]
161    pub directories: usize,
162    /// `key=value` subdirectories, which are also counted in `directories`.
163    #[serde(default)]
164    pub partitions: usize,
165    /// Files datui has no reader for: a README, a script, a notebook. Neither data nor a
166    /// writer's own, and without a count of their own they were in nothing — a directory
167    /// of twenty of them read `dir` with no line at all, the same as an empty one.
168    #[serde(default)]
169    pub not_read: usize,
170    /// Files with no extension. No name says what they are, so they are neither data
171    /// nor `not_read`: Spark and GBIF write their part files this way, and the open
172    /// reads them by their bytes.
173    #[serde(default)]
174    pub unnamed: usize,
175    /// Entries skipped as a writer's own, and the first few by name for the pane.
176    #[serde(default)]
177    pub skipped: usize,
178    #[serde(default)]
179    pub skipped_names: Vec<String>,
180    /// Whether the listing stopped at [`MAX_ENTRIES_PER_DIR`], so every count is a
181    /// floor. Shown as `5000+`.
182    #[serde(default)]
183    pub truncated: bool,
184    /// A Hugging Face DatasetDict saved with `save_to_disk`: `dataset_dict.json`
185    /// beside directories, its splits. Read as Arrow, one split at a time.
186    #[serde(default)]
187    pub dataset_dict: bool,
188}
189
190/// How many skipped names are kept for the pane. Enough to recognise the convention.
191pub(crate) const SKIPPED_NAMES_SHOWN: usize = 4;
192
193/// A name cut to `width`, keeping both ends and marking the middle.
194pub(crate) fn shorten(name: &str, width: usize) -> String {
195    let chars: Vec<char> = name.chars().collect();
196    if chars.len() <= width {
197        return name.to_string();
198    }
199    let ellipsis = crate::glyphs::get().ellipsis;
200    let room = width.saturating_sub(ellipsis.chars().count());
201    let head = room.div_ceil(2);
202    let tail = room - head;
203    format!(
204        "{}{ellipsis}{}",
205        chars[..head].iter().collect::<String>(),
206        chars[chars.len() - tail..].iter().collect::<String>()
207    )
208}
209
210impl Holds {
211    /// Data files of every format.
212    pub fn data_files(&self) -> usize {
213        self.formats.iter().map(|(_, n)| n).sum()
214    }
215
216    /// The one format this directory holds, when it holds exactly one.
217    pub fn one_format(&self) -> Option<&str> {
218        match self.formats.as_slice() {
219            [(name, _)] => Some(name),
220            _ => None,
221        }
222    }
223
224    /// The weights' format and file count, when this directory is a model: weights of
225    /// one format with nothing beside them but JSON. See [`is_model_directory`].
226    pub fn model_weights(&self) -> Option<(&str, usize)> {
227        if !is_model_directory(counts_names(self)) {
228            return None;
229        }
230        self.formats
231            .iter()
232            .find(|(name, _)| is_weights(name))
233            .map(|(name, count)| (name.as_str(), *count))
234    }
235
236    /// The label a directory's row carries when its kind does not name itself: `12
237    /// parquet`, `mixed`, or `dir` for a directory with no data directly inside.
238    pub fn label(&self) -> String {
239        let more = if self.truncated { "+" } else { "" };
240        // A model's weights beside its config and tokenizer JSON: the directory is the
241        // model, and its label says so rather than `mixed`.
242        if let Some((name, count)) = self.model_weights() {
243            return format!("{count}{more} {name}");
244        }
245        match self.formats.as_slice() {
246            // `dir` says there is no data file inside. A listing cut short cannot say
247            // that — it found none among the entries it read, and more files can
248            // unmake it, which is what separates this from `mixed`.
249            // A directory of directories counts them: `3 dirs` says where to go next.
250            [] if self.directories == 1 => format!("1 dir{more}"),
251            [] if self.directories > 1 => format!("{}{more} dirs", self.directories),
252            [] => format!("dir{more}"),
253            // The `+` hedges the whole claim, not only the number: past the cap a
254            // second format may be among the entries that were not read, so `5000+
255            // parquet` and `mixed` are both answers this directory can give depending on
256            // the order it came back in. What is certain is that five thousand Parquet
257            // files are in there.
258            [(name, count)] => format!("{count}{more} {name}"),
259            // No `+`: `mixed` is not a count, and more files cannot unmake it. The
260            // pane's line carries the qualifier on each number it does report.
261            _ => "mixed".to_string(),
262        }
263    }
264
265    /// Whether nothing has been counted here: a file, or a directory nothing has looked
266    /// into. A directory that was looked into and found empty is not this — it has no
267    /// formats either, and `dir` is the right word for both.
268    pub fn is_empty(&self) -> bool {
269        self.formats.is_empty()
270            && self.directories == 0
271            && self.skipped == 0
272            && self.not_read == 0
273            && self.unnamed == 0
274            // Every field, including the two that are counted elsewhere as well: a
275            // partition is a directory and a skipped name is one of `skipped`, so on both
276            // routes today these are implied. This is a `skip_serializing_if` and the
277            // guard that stops a placeholder erasing a row's count, and neither should
278            // turn on an invariant two other functions have to keep.
279            && self.partitions == 0
280            && self.skipped_names.is_empty()
281            && !self.dataset_dict
282            // A listing cut short before it found anything still says something: that
283            // what it found is not all there is. Without this the row falls back to its
284            // kind and reads `dir`, where `label` would have said `dir+`.
285            && !self.truncated
286    }
287
288    /// The `contains` line in the details pane: the data files by format, the
289    /// directories, and the partitions — what there is to open, and nothing else.
290    ///
291    /// Files datui cannot read and a writer's markers are left out. Counted here they
292    /// read as a warning ("10 not read") about a directory with nothing wrong in it;
293    /// inside the directory, a row of its own says what is not shown.
294    pub fn line(&self, with_partitions: bool) -> Option<String> {
295        let more = if self.truncated { "+" } else { "" };
296        let mut parts: Vec<String> = self
297            .formats
298            .iter()
299            .map(|(name, count)| format!("{count}{more} {name}"))
300            .collect();
301        // Partitions are directories too, and counted in `directories`; naming both would
302        // count them twice. What is left is the directories that are not partitions.
303        let plain = self.directories.saturating_sub(self.partitions);
304        if plain > 0 {
305            let word = if plain == 1 {
306                "directory"
307            } else {
308                "directories"
309            };
310            parts.push(format!("{plain}{more} {word}"));
311        }
312        if with_partitions && self.partitions > 0 {
313            let word = if self.partitions == 1 {
314                "partition"
315            } else {
316                "partitions"
317            };
318            parts.push(format!("{}{more} {word}", self.partitions));
319        }
320        (!parts.is_empty()).then(|| parts.join(" · "))
321    }
322}
323
324impl Entry {
325    /// Whether Enter on this row lists the tables inside it: a file of several that is
326    /// no table itself. A file that opens one of its tables (a workbook's first sheet)
327    /// opens it, and a file a spec reads as several variants is one table too (each row
328    /// a variant, with a `type` column), which Enter opens; → lists them.
329    pub fn enter_lists_tables(&self) -> bool {
330        self.kind == EntryKind::File
331            && self.cost.tables.is_some_and(|n| n > 1)
332            && !self.cost.opens_one
333            && self.format_spec.is_none()
334    }
335
336    /// Whether the home screen leaves this row out until Ctrl+A: a file datui cannot
337    /// open, or a database's own table.
338    pub fn hidden_by_default(&self) -> bool {
339        self.kind == EntryKind::Other || self.table.as_ref().is_some_and(|t| t.internal)
340    }
341
342    /// The short label beside a row's name: what it holds, rather than what `Enter`
343    /// will do with it.
344    ///
345    /// A directory that has been looked into is described by the count — `12 parquet`,
346    /// `mixed`, `dir` — and a lake table or a hive root by the format's own name, which
347    /// is the thing it is. A row nothing has looked into has only its kind to go on.
348    pub fn label(&self) -> std::borrow::Cow<'static, str> {
349        match self.kind {
350            // See `opens_whole_directory`: the one row whose label would be about a
351            // different set of files than the row itself.
352            _ if self.opens_whole_directory => "".into(),
353            EntryKind::Directory | EntryKind::MultiFile if !self.holds.is_empty() => {
354                self.holds.label().into()
355            }
356            EntryKind::File if self.format_spec.is_some() => {
357                self.format_spec.clone().unwrap_or_default().into()
358            }
359            EntryKind::File if self.cost.tables.is_some() => {
360                let n = self.cost.tables.unwrap_or_default();
361                format!("{n} {}", if n == 1 { "table" } else { "tables" }).into()
362            }
363            // A file named for what it holds rather than by its file name, as a
364            // collection names one: its format, which the name no longer says.
365            EntryKind::File
366                if crate::FileFormat::from_path(Path::new(&self.name)).is_none()
367                    && crate::FileFormat::from_path(&self.path).is_some() =>
368            {
369                crate::FileFormat::from_path(&self.path)
370                    .map(crate::FileFormat::name)
371                    .unwrap_or_default()
372                    .into()
373            }
374            kind => kind.label().into(),
375        }
376    }
377}
378
379/// One row on the home screen.
380#[derive(Debug, Clone)]
381pub struct Entry {
382    pub path: PathBuf,
383    pub kind: EntryKind,
384    /// Display name — the file or directory name, not the full path.
385    pub name: String,
386    /// Size in bytes. For a multi-file or hive dataset this is the sum of the files
387    /// actually inspected, so it is a floor rather than an exact total.
388    pub size: Option<u64>,
389    pub modified: Option<std::time::SystemTime>,
390    /// Row count, when it can be had without reading data (Parquet footers only).
391    pub rows: Option<usize>,
392    /// Column count, same caveat.
393    pub cols: Option<usize>,
394    /// Whether `cols` came from a spread of the directory rather than all of it. A
395    /// directory past the footer budget is read at its ends and its middle, so the count
396    /// is a floor: shown as `6+` rather than `6`, the way the row count is already shown
397    /// as `?` when it is out of reach.
398    pub cols_sampled: bool,
399    /// Column names, when they were free to obtain. A Parquet footer carries them
400    /// alongside the row count, so knowing what is *in* a dataset costs nothing
401    /// beyond knowing how big it is.
402    pub columns: Vec<String>,
403    /// What opening this will cost: where it lives, how it is stored, how it is laid
404    /// out. All of it derived from bytes datui already reads.
405    pub cost: Cost,
406    /// What one listing of it found, for a directory. Empty for a file, and for a
407    /// directory nothing has looked into.
408    pub holds: Holds,
409    /// Whether this row is the door that opens the directory being browsed, rather than
410    /// something in it.
411    ///
412    /// It carries no label. Every other label counts what is directly inside a directory,
413    /// and this row is the one that reads the whole of it — so `dir` beside `(all
414    /// files)` would say there is no data here while offering to open it, and `2
415    /// parquet` beside it would name two of the twenty it is about to read. The name
416    /// says what it does; the numbers beside it, once measured, say how much.
417    pub opens_whole_directory: bool,
418    /// The format spec that reads this file: its glob names it, or its magic is at
419    /// the front of it.
420    pub format_spec: Option<String>,
421    /// A table inside a file of tables (a SQLite database, a NumPy archive), for the
422    /// rows listed inside one: its path is the file's with the table's name after it,
423    /// which nothing on disk has.
424    pub table: Option<TableOf>,
425}
426
427/// What a row inside a file of tables says about its table.
428#[derive(Debug, Clone, PartialEq, Eq)]
429pub struct TableOf {
430    /// The format of the file it is in; `None` for a format spec's variant, whose spec
431    /// [`Entry::format_spec`] names.
432    pub format: Option<crate::FileFormat>,
433    /// What the file calls it: SQLite's `table`, `view`, `virtual` or `shadow`, or a
434    /// NumPy archive's `array`.
435    pub kind: String,
436    /// SQLite's own (its schema, its statistics, a virtual table's shadows): hidden
437    /// like a file datui cannot open until Ctrl+A shows it, and opened like any other.
438    pub internal: bool,
439}
440
441/// What pressing Enter on a dataset will actually cost.
442///
443/// `rows`, `cols` and `size` say what a dataset *is*. None of them say what reading
444/// it will do, and the difference is large: 200 MB of zstd-compressed Parquet is two
445/// gigabytes in memory, and two gigabytes on a hotel-wifi NFS mount is a different
446/// afternoon than two gigabytes on tmpfs.
447///
448/// Every field here comes from something datui already reads — the mount table, and
449/// the same Parquet footer that yields the row count. Nothing here costs an extra
450/// byte of the dataset itself.
451#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
452pub struct Cost {
453    /// Filesystem or URL scheme: `nfs4`, `ext4`, `tmpfs`, `fuse.sshfs`, `s3`.
454    #[serde(default, skip_serializing_if = "Option::is_none")]
455    pub source: Option<String>,
456    /// Bytes once decompressed — what this will occupy, as against what it occupies
457    /// on disk.
458    #[serde(default, skip_serializing_if = "Option::is_none")]
459    pub uncompressed: Option<u64>,
460    /// Compression codec, as the file itself names it.
461    #[serde(default, skip_serializing_if = "Option::is_none")]
462    pub codec: Option<String>,
463    /// Row groups. One enormous row group cannot be read in parallel or skipped
464    /// through; a thousand tiny ones cost more in overhead than they save.
465    #[serde(default, skip_serializing_if = "Option::is_none")]
466    pub row_groups: Option<usize>,
467    /// Partition layout, for a hive dataset.
468    #[serde(default, skip_serializing_if = "Option::is_none")]
469    pub partitions: Option<Partitions>,
470    /// Tables of its own, for a file of tables: one opens, several are listed.
471    #[serde(default, skip_serializing_if = "Option::is_none")]
472    pub tables: Option<usize>,
473    /// Whether a file of several tables opens one of them (a workbook's first sheet),
474    /// so Enter opens it and → lists them, rather than Enter listing them.
475    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
476    pub opens_one: bool,
477    /// An Arrow file that is an IPC stream, which is converted before it is scanned,
478    /// rather than an IPC file, which is scanned where it is. From its first bytes.
479    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
480    pub ipc_stream: bool,
481}
482
483/// How opening a file row will read it. See [`how_read`].
484#[derive(Debug, Clone, Copy, PartialEq, Eq)]
485pub struct HowRead {
486    pub mode: crate::ReadMode,
487    /// A remote file that is downloaded whole before it is read.
488    pub download: bool,
489}
490
491/// How opening `entry` will read it, as [`crate::FileFormat::read_mode`] says for its
492/// format and how it is stored, and whether a remote one is downloaded first. `None`
493/// for anything but a file whose name says its format.
494pub fn how_read(entry: &Entry) -> Option<HowRead> {
495    use crate::Stored;
496    if entry.kind != EntryKind::File {
497        return None;
498    }
499    let stored = if crate::CompressionFormat::from_extension(&entry.path).is_some() {
500        Stored::Compressed { in_memory: false }
501    } else if entry.cost.ipc_stream {
502        Stored::Stream
503    } else {
504        Stored::Plain
505    };
506    let choice = match &entry.format_spec {
507        Some(name) => crate::cli::FormatChoice::Spec(name.clone()),
508        // A table inside a file of tables, at its path inside it (`shop.db/orders`).
509        None if entry.table.is_some() => {
510            crate::cli::FormatChoice::Builtin(entry.table.as_ref().and_then(|t| t.format)?)
511        }
512        None => crate::cli::FormatChoice::Builtin(data_format(&entry.path)?),
513    };
514    let mode = choice.read_mode(stored)?;
515    let download = match crate::source::input_source(&entry.path) {
516        crate::source::InputSource::Local(_) => false,
517        crate::source::InputSource::Http(_) => choice.http_file() == crate::RemoteRead::Downloaded,
518        // A remote Arrow file is not peeked at here, so it is taken for an IPC file.
519        _ => choice.bucket_object(stored) == crate::RemoteRead::Downloaded,
520    };
521    Some(HowRead { mode, download })
522}
523
524/// How a hive dataset is laid out on disk.
525///
526/// The shape of a partitioned dataset is the first thing anyone asks about it, and
527/// the answer is in the directory names — no file needs opening to know it.
528#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
529pub struct Partitions {
530    /// Partition keys, outermost first: `["year", "month"]`.
531    pub keys: Vec<String>,
532    /// Distinct values seen for the outermost key, in sorted order. Bounded, so this
533    /// is what was seen rather than necessarily all there is.
534    pub first_key_values: Vec<String>,
535    /// Directories counted at the outermost level.
536    pub count: usize,
537    /// The count stopped at a limit; there are more.
538    pub more: bool,
539}
540
541impl Entry {
542    /// A plain directory row — somewhere to step into, with nothing read from it.
543    pub fn directory(path: &Path) -> Self {
544        Self::new(path.to_path_buf(), EntryKind::Directory)
545    }
546
547    /// A file entry with a chosen display name, for tests that need a search result
548    /// without running a walk to produce one.
549    pub fn for_test(path: &Path, name: &str) -> Self {
550        let mut entry = Self::new(path.to_path_buf(), EntryKind::File);
551        entry.name = name.to_string();
552        entry
553    }
554
555    pub(crate) fn new(path: PathBuf, kind: EntryKind) -> Self {
556        let name = path
557            .file_name()
558            .map(|n| n.to_string_lossy().into_owned())
559            .unwrap_or_else(|| path.to_string_lossy().into_owned());
560        Self {
561            path,
562            kind,
563            name,
564            size: None,
565            modified: None,
566            rows: None,
567            cols: None,
568            cols_sampled: false,
569            columns: Vec::new(),
570            cost: Cost::default(),
571            holds: Default::default(),
572            opens_whole_directory: false,
573            format_spec: None,
574            table: None,
575        }
576    }
577
578    /// Attach size and mtime from a directory entry's metadata.
579    pub(crate) fn with_fs_metadata(mut self, meta: &std::fs::Metadata) -> Self {
580        if meta.is_file() {
581            self.size = Some(meta.len());
582        }
583        self.modified = meta.modified().ok();
584        self
585    }
586}
587
588/// Whether an object key or path is Parquet: named `.parquet`, or a part file with no
589/// extension inside a directory named `.parquet`, as Spark and GBIF write them
590/// (`occurrence.parquet/000001`). Hidden and job files (`_SUCCESS`, `.crc`) are not.
591pub fn is_parquet_key(key: &str) -> bool {
592    let key = key.trim_end_matches('/');
593    let (directory, name) = match key.rsplit_once('/') {
594        Some((directory, name)) => (directory, name),
595        None => ("", key),
596    };
597    if is_bookkeeping(name) {
598        return false;
599    }
600    if name.to_ascii_lowercase().ends_with(".parquet") {
601        return true;
602    }
603    let directory_name = directory.rsplit('/').next().unwrap_or(directory);
604    !name.contains('.') && directory_name.to_ascii_lowercase().ends_with(".parquet")
605}
606
607#[cfg(test)]
608mod parquet_key_tests {
609    use super::is_parquet_key;
610
611    #[test]
612    fn parquet_without_an_extension_is_known_by_its_directory() {
613        assert!(is_parquet_key(
614            "occurrence/2026-09-01/occurrence.parquet/000001"
615        ));
616        assert!(is_parquet_key("data/part-0.parquet"));
617        assert!(is_parquet_key("DATA/PART-0.PARQUET"));
618        assert!(!is_parquet_key(
619            "occurrence/2026-09-01/occurrence.parquet/_SUCCESS"
620        ));
621        assert!(!is_parquet_key("occurrence.parquet/.part-0.crc"));
622        assert!(!is_parquet_key("occurrence/2026-09-01/citation.txt"));
623        assert!(!is_parquet_key("notes/000001"));
624    }
625}
626
627/// What a file with no usable extension turns out to be, from the bytes at its start.
628///
629/// Every format datui reads as a directory puts a fixed signature at the front — Parquet
630/// at both ends, and the other three at the front alone. A name is the cheap answer and
631/// the one every listing uses; this is the expensive one, and it is asked only of a
632/// directory somebody is opening, never of a directory somebody is looking at.
633///
634/// Spark and GBIF both write part files with no extension — `occurrence.parquet/000001`
635/// is read by its directory's name, and the same files under a directory named anything
636/// else were not data at all as far as datui was concerned.
637///
638/// CSV and JSON are deliberately absent: they have no signature, and guessing from the
639/// first line is a parse rather than a look. Which signatures a listing believes is each
640/// format's to say ([`crate::readers::Trusted::listing`]).
641pub fn sniff_format(path: &Path) -> Option<crate::FileFormat> {
642    crate::readers::sniff_file(path, crate::readers::Asked::Listing)
643}
644
645/// What a listing finds a file to be by its first bytes.
646#[derive(Debug, Clone)]
647pub enum Sniffed {
648    /// A format datui reads.
649    Format(crate::FileFormat),
650    /// A format spec's, which reads it.
651    Spec(std::sync::Arc<crate::formats::Spec>),
652}
653
654/// [`sniff_format`], and when no format datui reads says it, the format spec that
655/// reads it as an open would pick one: by glob, else by magic and `match.where`. One
656/// read of the file's head answers both, so a listing reads nothing more for specs.
657pub fn sniff_listed(path: &Path, formats: &crate::formats::Registry) -> Option<Sniffed> {
658    use crate::readers::{Asked, HEAD, head_of, sniff};
659    let head = head_of(path)?;
660    if let Some(format) = sniff(&head, Some(path), Asked::Listing, |_| true) {
661        return Some(Sniffed::Format(format));
662    }
663    formats
664        .listed(path, &head, head.len() < HEAD)
665        .map(Sniffed::Spec)
666}
667
668/// Name `entry` a file of `spec`, which reads it; a spec that reads its records as
669/// several variants makes it a place too, whose tables → lists.
670pub fn name_spec_file(entry: &mut Entry, spec: &crate::formats::Spec) {
671    entry.kind = EntryKind::File;
672    entry.format_spec = Some(spec.name.clone());
673    if spec.lists_variants() {
674        entry.cost.tables = Some(spec.records.variants.len());
675    }
676}
677
678/// Name a local file row no listing classified (a recent) by the spec that reads it,
679/// as a listing names one: a spec's glob, else, when its name says nothing, the magic
680/// in its first bytes.
681pub fn name_unlisted_file(entry: &mut Entry, formats: &crate::formats::Registry) {
682    if formats.is_empty()
683        || entry.kind != EntryKind::File
684        || entry.table.is_some()
685        || is_data_file(&entry.path)
686    {
687        return;
688    }
689    let spec = match formats.by_glob(&entry.path, false).into_iter().next() {
690        Some(spec) => Some(spec),
691        None if worth_sniffing(&entry.path) && is_regular_file(&entry.path) => {
692            match sniff_listed(&entry.path, formats) {
693                Some(Sniffed::Spec(spec)) => Some(spec),
694                _ => None,
695            }
696        }
697        None => None,
698    };
699    if let Some(spec) = spec {
700        name_spec_file(entry, &spec);
701    }
702}
703
704/// How many extension-less files one listing looks inside. A directory of Spark output
705/// is a few hundred part files; past this the rest are listed by name alone.
706pub(crate) const MAX_SNIFFS_PER_DIR: usize = 256;
707
708/// Whether a file's name has no extension at all: `part-00000`, `LICENSE`.
709pub fn has_no_extension(path: &Path) -> bool {
710    path.extension().is_none()
711}
712
713/// Whether a listing looks inside a file to say what it is: one with no extension, or
714/// one whose extension says nothing (`.bin`, which ArduPilot's logs and model
715/// checkpoints share with everything else) or only text (`.log`, which candump
716/// writes; `.txt`).
717pub fn worth_sniffing(path: &Path) -> bool {
718    path.extension().is_none_or(|e| {
719        e.eq_ignore_ascii_case("bin") || data_format(path).is_some_and(crate::FileFormat::is_lines)
720    })
721}
722
723/// Whether a local file is Parquet by its contents: `PAR1` at both ends.
724pub fn has_parquet_magic(path: &Path) -> bool {
725    use std::io::{Read, Seek, SeekFrom};
726    let Ok(mut file) = std::fs::File::open(path) else {
727        return false;
728    };
729    let mut head = [0u8; 4];
730    let mut tail = [0u8; 4];
731    file.read_exact(&mut head).is_ok()
732        && file.seek(SeekFrom::End(-4)).is_ok()
733        && file.read_exact(&mut tail).is_ok()
734        && &head == b"PAR1"
735        && &tail == b"PAR1"
736}
737
738/// Whether a path names a Parquet file: by its extension, or by sitting as a part file
739/// with no extension inside a `.parquet` directory.
740///
741/// Not [`is_parquet_key`], which also answers "does this count toward what a directory
742/// holds" and so says no to a writer's own name. `_manifest.parquet` is a file somebody
743/// may open and the listing shows it; reading its footer is a different question from
744/// whether it makes the directory around it a dataset.
745pub fn is_parquet_path(path: &Path) -> bool {
746    path.extension()
747        .and_then(|e| e.to_str())
748        .is_some_and(|e| e.eq_ignore_ascii_case("parquet"))
749        || is_parquet_key(&directory_and_name(path))
750}
751
752/// Whether a path looks like something datui can open.
753///
754/// Its name, or its place: a part file with no extension inside a `.parquet` directory is
755/// Parquet, as Spark and GBIF write them. Every route that asks what a name means asks
756/// here — the listing, the search, `~` input, the counts and the schema pane — because
757/// the one that did not was always the one that disagreed.
758pub fn is_data_file(path: &Path) -> bool {
759    data_extension(path).is_some() || is_parquet_key(&directory_and_name(path))
760}
761
762/// The extension that says what a file *is*, with any compression suffix walked past.
763///
764/// `sales.csv.gz` is a CSV: `Path::extension` answers `gz`, which is how it is stored
765/// rather than what it holds. Anything deciding a *format* wants this one — two files
766/// named `.csv.gz` and `.json.gz` agree on their extension and on nothing that
767/// matters.
768///
769/// `None` when the name does not end in something datui reads.
770pub fn data_extension(path: &Path) -> Option<String> {
771    let name = path.file_name().and_then(|n| n.to_str())?;
772    let lower = name.to_ascii_lowercase();
773    let mut parts: Vec<&str> = lower.rsplit('.').collect();
774    parts.reverse();
775    if parts.len() < 2 {
776        return None;
777    }
778    // Walk back past a compression suffix so `sales.csv.gz` still reads as CSV.
779    let mut idx = parts.len() - 1;
780    if COMPRESSION_EXTENSIONS.contains(&parts[idx]) && idx > 1 {
781        idx -= 1;
782    }
783    crate::FileFormat::from_extension(parts[idx]).map(|_| parts[idx].to_string())
784}
785
786/// What the home screen says of a file [`unreadable_by_name`] turns away.
787pub const NO_READER: &str = "datui has no reader for this file";
788
789/// Whether a file's name already says datui will not open it: an extension no reader
790/// takes, under any compression suffix. A bare `data.gz` is left to the open, which
791/// looks inside, and so is a name with no extension.
792pub fn unreadable_by_name(path: &Path) -> bool {
793    let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
794        return false;
795    };
796    let name = name.to_ascii_lowercase();
797    let parts: Vec<&str> = name.rsplit('.').collect();
798    let compressed = |last: &str| COMPRESSION_EXTENSIONS.contains(&last);
799    let ext = match parts[..] {
800        [last, inner, _, ..] if compressed(last) => inner,
801        [last, _] if compressed(last) => return false,
802        [last, _, ..] => last,
803        _ => return false,
804    };
805    crate::FileFormat::from_extension(ext).is_none()
806}
807
808/// The format a file's name says it holds, compression suffix walked past.
809///
810/// The question every listing actually asks. Named extensions are not formats: `.ipc`,
811/// `.arrow`, `.arrows` and `.feather` are one format under four names, and a directory holding two
812/// of them is one kind of thing. Asking [`crate::FileFormat`] rather than a list of its
813/// own is what keeps the home screen from offering a file the reader has no route for,
814/// which is how `.txt` came to be listed and refused and `.psv` readable and invisible.
815pub fn data_format(path: &Path) -> Option<crate::FileFormat> {
816    // A sharded checkpoint's index is the model, not a JSON table.
817    crate::FileFormat::from_name_ending(path)
818        .or_else(|| crate::FileFormat::from_extension(&data_extension(path)?))
819}
820
821/// The two path segments `is_parquet_key` needs, as it splits them.
822///
823/// A whole path would reach it with backslashes on Windows, which it does not split on,
824/// so `occurrence.parquet\000001` would arrive as one name that contains a dot and be
825/// read as an ordinary file. The same reason `DataTableState::directory_and_name` exists.
826pub(crate) fn directory_and_name(path: &Path) -> String {
827    let name = path.file_name().unwrap_or_default().to_string_lossy();
828    match path.parent().and_then(|p| p.file_name()) {
829        Some(directory) => format!("{}/{name}", directory.to_string_lossy()),
830        None => name.into_owned(),
831    }
832}
833
834/// Commonest first, Parquet ahead of anything it ties with, then by name, so the line
835/// reads the same way twice running.
836///
837/// One order, by name of format, for everything that ranks a directory's formats: the
838/// local label, the local read that picks a reader, and the cloud label. They agreed on
839/// the common case and not on a tie — a directory of two CSV and two Parquet was
840/// *labelled* `2 csv · 2 parquet` and *read* as Parquet, so the row said one thing and
841/// `Enter` did another. Parquet wins the tie because it is the format a directory of data
842/// files is most likely to be about and the one every other route reads in place.
843///
844/// Named rather than written inline because `read_dir` order is exactly what it exists
845/// to remove, and a fixture on disk cannot pin an order that depends on it: the tie is
846/// the whole point and only a caller choosing the input order can put one there.
847pub(crate) fn rank_formats(a: (&str, usize), b: (&str, usize)) -> std::cmp::Ordering {
848    // Text is what is read when nothing else is: a README among data files is not
849    // the table, however many there are.
850    let text = crate::FileFormat::Text.name();
851    (a.0 == text)
852        .cmp(&(b.0 == text))
853        .then_with(|| b.1.cmp(&a.1))
854        .then_with(|| (a.0 != "parquet").cmp(&(b.0 != "parquet")))
855        .then_with(|| a.0.cmp(b.0))
856}
857
858/// Whether a file is one Hugging Face `datasets` writes beside a dataset's Arrow
859/// shards to describe them: `save_to_disk` writes both, and its cache the first. They
860/// are the dataset's metadata, not its data, where `.arrow` files sit beside them; two
861/// JSON files would otherwise outnumber a dataset of one shard and be read instead of
862/// it.
863pub(crate) fn is_hugging_face_metadata(name: &str) -> bool {
864    matches!(name, "dataset_info.json" | "state.json")
865}
866
867fn order_formats(counts: &mut [(crate::FileFormat, usize)]) {
868    counts.sort_by(|a, b| rank_formats((a.0.name(), a.1), (b.0.name(), b.1)));
869}
870
871/// Whether a listing entry is bookkeeping rather than data.
872///
873/// The one convention datui knows, and the only one: a leading `_` or `.`, which every
874/// engine in the table uses for the files it leaves beside its output — `_SUCCESS`,
875/// `_committed_*`, `_started_*`, `_metadata.json`, `.crc` — and the `_$folder$` marker
876/// some tools write to stand in for a folder in a flat store.
877///
878/// One predicate rather than the five that had drifted apart: a local listing skipped
879/// dotfiles and the literal `_SUCCESS`, a local open skipped both prefixes, and the
880/// cloud listing knew three more names. A directory whose ninth entry is `_metadata.json`
881/// answered `multi` locally and `dir` in a bucket for no better reason than that.
882pub fn is_bookkeeping(name: &str) -> bool {
883    // The `_$folder$` marker is asked about first, because it is a suffix and the
884    // folder it stands in for can itself be a partition: legacy s3n and EMR write
885    // `year=2024_$folder$` beside `year=2024/`, and a partition test looking only for
886    // an `=` calls that marker data.
887    if name.ends_with("_$folder$") {
888        return true;
889    }
890    // A `key=value` name is a partition wherever it appears, whatever it starts with.
891    // Spark and Hive partition on internal columns — `_date=2024-01-01`, `_c0=…` — and
892    // reading those as a writer's own files loses the whole dataset.
893    if is_partition_name(name) {
894        return false;
895    }
896    name.starts_with(['_', '.'])
897}
898
899/// Whether a format, by name, is model weights.
900fn is_weights(name: &str) -> bool {
901    name == crate::FileFormat::Safetensors.name() || name == crate::FileFormat::Gguf.name()
902}
903
904/// Whether formats found side by side in one directory are a model: weights of one
905/// format, with nothing else beside them but JSON (a config, a tokenizer). Such a
906/// directory is the model, however many JSON files outnumber the shards.
907pub(crate) fn is_model_directory<'a>(names: impl IntoIterator<Item = &'a str>) -> bool {
908    let mut weights = None;
909    for name in names {
910        if is_weights(name) {
911            if weights.is_some_and(|w| w != name) {
912                return false;
913            }
914            weights = Some(name);
915        } else if name != crate::FileFormat::Json.name() {
916            return false;
917        }
918    }
919    weights.is_some()
920}
921
922/// The format names `holds` counted.
923fn counts_names(holds: &Holds) -> impl Iterator<Item = &str> {
924    holds.formats.iter().map(|(name, _)| name.as_str())
925}
926
927/// Whether a name is a hive partition (`year=2024`): `key=value`, with a non-empty key.
928/// The value may be empty in practice.
929pub fn is_partition_name(name: &str) -> bool {
930    matches!(name.find('='), Some(i) if i > 0)
931}
932
933/// What the data files sitting directly in a directory say about how to read it.
934///
935/// A directory opened as one dataset has to be read by *something*, and the only honest
936/// source for that is the files in it. Before this existed the answer was assumed:
937/// any directory was scanned as Parquet, so a directory of `.json.gz` was opened by
938/// seeking to the end of each file for a `PAR1` that was never going to be there.
939#[derive(Debug, Clone, PartialEq, Eq)]
940pub enum DirectoryFormat {
941    /// Every data file directly in the directory reads as this one format, and these are
942    /// the files. Sorted, because a concatenation's row order is its file order.
943    One(crate::FileFormat, Vec<PathBuf>),
944    /// The files name more than one format. The commonest of them is the table — a
945    /// directory of a thousand CSVs and one stray JSON is a directory of CSVs — and the
946    /// rest are counted by format so the read can say what it passed over. Parquet wins a
947    /// tie, because it is the format a directory of data files is most likely to be about
948    /// and the one every other route here reads in place.
949    Mixed {
950        format: crate::FileFormat,
951        files: Vec<PathBuf>,
952        passed_over: Vec<(crate::FileFormat, usize)>,
953    },
954    /// The directory settles nothing by itself: it holds no readable data file directly,
955    /// or it has subdirectories and so may hold its data below. A hive dataset looks like
956    /// this — its files are a level down, under `key=value`.
957    Deeper,
958}
959
960/// What [`DirectoryFormat`] the data files directly in `dir` amount to.
961///
962/// One level only, and no file is opened: this reads names, exactly as the rest of
963/// this module does. A directory of two hundred thousand files costs one listing
964/// and nothing per file beyond it — the entry's own type comes back with the name, so
965/// there is no `stat` to bound.
966///
967/// Deliberately *not* bounded by [`MAX_ENTRIES_PER_DIR`], unlike every listing in this
968/// module. The files this returns are not a menu to show, they are the table to read:
969/// stopping at five thousand of a directory's six thousand CSVs would open it with a row
970/// count, a schema union and every aggregate quietly computed over a subset, and the
971/// `take` running before the sort would drop whichever file the directory read
972/// happened to return last. The Parquet route this mirrors hands the directory to a scan
973/// that enumerates it, with no cap either.
974pub fn directory_format(dir: &Path) -> DirectoryFormat {
975    let Ok(iter) = std::fs::read_dir(dir) else {
976        return DirectoryFormat::Deeper;
977    };
978
979    // Every format the directory names, with the files of each. A directory of one format
980    // takes the only entry; a directory of several takes the commonest and reports the
981    // rest, which is what stops one stray file deciding a directory cannot be read.
982    let mut by_format: Vec<(crate::FileFormat, Vec<PathBuf>)> = Vec::new();
983    // Files whose names say nothing, kept in case their bytes do. See below.
984    let mut nameless: Vec<PathBuf> = Vec::new();
985    let mut partitioned = false;
986    for entry in iter.flatten() {
987        let path = entry.path();
988        let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
989            continue;
990        };
991        // A table format's own files are not the table's, and a dotfile is nobody's.
992        // The same test every other route makes, so they agree on what is data.
993        if is_bookkeeping(name) {
994            continue;
995        }
996        // The type the directory read already returned, rather than a `stat` per
997        // entry: on a share that is a round trip per entry, and the question is only
998        // whether this is a file. A symlink still gets the stat, because `d_type`
999        // cannot say what is on the other end of one.
1000        let is_file = match entry.file_type() {
1001            Ok(kind) if kind.is_symlink() => is_regular_file(&path),
1002            Ok(kind) => kind.is_file(),
1003            Err(_) => is_regular_file(&path),
1004        };
1005        if !is_file {
1006            // Only a partition, not any subdirectory: a directory of CSVs with some
1007            // unrelated directory beside them is still a directory of CSVs, and handing
1008            // that to a scanner that walks trees is the very thing this function exists
1009            // to stop.
1010            partitioned |= is_partition_dir(&path);
1011            continue;
1012        }
1013        let Some(found) = data_format(&path) else {
1014            // A name that says nothing. Kept rather than dropped, because the bytes may
1015            // still say what it is — see below, where they are asked.
1016            if path.extension().is_none() {
1017                nameless.push(path);
1018            }
1019            continue;
1020        };
1021        match by_format.iter_mut().find(|(f, _)| *f == found) {
1022            Some((_, of_that_format)) => of_that_format.push(path),
1023            None => by_format.push((found, vec![path])),
1024        }
1025    }
1026
1027    // Files with no extension, in a directory whose names settled nothing. Spark and GBIF
1028    // both write part files this way; `occurrence.parquet/000001` is read by its
1029    // directory's name, and the same files under a directory named anything else were not
1030    // data at all as far as datui was concerned — the directory was `dir` and its files
1031    // were not listed.
1032    //
1033    // Only when the names have nothing to say. A directory of Parquet with a `LICENSE` in
1034    // it is a directory of Parquet, and opening the `LICENSE` to find out is a read per
1035    // file for an answer already given.
1036    //
1037    // A spread rather than every one, for the reason `sample_footers` takes a spread:
1038    // the cost is one open per file, and a directory written by one job holds one kind of
1039    // thing. They have to agree — a directory where the ends disagree is not one table by
1040    // any reading — and then all of them are taken as that format, because a scan that
1041    // reads what it can and says what it could not is what happens to the odd one out.
1042    if by_format.is_empty() && !nameless.is_empty() {
1043        nameless.sort();
1044        let mut picks = vec![0, nameless.len() / 2, nameless.len() - 1];
1045        picks.dedup();
1046        let sniffed: Vec<crate::FileFormat> = picks
1047            .iter()
1048            .filter_map(|i| nameless.get(*i))
1049            .filter_map(|f| sniff_format(f))
1050            .collect();
1051        if sniffed.len() == picks.len()
1052            && let Some(found) = sniffed.first().copied()
1053            && sniffed.iter().all(|f| *f == found)
1054        {
1055            by_format.push((found, nameless));
1056        }
1057    }
1058
1059    // Text is data only where nothing else is: a README beside Parquet is not a
1060    // candidate, as the listing does not count it.
1061    if by_format.iter().any(|(f, _)| !f.is_lines()) {
1062        by_format.retain(|(f, _)| !f.is_lines());
1063    }
1064
1065    // A Hugging Face dataset's own JSON files, beside its shards.
1066    if by_format
1067        .iter()
1068        .any(|(f, _)| *f == crate::FileFormat::Arrow)
1069    {
1070        for (format, files) in &mut by_format {
1071            if *format == crate::FileFormat::Json {
1072                files.retain(|f| {
1073                    !f.file_name()
1074                        .and_then(|n| n.to_str())
1075                        .is_some_and(is_hugging_face_metadata)
1076                });
1077            }
1078        }
1079        by_format.retain(|(_, files)| !files.is_empty());
1080    }
1081
1082    // Decided after the whole listing rather than at the first entry that could settle
1083    // it, so the answer does not depend on the order a directory read happens to
1084    // return. One `key=value` below and the directory stops being the whole story: a hive
1085    // dataset's data is down there, whatever strays are lying at the top.
1086    if partitioned {
1087        return DirectoryFormat::Deeper;
1088    }
1089
1090    // The one order every route ranks a directory's formats by, so the reader this picks
1091    // is the format the label names.
1092    by_format.sort_by(|a, b| rank_formats((a.0.name(), a.1.len()), (b.0.name(), b.1.len())));
1093    // A directory holding model weights is the model. Its config and tokenizer JSON
1094    // sit beside the shards and often outnumber them, which does not make it a table
1095    // of JSON; the JSON is what the read passes over.
1096    if is_model_directory(by_format.iter().map(|(f, _)| f.name()))
1097        && let Some(at) = by_format.iter().position(|(f, _)| is_weights(f.name()))
1098    {
1099        let weights = by_format.remove(at);
1100        by_format.insert(0, weights);
1101    }
1102    let mut by_format = by_format.into_iter();
1103    let Some((format, mut files)) = by_format.next() else {
1104        return DirectoryFormat::Deeper;
1105    };
1106    files.sort();
1107    let passed_over: Vec<(crate::FileFormat, usize)> =
1108        by_format.map(|(f, of_that)| (f, of_that.len())).collect();
1109    if passed_over.is_empty() {
1110        DirectoryFormat::One(format, files)
1111    } else {
1112        DirectoryFormat::Mixed {
1113            format,
1114            files,
1115            passed_over,
1116        }
1117    }
1118}
1119
1120/// How far down a hive root is followed looking for the files it partitions.
1121///
1122/// A dataset partitioned by year, month, day and hour is four; past this the directory
1123/// is something other than a hive dataset, and guessing further costs a directory
1124/// read per level on a share.
1125const MAX_HIVE_DEPTH: usize = 16;
1126
1127/// What the files under a hive root's `key=value` partitions actually are.
1128///
1129/// A hive root holds no data itself, so [`directory_format`] can only say `Deeper` about
1130/// one. This follows a single spine down — the same one path through the tree a hive
1131/// scan reads its schema from — and reports what it finds at the bottom.
1132///
1133/// One spine, and the first partition at each level, so a dataset of ten thousand
1134/// partitions costs what one of two costs. That makes it a sample: a tree whose
1135/// partitions disagree is reported as whatever the first one holds. The alternative
1136/// is walking the dataset to answer a question asked before it is opened.
1137pub fn hive_leaf_format(dir: &Path) -> DirectoryFormat {
1138    let mut at = dir.to_path_buf();
1139    for _ in 0..MAX_HIVE_DEPTH {
1140        match directory_format(&at) {
1141            // Nothing here settles it. Follow the partitions down, if there are any.
1142            DirectoryFormat::Deeper => match first_partition(&at) {
1143                Some(next) => at = next,
1144                None => return DirectoryFormat::Deeper,
1145            },
1146            settled => return settled,
1147        }
1148    }
1149    DirectoryFormat::Deeper
1150}
1151
1152/// The first `key=value` subdirectory of `dir`, by name.
1153///
1154/// By name rather than in directory order: two runs asking what a dataset holds must
1155/// not look at different partitions and give different answers.
1156fn first_partition(dir: &Path) -> Option<PathBuf> {
1157    let iter = std::fs::read_dir(dir).ok()?;
1158    iter.flatten()
1159        .take(MAX_ENTRIES_PER_DIR)
1160        .map(|entry| entry.path())
1161        .filter(|path| is_partition_dir(path) && path.is_dir())
1162        .min()
1163}
1164
1165/// Whether a directory name is a hive partition (`year=2024`).
1166fn is_partition_dir(path: &Path) -> bool {
1167    path.file_name()
1168        .and_then(|n| n.to_str())
1169        .is_some_and(is_partition_name)
1170}
1171
1172/// Classify a directory without walking it.
1173///
1174/// Reads one listing, bounded by [`MAX_ENTRIES_PER_DIR`] rather than by a probe of the
1175/// first few entries. A probe makes the answer depend on the order the filesystem hands
1176/// entries back: a directory of eight Parquet files followed by `_metadata.json` answered
1177/// `multi` locally, where a bucket listing the same directory sorts the JSON first and
1178/// answered `dir`.
1179///
1180/// It costs more than the probe did: a plain directory row is enriched with nothing, so
1181/// its listing is read for this alone, and a directory of five thousand entries is read
1182/// whole where eight used to settle it — six hundred times the entries, for the worst
1183/// row, and `look_into_batch` walks a batch of sixteen of them one at a time.
1184///
1185/// That includes rows on a network mount: `network_check` gates listing a directory you
1186/// have browsed into, not classifying the rows of one. It is one `getdents` walk, with
1187/// a `stat` only for a symlink, since `d_type` cannot say what is on the far end of one
1188/// — so a directory of symlinks is the expensive case. `home_open_selected` makes the
1189/// call on the thread reading keys; every other caller is on a worker.
1190///
1191/// Capping it lower again would put the order-dependence back exactly where the
1192/// directories are biggest.
1193pub fn classify_directory(path: &Path) -> EntryKind {
1194    look_at_directory(path).0
1195}
1196
1197/// The kind *and* what the listing found, from one read of it.
1198///
1199/// Two answers to two questions. The kind decides what `Enter` does with the directory;
1200/// the count says what is in it, and the row's label is written from that — so a label
1201/// that is wrong about the first is still true about the second.
1202pub fn look_at_directory(path: &Path) -> (EntryKind, Holds) {
1203    let mut holds = Holds::default();
1204    // Before anything is counted: a lake table's data files genuinely do agree on a
1205    // schema, so every rule below says "one table" and is right about the schema and
1206    // wrong about the rows.
1207    //
1208    // And before the listing, which it does not need: three `join` tests answer it, and
1209    // counting would walk up to `MAX_ENTRIES_PER_DIR` entries of every table in a
1210    // warehouse, on every pass, for a `holds` line beside a table whose files `enrich`
1211    // then refuses to read. A prefix in a bucket does carry one, because the listing it
1212    // is counted from had already been paid for.
1213    if let Some(lake) = lake_table(path) {
1214        return (lake, holds);
1215    }
1216    let Ok(iter) = std::fs::read_dir(path) else {
1217        return (EntryKind::Directory, holds);
1218    };
1219
1220    let mut partitions = 0usize;
1221    let mut data_files = 0usize;
1222    let mut seen = 0usize;
1223    // As `scan_dir_bounded` does, so the tally agrees with the rows inside.
1224    let mut sniffs_left = if crate::home::is_remote_path(path) {
1225        0
1226    } else {
1227        MAX_SNIFFS_PER_DIR
1228    };
1229    let mut counts: Vec<(crate::FileFormat, usize)> = Vec::new();
1230    // A multi-file dataset is homogeneous by definition; a directory that merely
1231    // contains two different spreadsheets is not one. Compared as formats rather than
1232    // as extensions, so `.ipc` beside `.arrow` is one kind of thing and not two.
1233    let mut format: Option<crate::FileFormat> = None;
1234    let mut mixed_formats = false;
1235    // Counted as data until the listing is done; see `is_hugging_face_metadata`.
1236    let mut hugging_face: Vec<String> = Vec::new();
1237
1238    // Bounded where the entries come from rather than after they are counted: a
1239    // Hadoop-style output directory is a `.crc` per data file, and skipping those before
1240    // the count would let the walk run to twice the cap. The cap is a cost bound and not
1241    // a correctness one either way — a directory past it is decided by whichever entries
1242    // came back first, whether they were data or a writer's own.
1243    // One past the cap, so "there is more" is known without paying to process it —
1244    // the same shape `scan_dir_bounded` uses, and the reason a directory of exactly five
1245    // thousand entries is a total rather than a floor.
1246    for (entries, entry) in iter.flatten().take(MAX_ENTRIES_PER_DIR + 1).enumerate() {
1247        if entries >= MAX_ENTRIES_PER_DIR {
1248            holds.truncated = true;
1249            break;
1250        }
1251        let entry_path = entry.path();
1252        let name = entry.file_name();
1253        // The markers and job files tools leave beside their output, by the one test
1254        // every route makes.
1255        let name = name.to_string_lossy().into_owned();
1256        if is_bookkeeping(&name) {
1257            holds.skipped += 1;
1258            // The first few by name, not the first few the filesystem returned: a line
1259            // in the pane that reads differently on two runs of the same directory is the
1260            // order-dependence this module just spent a release removing. Past
1261            // `MAX_ENTRIES_PER_DIR` it is the first few by name *of what was read*, and
1262            // where the walk stopped is the filesystem's order again — which the `+` on
1263            // every count beside them says.
1264            if holds.skipped_names.last().is_none_or(|last| &name < last)
1265                || holds.skipped_names.len() < SKIPPED_NAMES_SHOWN
1266            {
1267                holds.skipped_names.push(name);
1268                holds.skipped_names.sort();
1269                holds.skipped_names.truncate(SKIPPED_NAMES_SHOWN);
1270            }
1271            continue;
1272        }
1273        // The type the directory read already returned, rather than a `stat` per entry:
1274        // this walks the whole listing now, and on a share every stat is a round trip.
1275        // A symlink still gets one, because `d_type` cannot say what is on the far end.
1276        //
1277        // A regular file rather than "not a directory", the test `directory_format`
1278        // makes: a FIFO named `a.csv` blocks whoever opens it until a writer appears, and
1279        // a broken symlink named `b.csv` opens as nothing. Counting either as data offers
1280        // a directory that cannot be read.
1281        let followed = |path: &Path| {
1282            // One `stat`, not two: what is on the far end is one question, and asking
1283            // it twice is a second round trip on a share.
1284            std::fs::metadata(path).map_or((false, false), |m| (m.is_dir(), m.is_file()))
1285        };
1286        let (is_dir, is_file) = match entry.file_type() {
1287            Ok(kind) if kind.is_symlink() => followed(&entry_path),
1288            Ok(kind) => (kind.is_dir(), kind.is_file()),
1289            Err(_) => followed(&entry_path),
1290        };
1291        if is_dir {
1292            holds.directories += 1;
1293            if is_partition_dir(&entry_path) {
1294                partitions += 1;
1295            }
1296        } else if let Some(found) = data_format(&entry_path)
1297            // Text by its name, unless its bytes say more: below.
1298            .filter(|f| !f.is_lines())
1299            // A sharded checkpoint's index is counted as the JSON it is, so the label
1300            // counts the shards; the read still takes it, for the metadata it carries.
1301            .map(
1302                |found| match crate::model_files::is_safetensors_index(&entry_path) {
1303                    true => crate::FileFormat::Json,
1304                    false => found,
1305                },
1306            )
1307            // Data by where it sits rather than by its name: see [`is_data_file`].
1308            .or_else(|| {
1309                is_parquet_key(&directory_and_name(&entry_path))
1310                    .then_some(crate::FileFormat::Parquet)
1311            })
1312            .or_else(|| {
1313                (is_file && sniffs_left > 0 && worth_sniffing(&entry_path)).then(|| {
1314                    sniffs_left -= 1;
1315                    sniff_format(&entry_path)
1316                })?
1317            })
1318            .or_else(|| data_format(&entry_path))
1319            .filter(|_| is_file)
1320        {
1321            if found == crate::FileFormat::Json && is_hugging_face_metadata(&name) {
1322                hugging_face.push(name.clone());
1323            }
1324            data_files += 1;
1325            match counts.iter_mut().find(|(f, _)| *f == found) {
1326                Some((_, n)) => *n += 1,
1327                None => counts.push((found, 1)),
1328            }
1329            match format {
1330                None => format = Some(found),
1331                Some(first) if first != found => mixed_formats = true,
1332                Some(_) => {}
1333            }
1334        } else if is_file && has_no_extension(&entry_path) {
1335            holds.unnamed += 1;
1336        } else {
1337            // Everything else in the listing: a file with no reader, and a name with
1338            // nothing behind it — a FIFO, a socket, a broken symlink. Named like data
1339            // or not, none of them can be read.
1340            holds.not_read += 1;
1341        }
1342        seen += 1;
1343    }
1344
1345    // A Hugging Face dataset's own JSON files are its writer's, like `_SUCCESS`.
1346    if !hugging_face.is_empty() && counts.iter().any(|(f, _)| *f == crate::FileFormat::Arrow) {
1347        let n = hugging_face.len();
1348        for (format, count) in &mut counts {
1349            if *format == crate::FileFormat::Json {
1350                *count -= n;
1351            }
1352        }
1353        counts.retain(|(_, count)| *count > 0);
1354        data_files -= n;
1355        seen -= n;
1356        holds.skipped += n;
1357        holds.skipped_names.extend(hugging_face);
1358        holds.skipped_names.sort();
1359        holds.skipped_names.truncate(SKIPPED_NAMES_SHOWN);
1360        mixed_formats = counts.len() > 1;
1361        format = counts.first().map(|(f, _)| *f);
1362    }
1363
1364    // Text is data only where nothing else is: a README beside Parquet is a file
1365    // nothing reads as the directory's table, as it is to the cloud route.
1366    if counts.iter().any(|(f, _)| !f.is_lines())
1367        && let Some(at) = counts.iter().position(|(f, _)| f.is_lines())
1368    {
1369        let (_, n) = counts.remove(at);
1370        data_files -= n;
1371        holds.not_read += n;
1372        mixed_formats = counts.len() > 1;
1373        format = counts.first().map(|(f, _)| *f);
1374    }
1375
1376    holds.partitions = partitions;
1377    order_formats(&mut counts);
1378    holds.formats = counts
1379        .into_iter()
1380        .map(|(f, n)| (f.name().to_string(), n))
1381        .collect();
1382
1383    // Unchanged from before the probe went, deliberately. Reading the whole listing
1384    // makes one `notes=old` among twenty ordinary subdirectories a hive root every time
1385    // rather than only when it came back first, and two attempts at a majority to rule
1386    // that out each refused a real hive root instead — against everything present, one
1387    // with a README beside it; against the other directories, one with a `scripts/` and a
1388    // `docs/`. Refusing a dataset is the worse direction, and a rule per case is what
1389    // #275 exists to stop. The label stops deciding what `Enter` does in phase 3, and
1390    // the question goes with it.
1391    // Deterministic now rather than occasional, which is the cost of the whole listing:
1392    // a source tree with a `cfg=debug/` in it reads `hive` on every pass, and `enrich`
1393    // then walks it to depth four looking for footers. Left alone all the same — see
1394    // above for the two majorities that refused real hive roots instead.
1395    if partitions > 0 && partitions >= data_files {
1396        return (EntryKind::Hive, holds);
1397    }
1398
1399    // Require a format that can actually be read as many files. Without this the home
1400    // screen offers a directory of `.tsv` or `.xlsx` as one dataset and the open refuses
1401    // it — the same "offered but unreadable" the one vocabulary exists to stop, one
1402    // layer up.
1403    let readable_as_one = format.is_some_and(crate::FileFormat::reads_many_files);
1404    // Require homogeneity *and* that data is what this directory is mostly for.
1405    // Without the majority test, any directory with a couple of stray CSVs in it would
1406    // be offered as a dataset, which is worse than useless: it hides the directory.
1407    let homogeneous = data_files > 1 && !mixed_formats && readable_as_one;
1408    let mostly_data = data_files * 2 >= seen;
1409    // A model directory opens as the model: its shards as one table, the JSON beside
1410    // them left out. One file of weights is a model too.
1411    let model = partitions == 0 && is_model_directory(counts_names(&holds));
1412    let kind = if (homogeneous && mostly_data) || model {
1413        EntryKind::MultiFile
1414    } else {
1415        // Everything else — including a directory holding a single data file — is a
1416        // place to look inside, not a dataset in its own right.
1417        EntryKind::Directory
1418    };
1419    (kind, holds)
1420}
1421
1422/// Entries under `metadata/` to look at before giving up on Iceberg. A table with a
1423/// long history has thousands, and the newest are not first in any order a directory
1424/// read promises — but `vN.metadata.json` is written on the first commit and never
1425/// removed, so one is always there to find.
1426const ICEBERG_METADATA_PROBE: usize = 64;
1427
1428/// Whether `path` is the root of a lake table, and which.
1429///
1430/// Marker directory names are convention knowledge, which the one-table rule
1431/// deliberately keeps out: inferring a dataset from filenames is a list that is never
1432/// finished. These three are a different thing — a declared format with a specified
1433/// layout, where the marker is part of the spec.
1434///
1435/// Named directly rather than found by walking the listing: three `join` tests answer it
1436/// whatever the directory holds, where a walk pays for every entry of a table with a
1437/// hundred thousand data files to find one name it already knows.
1438fn lake_table(path: &Path) -> Option<EntryKind> {
1439    if path.join("_delta_log").is_dir() {
1440        return Some(EntryKind::Delta);
1441    }
1442    if path.join(".hoodie").is_dir() {
1443        return Some(EntryKind::Hudi);
1444    }
1445    // Iceberg's marker is a plain name, so it takes the whole shape: metadata beside
1446    // data, and a metadata file actually in it. `metadata/` alone is a directory anybody
1447    // may have.
1448    let metadata = path.join("metadata");
1449    if path.join("data").is_dir()
1450        && metadata.is_dir()
1451        && std::fs::read_dir(&metadata).is_ok_and(|entries| {
1452            entries
1453                .flatten()
1454                .take(ICEBERG_METADATA_PROBE)
1455                .any(|e| e.file_name().to_string_lossy().ends_with(".metadata.json"))
1456        })
1457    {
1458        return Some(EntryKind::Iceberg);
1459    }
1460    None
1461}
1462
1463/// List one directory level, classified. Never recurses.
1464///
1465/// Errors are swallowed deliberately: an unreadable or unmounted directory yields an
1466/// empty listing rather than failing the home screen, and the caller reports
1467/// availability separately.
1468pub fn scan_dir(dir: &Path) -> Vec<Entry> {
1469    scan_dir_bounded(dir).entries
1470}
1471
1472/// What one directory listing produced, and whether it saw all of it.
1473#[derive(Debug, Clone, Default)]
1474pub struct Scan {
1475    pub entries: Vec<Entry>,
1476    /// The directory held more than `MAX_ENTRIES_PER_DIR`; `entries` is a prefix of
1477    /// it. Worth saying out loud: a listing that silently stops at five thousand
1478    /// looks identical to a directory that simply has five thousand things in it.
1479    pub truncated: bool,
1480}
1481
1482/// List one directory, doing a bounded amount of work regardless of what is in it.
1483///
1484/// The cost is one `read_dir` and a `stat` per entry, bounded by
1485/// [`MAX_ENTRIES_PER_DIR`], and nothing per subdirectory at all.
1486///
1487/// **Nothing here is classified.** Telling a hive dataset from a plain directory means
1488/// reading the directory, which is a round trip apiece on a share — so no listing pays
1489/// for it, however small. Every subdirectory comes back [`EntryKind::Unknown`], which
1490/// claims nothing, and is looked into later from the viewport, a batch at a time, by
1491/// whoever is actually reading the rows.
1492///
1493/// That is what makes a row's label a fact about the row. Classifying the first
1494/// sixty-four subdirectories and calling every identical one after them a plain directory
1495/// made it a fact about position instead; classifying them only when a listing is small
1496/// enough moved the arbitrariness rather than removing it, since two directories holding
1497/// the same subdirectories would still disagree about what to call them.
1498pub fn scan_dir_bounded(dir: &Path) -> Scan {
1499    scan_dir_progressive(dir, |_| {})
1500}
1501
1502/// [`scan_dir_bounded`], naming the files it looks inside by `formats` too: a file
1503/// whose first bytes carry a spec's magic is listed as that spec's.
1504pub fn scan_dir_specs(dir: &Path, formats: &crate::formats::Registry) -> Scan {
1505    scan_dir_with(dir, formats, |_| {})
1506}
1507
1508/// How often a listing still being read shows what it has so far.
1509const LISTING_PROGRESS_EVERY: std::time::Duration = std::time::Duration::from_millis(250);
1510
1511/// [`scan_dir_bounded`], handing `progress` the rows read so far, sorted, every
1512/// [`LISTING_PROGRESS_EVERY`] while the read goes on. A directory a share takes seconds
1513/// to list shows its first rows as they arrive rather than a spinner until the last.
1514pub fn scan_dir_progressive(dir: &Path, progress: impl FnMut(&[Entry])) -> Scan {
1515    scan_dir_with(dir, &crate::formats::Registry::default(), progress)
1516}
1517
1518fn scan_dir_with(
1519    dir: &Path,
1520    formats: &crate::formats::Registry,
1521    mut progress: impl FnMut(&[Entry]),
1522) -> Scan {
1523    let Ok(iter) = std::fs::read_dir(dir) else {
1524        return Scan::default();
1525    };
1526    let mut shown = std::time::Instant::now();
1527
1528    let mut entries = Vec::new();
1529    let mut seen = 0usize;
1530    let mut truncated = false;
1531    // Files with no extension are looked at, a few bytes each, so a Spark part file
1532    // is listed as the data it is while a LICENSE stays out of the way. Never on a
1533    // share, where each open is a round trip and one that may not come back.
1534    let mut sniffs_left = if crate::home::is_remote_path(dir) {
1535        0
1536    } else {
1537        MAX_SNIFFS_PER_DIR
1538    };
1539
1540    // One past the cap: enough to know more exists without paying to process it.
1541    for dir_entry in iter.flatten().take(MAX_ENTRIES_PER_DIR + 1) {
1542        seen += 1;
1543        if seen > MAX_ENTRIES_PER_DIR {
1544            truncated = true;
1545            break;
1546        }
1547
1548        let path = dir_entry.path();
1549        let name = dir_entry.file_name();
1550        if name.to_string_lossy().starts_with('.') {
1551            continue;
1552        }
1553
1554        let Ok(meta) = dir_entry.metadata() else {
1555            continue;
1556        };
1557
1558        let mut spec = None;
1559        let kind = if meta.is_dir() {
1560            EntryKind::Unknown
1561        } else if meta.is_file() && is_data_file(&path) {
1562            EntryKind::File
1563        } else if meta.is_file() && sniffs_left > 0 && worth_sniffing(&path) {
1564            sniffs_left -= 1;
1565            match sniff_listed(&path, formats) {
1566                Some(Sniffed::Format(_)) => EntryKind::File,
1567                Some(Sniffed::Spec(found)) => {
1568                    spec = Some(found);
1569                    EntryKind::File
1570                }
1571                None => EntryKind::Other,
1572            }
1573        } else if meta.is_file() {
1574            EntryKind::Other
1575        } else {
1576            // Not a directory or a regular file. A FIFO named `x.parquet` is a
1577            // listing entry datui must never offer to open.
1578            continue;
1579        };
1580
1581        let mut entry = Entry::new(path, kind).with_fs_metadata(&meta);
1582        if let Some(spec) = spec {
1583            name_spec_file(&mut entry, &spec);
1584        }
1585        entries.push(entry);
1586        if shown.elapsed() >= LISTING_PROGRESS_EVERY {
1587            let mut so_far = entries.clone();
1588            sort_entries(&mut so_far);
1589            progress(&so_far);
1590            shown = std::time::Instant::now();
1591        }
1592    }
1593
1594    sort_entries(&mut entries);
1595    Scan { entries, truncated }
1596}
1597
1598/// Datasets first, then directories; each group alphabetical.
1599///
1600/// Recency is a better sort for recents, but a directory listing is a place you
1601/// scan by name, so name order wins here.
1602///
1603/// A row nothing has looked into yet sorts with the directories, although
1604/// [`EntryKind::is_dataset`] offers it as openable. In a fresh listing that is every
1605/// subdirectory, so what this amounts to there is files first and directories after —
1606/// and it is the one ordering a directory can be given before anything is known about
1607/// it, since it is where the row lands if the directory turns out to be a plain one.
1608///
1609/// Which is the point: a kind arriving later never moves the row, because a row that
1610/// moves out from under the cursor while you are scrolling is worse than a label that
1611/// is late.
1612pub(crate) fn sort_entries(entries: &mut [Entry]) {
1613    entries.sort_by(|a, b| {
1614        // Data, then directories, then what datui cannot read.
1615        let group = |k: EntryKind| match k {
1616            k if k.is_known_dataset() => 0,
1617            EntryKind::Other => 2,
1618            _ => 1,
1619        };
1620        group(a.kind).cmp(&group(b.kind)).then_with(|| {
1621            a.name
1622                .to_ascii_lowercase()
1623                .cmp(&b.name.to_ascii_lowercase())
1624        })
1625    });
1626}
1627
1628/// How far below a directory the footer walk goes. A hive dataset partitioned by year,
1629/// month, day and hour is four; past this the files belong to something else.
1630const MAX_WALK_DEPTH: u8 = 4;
1631
1632/// Upper bound on Parquet footers read to size a multi-file or hive dataset.
1633///
1634/// Two partitions is cheap; five thousand is not, and a home screen that stalls on
1635/// the biggest dataset is worse than one that admits it does not know. Past this
1636/// bound the count is left blank rather than reported as a partial total.
1637const MAX_FOOTERS_PER_DATASET: usize = 64;
1638
1639/// Fill in row and column counts for a dataset, from Parquet footers only.
1640///
1641/// Handles a single file, and sums a bounded number of files for hive and multi-file
1642/// datasets. Anything not backed by Parquet keeps `None`, which the UI shows as an
1643/// honest blank.
1644pub fn enrich(entry: &mut Entry) {
1645    enrich_as(entry, &crate::schema_union::ReadAs::default())
1646}
1647
1648/// As [`enrich`], reading each file the way the open that follows will read it.
1649///
1650/// The rule that decides whether a directory's files are one table reads the names at the
1651/// front of them, and where those names are is a reader setting. A pass that used its own
1652/// answers would judge a directory by a reading nobody is going to make — which is how
1653/// `datui --no-header directory/` came to open the home screen for a directory the flag
1654/// reads perfectly as one table.
1655pub fn enrich_as(entry: &mut Entry, as_read: &crate::schema_union::ReadAs) {
1656    enrich_with(entry, as_read, None)
1657}
1658
1659/// As [`enrich_as`], taking a dataset's measure from the shape an open kept of it in
1660/// `remembered`, where its files are as they were then, rather than from a sample of
1661/// its footers.
1662pub fn enrich_with(
1663    entry: &mut Entry,
1664    as_read: &crate::schema_union::ReadAs,
1665    remembered: Option<&crate::cache::CacheManager>,
1666) {
1667    match entry.kind {
1668        EntryKind::File => {
1669            enrich_parquet(entry);
1670            enrich_tables(entry);
1671            enrich_arrow(entry);
1672        }
1673        EntryKind::Hive | EntryKind::MultiFile => enrich_dataset(entry, as_read, remembered),
1674        // Nothing to read for a plain directory, and nothing that *may* be read for
1675        // one that has not been looked at. Nor for a lake table: summing the footers
1676        // under one counts tombstoned rows, every rewritten version and both sides of
1677        // a compaction, which is the whole reason it is not offered as a dataset.
1678        EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => {}
1679        EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => {}
1680    }
1681}
1682
1683/// Sum footers across a bounded set of Parquet files under `entry`.
1684fn enrich_dataset(
1685    entry: &mut Entry,
1686    as_read: &crate::schema_union::ReadAs,
1687    remembered: Option<&crate::cache::CacheManager>,
1688) {
1689    // A directory of JSON is not described by the Parquet under it. The walk below
1690    // recurses — it has to, because that is what opening the directory reads — so for a
1691    // directory whose own files are a format this cannot count, every number it produced
1692    // belonged to something the row does not name: `6 json` reported the sixty-one
1693    // columns of the Parquet in its subdirectories.
1694    //
1695    // A directory of Parquet with more Parquet beneath it is the opposite case and keeps
1696    // the walk. The counts are a promise about what `Enter` gives, and `Enter` reads
1697    // the subtree; measuring only the top would promise three files and open
1698    // twenty-three, and would ask `is_one_table` about three files while unioning all
1699    // twenty-three. The `holds` line names the directory that explains the difference.
1700
1701    // The partition layout comes from directory names, so it is knowable even for a
1702    // dataset far too large to count the rows of — which is exactly the dataset whose
1703    // shape you most want described before opening it.
1704    if entry.kind == EntryKind::Hive {
1705        entry.cost.partitions = partition_layout(&entry.path);
1706    }
1707
1708    // Whether the footers below are this directory's own shape, or something else's. A
1709    // directory's own format is counted exactly, so this is exact for one.
1710    //
1711    // Not asked of a hive root at all. Its own files are strays beside the partitions —
1712    // a `schema.json` or a `manifest.csv` left at the top — so its counted format is
1713    // not its data's, and one such file would blank the whole dataset. Its data is down
1714    // in the partitions, where the format can only be sampled, and one spine tells the
1715    // two cases apart in neither direction: a CSV tree with a stray `snapshot.parquet`
1716    // in the sampled partition and a Parquet tree with a stray `notes.csv` in it both
1717    // come back `NotOneTable`. A stray Parquet in a CSV tree is still counted as the
1718    // dataset's, which #275 phase 4 settles by making the tree readable in its own
1719    // format.
1720    let reads_as_parquet = entry.kind == EntryKind::Hive
1721        || match entry.holds.one_format() {
1722            // No single format to object with, so nothing to object. No row reaches
1723            // this today — a directory of more than one format is a `Directory` and
1724            // `enrich` leaves those alone — so it is a default, and the safe one:
1725            // leaving the counts off a directory is a mistake opening it undoes.
1726            None => true,
1727            // A name this build cannot read back is not Parquet as far as anything here
1728            // knows. Leaving the counts off a directory is the mistake that can be undone
1729            // by opening it; giving it another format's numbers is not.
1730            Some(name) => crate::FileFormat::from_name(name) == Some(crate::FileFormat::Parquet),
1731        };
1732    if !reads_as_parquet {
1733        entry.size = None;
1734        judge_by_names(entry, as_read);
1735        return;
1736    }
1737
1738    // The stat'ed size of a dataset directory is its own inode: a couple of hundred
1739    // bytes that have nothing to do with the terabyte inside it. Dropped up front and
1740    // restored only if the files are actually totalled, so no path out of here can
1741    // leave it behind to be read as an answer.
1742    entry.size = None;
1743
1744    let files = parquet_files_under(&entry.path);
1745    // Past the budget the footers are not read here, but an open that read them all
1746    // kept them: listing the dataset again says whether they still describe it.
1747    if files.len() > MAX_FOOTERS_PER_DATASET
1748        && let Some((listed, footers)) = remembered
1749            .and_then(|cache| crate::dataset_files::remembered_footers(&entry.path, cache))
1750    {
1751        measure_from_footers(entry, &listed, &footers);
1752        return;
1753    }
1754    if files.is_empty() || files.len() > MAX_FOOTERS_PER_DATASET {
1755        // Whether these are one table is still worth asking, and it does not need
1756        // every footer: three files spread across the directory answer it. Without this a
1757        // directory large enough to be past the counting limit would skip the check
1758        // entirely, which is backwards — the more tables it holds, the more a union of
1759        // them costs.
1760        let sampled = sample_footers(&files);
1761        let names: Vec<Vec<String>> = sampled.iter().map(column_names).collect();
1762        if entry.kind == EntryKind::MultiFile && !files_nest(&names) {
1763            // Whether the directory is one table is asked of everything under it, because
1764            // that is what opening it would union. What it *holds* is the files the
1765            // label counts — the ones directly inside — and a downgraded row is never
1766            // opened as one table, so a *count* spanning the subtree would be a width
1767            // nothing produces. Three more footers, on a directory being downgraded, to
1768            // say `2 parquet` and mean those two.
1769            let own_files = direct_children(&files, &entry.path);
1770            let own = sample_footers(&own_files);
1771            // The names, though, are every one sampled under it, the same as the arm
1772            // below: they are the home screen's search index, and a directory is found by
1773            // a column that looking inside it will reach. Narrowing these to the direct
1774            // children made a big directory unfindable by a column a small one is found
1775            // by.
1776            entry.columns = union_of(&names);
1777            // A floor only when a footer was left unread. The directory is past the
1778            // counting budget, but its *own* files may be three of the seventy — and
1779            // then `5+ cols` claims a sample that did not happen.
1780            entry.cols_sampled = own.len() < own_files.len();
1781            // The columns a reader sees, from the schema rather than by splitting leaf
1782            // paths on a dot: a column named `user.id` and a struct `user` with a field
1783            // `id` are not the same thing, and a string cannot tell them apart.
1784            let top = union_of(&own.iter().map(top_level_names).collect::<Vec<_>>());
1785            downgrade_to_directory(entry, (!top.is_empty()).then_some(top.len()));
1786            return;
1787        }
1788        // Still worth knowing the shape, even when the row count is out of reach.
1789        if let Some(meta) = sampled.first() {
1790            // Three files rather than the first, because a directory written over time
1791            // keeps its newest columns in its last file — and the first is where a
1792            // dataset that grew is narrowest. Still a sample and not a total: the
1793            // count beside it is already `?`.
1794            entry.columns = union_of(&names);
1795            let top = union_of(&sampled.iter().map(top_level_names).collect::<Vec<_>>());
1796            entry.cols = Some(top.len() + partition_columns_beyond(entry, &top));
1797            entry.cols_sampled = true;
1798            // From one file, so it describes how the dataset is written rather
1799            // than its total: codec and row-group sizing are a property of the
1800            // writer and are uniform in practice.
1801            physical_facts(meta, &mut entry.cost);
1802            entry.cost.uncompressed = None;
1803        }
1804        return;
1805    }
1806
1807    let mut rows = 0usize;
1808    let mut bytes = 0u64;
1809    // Every column any file has, in the order they first appear — not the first
1810    // file's. A dataset whose columns grew over time reported the shape it was born
1811    // with: Bitcoin transactions, whose `inputs` gained `address` and then
1812    // `txinwitness`, answered no to "which of these has `txinwitness`?".
1813    let mut columns: Vec<String> = Vec::new();
1814    let mut seen_columns = std::collections::HashSet::new();
1815    // The columns a reader sees, unioned the same way. Kept beside the leaves rather
1816    // than derived from them, because a leaf path cannot say whether its dots are
1817    // nesting or part of a name. See [`top_level_names`].
1818    let mut top_level: Vec<String> = Vec::new();
1819    let mut seen_top_level = std::collections::HashSet::new();
1820    let mut per_file: Vec<Vec<String>> = Vec::with_capacity(files.len());
1821    // The width and the size, restricted to the directory's own files. A directory the
1822    // footers downgrade is never opened as one table, so a *count* spanning the subtree
1823    // would be a width nothing produces — and the label beside it counts only what is
1824    // inside. The column names stay the subtree's: they are the search index, not the
1825    // label.
1826    let mut own_bytes = 0u64;
1827    let mut own_top_level: Vec<String> = Vec::new();
1828    let mut own_seen_top = std::collections::HashSet::new();
1829    let mut cost = Cost::default();
1830    let mut uncompressed = 0u64;
1831    let mut row_groups = 0usize;
1832    for file in &files {
1833        let Some(meta) = crate::parquet_footer::read_parquet_metadata(file) else {
1834            return; // A file we cannot read makes the total a guess; report nothing.
1835        };
1836        rows += meta.num_rows;
1837        let names = column_names(&meta);
1838        for name in &names {
1839            if seen_columns.insert(name.clone()) {
1840                columns.push(name.clone());
1841            }
1842        }
1843        for name in top_level_names(&meta) {
1844            if seen_top_level.insert(name.clone()) {
1845                top_level.push(name);
1846            }
1847        }
1848        // The columns a reader sees, not the leaves the footer names: see
1849        // [`crate::schema_union::top_level_columns`].
1850        per_file.push(crate::schema_union::top_level_columns(&names));
1851        // And the same again for this directory's own files, which is what a downgraded
1852        // row is labelled from: `2 parquet` must mean those two.
1853        // One stat, feeding both totals: on a share each is a round trip, and a directory
1854        // of sixty-four files directly inside would have paid twice for every one.
1855        let file_bytes = std::fs::metadata(file).map(|m| m.len()).unwrap_or(0);
1856        bytes += file_bytes;
1857        if file.parent() == Some(entry.path.as_path()) {
1858            own_bytes += file_bytes;
1859            for name in top_level_names(&meta) {
1860                if own_seen_top.insert(name.clone()) {
1861                    own_top_level.push(name);
1862                }
1863            }
1864        }
1865        let mut per_file = Cost::default();
1866        physical_facts(&meta, &mut per_file);
1867        uncompressed += per_file.uncompressed.unwrap_or(0);
1868        row_groups += per_file.row_groups.unwrap_or(0);
1869        if cost.codec.is_none() {
1870            cost.codec = per_file.codec;
1871        }
1872    }
1873    // The footers are read by now, so whether these files are one table is known
1874    // rather than guessed. A directory of separate tables is a place to look inside: its
1875    // row count is the sum of unrelated things, its column count belongs to whichever
1876    // file happened to be read first, and opening it unions tables that share nothing.
1877    //
1878    // Only `multi` is reconsidered. A `key=value` layout says what the writer meant,
1879    // and a hive directory's files hold the same table by construction.
1880    if entry.kind == EntryKind::MultiFile && !crate::schema_union::is_nested(&per_file) {
1881        // Its own files' bytes, not the subtree's. The label counts what is directly
1882        // inside and so do the columns beside it; a size summed over a different set of
1883        // files is a third number on one row measured against neither of the other two.
1884        entry.size = Some(own_bytes);
1885        // Nothing here is one table's shape, but the names are what the directory holds,
1886        // and searching the home screen by column should still find the directory that
1887        // has one. The count is the directory's own files, which is what the label names.
1888        // The column *names* are every one under it: they are the home screen's search
1889        // index, and "which of these has a `txinwitness`?" is answered by the directory
1890        // that has one anywhere, which is where looking inside will find it.
1891        entry.columns = columns;
1892        entry.cols_sampled = false;
1893        downgrade_to_directory(
1894            entry,
1895            (!own_top_level.is_empty()).then_some(own_top_level.len()),
1896        );
1897        return;
1898    }
1899
1900    entry.rows = Some(rows);
1901    entry.cols = Some(top_level.len() + partition_columns_beyond(entry, &top_level));
1902    entry.size = Some(bytes);
1903    entry.columns = columns;
1904    cost.uncompressed = (uncompressed > 0).then_some(uncompressed);
1905    cost.row_groups = (row_groups > 0).then_some(row_groups);
1906    cost.partitions = entry.cost.partitions.take();
1907    entry.cost = cost;
1908}
1909
1910/// Partition keys the files do not carry themselves. The open hoists them in as
1911/// columns, so a hive table's width counts them: `12 × 4`, not the `12 × 2` its footers
1912/// say.
1913fn partition_columns_beyond(entry: &Entry, top_level: &[String]) -> usize {
1914    entry.cost.partitions.as_ref().map_or(0, |layout| {
1915        layout
1916            .keys
1917            .iter()
1918            .filter(|key| !top_level.contains(key))
1919            .count()
1920    })
1921}
1922
1923/// The files of `dir` itself, out of a walk that went below it.
1924fn direct_children(files: &[PathBuf], dir: &Path) -> Vec<PathBuf> {
1925    files
1926        .iter()
1927        .filter(|f| f.parent() == Some(dir))
1928        .cloned()
1929        .collect()
1930}
1931
1932/// The footers at the ends and the middle of a directory too large to read every one of.
1933///
1934/// The ends and the middle, because keys and filenames sort: a directory written table by
1935/// table can easily start with several files of the same table, so its head answers
1936/// nothing. The last file earns its place twice over — in a directory written over time
1937/// it is the newest, which is where a column added last year is.
1938fn sample_footers(files: &[PathBuf]) -> Vec<crate::parquet_footer::Footer> {
1939    if files.is_empty() {
1940        return Vec::new();
1941    }
1942    let mut picks = vec![0, files.len() / 2, files.len() - 1];
1943    picks.dedup();
1944    picks
1945        .iter()
1946        .filter_map(|i| files.get(*i))
1947        .filter_map(|file| crate::parquet_footer::read_parquet_metadata(file))
1948        .collect()
1949}
1950
1951/// Whether a spread of a directory's files agree on a schema.
1952///
1953/// Fewer than two readable footers decide nothing, and the directory keeps the kind its
1954/// names suggested.
1955fn files_nest(sampled: &[Vec<String>]) -> bool {
1956    let per_file: Vec<Vec<String>> = sampled
1957        .iter()
1958        .map(|names| crate::schema_union::top_level_columns(names))
1959        .collect();
1960    per_file.len() < 2 || crate::schema_union::is_nested(&per_file)
1961}
1962
1963/// Ask a directory with no footers whether its files are one table, by the names at the
1964/// front of them.
1965///
1966/// The same rule as [`files_nest`] on the same evidence — the column names — from the
1967/// only place a CSV or an NDJSON file keeps them. Without this a directory of forty
1968/// unrelated CSVs was labelled `40 csv`, `Enter` promised one table because nothing had
1969/// looked, and the read then refused it: the permissive rule with the strict reader,
1970/// which is the pairing #275 exists to stop. Parquet has had the test since phase 3;
1971/// this is the rest of the formats catching up.
1972///
1973/// Silence is optimism, as it is for an unreadable footer: too few files, a format whose
1974/// schema costs a whole read, or a file that would not parse all leave the directory as
1975/// its names suggested. That is only safe because the read behind it unions by name and
1976/// widens types rather than failing — see `DataTableState::union_of_files`.
1977fn judge_by_names(entry: &mut Entry, as_read: &crate::schema_union::ReadAs) {
1978    if entry.kind != EntryKind::MultiFile {
1979        return;
1980    }
1981    let Some(format) = entry
1982        .holds
1983        .one_format()
1984        .and_then(crate::FileFormat::from_name)
1985    else {
1986        return;
1987    };
1988    // The directory's own files, which is what the label counts and what the open reads.
1989    // A `MultiFile` directory is flat by construction — a `key=value` below it would have
1990    // made it `Hive` — so there is no subtree to walk for these.
1991    //
1992    // `One` and nothing else. `one_format` above already returned for a directory of more
1993    // than one format, and `look_at_directory` only calls a directory `MultiFile` when
1994    // its formats agree, so `Mixed` cannot arrive here — matching it as well read as
1995    // coverage this does not have. A directory of forty disjoint CSVs beside one stray
1996    // `.json` is a `Directory` before it reaches this, and goes inside for that reason
1997    // rather than for this one.
1998    let DirectoryFormat::One(_, files) = directory_format(&entry.path) else {
1999        return;
2000    };
2001    // Read the way the open that follows will read it: where the header is decides
2002    // what these names are, and a verdict reached by another reading is about a directory
2003    // nobody is going to open.
2004    let sampled = crate::schema_union::sample_files(&files, format, as_read);
2005    if sampled.nests == Some(false) {
2006        // The columns the sample found, so searching the home screen by column still
2007        // finds the directory that has one — the same thing the Parquet path keeps when
2008        // it downgrades. From the spread that was read rather than from every file: a
2009        // directory of forty thousand CSVs must cost what a directory of four costs, and
2010        // this runs on the thread that opens a path named on the command line.
2011        // `cols_sampled` is what says the count is a floor.
2012        let cols = (!sampled.columns.is_empty()).then_some(sampled.columns.len());
2013        // A floor only when there were files the sample did not open. A directory of two
2014        // or three had every one read, and `N+ cols` on that row claims a hedge the
2015        // count does not need — the Parquet path next door works this out the same way.
2016        entry.cols_sampled = sampled.read < files.len();
2017        entry.columns = sampled.columns;
2018        downgrade_to_directory(entry, cols);
2019    }
2020}
2021
2022/// A directory whose files turned out to be separate tables is a place to look inside.
2023///
2024/// Its row count would be the sum of unrelated things, so it is not reported. The column
2025/// count is: the union of what the directory's files hold is a true answer to "what is in
2026/// here" even when "how many rows" has none, so a directory of fifteen tables reads
2027/// `15 parquet · 72 columns` and no row count. Passed in rather than derived from
2028/// `columns`, which names leaves: see [`top_level_names`] for why a leaf path cannot be
2029/// split back into the columns a reader sees.
2030fn downgrade_to_directory(entry: &mut Entry, cols: Option<usize>) {
2031    entry.kind = EntryKind::Directory;
2032    entry.rows = None;
2033    entry.cols = cols;
2034    entry.cost = Cost {
2035        partitions: entry.cost.partitions.take(),
2036        ..Cost::default()
2037    };
2038}
2039
2040/// The columns a reader sees: the schema's own top-level fields.
2041///
2042/// Not the leaves a footer names, and not those leaves split on a dot either. Leaves
2043/// counted directly double for a directory whose writer changed — the same nested column
2044/// written by parquet-mr and by Arrow gives `inputs.list.element.address` in one file and
2045/// `inputs.bag.array_element.address` in the other, and a union of leaf paths holds both.
2046/// Splitting the dotted path fixes that and breaks something else: a column literally
2047/// named `user.id` is one column, and so is a struct `user` with a field `id`, and the
2048/// string cannot tell them apart.
2049///
2050/// The schema knows. `fields()` is the root's own children, which is what a struct counts
2051/// as here, what `schema_preview` lists in the details pane, and what the table shows.
2052fn top_level_names(meta: &crate::parquet_footer::Footer) -> Vec<String> {
2053    meta.schema_descr
2054        .fields()
2055        .iter()
2056        .map(|field| field.name().to_string())
2057        .collect()
2058}
2059
2060/// Every column name any of the files has, in the order they first appear.
2061fn union_of(per_file: &[Vec<String>]) -> Vec<String> {
2062    let mut seen = std::collections::HashSet::new();
2063    per_file
2064        .iter()
2065        .flatten()
2066        .filter(|name| seen.insert(name.as_str()))
2067        .cloned()
2068        .collect()
2069}
2070
2071/// The Parquet files under `dir`, sorted, as far down as a dataset goes: one past the
2072/// budget when there are more, which is what says there are too many to count. The
2073/// open's own walk, stopped once it has seen enough.
2074fn parquet_files_under(dir: &Path) -> Vec<PathBuf> {
2075    let mut files = crate::dataset_files::LocalFiles::new(dir)
2076        .first_files(MAX_WALK_DEPTH as usize + 1, MAX_FOOTERS_PER_DATASET);
2077    // The walk takes a directory entry's own type; a file this reads must be one.
2078    files.retain(|p| is_regular_file(p));
2079    files
2080}
2081
2082/// Measure a dataset from every file's footer as an open kept them: its rows, width,
2083/// size and row groups, and whether its files are one table, with none read here.
2084fn measure_from_footers(
2085    entry: &mut Entry,
2086    files: &[crate::dataset_files::DatasetFile],
2087    footers: &[Option<crate::schema_union::FileFooter>],
2088) {
2089    let per_file: Vec<Vec<String>> = footers
2090        .iter()
2091        .flatten()
2092        .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
2093        .collect();
2094    let columns = union_of(&per_file);
2095    if entry.kind == EntryKind::MultiFile && !crate::schema_union::is_nested(&per_file) {
2096        // As a read of every footer judges it: the label counts the directory's own
2097        // files, and so do the width and the size beside it.
2098        let own: Vec<usize> = files
2099            .iter()
2100            .enumerate()
2101            .filter(|(_, f)| Path::new(&f.key).parent() == Some(entry.path.as_path()))
2102            .map(|(i, _)| i)
2103            .collect();
2104        entry.size = Some(own.iter().map(|&i| files[i].size).sum());
2105        let own_columns = union_of(
2106            &own.iter()
2107                .filter_map(|&i| footers[i].as_ref())
2108                .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
2109                .collect::<Vec<_>>(),
2110        );
2111        entry.columns = columns;
2112        entry.cols_sampled = false;
2113        downgrade_to_directory(
2114            entry,
2115            (!own_columns.is_empty()).then_some(own_columns.len()),
2116        );
2117        return;
2118    }
2119    let footers: Vec<&crate::schema_union::FileFooter> = footers.iter().flatten().collect();
2120    let uncompressed: u64 = footers
2121        .iter()
2122        .flat_map(|f| &f.column_bytes)
2123        .map(|(_, bytes)| *bytes as u64)
2124        .sum();
2125    let row_groups: usize = footers.iter().map(|f| f.row_group_rows.len()).sum();
2126    entry.rows = Some(footers.iter().map(|f| f.rows()).sum());
2127    entry.cols = Some(columns.len() + partition_columns_beyond(entry, &columns));
2128    entry.size = Some(files.iter().map(|f| f.size).sum());
2129    entry.columns = columns;
2130    entry.cols_sampled = false;
2131    entry.cost = Cost {
2132        uncompressed: (uncompressed > 0).then_some(uncompressed),
2133        row_groups: (row_groups > 0).then_some(row_groups),
2134        partitions: entry.cost.partitions.take(),
2135        ..Cost::default()
2136    };
2137}
2138
2139/// Fill in row and column counts for a Parquet file from its footer.
2140///
2141/// Free in the sense that matters: no column data is read. Non-Parquet formats have
2142/// no equivalent — a CSV's row count cannot be known without scanning it — so those
2143/// entries keep `None`, and the UI shows the absence honestly rather than guessing.
2144pub fn enrich_parquet(entry: &mut Entry) {
2145    if entry.kind != EntryKind::File {
2146        return;
2147    }
2148    if !is_parquet_path(&entry.path) {
2149        return;
2150    }
2151    if !is_regular_file(&entry.path) {
2152        return;
2153    }
2154    if let Some(meta) = crate::parquet_footer::read_parquet_metadata(&entry.path) {
2155        entry.rows = Some(meta.num_rows);
2156        entry.columns = column_names(&meta);
2157        // The columns a reader sees, as a directory's row reports them: `schema_descr`
2158        // names the leaves, so a file with one struct of three fields counted four and
2159        // then listed two in the pane beside it. See [`top_level_names`].
2160        entry.cols = Some(top_level_names(&meta).len());
2161        physical_facts(&meta, &mut entry.cost);
2162    }
2163}
2164
2165/// A file of tables' tables (a SQLite database's schema, a NumPy archive's directory):
2166/// how many of its own, whether Enter opens one of them, and the columns of the one when
2167/// there is one. A file whose name says a format its bytes must say (a `.db` file that
2168/// is not SQLite) is one datui cannot open.
2169pub fn enrich_tables(entry: &mut Entry) {
2170    if entry.kind != EntryKind::File || entry.table.is_some() {
2171        return;
2172    }
2173    let named = data_format(&entry.path);
2174    if !is_regular_file(&entry.path) {
2175        return;
2176    }
2177    let Some(format) = crate::members::holder(&entry.path) else {
2178        if named.is_some_and(|f| f.holds_tables() && crate::readers::of(f).bytes_decide) {
2179            entry.kind = EntryKind::Other;
2180        }
2181        return;
2182    };
2183    let Ok(tables) = crate::members::tables(&entry.path, format) else {
2184        return;
2185    };
2186    let own: Vec<&crate::sqlite::Table> = tables.iter().filter(|t| !t.internal).collect();
2187    entry.cost.tables = Some(own.len());
2188    entry.cost.opens_one = format.opens_one_table();
2189    if let [one] = own.as_slice()
2190        && !one.columns.is_empty()
2191    {
2192        entry.columns = one.columns.iter().map(|(name, _)| name.clone()).collect();
2193        entry.cols = Some(entry.columns.len());
2194    }
2195}
2196
2197/// The rows of a file of tables' listing on the home screen: a database's tables and
2198/// views, or an archive's arrays, by name, as a directory lists its files, SQLite's own
2199/// marked to be hidden, each at its path inside the file.
2200pub fn database_rows(file: &Path) -> Vec<Entry> {
2201    let Some(format) = crate::members::holder(file) else {
2202        return Vec::new();
2203    };
2204    let Ok(mut tables) = crate::members::tables(file, format) else {
2205        return Vec::new();
2206    };
2207    // A database's tables by name; an archive's arrays in the order they were saved.
2208    if format
2209        .descriptor()
2210        .tables
2211        .as_ref()
2212        .is_some_and(|t| t.by_name)
2213    {
2214        tables.sort_by_cached_key(|t| t.name.to_lowercase());
2215    }
2216    let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
2217    tables
2218        .into_iter()
2219        .map(|table| table_entry(file, format, table, modified))
2220        .collect()
2221}
2222
2223/// The rows of a Hugging Face cache directory's splits, each at its path inside the
2224/// directory (`cache/test`): opened, it is the directory read with `--table`. Empty
2225/// for any other directory, and for a cache of one split, which its door opens.
2226pub fn split_rows(dir: &Path) -> Vec<Entry> {
2227    let splits = crate::hf_splits::cache_splits(dir);
2228    if splits.len() < 2 {
2229        return Vec::new();
2230    }
2231    splits
2232        .into_iter()
2233        .map(|split| split_entry(dir, split))
2234        .collect()
2235}
2236
2237/// The row of a split named by its path inside its cache directory, as a recent is
2238/// listed: `None` when the path names no split of one.
2239pub fn split_row(path: &Path) -> Option<Entry> {
2240    let (dir, split) = crate::hf_splits::split_place(path)?;
2241    Some(split_entry(&dir, split))
2242}
2243
2244fn split_entry(dir: &Path, split: String) -> Entry {
2245    let mut entry = Entry::new(dir.join(&split), EntryKind::File);
2246    entry.name = split;
2247    entry.table = Some(TableOf {
2248        format: Some(crate::FileFormat::Arrow),
2249        kind: "split".to_string(),
2250        internal: false,
2251    });
2252    entry
2253}
2254
2255/// The rows of a file a format spec reads as several variants, one a variant, each at
2256/// its path inside the file (`day.itch/add`): opened, it is the file read with
2257/// `--table`. Empty for any other file.
2258pub fn variant_rows(file: &Path, formats: &crate::formats::Registry) -> Vec<Entry> {
2259    let Some((spec, tables)) = crate::members::variants(file, formats) else {
2260        return Vec::new();
2261    };
2262    let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
2263    tables
2264        .into_iter()
2265        .map(|table| variant_entry(file, &spec, table, modified))
2266        .collect()
2267}
2268
2269/// The row of a variant named by its path inside its file (`day.itch/add`), as a
2270/// recent is listed: `None` when the path names no variant of such a file.
2271pub fn variant_row(path: &Path, formats: &crate::formats::Registry) -> Option<Entry> {
2272    let (file, name) = crate::members::split_variant(path, formats)?;
2273    let (spec, tables) = crate::members::variants(&file, formats)?;
2274    let table = tables.into_iter().find(|t| t.name == name)?;
2275    let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
2276    let mut entry = variant_entry(&file, &spec, table, modified);
2277    entry.path = path.to_path_buf();
2278    Some(entry)
2279}
2280
2281fn variant_entry(
2282    file: &Path,
2283    spec: &str,
2284    table: crate::sqlite::Table,
2285    modified: Option<std::time::SystemTime>,
2286) -> Entry {
2287    let mut entry = Entry::new(crate::members::place(file, &table.name), EntryKind::File);
2288    entry.name = table.name;
2289    entry.modified = modified;
2290    entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
2291    entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
2292    entry.format_spec = Some(spec.to_string());
2293    entry.table = Some(TableOf {
2294        format: None,
2295        kind: table.kind,
2296        internal: false,
2297    });
2298    entry
2299}
2300
2301/// The row of a table inside a file of tables named by its path (`app.db/users`), as a
2302/// recent is listed: `None` when the path names no table of such a file.
2303pub fn table_row(path: &Path) -> Option<Entry> {
2304    let (file, name) = crate::members::split(path)?;
2305    let format = crate::members::holder(&file)?;
2306    let table = crate::members::tables(&file, format)
2307        .ok()?
2308        .into_iter()
2309        .find(|t| t.name == name)?;
2310    let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
2311    let mut entry = table_entry(&file, format, table, modified);
2312    entry.path = path.to_path_buf();
2313    Some(entry)
2314}
2315
2316fn table_entry(
2317    file: &Path,
2318    format: crate::FileFormat,
2319    table: crate::sqlite::Table,
2320    modified: Option<std::time::SystemTime>,
2321) -> Entry {
2322    let mut entry = Entry::new(crate::members::place(file, &table.name), EntryKind::File);
2323    entry.name = table.name;
2324    entry.modified = modified;
2325    entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
2326    entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
2327    entry.table = Some(TableOf {
2328        format: Some(format),
2329        kind: table.kind,
2330        internal: table.internal,
2331    });
2332    entry
2333}
2334
2335/// The first bytes of `path`, as many as fit in `buf`. A short read is the whole file.
2336fn read_head<'a>(path: &Path, buf: &'a mut [u8]) -> Option<&'a [u8]> {
2337    use std::io::Read;
2338    let mut file = std::fs::File::open(path).ok()?;
2339    let mut filled = 0;
2340    loop {
2341        match file.read(&mut buf[filled..]) {
2342            Ok(0) => break,
2343            Ok(n) => filled += n,
2344            Err(_) => return None,
2345        }
2346        if filled == buf.len() {
2347            break;
2348        }
2349    }
2350    Some(&buf[..filled])
2351}
2352
2353/// Whether an Arrow file is an IPC stream: an IPC file starts `ARROW1`, a stream with
2354/// its schema message. Eight bytes, so a listing can say which will be converted.
2355fn enrich_arrow(entry: &mut Entry) {
2356    if entry.kind != EntryKind::File
2357        || data_format(&entry.path) != Some(crate::FileFormat::Arrow)
2358        || crate::CompressionFormat::from_extension(&entry.path).is_some()
2359        || !is_regular_file(&entry.path)
2360    {
2361        return;
2362    }
2363    let mut head = [0u8; 8];
2364    if let Some(head) = read_head(&entry.path, &mut head) {
2365        entry.cost.ipc_stream = !head.starts_with(b"ARROW1");
2366    }
2367}
2368
2369/// Pull layout and compression out of a footer that has already been read.
2370///
2371/// Every one of these was being parsed and thrown away. They are the difference
2372/// between knowing how big a file is and knowing what reading it will do.
2373pub fn physical_facts(meta: &crate::parquet_footer::Footer, cost: &mut Cost) {
2374    if meta.row_groups.is_empty() {
2375        return;
2376    }
2377    cost.row_groups = Some(meta.row_groups.len());
2378
2379    let mut uncompressed: u64 = 0;
2380    let mut codecs: Vec<String> = Vec::new();
2381    for rg in &meta.row_groups {
2382        uncompressed = uncompressed.saturating_add(rg.total_byte_size() as u64);
2383        for cc in rg.parquet_columns() {
2384            let codec = format!("{:?}", cc.compression()).to_lowercase();
2385            if !codecs.contains(&codec) {
2386                codecs.push(codec);
2387            }
2388        }
2389    }
2390    if uncompressed > 0 {
2391        cost.uncompressed = Some(uncompressed);
2392    }
2393    // A file usually uses one codec throughout. When it does not, say so rather than
2394    // picking one and implying uniformity that is not there.
2395    cost.codec = match codecs.len() {
2396        0 => None,
2397        1 => Some(codecs.remove(0)),
2398        n => Some(format!("mixed ({n})")),
2399    };
2400}
2401
2402/// Outermost directories to look at when describing a hive dataset's partitioning.
2403///
2404/// Enough to name the keys and show the shape of the first one; bounded because a
2405/// dataset partitioned by day over a decade has thousands, and counting all of them
2406/// to print "3,653" is not worth a second of anyone's time on a network share.
2407const MAX_PARTITION_DIRS: usize = 512;
2408
2409/// Describe how a hive dataset is partitioned, from directory names alone.
2410pub fn partition_layout(dir: &Path) -> Option<Partitions> {
2411    let iter = std::fs::read_dir(dir).ok()?;
2412    let mut values: Vec<String> = Vec::new();
2413    let mut keys: Vec<String> = Vec::new();
2414    let mut count = 0usize;
2415    let mut more = false;
2416
2417    for entry in iter.flatten() {
2418        if count >= MAX_PARTITION_DIRS {
2419            more = true;
2420            break;
2421        }
2422        let name = entry.file_name().to_string_lossy().into_owned();
2423        let Some((key, value)) = name.split_once('=') else {
2424            continue;
2425        };
2426        if !entry.path().is_dir() {
2427            continue;
2428        }
2429        if keys.is_empty() {
2430            keys.push(key.to_string());
2431            // Only the first partition directory is descended into, for the nested
2432            // key names. One is representative, and a hive dataset that disagrees
2433            // with itself about its own schema is not a dataset datui can help with.
2434            keys.extend(nested_keys(&entry.path()));
2435        }
2436        values.push(value.to_string());
2437        count += 1;
2438    }
2439
2440    if keys.is_empty() {
2441        return None;
2442    }
2443    values.sort();
2444    values.dedup();
2445    Some(Partitions {
2446        keys,
2447        first_key_values: values,
2448        count,
2449        more,
2450    })
2451}
2452
2453/// Partition keys below `dir`, following the first child at each level.
2454fn nested_keys(dir: &Path) -> Vec<String> {
2455    let mut keys = Vec::new();
2456    let mut current = dir.to_path_buf();
2457    // Bounded: a hive path deeper than this is pathological, and each level costs a
2458    // directory read.
2459    for _ in 0..6 {
2460        let Ok(iter) = std::fs::read_dir(&current) else {
2461            break;
2462        };
2463        let Some(child) = iter
2464            .flatten()
2465            .find(|e| e.file_name().to_string_lossy().contains('=') && e.path().is_dir())
2466        else {
2467            break;
2468        };
2469        let name = child.file_name().to_string_lossy().into_owned();
2470        let Some((key, _)) = name.split_once('=') else {
2471            break;
2472        };
2473        keys.push(key.to_string());
2474        current = child.path();
2475    }
2476    keys
2477}
2478
2479/// Render a byte count compactly for a listing (`340 MB`).
2480pub fn format_size(bytes: u64) -> String {
2481    const UNITS: &[&str] = &["B", "KB", "MB", "GB", "TB"];
2482    let mut value = bytes as f64;
2483    let mut unit = 0;
2484    while value >= 1024.0 && unit < UNITS.len() - 1 {
2485        value /= 1024.0;
2486        unit += 1;
2487    }
2488    if unit == 0 {
2489        format!("{} {}", bytes, UNITS[0])
2490    } else if value >= 100.0 {
2491        format!("{:.0} {}", value, UNITS[unit])
2492    } else {
2493        format!("{:.1} {}", value, UNITS[unit])
2494    }
2495}
2496
2497/// Render a row count compactly (`2.4M`).
2498pub fn format_rows(rows: usize) -> String {
2499    let r = rows as f64;
2500    if rows >= 1_000_000_000 {
2501        format!("{:.1}B", r / 1e9)
2502    } else if rows >= 1_000_000 {
2503        format!("{:.1}M", r / 1e6)
2504    } else if rows >= 10_000 {
2505        format!("{:.0}k", r / 1e3)
2506    } else if rows >= 1_000 {
2507        // Below ten thousand the exact count fits and rounding actively misleads:
2508        // 3,653 daily observations is ten years of data, and "4k" is not.
2509        let mut out = String::new();
2510        let digits = rows.to_string();
2511        for (i, c) in digits.chars().enumerate() {
2512            if i > 0 && (digits.len() - i).is_multiple_of(3) {
2513                out.push(',');
2514            }
2515            out.push(c);
2516        }
2517        out
2518    } else {
2519        rows.to_string()
2520    }
2521}
2522
2523/// Render "how long ago" compactly (`2d`, `3h`).
2524pub fn format_age(t: std::time::SystemTime) -> String {
2525    let Ok(elapsed) = t.elapsed() else {
2526        return String::new();
2527    };
2528    let secs = elapsed.as_secs();
2529    if secs < 60 {
2530        "now".to_string()
2531    } else if secs < 3600 {
2532        format!("{}m", secs / 60)
2533    } else if secs < 86_400 {
2534        format!("{}h", secs / 3600)
2535    } else if secs < 86_400 * 365 {
2536        format!("{}d", secs / 86_400)
2537    } else {
2538        format!("{}y", secs / (86_400 * 365))
2539    }
2540}
2541
2542/// Column name and type, for the home screen's preview pane.
2543pub type SchemaPreview = Vec<(String, polars::prelude::DataType)>;
2544
2545/// The preview of a table of a file of tables: a row inside a database or an archive, or
2546/// a file of tables (or a NumPy array file) as it opens. `None` when the entry is none of
2547/// these, `Some(None)` when it is and has nothing to show.
2548fn table_preview(entry: &Entry) -> Option<Option<SchemaPreview>> {
2549    let (file, format, name) = match &entry.table {
2550        Some(table) => match (crate::members::split(&entry.path), table.format) {
2551            (Some((file, _)), Some(format)) => (file, format, Some(entry.name.as_str())),
2552            _ => return Some(None),
2553        },
2554        None if is_regular_file(&entry.path) => {
2555            let format = crate::members::holder(&entry.path).or_else(|| {
2556                data_format(&entry.path)
2557                    .filter(|f| f.holds_tables() && crate::readers::of(*f).table_schema.is_some())
2558            })?;
2559            (entry.path.clone(), format, None)
2560        }
2561        None => return None,
2562    };
2563    Some(
2564        crate::readers::of(format)
2565            .table_schema
2566            .and_then(|schema| schema(&file, name)),
2567    )
2568}
2569
2570/// Find the first Parquet file at or under `dir`, without walking the whole tree.
2571///
2572/// Bounded on both breadth and depth so a hive dataset with thousands of partitions
2573/// costs the same as one with three.
2574fn first_parquet_under(dir: &Path, depth: u8) -> Option<PathBuf> {
2575    if depth > MAX_WALK_DEPTH {
2576        return None;
2577    }
2578    let mut subdirs = Vec::new();
2579    for entry in std::fs::read_dir(dir).ok()?.flatten().take(64) {
2580        let path = entry.path();
2581        if path.is_dir() {
2582            subdirs.push(path);
2583        } else if is_parquet_path(&path) && is_regular_file(&path) {
2584            return Some(path);
2585        }
2586    }
2587    subdirs.sort();
2588    subdirs
2589        .into_iter()
2590        .take(4)
2591        .find_map(|d| first_parquet_under(&d, depth + 1))
2592}
2593
2594/// Column names from a Parquet footer.
2595pub fn column_names(meta: &crate::parquet_footer::Footer) -> Vec<String> {
2596    meta.schema_descr
2597        .columns()
2598        .iter()
2599        .map(|c| c.path_in_schema.join("."))
2600        .collect()
2601}
2602
2603/// Whether `path` is a regular file that is safe to open.
2604///
2605/// Opening a FIFO blocks until a writer appears — indefinitely, for a named pipe
2606/// nobody is writing to — and opening a device or a socket does something stranger
2607/// still. A directory listing happily reports any of these with a `.parquet` name,
2608/// so every read here is gated on the kind first. `symlink_metadata` follows nothing
2609/// and `metadata` only stats, so neither can block the way an open can.
2610fn is_regular_file(path: &Path) -> bool {
2611    std::fs::metadata(path)
2612        .map(|m| m.file_type().is_file())
2613        .unwrap_or(false)
2614}
2615
2616/// Read a dataset's column names and types without reading any data.
2617///
2618/// Parquet only — a CSV's schema cannot be known without scanning it, and doing that
2619/// for every row the cursor passes over would defeat the point of a preview. Returns
2620/// `None` for anything else, and the UI says so rather than guessing.
2621pub fn schema_preview(entry: &Entry) -> Option<SchemaPreview> {
2622    // `SerReader` is what brings `ParquetReader::new` into scope.
2623    use polars::prelude::{ParquetReader, Schema, SchemaExt, SerReader};
2624
2625    let file_path = match entry.kind {
2626        EntryKind::File => {
2627            if let Some(preview) = table_preview(entry) {
2628                return preview;
2629            }
2630            if !is_parquet_path(&entry.path) {
2631                return None;
2632            }
2633            entry.path.clone()
2634        }
2635        EntryKind::Hive | EntryKind::MultiFile => first_parquet_under(&entry.path, 0)?,
2636        EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => return None,
2637        // One data file's schema is not the table's: Iceberg field IDs and Delta
2638        // column mapping both mean a renamed column reads as two.
2639        EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => return None,
2640    };
2641
2642    if !is_regular_file(&file_path) {
2643        return None;
2644    }
2645    let file = std::fs::File::open(&file_path).ok()?;
2646    let mut reader = ParquetReader::new(file);
2647    let arrow_schema = reader.schema().ok()?;
2648    let schema = Schema::from_arrow_schema(arrow_schema.as_ref());
2649    let mut preview: SchemaPreview = Vec::new();
2650    // A hive table opens with its partition keys hoisted to the front, so the pane lists
2651    // them there too, typed from the one path already in hand the way the scan infers
2652    // them.
2653    if entry.kind == EntryKind::Hive
2654        && let Ok(below) = file_path.strip_prefix(&entry.path)
2655    {
2656        for part in below.parent().into_iter().flat_map(Path::components) {
2657            let part = part.as_os_str().to_string_lossy();
2658            if let Some((key, value)) = part.split_once('=')
2659                && !key.is_empty()
2660                && schema.get(key).is_none()
2661            {
2662                // The scan's own inference, dates and booleans included.
2663                let dtype = if value.is_empty() || value == "__HIVE_DEFAULT_PARTITION__" {
2664                    polars::prelude::DataType::String
2665                } else {
2666                    polars::io::csv::read::schema_inference::infer_field_schema(value, true, false)
2667                };
2668                preview.push((key.to_string(), dtype));
2669            }
2670        }
2671    }
2672    preview.extend(
2673        schema
2674            .iter()
2675            .map(|(name, dtype)| (name.to_string(), dtype.clone())),
2676    );
2677    Some(preview)
2678}
2679
2680#[cfg(test)]
2681mod classification_tests {
2682    use super::*;
2683    use polars::prelude::*;
2684
2685    /// A file row's read follows its format and how it is stored, and a remote file
2686    /// other than a Parquet object or a model file is downloaded first. Directories say nothing.
2687    #[test]
2688    fn how_a_row_is_read() {
2689        use crate::ReadMode::*;
2690        let how = |path: &str| how_read(&Entry::for_test(Path::new(path), path));
2691        let at = |path: &str, mode, download| {
2692            assert_eq!(how(path), Some(HowRead { mode, download }), "{path}");
2693        };
2694        at("/d/a.parquet", Lazy, false);
2695        at("/d/a.csv", Lazy, false);
2696        at("/d/a.csv.gz", Decompressed, false);
2697        at("/d/a.json", InMemory, false);
2698        at("/d/a.gpx", Converted, false);
2699        at("/d/a.arrow", Lazy, false);
2700        at("s3://b/a.parquet", Lazy, false);
2701        at("s3://b/a.csv", Lazy, true);
2702        at("gs://b/a.json", InMemory, true);
2703        at("https://example.com/a.parquet", Lazy, true);
2704        at("s3://b/m.safetensors", InMemory, false);
2705        at("https://example.com/m.gguf", InMemory, false);
2706        assert_eq!(how("/d/a.parquet.gz"), None, "does not open");
2707        assert_eq!(how("/d/README"), None);
2708
2709        let mut stream = Entry::for_test(Path::new("/d/x.arrow"), "x.arrow");
2710        stream.cost.ipc_stream = true;
2711        assert_eq!(how_read(&stream).map(|h| h.mode), Some(Converted));
2712        let mut spec = Entry::for_test(Path::new("/d/day.l2.zst"), "day.l2.zst");
2713        spec.format_spec = Some("acme.l2feed".into());
2714        assert_eq!(how_read(&spec).map(|h| h.mode), Some(Decompressed));
2715        at("/d/shop.db", Lazy, false);
2716        at("s3://b/shop.sqlite", Lazy, true);
2717        let mut table = Entry::for_test(Path::new("/d/shop.db/orders"), "orders");
2718        table.table = Some(TableOf {
2719            format: Some(crate::FileFormat::Sqlite),
2720            kind: "table".into(),
2721            internal: false,
2722        });
2723        assert_eq!(how_read(&table).map(|h| h.mode), Some(Lazy));
2724        assert_eq!(how_read(&Entry::directory(Path::new("/d/x"))), None);
2725    }
2726
2727    /// An Arrow file is told a stream by its first bytes when it is measured.
2728    #[test]
2729    fn measuring_an_arrow_file_tells_a_stream() {
2730        let dir = tempfile::tempdir().unwrap();
2731        let file = dir.path().join("file.arrow");
2732        std::fs::write(&file, b"ARROW1\0\0rest").unwrap();
2733        let stream = dir.path().join("stream.arrow");
2734        std::fs::write(&stream, b"\xff\xff\xff\xff\x10\x01\0\0").unwrap();
2735        for (path, is_stream) in [(file, false), (stream, true)] {
2736            let mut entry = Entry::for_test(&path, "x.arrow");
2737            enrich(&mut entry);
2738            assert_eq!(entry.cost.ipc_stream, is_stream, "{}", path.display());
2739        }
2740    }
2741
2742    /// Every extension the home screen offers has a reader behind it, and every
2743    /// extension a reader knows is offered. The two lists had drifted: `.psv` and
2744    /// `.xlsb` opened but were invisible, and `.txt` was listed and then refused.
2745    #[test]
2746    fn what_is_offered_and_what_opens_are_one_list() {
2747        for ext in [
2748            "parquet", "csv", "tsv", "psv", "json", "jsonl", "ndjson", "arrow", "arrows", "ipc",
2749            "feather", "avro", "orc", "xls", "xlsx", "xlsm", "xlsb",
2750        ] {
2751            let named = PathBuf::from(format!("sales.{ext}"));
2752            assert!(
2753                data_format(&named).is_some(),
2754                ".{ext} opens, so the home screen must offer it"
2755            );
2756        }
2757        // A README is text, read as lines; beside data it is not the directory's
2758        // table (`a_file_datui_does_not_read_does_not_disqualify_a_directory`).
2759        assert_eq!(
2760            data_format(Path::new("README.txt")),
2761            Some(crate::FileFormat::Text)
2762        );
2763        assert!(data_format(Path::new("notes")).is_none());
2764    }
2765
2766    /// A format's name is not an extension, and the one place that stores a name has
2767    /// to read it back with the inverse of what wrote it. `excel` is a name no
2768    /// extension spells, so parsing it as one answers `None` — and `None` there means
2769    /// "not Parquet", which leaves a directory's counts off rather than filling them from
2770    /// whatever Parquet is under it.
2771    #[test]
2772    fn a_format_name_round_trips_only_through_from_name() {
2773        use crate::FileFormat;
2774        for format in FileFormat::ALL {
2775            assert_eq!(
2776                FileFormat::from_name(format.name()),
2777                Some(format),
2778                "{} is a name",
2779                format.name()
2780            );
2781        }
2782        assert_eq!(FileFormat::from_extension("excel"), None);
2783        assert_eq!(FileFormat::from_name("xlsx"), None);
2784    }
2785
2786    /// A compression suffix is how a file is stored, not what it holds, on both routes.
2787    #[test]
2788    fn a_compressed_name_reads_as_the_format_under_it() {
2789        assert_eq!(
2790            data_format(Path::new("sales.csv.gz")),
2791            Some(crate::FileFormat::Csv)
2792        );
2793        assert_eq!(
2794            data_format(Path::new("events.json.zst")),
2795            Some(crate::FileFormat::Json)
2796        );
2797    }
2798
2799    /// `.ipc`, `.arrow` and `.feather` are one format under three names, so a directory
2800    /// holding two of them is one kind of thing rather than a mixture.
2801    #[test]
2802    fn one_format_under_several_names_is_not_a_mixture() {
2803        let dir = tempfile::tempdir().unwrap();
2804        std::fs::write(dir.path().join("a.arrow"), b"x").unwrap();
2805        std::fs::write(dir.path().join("b.ipc"), b"x").unwrap();
2806        assert_eq!(classify_directory(dir.path()), EntryKind::MultiFile);
2807    }
2808
2809    /// A README is neither a marker nor data. Locally it counts toward the majority
2810    /// and does not disqualify the directory; the cloud route counted it as data and
2811    /// answered `dir` where the local one said `multi`.
2812    #[test]
2813    fn a_file_datui_does_not_read_does_not_disqualify_a_directory() {
2814        let dir = tempfile::tempdir().unwrap();
2815        write(dir.path(), "a.parquet", &["id"]);
2816        write(dir.path(), "b.parquet", &["id"]);
2817        std::fs::write(dir.path().join("README.txt"), b"notes").unwrap();
2818
2819        #[cfg(feature = "cloud")]
2820        let objects: Vec<(String, u64)> = [
2821            ("out/a.parquet", 100u64),
2822            ("out/b.parquet", 100),
2823            ("out/README.txt", 12),
2824        ]
2825        .iter()
2826        .map(|(k, s)| ((*k).to_string(), *s))
2827        .collect();
2828
2829        #[cfg(feature = "cloud")]
2830        assert_eq!(
2831            classify_directory(dir.path()),
2832            crate::cloud_browse::look_at_listing("out/", &[], &objects).0,
2833            "the two routes answer the same directory alike"
2834        );
2835        assert_eq!(classify_directory(dir.path()), EntryKind::MultiFile);
2836    }
2837
2838    /// A label says what is inside, so it is true whatever `Enter` then does. The same
2839    /// directory of three tables reads `3 parquet` and is one to look inside.
2840    #[test]
2841    fn a_label_counts_what_is_there_rather_than_naming_a_decision() {
2842        let dir = tempfile::tempdir().unwrap();
2843        for name in ["a.parquet", "b.parquet", "c.parquet"] {
2844            write(dir.path(), name, &["id"]);
2845        }
2846        std::fs::create_dir_all(dir.path().join("archive")).unwrap();
2847        std::fs::write(dir.path().join("notes.csv"), b"x").unwrap();
2848        std::fs::write(dir.path().join("_SUCCESS"), b"").unwrap();
2849        std::fs::write(dir.path().join(".part.crc"), b"").unwrap();
2850
2851        let entry = measured(dir.path());
2852        assert_eq!(entry.label(), "mixed", "two formats is two formats");
2853        assert_eq!(
2854            entry.holds.line(true).as_deref(),
2855            Some("3 parquet · 1 csv · 1 directory"),
2856            "and the pane says what the label boiled down"
2857        );
2858        assert_eq!(entry.holds.data_files(), 4);
2859        assert_eq!(entry.holds.directories, 1);
2860    }
2861
2862    /// One format, and the count is the files.
2863    #[test]
2864    fn a_directory_of_one_format_is_labelled_by_it() {
2865        let dir = tempfile::tempdir().unwrap();
2866        for i in 0..12 {
2867            write(dir.path(), &format!("part-{i:05}.parquet"), &["id", "ts"]);
2868        }
2869        let entry = measured(dir.path());
2870        assert_eq!(entry.label(), "12 parquet");
2871        assert_eq!(entry.holds.line(true).as_deref(), Some("12 parquet"));
2872
2873        // A directory with nothing in it datui reads is a place to look inside. The
2874        // files are counted, but the pane's line is about what can be opened, and
2875        // "20 not read" read as a fault in a directory with nothing wrong in it.
2876        let plain = tempfile::tempdir().unwrap();
2877        for i in 0..20 {
2878            std::fs::write(plain.path().join(format!("note{i}.md")), b"x").unwrap();
2879        }
2880        let plain = measured(plain.path());
2881        assert_eq!(plain.label(), "dir");
2882        assert_eq!(plain.holds.not_read, 20);
2883        assert_eq!(plain.holds.line(true), None);
2884    }
2885
2886    /// A row nothing has looked into has only its kind to go on, and a hive root or a
2887    /// lake table is named by the thing it is rather than counted.
2888    #[test]
2889    fn a_kind_that_names_itself_keeps_its_name() {
2890        let dir = tempfile::tempdir().unwrap();
2891        std::fs::create_dir_all(dir.path().join("year=2024")).unwrap();
2892        std::fs::create_dir_all(dir.path().join("year=2025")).unwrap();
2893        assert_eq!(measured(dir.path()).label(), "hive");
2894
2895        let lake = tempfile::tempdir().unwrap();
2896        std::fs::create_dir_all(lake.path().join("_delta_log")).unwrap();
2897        assert_eq!(measured(lake.path()).label(), "delta");
2898
2899        let unlooked = Entry::new(PathBuf::from("/nowhere"), EntryKind::Unknown);
2900        assert_eq!(unlooked.label(), "");
2901    }
2902
2903    /// A directory of exactly the cap is a total, not a floor. `5000+` claims there is
2904    /// more; saying so about a directory that was read whole is a lie in the direction
2905    /// nobody can check.
2906    #[test]
2907    fn a_directory_read_whole_does_not_claim_there_is_more() {
2908        let dir = tempfile::tempdir().unwrap();
2909        for i in 0..MAX_ENTRIES_PER_DIR {
2910            std::fs::write(dir.path().join(format!("f{i:05}.csv")), b"x").unwrap();
2911        }
2912        let holds = look_at_directory(dir.path()).1;
2913        assert!(!holds.truncated, "every entry was read");
2914        assert_eq!(holds.label(), format!("{MAX_ENTRIES_PER_DIR} csv"));
2915
2916        std::fs::write(dir.path().join("one-more.csv"), b"x").unwrap();
2917        let holds = look_at_directory(dir.path()).1;
2918        assert!(holds.truncated, "and now there is more than was read");
2919        assert!(holds.label().contains('+'));
2920    }
2921
2922    /// A lake table is answered by three `join` tests and costs no listing. Counting
2923    /// one would walk every table in a warehouse on every pass, for a line beside a
2924    /// table whose files `enrich` then refuses to read anyway.
2925    #[test]
2926    fn a_lake_table_is_not_counted() {
2927        let dir = tempfile::tempdir().unwrap();
2928        std::fs::create_dir_all(dir.path().join("_delta_log")).unwrap();
2929        write(dir.path(), "part-00000.parquet", &["id"]);
2930        write(dir.path(), "part-00001.parquet", &["id"]);
2931
2932        let (kind, holds) = look_at_directory(dir.path());
2933        assert_eq!(kind, EntryKind::Delta);
2934        assert!(holds.is_empty(), "and its label is the format's own name");
2935        let entry = measured(dir.path());
2936        assert_eq!(entry.label(), "delta");
2937    }
2938
2939    /// A listing cut short cannot say there is no data in a directory, only that it found
2940    /// none among the entries it read. `mixed` needs no such qualifier — more files
2941    /// cannot unmake it — and `dir` does, because they can.
2942    #[test]
2943    fn a_cut_short_listing_does_not_claim_a_directory_is_empty() {
2944        let seen = Holds {
2945            skipped: 5000,
2946            truncated: true,
2947            ..Default::default()
2948        };
2949        assert_eq!(seen.label(), "dir+");
2950
2951        let whole = Holds {
2952            skipped: 3,
2953            ..Default::default()
2954        };
2955        assert_eq!(whole.label(), "dir");
2956
2957        // And a listing cut short before it found anything at all still says so: it is
2958        // not an empty tally, or the row falls back to its kind and reads `dir`.
2959        let nothing_yet = Holds {
2960            truncated: true,
2961            ..Default::default()
2962        };
2963        assert!(!nothing_yet.is_empty());
2964        assert_eq!(nothing_yet.label(), "dir+");
2965        assert!(Holds::default().is_empty());
2966
2967        let mixed = Holds {
2968            formats: vec![("parquet".to_string(), 3), ("csv".to_string(), 2)],
2969            truncated: true,
2970            ..Default::default()
2971        };
2972        assert_eq!(mixed.label(), "mixed", "more files cannot unmake it");
2973    }
2974
2975    /// A name cut to fit keeps both ends. A Hadoop output directory's `.crc` files are
2976    /// named for the file they check, and the head and the tail are what say so.
2977    #[test]
2978    fn a_long_name_keeps_both_ends() {
2979        let name = ".part-00000-8f3a91c2-7b4d-4e19-a6f0-c1d2e3f4a5b6-c000.snappy.parquet.crc";
2980        let line = shorten(name, 24);
2981        assert!(line.starts_with(".part-00000"), "the head: {line}");
2982        // The tail, as much of it as the ellipsis leaves: it takes three characters of
2983        // the twenty-four in the ASCII glyph set and one in the Unicode one.
2984        assert!(line.ends_with(".crc"), "and the tail: {line}");
2985        assert!(!line.contains("8f3a91c2"), "the middle goes: {line}");
2986        assert!(line.chars().count() <= 24, "{line}");
2987    }
2988
2989    /// The same directory read twice reads the same. Skipped names come back in whatever
2990    /// order the filesystem holds them, so the pane takes the first few *by name*.
2991    #[test]
2992    fn what_a_directory_holds_reads_the_same_twice() {
2993        let dir = tempfile::tempdir().unwrap();
2994        write(dir.path(), "a.parquet", &["id"]);
2995        for marker in [
2996            "_SUCCESS",
2997            "_committed_9",
2998            "_committed_1",
2999            ".crc",
3000            "_started_4",
3001        ] {
3002            std::fs::write(dir.path().join(marker), b"").unwrap();
3003        }
3004        let first = look_at_directory(dir.path()).1;
3005        for _ in 0..8 {
3006            assert_eq!(look_at_directory(dir.path()).1, first);
3007        }
3008        assert_eq!(
3009            first.skipped_names,
3010            vec![".crc", "_SUCCESS", "_committed_1", "_committed_9"],
3011            "the first four by name, of five"
3012        );
3013        assert_eq!(first.skipped, 5);
3014    }
3015
3016    /// The label counts what is directly inside; the numbers beside it are a promise
3017    /// about what `Enter` gives, and `Enter` reads the subtree. Measuring only the top
3018    /// would promise three files and open twenty-three — and would ask `is_one_table`
3019    /// about three files while unioning all twenty-three, which is the union the
3020    /// downgrade exists to prevent. The `holds` line names the directory that explains
3021    /// it.
3022    #[test]
3023    fn a_directories_numbers_are_what_opening_it_gives() {
3024        let dir = tempfile::tempdir().unwrap();
3025        for name in ["a.parquet", "b.parquet", "c.parquet"] {
3026            write(dir.path(), name, &["id", "legacy"]);
3027        }
3028        let archive = dir.path().join("archive");
3029        std::fs::create_dir_all(&archive).unwrap();
3030        for i in 0..20 {
3031            write(&archive, &format!("old-{i}.parquet"), &["id", "legacy"]);
3032        }
3033
3034        let entry = measured(dir.path());
3035        assert_eq!(
3036            entry.label(),
3037            "3 parquet",
3038            "three files are directly inside"
3039        );
3040        assert_eq!(
3041            entry.holds.line(true).as_deref(),
3042            Some("3 parquet · 1 directory")
3043        );
3044        assert_eq!(entry.rows, Some(23), "and opening it reads all of them");
3045    }
3046
3047    /// A directory past the counting budget whose *own* files were all read says an exact
3048    /// width. The budget is about the subtree; three files at the top are three
3049    /// footers, and `5+ cols` claims a sample that did not happen.
3050    #[test]
3051    fn a_width_is_a_floor_only_when_a_footer_went_unread() {
3052        let dir = tempfile::tempdir().unwrap();
3053        for name in ["a.parquet", "b.parquet", "c.parquet"] {
3054            write(dir.path(), name, &["id", "ts"]);
3055        }
3056        let archive = dir.path().join("archive");
3057        std::fs::create_dir_all(&archive).unwrap();
3058        for i in 0..MAX_FOOTERS_PER_DATASET + 6 {
3059            write(
3060                &archive,
3061                &format!("old-{i:03}.parquet"),
3062                &["wholly", "different"],
3063            );
3064        }
3065
3066        let entry = measured(dir.path());
3067        assert_eq!(entry.kind, EntryKind::Directory, "not one table");
3068        assert_eq!(entry.label(), "3 parquet");
3069        assert_eq!(entry.cols, Some(2), "id and ts");
3070        assert!(
3071            !entry.cols_sampled,
3072            "all three of its own footers were read"
3073        );
3074    }
3075
3076    #[test]
3077    fn a_big_directory_is_still_found_by_a_column_one_level_down() {
3078        let dir = tempfile::tempdir().unwrap();
3079        for name in ["a.parquet", "b.parquet", "c.parquet"] {
3080            write(dir.path(), name, &["id", "ts"]);
3081        }
3082        let archive = dir.path().join("archive");
3083        std::fs::create_dir_all(&archive).unwrap();
3084        for i in 0..MAX_FOOTERS_PER_DATASET + 6 {
3085            write(
3086                &archive,
3087                &format!("old-{i:03}.parquet"),
3088                &["wholly", "different"],
3089            );
3090        }
3091
3092        let entry = measured(dir.path());
3093        assert_eq!(entry.kind, EntryKind::Directory);
3094        // The label and the width are the three files directly inside.
3095        assert_eq!(entry.label(), "3 parquet");
3096        assert_eq!(entry.cols, Some(2), "id and ts");
3097        // The names are not: they are the home screen's search index, and `wholly` has
3098        // to reach the directory that holds one whether the directory was small enough to
3099        // read every footer or, as here, too big and sampled instead. Narrowing these
3100        // to the directory's own files made the answer depend on the directory's size.
3101        assert!(
3102            entry.columns.contains(&"wholly".to_string()),
3103            "{:?}",
3104            entry.columns
3105        );
3106        assert!(entry.columns.contains(&"id".to_string()));
3107    }
3108
3109    #[test]
3110    fn a_width_over_a_directories_own_files_is_a_floor_when_there_are_too_many() {
3111        let dir = tempfile::tempdir().unwrap();
3112        // Past the footer budget with the directory's *own* files, and no two of them one
3113        // table, so the downgrade samples its own files as well and says so. Every file
3114        // gets its own column, because which three get sampled is `read_dir` order.
3115        for i in 0..MAX_FOOTERS_PER_DATASET + 6 {
3116            write(
3117                dir.path(),
3118                &format!("f-{i:03}.parquet"),
3119                &[&format!("c{i}")],
3120            );
3121        }
3122
3123        let entry = measured(dir.path());
3124        assert_eq!(entry.kind, EntryKind::Directory, "not one table");
3125        assert_eq!(entry.label(), "70 parquet");
3126        assert!(
3127            entry.cols_sampled,
3128            "three of seventy footers were read, so the width is a floor"
3129        );
3130    }
3131
3132    #[test]
3133    fn a_directory_read_as_one_table_is_sized_by_everything_under_it() {
3134        let dir = tempfile::tempdir().unwrap();
3135        write(dir.path(), "a.parquet", &["id", "ts"]);
3136        write(dir.path(), "b.parquet", &["id", "ts"]);
3137        let more = dir.path().join("more");
3138        std::fs::create_dir_all(&more).unwrap();
3139        write(&more, "c.parquet", &["id", "ts"]);
3140
3141        let entry = measured(dir.path());
3142        assert_eq!(entry.kind, EntryKind::MultiFile, "one table");
3143        // `Enter` unions the subtree, so the size and the rows beside it are the
3144        // subtree's — the opposite of a downgraded directory, whose numbers are its own
3145        // files because it is never opened as one table.
3146        let all: u64 = [
3147            dir.path().join("a.parquet"),
3148            dir.path().join("b.parquet"),
3149            more.join("c.parquet"),
3150        ]
3151        .iter()
3152        .map(|p| std::fs::metadata(p).unwrap().len())
3153        .sum();
3154        assert_eq!(entry.size, Some(all));
3155        assert_eq!(entry.rows, Some(3));
3156    }
3157
3158    #[test]
3159    fn nothing_counted_is_the_only_thing_holds_calls_empty() {
3160        // `is_empty` stops a peek's answer reaching a row and keeps a `Holds` out of
3161        // the cache, so anything it calls empty is thrown away. Asked of each field on
3162        // its own, because the contract is the function's and not its callers': both
3163        // routes happen to set `directories` beside `partitions` and `skipped` beside
3164        // `skipped_names` today, which is exactly the kind of agreement that stops
3165        // holding one refactor later.
3166        assert!(Holds::default().is_empty());
3167        let one = |f: fn(&mut Holds)| {
3168            let mut h = Holds::default();
3169            f(&mut h);
3170            h
3171        };
3172        for (what, holds) in [
3173            ("a data file", one(|h| h.formats.push(("csv".into(), 1)))),
3174            ("a directory", one(|h| h.directories = 1)),
3175            ("a partition", one(|h| h.partitions = 1)),
3176            ("a file it cannot read", one(|h| h.not_read = 1)),
3177            ("a writer's own file", one(|h| h.skipped = 1)),
3178            (
3179                "the name of one",
3180                one(|h| h.skipped_names.push("_SUCCESS".into())),
3181            ),
3182            ("a listing cut short", one(|h| h.truncated = true)),
3183        ] {
3184            assert!(!holds.is_empty(), "{what} is something to say");
3185        }
3186    }
3187
3188    #[test]
3189    fn formats_that_tie_are_ordered_by_name_whatever_order_they_arrived_in() {
3190        use crate::FileFormat;
3191        // Given in the order that is wrong on both counts, so neither clause of the
3192        // comparison can be the one doing nothing. Without the tie-break a directory of
3193        // two CSV and two JSON reads `2 csv · 2 json` on one pass and `2 json · 2 csv`
3194        // on the next, which is the `read_dir` order this release exists to remove.
3195        let mut counts = vec![
3196            (FileFormat::Json, 2),
3197            (FileFormat::Csv, 2),
3198            (FileFormat::Parquet, 5),
3199        ];
3200        order_formats(&mut counts);
3201        assert_eq!(
3202            counts,
3203            vec![
3204                (FileFormat::Parquet, 5),
3205                (FileFormat::Csv, 2),
3206                (FileFormat::Json, 2)
3207            ]
3208        );
3209    }
3210
3211    /// The label and the read name the same format, including on a tie.
3212    ///
3213    /// They agreed on the common case and not on a tie: the label sorted equal counts
3214    /// by name and the read put Parquet first, so a directory of two CSV and two Parquet
3215    /// was labelled `2 csv · 2 parquet` and opened as Parquet. One order now, and this
3216    /// is the case that tells the two orders apart.
3217    #[test]
3218    fn the_label_and_the_read_pick_the_same_format_on_a_tie() {
3219        use crate::FileFormat;
3220        let tmp = tempfile::TempDir::new().unwrap();
3221        for name in ["a.csv", "b.csv", "c.parquet", "d.parquet"] {
3222            std::fs::write(tmp.path().join(name), b"x").unwrap();
3223        }
3224
3225        let (_, holds) = look_at_directory(tmp.path());
3226        assert_eq!(
3227            holds.formats.first().map(|(f, n)| (f.as_str(), *n)),
3228            Some(("parquet", 2)),
3229            "the label names Parquet first: {:?}",
3230            holds.formats
3231        );
3232
3233        match directory_format(tmp.path()) {
3234            DirectoryFormat::Mixed { format, .. } => assert_eq!(
3235                format,
3236                FileFormat::Parquet,
3237                "and so does the reader the open picks"
3238            ),
3239            other => panic!("a directory of two formats is mixed, got {other:?}"),
3240        }
3241    }
3242
3243    /// A Hugging Face dataset saved to disk is one shard and two JSON files that
3244    /// describe it: a dataset of Arrow, labelled and read as one, the JSON its writer's
3245    /// own. Beside no Arrow, the same names are data.
3246    #[test]
3247    fn a_hugging_face_dataset_is_its_shards() {
3248        use crate::FileFormat;
3249        let tmp = tempfile::TempDir::new().unwrap();
3250        for name in [
3251            "data-00000-of-00002.arrow",
3252            "data-00001-of-00002.arrow",
3253            "dataset_info.json",
3254            "state.json",
3255        ] {
3256            std::fs::write(tmp.path().join(name), b"x").unwrap();
3257        }
3258        let (kind, holds) = look_at_directory(tmp.path());
3259        assert_eq!(kind, EntryKind::MultiFile);
3260        assert_eq!(holds.formats, [("arrow".to_string(), 2)]);
3261        assert_eq!(holds.skipped, 2);
3262        assert_eq!(holds.skipped_names, ["dataset_info.json", "state.json"]);
3263        match directory_format(tmp.path()) {
3264            DirectoryFormat::One(FileFormat::Arrow, files) => assert_eq!(files.len(), 2),
3265            other => panic!("the shards are the dataset, got {other:?}"),
3266        }
3267
3268        std::fs::remove_file(tmp.path().join("data-00001-of-00002.arrow")).unwrap();
3269        assert!(matches!(
3270            directory_format(tmp.path()),
3271            DirectoryFormat::One(FileFormat::Arrow, _)
3272        ));
3273
3274        let json = tempfile::TempDir::new().unwrap();
3275        for name in ["state.json", "other.json"] {
3276            std::fs::write(json.path().join(name), b"{}").unwrap();
3277        }
3278        assert_eq!(
3279            look_at_directory(json.path()).1.formats,
3280            [("json".to_string(), 2)]
3281        );
3282    }
3283
3284    #[test]
3285    fn partitions_carry_a_directory_only_while_they_are_the_most_of_it() {
3286        // The boundary the local rule turns on, and the twin of the cloud route's
3287        // `partitions_carry_a_prefix_only_while_they_are_the_most_of_it`. Every directory
3288        // on disk goes through this one.
3289        let laid_out = |strays: usize| {
3290            let dir = tempfile::tempdir().unwrap();
3291            for year in ["year=2024", "year=2025"] {
3292                let part = dir.path().join(year);
3293                std::fs::create_dir_all(&part).unwrap();
3294                write(&part, "data.parquet", &["id"]);
3295            }
3296            for i in 0..strays {
3297                write(dir.path(), &format!("stray-{i}.parquet"), &["id"]);
3298            }
3299            classify_directory(dir.path())
3300        };
3301        assert_eq!(
3302            laid_out(2),
3303            EntryKind::Hive,
3304            "two partitions against two files beside them"
3305        );
3306        assert_ne!(
3307            laid_out(3),
3308            EntryKind::Hive,
3309            "one more file than partitions is a directory that holds a key=value"
3310        );
3311    }
3312
3313    #[test]
3314    fn a_folder_marker_is_bookkeeping_even_beside_a_partition() {
3315        // Legacy s3n and EMR write a zero-byte `<name>_$folder$` object beside every
3316        // prefix. Where the prefix is a partition the marker carries the `=` too, so a
3317        // partition test that only looks for one calls the marker data and the pane
3318        // reports one unreadable file per partition.
3319        assert!(is_bookkeeping("year=2024_$folder$"));
3320        assert!(is_bookkeeping("alpha_$folder$"));
3321        assert!(!is_bookkeeping("year=2024"), "the partition itself is data");
3322        assert!(
3323            !is_bookkeeping("_date=2024-01-01"),
3324            "Spark partitions on internal columns"
3325        );
3326    }
3327
3328    /// And the files under it are what the one-table test is asked about, since they
3329    /// are what the union would hold.
3330    #[test]
3331    fn a_table_hidden_under_a_directory_still_downgrades_it() {
3332        let dir = tempfile::tempdir().unwrap();
3333        for name in ["a.parquet", "b.parquet"] {
3334            write(dir.path(), name, &["id", "ts"]);
3335        }
3336        let archive = dir.path().join("archive");
3337        std::fs::create_dir_all(&archive).unwrap();
3338        write(
3339            &archive,
3340            "other.parquet",
3341            &["wholly", "different", "columns"],
3342        );
3343
3344        let entry = measured(dir.path());
3345        assert_eq!(
3346            entry.kind,
3347            EntryKind::Directory,
3348            "a union over these is not one table"
3349        );
3350        assert_eq!(entry.rows, None);
3351        // And the width beside `2 parquet` is those two files. The check is asked of
3352        // everything under the directory, because that is what opening it would union;
3353        // a downgraded row is never opened as one, so reporting the subtree's union
3354        // would be a set of columns nothing produces.
3355        assert_eq!(entry.label(), "2 parquet");
3356        assert_eq!(entry.cols, Some(2), "id and ts");
3357        // The names are every column under the directory, because they are what the home
3358        // screen searches: the directory does hold a `wholly`, one level down.
3359        assert!(entry.columns.contains(&"wholly".to_string()));
3360        assert!(entry.columns.contains(&"id".to_string()));
3361
3362        // And the size is those two files, not the subtree's: three numbers on one row
3363        // measured over three different sets of files is no row at all.
3364        let own: u64 = ["a.parquet", "b.parquet"]
3365            .iter()
3366            .map(|n| std::fs::metadata(dir.path().join(n)).unwrap().len())
3367            .sum();
3368        assert_eq!(entry.size, Some(own));
3369    }
3370
3371    /// A hive tree of CSV is still laid out, whatever its rows cannot say. The layout
3372    /// is directory names — no footers, no opens — and it is the thing you most want
3373    /// before opening a dataset too large to count.
3374    #[test]
3375    fn a_hive_tree_of_another_format_is_still_laid_out() {
3376        let dir = tempfile::tempdir().unwrap();
3377        for year in ["year=2024", "year=2025"] {
3378            let part = dir.path().join(year);
3379            std::fs::create_dir_all(&part).unwrap();
3380            std::fs::write(part.join("data.csv"), b"id\n1\n").unwrap();
3381        }
3382        // A stray data file at the root, which is a hive root's ordinary furniture.
3383        std::fs::write(dir.path().join("summary.csv"), b"id\n1\n").unwrap();
3384
3385        let entry = measured(dir.path());
3386        assert_eq!(entry.kind, EntryKind::Hive);
3387        assert!(entry.cost.partitions.is_some(), "the layout is named");
3388        assert_eq!(entry.rows, None, "and nothing is invented about its rows");
3389    }
3390
3391    /// A hive root's own files are strays beside the partitions — a `schema.json` or a
3392    /// `manifest.csv` left at the top — so its counted format is not its data's, and
3393    /// asking it would blank the whole dataset for one such file.
3394    #[test]
3395    fn a_hive_dataset_is_described_despite_a_stray_file_at_its_root() {
3396        let dir = tempfile::tempdir().unwrap();
3397        for year in ["year=2024", "year=2025"] {
3398            let part = dir.path().join(year);
3399            std::fs::create_dir_all(&part).unwrap();
3400            write(&part, "data.parquet", &["id"]);
3401        }
3402        std::fs::write(dir.path().join("schema.json"), b"{}").unwrap();
3403
3404        let entry = measured(dir.path());
3405        assert_eq!(entry.kind, EntryKind::Hive);
3406        assert_eq!(entry.holds.one_format(), Some("json"), "its own only file");
3407        assert_eq!(entry.rows, Some(2), "and the dataset is still counted");
3408        assert_eq!(entry.cols, Some(2), "`id` and the partition column `year`");
3409    }
3410
3411    /// A hive dataset is described whatever odd file is lying in a partition. One
3412    /// spine cannot tell a Parquet tree with a stray CSV in it from a CSV tree with a
3413    /// stray Parquet, and blanking a dataset that opens perfectly is the worse of the
3414    /// two mistakes.
3415    #[test]
3416    fn a_hive_dataset_is_described_despite_a_stray_file() {
3417        let dir = tempfile::tempdir().unwrap();
3418        for year in ["year=2024", "year=2025"] {
3419            let part = dir.path().join(year);
3420            std::fs::create_dir_all(&part).unwrap();
3421            write(&part, "data.parquet", &["id"]);
3422        }
3423        // Somebody's notes, dropped in beside the data.
3424        std::fs::write(dir.path().join("year=2024/notes.csv"), b"x").unwrap();
3425
3426        let entry = measured(dir.path());
3427        assert_eq!(entry.kind, EntryKind::Hive);
3428        assert_eq!(entry.rows, Some(2), "the dataset is still counted");
3429        assert!(
3430            entry.cost.partitions.is_some(),
3431            "and its layout still named"
3432        );
3433    }
3434
3435    /// A directory is described by its own files, not by what is under them. The footer
3436    /// walk recurses, which is right for a hive root and wrong for a directory of JSON
3437    /// that happens to have Parquet in a subdirectory.
3438    #[test]
3439    fn a_directory_is_not_described_by_files_it_does_not_name() {
3440        let dir = tempfile::tempdir().unwrap();
3441        for name in ["a.json", "b.json", "c.json"] {
3442            std::fs::write(dir.path().join(name), b"{}").unwrap();
3443        }
3444        let under = dir.path().join("derived");
3445        std::fs::create_dir_all(&under).unwrap();
3446        write(&under, "one.parquet", &["id", "ts", "amount"]);
3447        write(&under, "two.parquet", &["id", "ts", "amount"]);
3448
3449        let entry = measured(dir.path());
3450        assert_eq!(entry.label(), "3 json");
3451        assert_eq!(
3452            entry.cols, None,
3453            "the Parquet below it is not this directory's shape"
3454        );
3455        assert_eq!(entry.rows, None);
3456        assert!(entry.columns.is_empty());
3457    }
3458
3459    /// The gate's default, for a dataset row that counted nothing.
3460    ///
3461    /// **Nothing produces this row today.** `enrich` only reaches the gate for `Hive`
3462    /// and `MultiFile`; `look_at_directory` cannot answer `MultiFile` without counting
3463    /// a format, a hive root skips the gate outright, `CLASSIFIER_VERSION` 4 refuses a
3464    /// cached kind that arrives without a tally, and the cloud `(all files)` row is
3465    /// built from a listing and never measured. So this constructs the row by hand, and
3466    /// it pins a default rather than a path.
3467    ///
3468    /// It is worth pinning because the default is the arguable one. Turning it away
3469    /// would blank the size, the width and the row count of any such row the moment one
3470    /// appeared, and leaving the counts off a directory is a mistake opening it undoes —
3471    /// giving it another format's numbers is not.
3472    #[test]
3473    fn a_dataset_row_that_counted_nothing_is_still_described() {
3474        let dir = tempfile::tempdir().unwrap();
3475        write(dir.path(), "a.parquet", &["id", "ts"]);
3476        write(dir.path(), "b.parquet", &["id", "ts"]);
3477
3478        let mut entry = Entry {
3479            kind: EntryKind::MultiFile,
3480            ..Entry::for_test(dir.path(), "data")
3481        };
3482        assert!(entry.holds.one_format().is_none(), "nothing counted");
3483
3484        enrich(&mut entry);
3485        assert!(entry.size.is_some(), "the footers were read");
3486        assert_eq!(entry.rows, Some(2));
3487        assert_eq!(entry.cols, Some(2), "id and ts");
3488    }
3489
3490    /// A Parquet file whose name begins with `_` is still a Parquet file. It does not
3491    /// count toward what the directory around it holds — that is what `is_bookkeeping` is
3492    /// for — but the listing shows it, `Enter` opens it, and the row beside it must say
3493    /// how many rows it has rather than nothing at all.
3494    #[test]
3495    fn a_parquet_file_named_like_a_writers_file_is_still_measured() {
3496        let dir = tempfile::tempdir().unwrap();
3497        write(dir.path(), "_2024_sales.parquet", &["id", "amount"]);
3498
3499        let mut entry = Entry {
3500            path: dir.path().join("_2024_sales.parquet"),
3501            kind: EntryKind::File,
3502            name: "_2024_sales.parquet".into(),
3503            size: None,
3504            modified: None,
3505            rows: None,
3506            cols: None,
3507            cols_sampled: false,
3508            columns: Vec::new(),
3509            cost: Cost::default(),
3510            holds: Default::default(),
3511            opens_whole_directory: false,
3512            format_spec: None,
3513            table: None,
3514        };
3515        enrich(&mut entry);
3516        assert_eq!(entry.rows, Some(1), "its footer was read");
3517        assert_eq!(entry.cols, Some(2));
3518        assert!(schema_preview(&entry).is_some(), "and the pane shows it");
3519
3520        // And it still does not make the directory around it a dataset.
3521        assert!(is_bookkeeping("_2024_sales.parquet"));
3522        assert_eq!(classify_directory(dir.path()), EntryKind::Directory);
3523    }
3524
3525    /// A hive table's pane lists its partition keys typed the way the scan types them:
3526    /// a date is a date and `true` a boolean, not text.
3527    #[test]
3528    fn a_hive_preview_types_its_keys_as_the_scan_does() {
3529        use polars::prelude::DataType;
3530        let dir = tempfile::tempdir().unwrap();
3531        let leaf = dir.path().join("day=2024-01-02/flag=true/n=3/x=1.5");
3532        std::fs::create_dir_all(&leaf).unwrap();
3533        write(&leaf, "part.parquet", &["id"]);
3534        let mut entry = Entry::directory(dir.path());
3535        entry.kind = EntryKind::Hive;
3536        let preview = schema_preview(&entry).expect("a footer to read");
3537        let types: Vec<(&str, &DataType)> = preview.iter().map(|(n, t)| (n.as_str(), t)).collect();
3538        assert_eq!(
3539            types[..4],
3540            [
3541                ("day", &DataType::Date),
3542                ("flag", &DataType::Boolean),
3543                ("n", &DataType::Int64),
3544                ("x", &DataType::Float64),
3545            ]
3546        );
3547        assert_eq!(types[4].0, "id");
3548    }
3549
3550    /// Part files with no extension inside a `.parquet` directory are data by where they
3551    /// sit. The cloud route has always counted them; the local one said `dir`.
3552    #[cfg(feature = "cloud")]
3553    #[test]
3554    fn extensionless_part_files_are_data_on_both_routes() {
3555        let dir = tempfile::tempdir().unwrap();
3556        let table = dir.path().join("occurrence.parquet");
3557        std::fs::create_dir_all(&table).unwrap();
3558        std::fs::write(table.join("000001"), b"PAR1").unwrap();
3559        std::fs::write(table.join("000002"), b"PAR1").unwrap();
3560
3561        let objects: Vec<(String, u64)> = [
3562            "gbif/occurrence.parquet/000001",
3563            "gbif/occurrence.parquet/000002",
3564        ]
3565        .iter()
3566        .map(|k| ((*k).to_string(), 10u64))
3567        .collect();
3568
3569        assert_eq!(
3570            classify_directory(&table),
3571            crate::cloud_browse::look_at_listing("gbif/occurrence.parquet/", &[], &objects).0,
3572            "the two routes answer the same directory alike"
3573        );
3574        assert_eq!(classify_directory(&table), EntryKind::MultiFile);
3575    }
3576
3577    /// And a directory offered as a dataset is one whose files can be counted. The same
3578    /// name test decides both, or the row promises a dataset and shows `?` rows and an
3579    /// empty schema for the rest of the session.
3580    #[test]
3581    fn extensionless_part_files_are_measured_not_just_offered() {
3582        let dir = tempfile::tempdir().unwrap();
3583        let table = dir.path().join("occurrence.parquet");
3584        std::fs::create_dir_all(&table).unwrap();
3585        // Named as GBIF and Spark leave them: no extension, inside a `.parquet`
3586        // directory.
3587        write(&table, "000001", &["id", "species"]);
3588        write(&table, "000002", &["id", "species"]);
3589
3590        let entry = measured(&table);
3591        assert_eq!(entry.kind, EntryKind::MultiFile);
3592        assert_eq!(entry.rows, Some(2), "both footers were read");
3593        assert_eq!(entry.cols, Some(2));
3594        assert!(
3595            schema_preview(&entry).is_some(),
3596            "and the schema pane shows what those footers said, rather than asking \
3597             for a full read of files already read"
3598        );
3599
3600        // → goes inside a directory labelled `multi`, so the listing has to show the
3601        // files the label was counted from — and each is a Parquet file in its own right.
3602        let mut listed = scan_dir(&table);
3603        assert_eq!(
3604            listed.iter().map(|e| e.name.as_str()).collect::<Vec<_>>(),
3605            vec!["000001", "000002"],
3606            "the directory the label promises is not an empty listing"
3607        );
3608        let part = listed.first_mut().expect("a part file is listed");
3609        enrich(part);
3610        assert_eq!(part.rows, Some(1), "a part file counts its own rows");
3611        assert_eq!(part.cols, Some(2));
3612    }
3613
3614    /// One `key=value` prefix among files datui does not read is a hive root on both
3615    /// routes. It is not much of one — but the local route has always said so, and the
3616    /// cloud route disagreeing was the divergence. Pinned rather than left to be
3617    /// rediscovered: #275 phase 3 takes the consequence off the label.
3618    #[cfg(feature = "cloud")]
3619    #[test]
3620    fn one_partition_beside_files_datui_cannot_read_answers_alike() {
3621        let dir = tempfile::tempdir().unwrap();
3622        std::fs::create_dir_all(dir.path().join("notes=old")).unwrap();
3623        for note in ["README.md", "LICENSE", "logo.png"] {
3624            std::fs::write(dir.path().join(note), b"x").unwrap();
3625        }
3626        let objects: Vec<(String, u64)> = ["out/README.md", "out/LICENSE", "out/logo.png"]
3627            .iter()
3628            .map(|k| ((*k).to_string(), 12u64))
3629            .collect();
3630
3631        assert_eq!(
3632            classify_directory(dir.path()),
3633            crate::cloud_browse::look_at_listing("out/", &["out/notes=old/".to_string()], &objects)
3634                .0,
3635            "the two routes answer the same directory alike"
3636        );
3637    }
3638
3639    /// A partition is a partition whatever it starts with. Spark and Hive partition on
3640    /// internal columns — `_date=2024-01-01`, `_c0=…` — and reading those as a writer's
3641    /// own files loses the whole dataset.
3642    #[test]
3643    fn a_partition_named_like_a_writers_file_is_still_a_partition() {
3644        let dir = tempfile::tempdir().unwrap();
3645        let mut directories = Vec::new();
3646        for day in ["2024-01-01", "2024-01-02", "2024-01-03"] {
3647            std::fs::create_dir_all(dir.path().join(format!("_date={day}"))).unwrap();
3648            directories.push(format!("events/_date={day}/"));
3649        }
3650
3651        #[cfg(feature = "cloud")]
3652        assert_eq!(
3653            classify_directory(dir.path()),
3654            crate::cloud_browse::look_at_listing("events/", &directories, &[]).0,
3655            "the two routes answer the same directory alike"
3656        );
3657        assert_eq!(classify_directory(dir.path()), EntryKind::Hive);
3658        assert!(!is_bookkeeping("_date=2024-01-01"));
3659        assert!(is_bookkeeping("_temporary"));
3660    }
3661
3662    /// A prefix a writer made for itself is not a directory somebody put data in, on
3663    /// either route. `_temporary/` counted toward the majority in a bucket and not
3664    /// locally, so the same directory came back two different kinds.
3665    #[cfg(feature = "cloud")]
3666    #[test]
3667    fn a_writers_own_directory_is_skipped_on_both_routes() {
3668        let dir = tempfile::tempdir().unwrap();
3669        write(dir.path(), "part-00000.parquet", &["id"]);
3670        write(dir.path(), "part-00001.parquet", &["id"]);
3671        std::fs::create_dir_all(dir.path().join("_temporary")).unwrap();
3672        std::fs::create_dir_all(dir.path().join("notes")).unwrap();
3673        std::fs::create_dir_all(dir.path().join("archive")).unwrap();
3674
3675        let local = classify_directory(dir.path());
3676        let directories: Vec<String> = ["out/_temporary/", "out/notes/", "out/archive/"]
3677            .iter()
3678            .map(|f| (*f).to_string())
3679            .collect();
3680        let objects: Vec<(String, u64)> = [
3681            ("out/part-00000.parquet", 100u64),
3682            ("out/part-00001.parquet", 100),
3683        ]
3684        .iter()
3685        .map(|(k, s)| ((*k).to_string(), *s))
3686        .collect();
3687        let cloud = crate::cloud_browse::look_at_listing("out/", &directories, &objects).0;
3688
3689        assert_eq!(
3690            local, cloud,
3691            "the two routes answer the same directory alike"
3692        );
3693        assert_eq!(local, EntryKind::MultiFile);
3694    }
3695
3696    /// Every entry is in exactly one count, including the ones with nothing behind
3697    /// them. A FIFO and a broken symlink named like data are not data and are not a
3698    /// writer's own; without a count they were in nothing, and the pane said `1 csv`
3699    /// about a directory of three entries.
3700    #[cfg(unix)]
3701    #[test]
3702    fn every_entry_is_in_one_count() {
3703        let dir = tempfile::tempdir().unwrap();
3704        std::fs::write(dir.path().join("real.csv"), b"id\n1\n").unwrap();
3705        std::os::unix::fs::symlink(dir.path().join("gone"), dir.path().join("broken.csv")).unwrap();
3706        std::fs::write(dir.path().join("notes.md"), b"x").unwrap();
3707        std::fs::write(dir.path().join("_SUCCESS"), b"").unwrap();
3708
3709        let holds = look_at_directory(dir.path()).1;
3710        assert_eq!(holds.data_files(), 1);
3711        assert_eq!(holds.not_read, 2, "the note and the broken link");
3712        assert_eq!(holds.skipped, 1);
3713        assert_eq!(holds.line(true).as_deref(), Some("1 csv"));
3714    }
3715
3716    /// Named like data and impossible to read: a FIFO blocks whoever opens it until a
3717    /// writer appears, and a broken symlink opens as nothing. `directory_format` has
3718    /// always skipped both; the listing now agrees.
3719    #[cfg(unix)]
3720    #[test]
3721    fn a_name_with_nothing_behind_it_is_not_a_data_file() {
3722        let dir = tempfile::tempdir().unwrap();
3723        std::os::unix::fs::symlink(dir.path().join("gone.csv"), dir.path().join("a.csv")).unwrap();
3724        std::os::unix::fs::symlink(dir.path().join("gone.csv"), dir.path().join("b.csv")).unwrap();
3725        assert_eq!(
3726            classify_directory(dir.path()),
3727            EntryKind::Directory,
3728            "two broken symlinks are not a dataset"
3729        );
3730    }
3731
3732    /// A checkpoint directory is the model: its shards are the table, the config and
3733    /// tokenizer JSON beside them are passed over however many there are, and the label
3734    /// names the weights rather than calling the directory mixed.
3735    #[test]
3736    fn a_model_directory_is_its_weights() {
3737        let dir = tempfile::tempdir().unwrap();
3738        for name in [
3739            "model-00001-of-00002.safetensors",
3740            "model-00002-of-00002.safetensors",
3741            "config.json",
3742            "generation_config.json",
3743            "tokenizer.json",
3744            "tokenizer_config.json",
3745            "model.safetensors.index.json",
3746        ] {
3747            std::fs::write(dir.path().join(name), b"x").unwrap();
3748        }
3749        let DirectoryFormat::Mixed {
3750            format,
3751            files,
3752            passed_over,
3753        } = directory_format(dir.path())
3754        else {
3755            panic!("weights and JSON are two formats");
3756        };
3757        assert_eq!(format, crate::FileFormat::Safetensors);
3758        assert_eq!(files.len(), 3, "the shards, and the index for its metadata");
3759        assert_eq!(passed_over, [(crate::FileFormat::Json, 4)]);
3760        let (kind, holds) = look_at_directory(dir.path());
3761        assert_eq!(kind, EntryKind::MultiFile, "opened as one");
3762        assert_eq!(holds.label(), "2 safetensors", "the shards, not the index");
3763
3764        // The index is the model too, named by what it is rather than its extension.
3765        assert_eq!(
3766            data_format(Path::new("model.safetensors.index.json")),
3767            Some(crate::FileFormat::Safetensors)
3768        );
3769        // Weights beside another table format are not a model directory.
3770        std::fs::write(dir.path().join("data.parquet"), b"x").unwrap();
3771        let (kind, holds) = look_at_directory(dir.path());
3772        assert_eq!(
3773            (kind, holds.label().as_str()),
3774            (EntryKind::Directory, "mixed")
3775        );
3776    }
3777
3778    /// A model or MIDI file is known by its first bytes under any name.
3779    #[test]
3780    fn signed_files_are_sniffed_by_their_first_bytes() {
3781        let dir = tempfile::tempdir().unwrap();
3782        let gguf = dir.path().join("weights");
3783        std::fs::write(&gguf, b"GGUF\x03\x00\x00\x00").unwrap();
3784        let st = dir.path().join("checkpoint.bin");
3785        let mut bytes = 2u64.to_le_bytes().to_vec();
3786        bytes.extend_from_slice(b"{}");
3787        std::fs::write(&st, &bytes).unwrap();
3788        let text = dir.path().join("notes");
3789        std::fs::write(&text, b"just some text").unwrap();
3790        assert_eq!(sniff_format(&gguf), Some(crate::FileFormat::Gguf));
3791        let opened = |path: &Path| crate::readers::sniff_open(path, None);
3792        assert_eq!(opened(&st), Some(crate::FileFormat::Safetensors));
3793        assert_eq!(opened(&text), None);
3794        let midi = dir.path().join("song.bin");
3795        std::fs::write(&midi, b"MThd\0\0\0\x06\0\0\0\x01\0\x60").unwrap();
3796        assert_eq!(opened(&midi), Some(crate::FileFormat::Midi));
3797    }
3798
3799    /// A directory is offered as one dataset only when its format can be read as many
3800    /// files. `.tsv`, `.psv` and Excel have a single-file reader and nothing that takes
3801    /// a list, so offering them puts the refusal one keystroke later instead of not
3802    /// making the promise.
3803    #[test]
3804    fn a_format_that_cannot_be_read_as_many_is_not_offered_as_one() {
3805        for ext in ["tsv", "psv", "xlsx", "xlsb"] {
3806            let dir = tempfile::tempdir().unwrap();
3807            std::fs::write(dir.path().join(format!("a.{ext}")), b"x").unwrap();
3808            std::fs::write(dir.path().join(format!("b.{ext}")), b"x").unwrap();
3809            assert_eq!(
3810                classify_directory(dir.path()),
3811                EntryKind::Directory,
3812                "a directory of .{ext} has no reader that takes a list"
3813            );
3814        }
3815        // The ones that do are unaffected — every arm the multi-path open handles.
3816        for ext in [
3817            "parquet", "csv", "json", "jsonl", "ndjson", "arrow", "arrows", "ipc", "feather",
3818            "avro", "orc",
3819        ] {
3820            let dir = tempfile::tempdir().unwrap();
3821            std::fs::write(dir.path().join(format!("a.{ext}")), b"x").unwrap();
3822            std::fs::write(dir.path().join(format!("b.{ext}")), b"x").unwrap();
3823            assert_eq!(
3824                classify_directory(dir.path()),
3825                EntryKind::MultiFile,
3826                ".{ext} reads as many files"
3827            );
3828        }
3829    }
3830
3831    /// The readdir-order bug: eight Parquet files and a ninth entry that is a writer's
3832    /// own file. A probe of the first eight entries never saw the JSON and said `multi`;
3833    /// a bucket listing sorts `_metadata.json` first and said `dir`. Same directory, two
3834    /// answers, decided by the order the filesystem happened to return.
3835    #[cfg(feature = "cloud")]
3836    #[test]
3837    fn a_writers_own_file_is_skipped_whatever_order_it_is_listed_in() {
3838        let dir = tempfile::tempdir().unwrap();
3839        for part in 0..8 {
3840            write(dir.path(), &format!("{part}.parquet"), &["season"]);
3841        }
3842        std::fs::write(dir.path().join("_metadata.json"), b"{}").unwrap();
3843
3844        let local = classify_directory(dir.path());
3845        // The same directory as a bucket lists it: lexicographic, so the JSON comes
3846        // first.
3847        let mut keys: Vec<(String, u64)> = vec![("jolpica/2000/_metadata.json".into(), 2)];
3848        for part in 0..8 {
3849            keys.push((format!("jolpica/2000/{part}.parquet"), 100));
3850        }
3851        keys.sort();
3852        let cloud = crate::cloud_browse::look_at_listing("jolpica/2000/", &[], &keys).0;
3853
3854        assert_eq!(
3855            local, cloud,
3856            "the two routes answer the same directory alike"
3857        );
3858        assert_eq!(local, EntryKind::MultiFile);
3859    }
3860
3861    /// The files a job leaves beside its output are skipped on every route, not just
3862    /// the two names each route happened to know.
3863    #[test]
3864    fn job_files_are_skipped_on_every_route() {
3865        let dir = tempfile::tempdir().unwrap();
3866        write(dir.path(), "part-00000.parquet", &["id"]);
3867        write(dir.path(), "part-00001.parquet", &["id"]);
3868        for marker in [
3869            "_SUCCESS",
3870            "_committed_1727",
3871            "_committed_1728",
3872            "_started_1727",
3873            ".part.crc",
3874        ] {
3875            std::fs::write(dir.path().join(marker), b"").unwrap();
3876        }
3877        assert_eq!(
3878            classify_directory(dir.path()),
3879            EntryKind::MultiFile,
3880            "five markers beside two data files do not outvote them"
3881        );
3882
3883        #[cfg(feature = "cloud")]
3884        let keys: Vec<(String, u64)> = [
3885            ("out/_SUCCESS", 0u64),
3886            ("out/_committed_1727", 12),
3887            ("out/_committed_1728", 12),
3888            ("out/_started_1727", 12),
3889            ("out/.part.crc", 8),
3890            ("out/part-00000.parquet", 100),
3891            ("out/part-00001.parquet", 100),
3892        ]
3893        .iter()
3894        .map(|(k, s)| ((*k).to_string(), *s))
3895        .collect();
3896        #[cfg(feature = "cloud")]
3897        assert_eq!(
3898            crate::cloud_browse::look_at_listing("out/", &[], &keys).0,
3899            EntryKind::MultiFile,
3900            "and the same in a bucket"
3901        );
3902    }
3903
3904    /// Write `columns` as a one-row Parquet file named `name` under `dir`.
3905    fn write(dir: &Path, name: &str, columns: &[&str]) {
3906        let mut frame = DataFrame::new(
3907            1,
3908            columns
3909                .iter()
3910                .map(|c| Column::new((*c).into(), &[1i32]))
3911                .collect::<Vec<_>>(),
3912        )
3913        .unwrap();
3914        let file = std::fs::File::create(dir.join(name)).unwrap();
3915        ParquetWriter::new(file).finish(&mut frame).unwrap();
3916    }
3917
3918    /// A one-row Parquet file with a struct column, so the leaves and the columns a
3919    /// reader sees are genuinely different things rather than dots in a name.
3920    fn write_nested(dir: &Path, name: &str, struct_name: &str, fields: &[&str]) {
3921        let inner = DataFrame::new(
3922            1,
3923            fields
3924                .iter()
3925                .map(|f| Column::new((*f).into(), &[1i32]))
3926                .collect::<Vec<_>>(),
3927        )
3928        .unwrap();
3929        let nested = inner
3930            .into_struct(struct_name.into())
3931            .into_series()
3932            .into_column();
3933        let mut frame = DataFrame::new(1, vec![Column::new("id".into(), &[1i32]), nested]).unwrap();
3934        let file = std::fs::File::create(dir.join(name)).unwrap();
3935        ParquetWriter::new(file).finish(&mut frame).unwrap();
3936    }
3937
3938    fn measured(dir: &Path) -> Entry {
3939        let (kind, holds) = look_at_directory(dir);
3940        let mut entry = Entry {
3941            path: dir.to_path_buf(),
3942            kind,
3943            name: dir.file_name().unwrap().to_string_lossy().into_owned(),
3944            size: None,
3945            modified: None,
3946            rows: None,
3947            cols: None,
3948            cols_sampled: false,
3949            columns: Vec::new(),
3950            cost: Cost::default(),
3951            holds,
3952            opens_whole_directory: false,
3953            format_spec: None,
3954            table: None,
3955        };
3956        enrich(&mut entry);
3957        entry
3958    }
3959
3960    /// The shape that prompted this: one Parquet file per table, sharing an extension
3961    /// and nothing else. Named for what it is rather than what it is called, because
3962    /// the filenames are exactly what cannot decide it.
3963    /// A record cached before the rename still says how many subdirectories it saw.
3964    #[test]
3965    fn holds_written_as_folders_still_reads() {
3966        let old: Holds = serde_json::from_str(r#"{"folders":3,"partitions":2}"#).unwrap();
3967        assert_eq!((old.directories, old.partitions), (3, 2));
3968        let new = serde_json::to_string(&old).unwrap();
3969        assert!(new.contains(r#""directories":3"#), "{new}");
3970    }
3971
3972    #[test]
3973    fn a_directory_of_separate_tables_is_not_a_dataset() {
3974        let dir = tempfile::tempdir().unwrap();
3975        write(
3976            dir.path(),
3977            "circuits.parquet",
3978            &["circuit_id", "lat", "lng"],
3979        );
3980        write(
3981            dir.path(),
3982            "drivers.parquet",
3983            &["driver_id", "code", "nationality"],
3984        );
3985        write(
3986            dir.path(),
3987            "laps.parquet",
3988            &["lap", "position", "time_millis"],
3989        );
3990
3991        assert_eq!(
3992            classify_directory(dir.path()),
3993            EntryKind::MultiFile,
3994            "the filenames alone still say multi"
3995        );
3996        let entry = measured(dir.path());
3997        assert_eq!(
3998            entry.kind,
3999            EntryKind::Directory,
4000            "reading the footers says otherwise"
4001        );
4002        assert_eq!(
4003            entry.rows, None,
4004            "a sum across separate tables is not a row count"
4005        );
4006        assert_eq!(
4007            entry.cols,
4008            Some(9),
4009            "the union of what the directory holds is still a true answer to what is in it"
4010        );
4011        assert_eq!(entry.label(), "3 parquet", "and the label counts the files");
4012    }
4013
4014    /// The rows of one table split across files, which is what `multi` is for.
4015    /// The directories the old threshold took as one table and nesting does not.
4016    ///
4017    /// Two files that each bring a column the other lacks — a renamed column is the
4018    /// everyday case — scored two thirds against a bar of a half, so they opened as one
4019    /// table and the union carried both spellings with nulls under each. Nothing datui
4020    /// can see tells that apart from two tables that share most of their columns, which
4021    /// is why the number moved rather than the question.
4022    ///
4023    /// The directory is not refused. It is a place to look inside, and the row inside it
4024    /// opens the union anyway.
4025    #[test]
4026    fn a_directory_whose_files_each_bring_a_column_is_a_place_to_look_inside() {
4027        let dir = tempfile::tempdir().unwrap();
4028        write(dir.path(), "old.parquet", &["id", "ts", "amount"]);
4029        write(dir.path(), "new.parquet", &["id", "ts", "amt"]);
4030
4031        assert_eq!(
4032            classify_directory(dir.path()),
4033            EntryKind::MultiFile,
4034            "the names alone still say two Parquet files"
4035        );
4036        let entry = measured(dir.path());
4037        assert_eq!(
4038            entry.kind,
4039            EntryKind::Directory,
4040            "and the footers say neither file's columns are in the other's"
4041        );
4042        assert_eq!(entry.label(), "2 parquet", "which the label still reports");
4043        assert_eq!(entry.rows, None, "a sum over two tables is not a number");
4044    }
4045
4046    #[test]
4047    fn a_directory_of_one_table_stays_a_dataset() {
4048        let dir = tempfile::tempdir().unwrap();
4049        for part in 0..3 {
4050            write(
4051                dir.path(),
4052                &format!("part-0000{part}.parquet"),
4053                &["id", "ts", "amount"],
4054            );
4055        }
4056
4057        let entry = measured(dir.path());
4058        assert_eq!(entry.kind, EntryKind::MultiFile);
4059        assert_eq!(entry.rows, Some(3));
4060        assert_eq!(entry.cols, Some(3));
4061    }
4062
4063    /// A dataset whose columns changed over time is still one dataset. This is the
4064    /// case a rule about shared columns gets wrong: the older files have a third of
4065    /// what the newest one does.
4066    #[test]
4067    fn a_dataset_that_gained_columns_stays_a_dataset() {
4068        let dir = tempfile::tempdir().unwrap();
4069        write(dir.path(), "2009.parquet", &["id", "ts"]);
4070        write(dir.path(), "2015.parquet", &["id", "ts", "fee"]);
4071        write(
4072            dir.path(),
4073            "2025.parquet",
4074            &["id", "ts", "fee", "witness", "address", "value"],
4075        );
4076
4077        let entry = measured(dir.path());
4078        assert_eq!(entry.kind, EntryKind::MultiFile);
4079        assert_eq!(entry.rows, Some(3));
4080        assert_eq!(
4081            entry.columns,
4082            vec!["id", "ts", "fee", "witness", "address", "value"],
4083            "every column any file has, in the order they first appear — not the \
4084             2009 shape"
4085        );
4086        assert_eq!(entry.cols, Some(6), "and the count is of those");
4087    }
4088
4089    /// The same, for a hive tree: the row is the dataset's columns, not one
4090    /// partition's.
4091    #[test]
4092    fn a_hive_dataset_that_gained_columns_reports_all_of_them() {
4093        let dir = tempfile::tempdir().unwrap();
4094        for (part, columns) in [
4095            ("year=2009", &["id", "ts"][..]),
4096            ("year=2025", &["id", "ts", "address"][..]),
4097        ] {
4098            let sub = dir.path().join(part);
4099            std::fs::create_dir_all(&sub).unwrap();
4100            write(&sub, "part-0.parquet", columns);
4101        }
4102
4103        let entry = measured(dir.path());
4104        assert_eq!(entry.kind, EntryKind::Hive);
4105        assert_eq!(entry.columns, vec!["id", "ts", "address"]);
4106        assert_eq!(entry.cols, Some(4), "and the partition column `year`");
4107    }
4108
4109    /// Past the counting limit the columns come from a spread of the directory rather
4110    /// than its head, because a directory written over time is narrowest at the start.
4111    #[test]
4112    fn a_directory_too_large_to_count_still_reports_the_columns_it_gained() {
4113        let dir = tempfile::tempdir().unwrap();
4114        for part in 0..MAX_FOOTERS_PER_DATASET + 1 {
4115            let mut columns = vec!["id".to_string(), "ts".to_string()];
4116            if part > MAX_FOOTERS_PER_DATASET / 2 {
4117                columns.push("address".to_string());
4118            }
4119            let refs: Vec<&str> = columns.iter().map(String::as_str).collect();
4120            write(dir.path(), &format!("part-{part:03}.parquet"), &refs);
4121        }
4122
4123        let entry = measured(dir.path());
4124        assert_eq!(entry.kind, EntryKind::MultiFile, "still one table");
4125        assert_eq!(entry.rows, None, "too many files to count");
4126        assert!(
4127            entry.columns.contains(&"address".to_string()),
4128            "the column the dataset gained is in the row: {:?}",
4129            entry.columns
4130        );
4131    }
4132
4133    /// A lake table's data files agree on a schema, so the one-table rule says `multi`
4134    /// and is right about the schema and wrong about the rows: the files a delete
4135    /// tombstoned are still on disk, every rewritten version is here together, and
4136    /// compaction leaves both sides in place.
4137    #[test]
4138    fn a_lake_table_is_not_a_directory_of_parquet_files() {
4139        for (marker, expected) in [
4140            ("_delta_log", EntryKind::Delta),
4141            (".hoodie", EntryKind::Hudi),
4142        ] {
4143            let dir = tempfile::tempdir().unwrap();
4144            write(dir.path(), "part-0.parquet", &["id", "amount"]);
4145            write(dir.path(), "part-1.parquet", &["id", "amount"]);
4146            write(dir.path(), "part-2.parquet", &["id", "amount"]);
4147            let log = dir.path().join(marker);
4148            std::fs::create_dir_all(&log).unwrap();
4149            std::fs::write(log.join("00000000000000000000.json"), b"{}").unwrap();
4150
4151            assert_eq!(
4152                classify_directory(dir.path()),
4153                expected,
4154                "{marker} says what this directory is"
4155            );
4156            let entry = measured(dir.path());
4157            assert_eq!(entry.kind, expected);
4158            assert_eq!(
4159                entry.rows, None,
4160                "and no row count is claimed for it: summing the footers would count \
4161                 the rows the log says are gone"
4162            );
4163            assert!(!entry.kind.is_dataset(), "it does not open as one table");
4164        }
4165    }
4166
4167    /// Iceberg's marker is a plain name, so it takes the whole shape rather than the
4168    /// name alone.
4169    #[test]
4170    fn an_iceberg_root_is_metadata_beside_data() {
4171        let iceberg = tempfile::tempdir().unwrap();
4172        let data = iceberg.path().join("data");
4173        let metadata = iceberg.path().join("metadata");
4174        std::fs::create_dir_all(&data).unwrap();
4175        std::fs::create_dir_all(&metadata).unwrap();
4176        write(&data, "00000-0-abc.parquet", &["id", "amount"]);
4177        write(&data, "00001-0-def.parquet", &["id", "amount"]);
4178        std::fs::write(metadata.join("v2.metadata.json"), b"{}").unwrap();
4179        std::fs::write(metadata.join("snap-1.avro"), b"x").unwrap();
4180        assert_eq!(classify_directory(iceberg.path()), EntryKind::Iceberg);
4181
4182        // A directory that merely has those names is not a table.
4183        let plain = tempfile::tempdir().unwrap();
4184        std::fs::create_dir_all(plain.path().join("data")).unwrap();
4185        std::fs::create_dir_all(plain.path().join("metadata")).unwrap();
4186        std::fs::write(plain.path().join("metadata/notes.txt"), b"x").unwrap();
4187        assert_eq!(
4188            classify_directory(plain.path()),
4189            EntryKind::Directory,
4190            "no *.metadata.json, so no Iceberg table"
4191        );
4192
4193        let no_data = tempfile::tempdir().unwrap();
4194        let metadata = no_data.path().join("metadata");
4195        std::fs::create_dir_all(&metadata).unwrap();
4196        std::fs::write(metadata.join("v1.metadata.json"), b"{}").unwrap();
4197        write(no_data.path(), "part-0.parquet", &["id"]);
4198        write(no_data.path(), "part-1.parquet", &["id"]);
4199        assert_eq!(
4200            classify_directory(no_data.path()),
4201            EntryKind::MultiFile,
4202            "metadata with no data/ beside it is somebody's directory, not a table root"
4203        );
4204    }
4205
4206    /// A single file counts its columns the same way a directory does, and both count
4207    /// what opening it shows.
4208    ///
4209    /// `enrich_parquet` read `schema_descr.columns()`, which is the leaf list — so a file
4210    /// with one struct of two fields said `columns 3` above a schema list of two, and a
4211    /// directory holding only that file said something different again.
4212    #[test]
4213    fn a_file_and_a_directory_of_it_count_the_same_columns() {
4214        let dir = tempfile::tempdir().unwrap();
4215        write_nested(dir.path(), "one.parquet", "inputs", &["address", "value"]);
4216
4217        let mut file = Entry::new(dir.path().join("one.parquet"), EntryKind::File);
4218        enrich(&mut file);
4219        assert_eq!(
4220            file.cols,
4221            Some(2),
4222            "`id` and `inputs`, which is what opening it shows: {:?}",
4223            file.columns
4224        );
4225        assert!(
4226            file.columns.iter().any(|c| c == "inputs.address"),
4227            "the leaves are still searchable: {:?}",
4228            file.columns
4229        );
4230
4231        write_nested(dir.path(), "two.parquet", "inputs", &["address", "value"]);
4232        let directory = measured(dir.path());
4233        assert_eq!(directory.kind, EntryKind::MultiFile);
4234        assert_eq!(
4235            directory.cols, file.cols,
4236            "and a directory of them says the same number"
4237        );
4238    }
4239
4240    /// Dots in a column's own name are not nesting, and are not counted as if they were.
4241    ///
4242    /// The obvious fix for the leaf problem — split the dotted path and count the roots —
4243    /// gets this wrong: `user.id` and `user.name` written by a flattening export are two
4244    /// columns, not one. The schema says which is which; the string cannot.
4245    #[test]
4246    fn a_dotted_column_name_is_its_own_column() {
4247        let dir = tempfile::tempdir().unwrap();
4248        write(dir.path(), "flat.parquet", &["id", "user.id", "user.name"]);
4249
4250        let mut file = Entry::new(dir.path().join("flat.parquet"), EntryKind::File);
4251        enrich(&mut file);
4252        assert_eq!(file.cols, Some(3), "three columns: {:?}", file.columns);
4253    }
4254
4255    /// A directory whose files encode the same nested column differently counts it once.
4256    ///
4257    /// The union is over leaf paths, and the same nested column written by parquet-mr and
4258    /// by Arrow gives different leaves — so the row reported roughly twice the width of a
4259    /// directory `is_one_table` had just called one dataset. Counted from each file's own
4260    /// root fields, the two spellings are one `inputs` whatever the leaves under it are.
4261    #[test]
4262    fn a_writer_change_does_not_double_the_column_count() {
4263        let dir = tempfile::tempdir().unwrap();
4264        write_nested(dir.path(), "old.parquet", "inputs", &["address"]);
4265        // The same column, one field wider, as a later writer left it.
4266        write_nested(dir.path(), "new.parquet", "inputs", &["address", "value"]);
4267
4268        let entry = measured(dir.path());
4269        assert_eq!(entry.kind, EntryKind::MultiFile, "still one table");
4270        assert_eq!(
4271            entry.cols,
4272            Some(2),
4273            "one `inputs`, not one per shape of it: {:?}",
4274            entry.columns
4275        );
4276        assert!(
4277            entry.columns.len() > 2,
4278            "while every leaf stays searchable: {:?}",
4279            entry.columns
4280        );
4281    }
4282
4283    /// A count read from a spread of a directory rather than all of it says it is a
4284    /// floor.
4285    #[test]
4286    fn a_sampled_column_count_says_it_is_a_floor() {
4287        let dir = tempfile::tempdir().unwrap();
4288        for part in 0..MAX_FOOTERS_PER_DATASET * 2 {
4289            write(
4290                dir.path(),
4291                &format!("part-{part:04}.parquet"),
4292                &["id", "ts"],
4293            );
4294        }
4295        let entry = measured(dir.path());
4296        assert_eq!(entry.rows, None, "too many files to count");
4297        assert!(entry.cols.is_some(), "but the width is still worth having");
4298        assert!(
4299            entry.cols_sampled,
4300            "and it is marked as the floor it is, not presented as a total"
4301        );
4302
4303        // A directory small enough to read every footer of claims no such thing.
4304        let small = tempfile::tempdir().unwrap();
4305        write(small.path(), "a.parquet", &["id", "ts"]);
4306        write(small.path(), "b.parquet", &["id", "ts"]);
4307        assert!(!measured(small.path()).cols_sampled);
4308    }
4309
4310    /// The files a directory offers come back in order, whatever order it was
4311    /// written in.
4312    ///
4313    /// Every caller reads order as meaning something — `sample_footers` takes the ends
4314    /// and the middle, and the union of the columns is built in the order the files
4315    /// appear. Unsorted, "the last file" was whichever one the filesystem happened to
4316    /// return last, which on the filesystems that return creation order is the one
4317    /// written first as often as not.
4318    #[test]
4319    fn the_files_a_directory_offers_come_back_in_order() {
4320        let dir = tempfile::tempdir().unwrap();
4321        for name in ["c.parquet", "a.parquet", "d.parquet", "b.parquet"] {
4322            write(dir.path(), name, &["id"]);
4323        }
4324        let files = parquet_files_under(dir.path());
4325        let names: Vec<String> = files
4326            .iter()
4327            .map(|p| p.file_name().unwrap().to_string_lossy().into_owned())
4328            .collect();
4329        assert_eq!(
4330            names,
4331            vec!["a.parquet", "b.parquet", "c.parquet", "d.parquet"],
4332            "sorted, not in the order the directory was written"
4333        );
4334    }
4335
4336    /// A directory past the budget still says so, and the files it keeps are the
4337    /// directory's first rather than the listing's.
4338    ///
4339    /// The ordering itself is `the_files_a_directory_offers_come_back_in_order`'s to
4340    /// prove: a directory read may return sorted entries of its own accord, so an
4341    /// assertion here about order could hold for the wrong reason. What this pins is
4342    /// *which* files survive the cap, and that the cap still says "too many to count".
4343    #[test]
4344    fn a_directory_past_the_budget_keeps_the_directories_first_files() {
4345        let dir = tempfile::tempdir().unwrap();
4346        for part in 0..MAX_FOOTERS_PER_DATASET * 3 {
4347            write(dir.path(), &format!("part-{part:04}.parquet"), &["id"]);
4348        }
4349        let files = parquet_files_under(dir.path());
4350
4351        assert_eq!(
4352            files.len(),
4353            MAX_FOOTERS_PER_DATASET + 1,
4354            "one past the budget, which is what says there are too many to count"
4355        );
4356        let names: Vec<String> = files
4357            .iter()
4358            .map(|p| p.file_name().unwrap().to_string_lossy().into_owned())
4359            .collect();
4360        let expected: Vec<String> = (0..=MAX_FOOTERS_PER_DATASET)
4361            .map(|part| format!("part-{part:04}.parquet"))
4362            .collect();
4363        // Not "sorted", which a directory read may be of its own accord, but the
4364        // directory's own first sixty-five. Sorting after truncating gives sixty-five
4365        // sorted names from wherever the read began, which is a different set.
4366        assert_eq!(
4367            names, expected,
4368            "the directory's first files, not the listing's"
4369        );
4370    }
4371
4372    /// The log is named rather than looked for, so a table's own data files cannot
4373    /// crowd it out of the listing however many of them there are.
4374    #[test]
4375    fn a_lake_table_is_recognized_among_its_data_files() {
4376        let dir = tempfile::tempdir().unwrap();
4377        for part in 0..32 {
4378            write(dir.path(), &format!("part-{part:03}.parquet"), &["id"]);
4379        }
4380        std::fs::create_dir_all(dir.path().join("_delta_log")).unwrap();
4381        assert_eq!(classify_directory(dir.path()), EntryKind::Delta);
4382    }
4383
4384    /// Past the counting limit the row count is out of reach, but whether the directory
4385    /// is one table is not — and a directory of a hundred tables is exactly where reading
4386    /// them as one costs most.
4387    #[test]
4388    fn a_directory_too_large_to_count_is_still_checked() {
4389        let dir = tempfile::tempdir().unwrap();
4390        for table in 0..MAX_FOOTERS_PER_DATASET + 1 {
4391            write(
4392                dir.path(),
4393                &format!("table_{table:03}.parquet"),
4394                &[&format!("{table}_id"), &format!("{table}_value")],
4395            );
4396        }
4397
4398        let entry = measured(dir.path());
4399        assert_eq!(entry.kind, EntryKind::Directory);
4400        assert_eq!(entry.rows, None, "too many files to count either way");
4401    }
4402
4403    /// The same directory size, but one table split across it.
4404    #[test]
4405    fn a_large_directory_of_one_table_stays_a_dataset() {
4406        let dir = tempfile::tempdir().unwrap();
4407        for part in 0..MAX_FOOTERS_PER_DATASET + 1 {
4408            write(
4409                dir.path(),
4410                &format!("part-{part:05}.parquet"),
4411                &["id", "ts"],
4412            );
4413        }
4414
4415        let entry = measured(dir.path());
4416        assert_eq!(entry.kind, EntryKind::MultiFile);
4417    }
4418
4419    /// Searching the home screen by column should still find a directory that holds one,
4420    /// even once the directory is no longer offered as a single table.
4421    #[test]
4422    fn a_downgraded_directory_keeps_every_column_its_files_have() {
4423        let dir = tempfile::tempdir().unwrap();
4424        write(
4425            dir.path(),
4426            "circuits.parquet",
4427            &["circuit_id", "lat", "lng"],
4428        );
4429        write(
4430            dir.path(),
4431            "drivers.parquet",
4432            &["driver_id", "code", "nationality"],
4433        );
4434        // And one a level down, so "every column its files have" is a claim about more
4435        // than the directory's own: the row's width is its own files, its names are
4436        // everything under it, and a fixture with no subdirectory cannot tell those
4437        // apart.
4438        let seasons = dir.path().join("seasons");
4439        std::fs::create_dir_all(&seasons).unwrap();
4440        write(&seasons, "2024.parquet", &["season_year", "round"]);
4441
4442        let entry = measured(dir.path());
4443        assert_eq!(entry.kind, EntryKind::Directory);
4444        for column in [
4445            "circuit_id",
4446            "lat",
4447            "lng",
4448            "driver_id",
4449            "code",
4450            "nationality",
4451            "season_year",
4452            "round",
4453        ] {
4454            assert!(
4455                entry.columns.iter().any(|c| c == column),
4456                "{column} in {:?}",
4457                entry.columns
4458            );
4459        }
4460    }
4461
4462    #[test]
4463    fn a_name_no_reader_takes_is_refused_before_opening() {
4464        let refused = |name: &str| unreadable_by_name(std::path::Path::new(name));
4465        assert!(refused("gs://b/ml/onnx/pipeline_rf.onnx"));
4466        assert!(refused("model.onnx.gz"));
4467        assert!(refused("README.md"));
4468        for readable in [
4469            "a.csv",
4470            "a.CSV",
4471            "a.csv.gz",
4472            "a.parquet",
4473            "a.xlsx",
4474            "data.gz",
4475            "part-0000",
4476        ] {
4477            assert!(!refused(readable), "{readable}");
4478        }
4479    }
4480}