datui_lib/discover.rs
1//! Dataset discovery for the home screen.
2//!
3//! This is deliberately *not* a catalogue. Nothing here is persisted: every listing
4//! is computed from the filesystem when asked for, and forgotten when the session
5//! ends. The only state datui keeps between runs is a list of recently opened paths.
6//!
7//! Discovery is also deliberately shallow. Interesting datasets tend to live on
8//! mounts — network filesystems, spinning disks, hive trees with a hundred thousand
9//! partition files — so a recursive walk would make the home screen slowest exactly
10//! where the data is most interesting. Every function here scans one directory level
11//! and stops.
12
13use std::path::{Path, PathBuf};
14
15/// Compression suffixes that may follow a data extension (`sales.csv.gz`).
16const COMPRESSION_EXTENSIONS: &[&str] = &["gz", "bz2", "xz", "zst", "zstd"];
17
18/// Upper bound on entries read from a single directory, so a pathological directory
19/// cannot hang the UI.
20pub const MAX_ENTRIES_PER_DIR: usize = 5_000;
21
22/// What a home-screen row represents.
23#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
24#[serde(rename_all = "lowercase")]
25pub enum EntryKind {
26 /// A single data file.
27 File,
28 /// A file datui has no reader for. Hidden on the home screen until `Ctrl+A` shows
29 /// it, dimmed, so a directory can be seen as it is.
30 Other,
31 /// A directory of `key=value` partitions — one dataset, not a tree to walk.
32 Hive,
33 /// A directory of similarly-shaped data files, openable as one table.
34 MultiFile,
35 /// A Delta Lake table: `_delta_log/` beside the data files.
36 Delta,
37 /// An Apache Iceberg table: `metadata/` holding the snapshots, `data/` the files.
38 Iceberg,
39 /// An Apache Hudi table: `.hoodie/` holding the timeline.
40 Hudi,
41 /// An ordinary directory, to descend into.
42 Directory,
43 /// Somewhere remote that has not been looked at yet. Classifying it would mean
44 /// reading it, which is the call that blocks when the network is gone — so it is
45 /// offered as openable and left unlabelled rather than guessed at.
46 ///
47 /// Also what a kind this build does not recognize reads back as. The dataset index
48 /// is one JSON map, and a value an older datui cannot parse would otherwise fail the
49 /// whole map and discard every dataset fact it had — see `CLASSIFIER_VERSION`, which
50 /// is why a new kind can appear in a file an older build reads.
51 #[serde(other)]
52 Unknown,
53}
54
55/// Bumped whenever a build starts classifying something differently.
56///
57/// A cached kind is the only thing a remote row has to go on — it was never stat'ed, and
58/// classifying it means reading it — so it is restored rather than re-derived. That makes
59/// it a way for an answer this build would not give to come back: a Delta root measured
60/// before lake tables were recognized was recorded as `multifile`, and restoring that
61/// opens it as one table, which is the whole of #237 read back off disk.
62///
63/// So the kind is restored only when the build that wrote it classified the way this one
64/// does. Everything else in the record — rows, columns, cost — is a measurement rather
65/// than a judgement, and survives.
66///
67/// 7: a file with no extension is a SQLite database when its first bytes say so.
68///
69/// 6: on a local disk, a file with no extension is data when its first bytes carry a
70/// Parquet, Arrow, Avro or ORC signature, so a directory of Spark part files a 5 called
71/// `dir` is one dataset. Unidentified ones are `unnamed` rather than `not_read`.
72///
73/// 5: a directory of CSV or NDJSON is judged by the names at the front of its files, the
74/// way a directory of Parquet is judged by its footers — so one a 4 called `multi` on its
75/// filenames alone may be a place to look inside. A cached kind is restored without
76/// looking again, so a record written by 4 would keep the answer this build exists to
77/// correct (#275 follow-up).
78///
79/// 4: a directory's row carries what one listing of it found, beside its kind, and the
80/// two are restored together — a record written by 3 carries the kind and not the count,
81/// and a row given a kind from the cache is never looked into again (#275, phase 2).
82///
83/// 3: one listing instead of a probe of the first eight entries, formats instead of
84/// extension strings, and one bookkeeping predicate. A directory of `.arrow` beside
85/// `.ipc` was `dir` and is now one dataset; a directory whose ninth entry decided it was
86/// answered by whatever the filesystem returned first (#275, phase 1).
87pub const CLASSIFIER_VERSION: u32 = 7;
88
89impl EntryKind {
90 /// Short label shown next to the entry name.
91 pub fn label(self) -> &'static str {
92 match self {
93 // A file with no reader says nothing: it is dimmed, and Enter shows its
94 // bytes. `binary` would be wrong for the README or log it often is.
95 EntryKind::File | EntryKind::Other => "",
96 EntryKind::Hive => "hive",
97 EntryKind::MultiFile => "multi",
98 EntryKind::Delta => "delta",
99 EntryKind::Iceberg => "iceberg",
100 EntryKind::Hudi => "hudi",
101 EntryKind::Directory => "dir",
102 EntryKind::Unknown => "",
103 }
104 }
105
106 /// Whether selecting this entry opens a dataset rather than navigating.
107 ///
108 /// An unexamined remote path counts: datui can open an object-store prefix or a
109 /// hive directory directly, and descending into one is not possible anyway
110 /// without the listing this deliberately has not fetched.
111 pub fn is_dataset(self) -> bool {
112 !matches!(self, EntryKind::Directory | EntryKind::Other) && !self.is_lake_table()
113 }
114
115 /// Whether this row is *known* to be a dataset.
116 ///
117 /// [`EntryKind::is_dataset`] answers "may this be opened", and a row nothing has
118 /// looked into answers yes: it is offered, and looked into before it is acted on.
119 /// This one answers "is this a dataset", which such a row cannot answer at all —
120 /// and that is the question counting asks. A directory of two hundred subdirectories
121 /// nobody has looked into is not two hundred datasets.
122 pub fn is_known_dataset(self) -> bool {
123 self != EntryKind::Unknown && self.is_dataset()
124 }
125
126 /// A Delta, Iceberg or Hudi table root: a log beside the data files that says which
127 /// of them are live, which datui does not read yet.
128 pub fn is_lake_table(self) -> bool {
129 matches!(
130 self,
131 EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi
132 )
133 }
134
135 /// The format's name for prose. `label` is the row's chip, and is lowercase like
136 /// `hive` and `multi` beside it.
137 pub fn lake_name(self) -> Option<&'static str> {
138 match self {
139 EntryKind::Delta => Some("Delta"),
140 EntryKind::Iceberg => Some("Iceberg"),
141 EntryKind::Hudi => Some("Hudi"),
142 _ => None,
143 }
144 }
145}
146
147/// What one listing of a directory found in it, counted rather than judged.
148///
149/// The label a directory's row carries comes from here, so it says what is inside rather
150/// than what `Enter` will do with it. A count that is wrong then costs a reader nothing:
151/// `12 parquet` is true of a directory whether or not its files are one table.
152#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
153pub struct Holds {
154 /// Data files by format, commonest first. The name is [`crate::FileFormat::name`],
155 /// kept as a string so a record written by one build reads in the next.
156 #[serde(default)]
157 pub formats: Vec<(String, usize)>,
158 /// Subdirectories, partitions among them. Development builds of 0.4.0 wrote it as
159 /// `folders`; the alias keeps a cache from one of those readable.
160 #[serde(default, alias = "folders")]
161 pub directories: usize,
162 /// `key=value` subdirectories, which are also counted in `directories`.
163 #[serde(default)]
164 pub partitions: usize,
165 /// Files datui has no reader for: a README, a script, a notebook. Neither data nor a
166 /// writer's own, and without a count of their own they were in nothing — a directory
167 /// of twenty of them read `dir` with no line at all, the same as an empty one.
168 #[serde(default)]
169 pub not_read: usize,
170 /// Files with no extension. No name says what they are, so they are neither data
171 /// nor `not_read`: Spark and GBIF write their part files this way, and the open
172 /// reads them by their bytes.
173 #[serde(default)]
174 pub unnamed: usize,
175 /// Entries skipped as a writer's own, and the first few by name for the pane.
176 #[serde(default)]
177 pub skipped: usize,
178 #[serde(default)]
179 pub skipped_names: Vec<String>,
180 /// Whether the listing stopped at [`MAX_ENTRIES_PER_DIR`], so every count is a
181 /// floor. Shown as `5000+`.
182 #[serde(default)]
183 pub truncated: bool,
184 /// A Hugging Face DatasetDict saved with `save_to_disk`: `dataset_dict.json`
185 /// beside directories, its splits. Read as Arrow, one split at a time.
186 #[serde(default)]
187 pub dataset_dict: bool,
188}
189
190/// How many skipped names are kept for the pane. Enough to recognise the convention.
191pub(crate) const SKIPPED_NAMES_SHOWN: usize = 4;
192
193/// A name cut to `width`, keeping both ends and marking the middle.
194pub(crate) fn shorten(name: &str, width: usize) -> String {
195 let chars: Vec<char> = name.chars().collect();
196 if chars.len() <= width {
197 return name.to_string();
198 }
199 let ellipsis = crate::glyphs::get().ellipsis;
200 let room = width.saturating_sub(ellipsis.chars().count());
201 let head = room.div_ceil(2);
202 let tail = room - head;
203 format!(
204 "{}{ellipsis}{}",
205 chars[..head].iter().collect::<String>(),
206 chars[chars.len() - tail..].iter().collect::<String>()
207 )
208}
209
210impl Holds {
211 /// Data files of every format.
212 pub fn data_files(&self) -> usize {
213 self.formats.iter().map(|(_, n)| n).sum()
214 }
215
216 /// The one format this directory holds, when it holds exactly one.
217 pub fn one_format(&self) -> Option<&str> {
218 match self.formats.as_slice() {
219 [(name, _)] => Some(name),
220 _ => None,
221 }
222 }
223
224 /// The weights' format and file count, when this directory is a model: weights of
225 /// one format with nothing beside them but JSON. See [`is_model_directory`].
226 pub fn model_weights(&self) -> Option<(&str, usize)> {
227 if !is_model_directory(counts_names(self)) {
228 return None;
229 }
230 self.formats
231 .iter()
232 .find(|(name, _)| is_weights(name))
233 .map(|(name, count)| (name.as_str(), *count))
234 }
235
236 /// The label a directory's row carries when its kind does not name itself: `12
237 /// parquet`, `mixed`, or `dir` for a directory with no data directly inside.
238 pub fn label(&self) -> String {
239 let more = if self.truncated { "+" } else { "" };
240 // A model's weights beside its config and tokenizer JSON: the directory is the
241 // model, and its label says so rather than `mixed`.
242 if let Some((name, count)) = self.model_weights() {
243 return format!("{count}{more} {name}");
244 }
245 match self.formats.as_slice() {
246 // `dir` says there is no data file inside. A listing cut short cannot say
247 // that — it found none among the entries it read, and more files can
248 // unmake it, which is what separates this from `mixed`.
249 // A directory of directories counts them: `3 dirs` says where to go next.
250 [] if self.directories == 1 => format!("1 dir{more}"),
251 [] if self.directories > 1 => format!("{}{more} dirs", self.directories),
252 [] => format!("dir{more}"),
253 // The `+` hedges the whole claim, not only the number: past the cap a
254 // second format may be among the entries that were not read, so `5000+
255 // parquet` and `mixed` are both answers this directory can give depending on
256 // the order it came back in. What is certain is that five thousand Parquet
257 // files are in there.
258 [(name, count)] => format!("{count}{more} {name}"),
259 // No `+`: `mixed` is not a count, and more files cannot unmake it. The
260 // pane's line carries the qualifier on each number it does report.
261 _ => "mixed".to_string(),
262 }
263 }
264
265 /// Whether nothing has been counted here: a file, or a directory nothing has looked
266 /// into. A directory that was looked into and found empty is not this — it has no
267 /// formats either, and `dir` is the right word for both.
268 pub fn is_empty(&self) -> bool {
269 self.formats.is_empty()
270 && self.directories == 0
271 && self.skipped == 0
272 && self.not_read == 0
273 && self.unnamed == 0
274 // Every field, including the two that are counted elsewhere as well: a
275 // partition is a directory and a skipped name is one of `skipped`, so on both
276 // routes today these are implied. This is a `skip_serializing_if` and the
277 // guard that stops a placeholder erasing a row's count, and neither should
278 // turn on an invariant two other functions have to keep.
279 && self.partitions == 0
280 && self.skipped_names.is_empty()
281 && !self.dataset_dict
282 // A listing cut short before it found anything still says something: that
283 // what it found is not all there is. Without this the row falls back to its
284 // kind and reads `dir`, where `label` would have said `dir+`.
285 && !self.truncated
286 }
287
288 /// The `contains` line in the details pane: the data files by format, the
289 /// directories, and the partitions — what there is to open, and nothing else.
290 ///
291 /// Files datui cannot read and a writer's markers are left out. Counted here they
292 /// read as a warning ("10 not read") about a directory with nothing wrong in it;
293 /// inside the directory, a row of its own says what is not shown.
294 pub fn line(&self, with_partitions: bool) -> Option<String> {
295 let more = if self.truncated { "+" } else { "" };
296 let mut parts: Vec<String> = self
297 .formats
298 .iter()
299 .map(|(name, count)| format!("{count}{more} {name}"))
300 .collect();
301 // Partitions are directories too, and counted in `directories`; naming both would
302 // count them twice. What is left is the directories that are not partitions.
303 let plain = self.directories.saturating_sub(self.partitions);
304 if plain > 0 {
305 let word = if plain == 1 {
306 "directory"
307 } else {
308 "directories"
309 };
310 parts.push(format!("{plain}{more} {word}"));
311 }
312 if with_partitions && self.partitions > 0 {
313 let word = if self.partitions == 1 {
314 "partition"
315 } else {
316 "partitions"
317 };
318 parts.push(format!("{}{more} {word}", self.partitions));
319 }
320 (!parts.is_empty()).then(|| parts.join(" · "))
321 }
322}
323
324impl Entry {
325 /// Whether Enter on this row lists the tables inside it: a file of several that is
326 /// no table itself. A file that opens one of its tables (a workbook's first sheet)
327 /// opens it, and a file a spec reads as several variants is one table too (each row
328 /// a variant, with a `type` column), which Enter opens; → lists them.
329 pub fn enter_lists_tables(&self) -> bool {
330 self.kind == EntryKind::File
331 && self.cost.tables.is_some_and(|n| n > 1)
332 && !self.cost.opens_one
333 && self.format_spec.is_none()
334 }
335
336 /// Whether the home screen leaves this row out until Ctrl+A: a file datui cannot
337 /// open, or a database's own table.
338 pub fn hidden_by_default(&self) -> bool {
339 self.kind == EntryKind::Other || self.table.as_ref().is_some_and(|t| t.internal)
340 }
341
342 /// The short label beside a row's name: what it holds, rather than what `Enter`
343 /// will do with it.
344 ///
345 /// A directory that has been looked into is described by the count — `12 parquet`,
346 /// `mixed`, `dir` — and a lake table or a hive root by the format's own name, which
347 /// is the thing it is. A row nothing has looked into has only its kind to go on.
348 pub fn label(&self) -> std::borrow::Cow<'static, str> {
349 match self.kind {
350 // See `opens_whole_directory`: the one row whose label would be about a
351 // different set of files than the row itself.
352 _ if self.opens_whole_directory => "".into(),
353 EntryKind::Directory | EntryKind::MultiFile if !self.holds.is_empty() => {
354 self.holds.label().into()
355 }
356 EntryKind::File if self.format_spec.is_some() => {
357 self.format_spec.clone().unwrap_or_default().into()
358 }
359 EntryKind::File if self.cost.tables.is_some() => {
360 let n = self.cost.tables.unwrap_or_default();
361 format!("{n} {}", if n == 1 { "table" } else { "tables" }).into()
362 }
363 // A file named for what it holds rather than by its file name, as a
364 // collection names one: its format, which the name no longer says.
365 EntryKind::File
366 if crate::FileFormat::from_path(Path::new(&self.name)).is_none()
367 && crate::FileFormat::from_path(&self.path).is_some() =>
368 {
369 crate::FileFormat::from_path(&self.path)
370 .map(crate::FileFormat::name)
371 .unwrap_or_default()
372 .into()
373 }
374 kind => kind.label().into(),
375 }
376 }
377}
378
379/// One row on the home screen.
380#[derive(Debug, Clone)]
381pub struct Entry {
382 pub path: PathBuf,
383 pub kind: EntryKind,
384 /// Display name — the file or directory name, not the full path.
385 pub name: String,
386 /// Size in bytes. For a multi-file or hive dataset this is the sum of the files
387 /// actually inspected, so it is a floor rather than an exact total.
388 pub size: Option<u64>,
389 pub modified: Option<std::time::SystemTime>,
390 /// Row count, when it can be had without reading data (Parquet footers only).
391 pub rows: Option<usize>,
392 /// Column count, same caveat.
393 pub cols: Option<usize>,
394 /// Whether `cols` came from a spread of the directory rather than all of it. A
395 /// directory past the footer budget is read at its ends and its middle, so the count
396 /// is a floor: shown as `6+` rather than `6`, the way the row count is already shown
397 /// as `?` when it is out of reach.
398 pub cols_sampled: bool,
399 /// Column names, when they were free to obtain. A Parquet footer carries them
400 /// alongside the row count, so knowing what is *in* a dataset costs nothing
401 /// beyond knowing how big it is.
402 pub columns: Vec<String>,
403 /// What opening this will cost: where it lives, how it is stored, how it is laid
404 /// out. All of it derived from bytes datui already reads.
405 pub cost: Cost,
406 /// What one listing of it found, for a directory. Empty for a file, and for a
407 /// directory nothing has looked into.
408 pub holds: Holds,
409 /// Whether this row is the door that opens the directory being browsed, rather than
410 /// something in it.
411 ///
412 /// It carries no label. Every other label counts what is directly inside a directory,
413 /// and this row is the one that reads the whole of it — so `dir` beside `(all
414 /// files)` would say there is no data here while offering to open it, and `2
415 /// parquet` beside it would name two of the twenty it is about to read. The name
416 /// says what it does; the numbers beside it, once measured, say how much.
417 pub opens_whole_directory: bool,
418 /// The format spec that reads this file: its glob names it, or its magic is at
419 /// the front of it.
420 pub format_spec: Option<String>,
421 /// A table inside a file of tables (a SQLite database, a NumPy archive), for the
422 /// rows listed inside one: its path is the file's with the table's name after it,
423 /// which nothing on disk has.
424 pub table: Option<TableOf>,
425}
426
427/// What a row inside a file of tables says about its table.
428#[derive(Debug, Clone, PartialEq, Eq)]
429pub struct TableOf {
430 /// The format of the file it is in; `None` for a format spec's variant, whose spec
431 /// [`Entry::format_spec`] names.
432 pub format: Option<crate::FileFormat>,
433 /// What the file calls it: SQLite's `table`, `view`, `virtual` or `shadow`, or a
434 /// NumPy archive's `array`.
435 pub kind: String,
436 /// SQLite's own (its schema, its statistics, a virtual table's shadows): hidden
437 /// like a file datui cannot open until Ctrl+A shows it, and opened like any other.
438 pub internal: bool,
439}
440
441/// What pressing Enter on a dataset will actually cost.
442///
443/// `rows`, `cols` and `size` say what a dataset *is*. None of them say what reading
444/// it will do, and the difference is large: 200 MB of zstd-compressed Parquet is two
445/// gigabytes in memory, and two gigabytes on a hotel-wifi NFS mount is a different
446/// afternoon than two gigabytes on tmpfs.
447///
448/// Every field here comes from something datui already reads — the mount table, and
449/// the same Parquet footer that yields the row count. Nothing here costs an extra
450/// byte of the dataset itself.
451#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
452pub struct Cost {
453 /// Filesystem or URL scheme: `nfs4`, `ext4`, `tmpfs`, `fuse.sshfs`, `s3`.
454 #[serde(default, skip_serializing_if = "Option::is_none")]
455 pub source: Option<String>,
456 /// Bytes once decompressed — what this will occupy, as against what it occupies
457 /// on disk.
458 #[serde(default, skip_serializing_if = "Option::is_none")]
459 pub uncompressed: Option<u64>,
460 /// Compression codec, as the file itself names it.
461 #[serde(default, skip_serializing_if = "Option::is_none")]
462 pub codec: Option<String>,
463 /// Row groups. One enormous row group cannot be read in parallel or skipped
464 /// through; a thousand tiny ones cost more in overhead than they save.
465 #[serde(default, skip_serializing_if = "Option::is_none")]
466 pub row_groups: Option<usize>,
467 /// Partition layout, for a hive dataset.
468 #[serde(default, skip_serializing_if = "Option::is_none")]
469 pub partitions: Option<Partitions>,
470 /// Tables of its own, for a file of tables: one opens, several are listed.
471 #[serde(default, skip_serializing_if = "Option::is_none")]
472 pub tables: Option<usize>,
473 /// Whether a file of several tables opens one of them (a workbook's first sheet),
474 /// so Enter opens it and → lists them, rather than Enter listing them.
475 #[serde(default, skip_serializing_if = "std::ops::Not::not")]
476 pub opens_one: bool,
477 /// An Arrow file that is an IPC stream, which is converted before it is scanned,
478 /// rather than an IPC file, which is scanned where it is. From its first bytes.
479 #[serde(default, skip_serializing_if = "std::ops::Not::not")]
480 pub ipc_stream: bool,
481}
482
483/// How opening a file row will read it. See [`how_read`].
484#[derive(Debug, Clone, Copy, PartialEq, Eq)]
485pub struct HowRead {
486 pub mode: crate::ReadMode,
487 /// A remote file that is downloaded whole before it is read.
488 pub download: bool,
489}
490
491/// How opening `entry` will read it, as [`crate::FileFormat::read_mode`] says for its
492/// format and how it is stored, and whether a remote one is downloaded first. `None`
493/// for anything but a file whose name says its format.
494pub fn how_read(entry: &Entry) -> Option<HowRead> {
495 use crate::Stored;
496 if entry.kind != EntryKind::File {
497 return None;
498 }
499 let stored = if crate::CompressionFormat::from_extension(&entry.path).is_some() {
500 Stored::Compressed { in_memory: false }
501 } else if entry.cost.ipc_stream {
502 Stored::Stream
503 } else {
504 Stored::Plain
505 };
506 let choice = match &entry.format_spec {
507 Some(name) => crate::cli::FormatChoice::Spec(name.clone()),
508 // A table inside a file of tables, at its path inside it (`shop.db/orders`).
509 None if entry.table.is_some() => {
510 crate::cli::FormatChoice::Builtin(entry.table.as_ref().and_then(|t| t.format)?)
511 }
512 None => crate::cli::FormatChoice::Builtin(data_format(&entry.path)?),
513 };
514 let mode = choice.read_mode(stored)?;
515 let download = match crate::source::input_source(&entry.path) {
516 crate::source::InputSource::Local(_) => false,
517 crate::source::InputSource::Http(_) => choice.http_file() == crate::RemoteRead::Downloaded,
518 // A remote Arrow file is not peeked at here, so it is taken for an IPC file.
519 _ => choice.bucket_object(stored) == crate::RemoteRead::Downloaded,
520 };
521 Some(HowRead { mode, download })
522}
523
524/// How a hive dataset is laid out on disk.
525///
526/// The shape of a partitioned dataset is the first thing anyone asks about it, and
527/// the answer is in the directory names — no file needs opening to know it.
528#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
529pub struct Partitions {
530 /// Partition keys, outermost first: `["year", "month"]`.
531 pub keys: Vec<String>,
532 /// Distinct values seen for the outermost key, in sorted order. Bounded, so this
533 /// is what was seen rather than necessarily all there is.
534 pub first_key_values: Vec<String>,
535 /// Directories counted at the outermost level.
536 pub count: usize,
537 /// The count stopped at a limit; there are more.
538 pub more: bool,
539}
540
541impl Entry {
542 /// A plain directory row — somewhere to step into, with nothing read from it.
543 pub fn directory(path: &Path) -> Self {
544 Self::new(path.to_path_buf(), EntryKind::Directory)
545 }
546
547 /// A file entry with a chosen display name, for tests that need a search result
548 /// without running a walk to produce one.
549 pub fn for_test(path: &Path, name: &str) -> Self {
550 let mut entry = Self::new(path.to_path_buf(), EntryKind::File);
551 entry.name = name.to_string();
552 entry
553 }
554
555 pub(crate) fn new(path: PathBuf, kind: EntryKind) -> Self {
556 let name = path
557 .file_name()
558 .map(|n| n.to_string_lossy().into_owned())
559 .unwrap_or_else(|| path.to_string_lossy().into_owned());
560 Self {
561 path,
562 kind,
563 name,
564 size: None,
565 modified: None,
566 rows: None,
567 cols: None,
568 cols_sampled: false,
569 columns: Vec::new(),
570 cost: Cost::default(),
571 holds: Default::default(),
572 opens_whole_directory: false,
573 format_spec: None,
574 table: None,
575 }
576 }
577
578 /// Attach size and mtime from a directory entry's metadata.
579 pub(crate) fn with_fs_metadata(mut self, meta: &std::fs::Metadata) -> Self {
580 if meta.is_file() {
581 self.size = Some(meta.len());
582 }
583 self.modified = meta.modified().ok();
584 self
585 }
586}
587
588/// Whether an object key or path is Parquet: named `.parquet`, or a part file with no
589/// extension inside a directory named `.parquet`, as Spark and GBIF write them
590/// (`occurrence.parquet/000001`). Hidden and job files (`_SUCCESS`, `.crc`) are not.
591pub fn is_parquet_key(key: &str) -> bool {
592 let key = key.trim_end_matches('/');
593 let (directory, name) = match key.rsplit_once('/') {
594 Some((directory, name)) => (directory, name),
595 None => ("", key),
596 };
597 if is_bookkeeping(name) {
598 return false;
599 }
600 if name.to_ascii_lowercase().ends_with(".parquet") {
601 return true;
602 }
603 let directory_name = directory.rsplit('/').next().unwrap_or(directory);
604 !name.contains('.') && directory_name.to_ascii_lowercase().ends_with(".parquet")
605}
606
607#[cfg(test)]
608mod parquet_key_tests {
609 use super::is_parquet_key;
610
611 #[test]
612 fn parquet_without_an_extension_is_known_by_its_directory() {
613 assert!(is_parquet_key(
614 "occurrence/2026-09-01/occurrence.parquet/000001"
615 ));
616 assert!(is_parquet_key("data/part-0.parquet"));
617 assert!(is_parquet_key("DATA/PART-0.PARQUET"));
618 assert!(!is_parquet_key(
619 "occurrence/2026-09-01/occurrence.parquet/_SUCCESS"
620 ));
621 assert!(!is_parquet_key("occurrence.parquet/.part-0.crc"));
622 assert!(!is_parquet_key("occurrence/2026-09-01/citation.txt"));
623 assert!(!is_parquet_key("notes/000001"));
624 }
625}
626
627/// What a file with no usable extension turns out to be, from the bytes at its start.
628///
629/// Every format datui reads as a directory puts a fixed signature at the front — Parquet
630/// at both ends, and the other three at the front alone. A name is the cheap answer and
631/// the one every listing uses; this is the expensive one, and it is asked only of a
632/// directory somebody is opening, never of a directory somebody is looking at.
633///
634/// Spark and GBIF both write part files with no extension — `occurrence.parquet/000001`
635/// is read by its directory's name, and the same files under a directory named anything
636/// else were not data at all as far as datui was concerned.
637///
638/// CSV and JSON are deliberately absent: they have no signature, and guessing from the
639/// first line is a parse rather than a look. Which signatures a listing believes is each
640/// format's to say ([`crate::readers::Trusted::listing`]).
641pub fn sniff_format(path: &Path) -> Option<crate::FileFormat> {
642 crate::readers::sniff_file(path, crate::readers::Asked::Listing)
643}
644
645/// What a listing finds a file to be by its first bytes.
646#[derive(Debug, Clone)]
647pub enum Sniffed {
648 /// A format datui reads.
649 Format(crate::FileFormat),
650 /// A format spec's, which reads it.
651 Spec(std::sync::Arc<crate::formats::Spec>),
652}
653
654/// [`sniff_format`], and when no format datui reads says it, the format spec that
655/// reads it as an open would pick one: by glob, else by magic and `match.where`. One
656/// read of the file's head answers both, so a listing reads nothing more for specs.
657pub fn sniff_listed(path: &Path, formats: &crate::formats::Registry) -> Option<Sniffed> {
658 use crate::readers::{Asked, HEAD, head_of, sniff};
659 let head = head_of(path)?;
660 if let Some(format) = sniff(&head, Some(path), Asked::Listing, |_| true) {
661 return Some(Sniffed::Format(format));
662 }
663 formats
664 .listed(path, &head, head.len() < HEAD)
665 .map(Sniffed::Spec)
666}
667
668/// Name `entry` a file of `spec`, which reads it; a spec that reads its records as
669/// several variants makes it a place too, whose tables → lists.
670pub fn name_spec_file(entry: &mut Entry, spec: &crate::formats::Spec) {
671 entry.kind = EntryKind::File;
672 entry.format_spec = Some(spec.name.clone());
673 if spec.lists_variants() {
674 entry.cost.tables = Some(spec.records.variants.len());
675 }
676}
677
678/// Name a local file row no listing classified (a recent) by the spec that reads it,
679/// as a listing names one: a spec's glob, else, when its name says nothing, the magic
680/// in its first bytes.
681pub fn name_unlisted_file(entry: &mut Entry, formats: &crate::formats::Registry) {
682 if formats.is_empty()
683 || entry.kind != EntryKind::File
684 || entry.table.is_some()
685 || is_data_file(&entry.path)
686 {
687 return;
688 }
689 let spec = match formats.by_glob(&entry.path, false).into_iter().next() {
690 Some(spec) => Some(spec),
691 None if worth_sniffing(&entry.path) && is_regular_file(&entry.path) => {
692 match sniff_listed(&entry.path, formats) {
693 Some(Sniffed::Spec(spec)) => Some(spec),
694 _ => None,
695 }
696 }
697 None => None,
698 };
699 if let Some(spec) = spec {
700 name_spec_file(entry, &spec);
701 }
702}
703
704/// How many extension-less files one listing looks inside. A directory of Spark output
705/// is a few hundred part files; past this the rest are listed by name alone.
706pub(crate) const MAX_SNIFFS_PER_DIR: usize = 256;
707
708/// Whether a file's name has no extension at all: `part-00000`, `LICENSE`.
709pub fn has_no_extension(path: &Path) -> bool {
710 path.extension().is_none()
711}
712
713/// Whether a listing looks inside a file to say what it is: one with no extension, or
714/// one whose extension says nothing (`.bin`, which ArduPilot's logs and model
715/// checkpoints share with everything else) or only text (`.log`, which candump
716/// writes; `.txt`).
717pub fn worth_sniffing(path: &Path) -> bool {
718 path.extension().is_none_or(|e| {
719 e.eq_ignore_ascii_case("bin") || data_format(path).is_some_and(crate::FileFormat::is_lines)
720 })
721}
722
723/// Whether a local file is Parquet by its contents: `PAR1` at both ends.
724pub fn has_parquet_magic(path: &Path) -> bool {
725 use std::io::{Read, Seek, SeekFrom};
726 let Ok(mut file) = std::fs::File::open(path) else {
727 return false;
728 };
729 let mut head = [0u8; 4];
730 let mut tail = [0u8; 4];
731 file.read_exact(&mut head).is_ok()
732 && file.seek(SeekFrom::End(-4)).is_ok()
733 && file.read_exact(&mut tail).is_ok()
734 && &head == b"PAR1"
735 && &tail == b"PAR1"
736}
737
738/// Whether a path names a Parquet file: by its extension, or by sitting as a part file
739/// with no extension inside a `.parquet` directory.
740///
741/// Not [`is_parquet_key`], which also answers "does this count toward what a directory
742/// holds" and so says no to a writer's own name. `_manifest.parquet` is a file somebody
743/// may open and the listing shows it; reading its footer is a different question from
744/// whether it makes the directory around it a dataset.
745pub fn is_parquet_path(path: &Path) -> bool {
746 path.extension()
747 .and_then(|e| e.to_str())
748 .is_some_and(|e| e.eq_ignore_ascii_case("parquet"))
749 || is_parquet_key(&directory_and_name(path))
750}
751
752/// Whether a path looks like something datui can open.
753///
754/// Its name, or its place: a part file with no extension inside a `.parquet` directory is
755/// Parquet, as Spark and GBIF write them. Every route that asks what a name means asks
756/// here — the listing, the search, `~` input, the counts and the schema pane — because
757/// the one that did not was always the one that disagreed.
758pub fn is_data_file(path: &Path) -> bool {
759 data_extension(path).is_some() || is_parquet_key(&directory_and_name(path))
760}
761
762/// The extension that says what a file *is*, with any compression suffix walked past.
763///
764/// `sales.csv.gz` is a CSV: `Path::extension` answers `gz`, which is how it is stored
765/// rather than what it holds. Anything deciding a *format* wants this one — two files
766/// named `.csv.gz` and `.json.gz` agree on their extension and on nothing that
767/// matters.
768///
769/// `None` when the name does not end in something datui reads.
770pub fn data_extension(path: &Path) -> Option<String> {
771 let name = path.file_name().and_then(|n| n.to_str())?;
772 let lower = name.to_ascii_lowercase();
773 let mut parts: Vec<&str> = lower.rsplit('.').collect();
774 parts.reverse();
775 if parts.len() < 2 {
776 return None;
777 }
778 // Walk back past a compression suffix so `sales.csv.gz` still reads as CSV.
779 let mut idx = parts.len() - 1;
780 if COMPRESSION_EXTENSIONS.contains(&parts[idx]) && idx > 1 {
781 idx -= 1;
782 }
783 crate::FileFormat::from_extension(parts[idx]).map(|_| parts[idx].to_string())
784}
785
786/// What the home screen says of a file [`unreadable_by_name`] turns away.
787pub const NO_READER: &str = "datui has no reader for this file";
788
789/// Whether a file's name already says datui will not open it: an extension no reader
790/// takes, under any compression suffix. A bare `data.gz` is left to the open, which
791/// looks inside, and so is a name with no extension.
792pub fn unreadable_by_name(path: &Path) -> bool {
793 let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
794 return false;
795 };
796 let name = name.to_ascii_lowercase();
797 let parts: Vec<&str> = name.rsplit('.').collect();
798 let compressed = |last: &str| COMPRESSION_EXTENSIONS.contains(&last);
799 let ext = match parts[..] {
800 [last, inner, _, ..] if compressed(last) => inner,
801 [last, _] if compressed(last) => return false,
802 [last, _, ..] => last,
803 _ => return false,
804 };
805 crate::FileFormat::from_extension(ext).is_none()
806}
807
808/// The format a file's name says it holds, compression suffix walked past.
809///
810/// The question every listing actually asks. Named extensions are not formats: `.ipc`,
811/// `.arrow`, `.arrows` and `.feather` are one format under four names, and a directory holding two
812/// of them is one kind of thing. Asking [`crate::FileFormat`] rather than a list of its
813/// own is what keeps the home screen from offering a file the reader has no route for,
814/// which is how `.txt` came to be listed and refused and `.psv` readable and invisible.
815pub fn data_format(path: &Path) -> Option<crate::FileFormat> {
816 // A sharded checkpoint's index is the model, not a JSON table.
817 crate::FileFormat::from_name_ending(path)
818 .or_else(|| crate::FileFormat::from_extension(&data_extension(path)?))
819}
820
821/// The two path segments `is_parquet_key` needs, as it splits them.
822///
823/// A whole path would reach it with backslashes on Windows, which it does not split on,
824/// so `occurrence.parquet\000001` would arrive as one name that contains a dot and be
825/// read as an ordinary file. The same reason `DataTableState::directory_and_name` exists.
826pub(crate) fn directory_and_name(path: &Path) -> String {
827 let name = path.file_name().unwrap_or_default().to_string_lossy();
828 match path.parent().and_then(|p| p.file_name()) {
829 Some(directory) => format!("{}/{name}", directory.to_string_lossy()),
830 None => name.into_owned(),
831 }
832}
833
834/// Commonest first, Parquet ahead of anything it ties with, then by name, so the line
835/// reads the same way twice running.
836///
837/// One order, by name of format, for everything that ranks a directory's formats: the
838/// local label, the local read that picks a reader, and the cloud label. They agreed on
839/// the common case and not on a tie — a directory of two CSV and two Parquet was
840/// *labelled* `2 csv · 2 parquet` and *read* as Parquet, so the row said one thing and
841/// `Enter` did another. Parquet wins the tie because it is the format a directory of data
842/// files is most likely to be about and the one every other route reads in place.
843///
844/// Named rather than written inline because `read_dir` order is exactly what it exists
845/// to remove, and a fixture on disk cannot pin an order that depends on it: the tie is
846/// the whole point and only a caller choosing the input order can put one there.
847pub(crate) fn rank_formats(a: (&str, usize), b: (&str, usize)) -> std::cmp::Ordering {
848 // Text is what is read when nothing else is: a README among data files is not
849 // the table, however many there are.
850 let text = crate::FileFormat::Text.name();
851 (a.0 == text)
852 .cmp(&(b.0 == text))
853 .then_with(|| b.1.cmp(&a.1))
854 .then_with(|| (a.0 != "parquet").cmp(&(b.0 != "parquet")))
855 .then_with(|| a.0.cmp(b.0))
856}
857
858/// Whether a file is one Hugging Face `datasets` writes beside a dataset's Arrow
859/// shards to describe them: `save_to_disk` writes both, and its cache the first. They
860/// are the dataset's metadata, not its data, where `.arrow` files sit beside them; two
861/// JSON files would otherwise outnumber a dataset of one shard and be read instead of
862/// it.
863pub(crate) fn is_hugging_face_metadata(name: &str) -> bool {
864 matches!(name, "dataset_info.json" | "state.json")
865}
866
867fn order_formats(counts: &mut [(crate::FileFormat, usize)]) {
868 counts.sort_by(|a, b| rank_formats((a.0.name(), a.1), (b.0.name(), b.1)));
869}
870
871/// Whether a listing entry is bookkeeping rather than data.
872///
873/// The one convention datui knows, and the only one: a leading `_` or `.`, which every
874/// engine in the table uses for the files it leaves beside its output — `_SUCCESS`,
875/// `_committed_*`, `_started_*`, `_metadata.json`, `.crc` — and the `_$folder$` marker
876/// some tools write to stand in for a folder in a flat store.
877///
878/// One predicate rather than the five that had drifted apart: a local listing skipped
879/// dotfiles and the literal `_SUCCESS`, a local open skipped both prefixes, and the
880/// cloud listing knew three more names. A directory whose ninth entry is `_metadata.json`
881/// answered `multi` locally and `dir` in a bucket for no better reason than that.
882pub fn is_bookkeeping(name: &str) -> bool {
883 // The `_$folder$` marker is asked about first, because it is a suffix and the
884 // folder it stands in for can itself be a partition: legacy s3n and EMR write
885 // `year=2024_$folder$` beside `year=2024/`, and a partition test looking only for
886 // an `=` calls that marker data.
887 if name.ends_with("_$folder$") {
888 return true;
889 }
890 // A `key=value` name is a partition wherever it appears, whatever it starts with.
891 // Spark and Hive partition on internal columns — `_date=2024-01-01`, `_c0=…` — and
892 // reading those as a writer's own files loses the whole dataset.
893 if is_partition_name(name) {
894 return false;
895 }
896 name.starts_with(['_', '.'])
897}
898
899/// Whether a format, by name, is model weights.
900fn is_weights(name: &str) -> bool {
901 name == crate::FileFormat::Safetensors.name() || name == crate::FileFormat::Gguf.name()
902}
903
904/// Whether formats found side by side in one directory are a model: weights of one
905/// format, with nothing else beside them but JSON (a config, a tokenizer). Such a
906/// directory is the model, however many JSON files outnumber the shards.
907pub(crate) fn is_model_directory<'a>(names: impl IntoIterator<Item = &'a str>) -> bool {
908 let mut weights = None;
909 for name in names {
910 if is_weights(name) {
911 if weights.is_some_and(|w| w != name) {
912 return false;
913 }
914 weights = Some(name);
915 } else if name != crate::FileFormat::Json.name() {
916 return false;
917 }
918 }
919 weights.is_some()
920}
921
922/// The format names `holds` counted.
923fn counts_names(holds: &Holds) -> impl Iterator<Item = &str> {
924 holds.formats.iter().map(|(name, _)| name.as_str())
925}
926
927/// Whether a name is a hive partition (`year=2024`): `key=value`, with a non-empty key.
928/// The value may be empty in practice.
929pub fn is_partition_name(name: &str) -> bool {
930 matches!(name.find('='), Some(i) if i > 0)
931}
932
933/// What the data files sitting directly in a directory say about how to read it.
934///
935/// A directory opened as one dataset has to be read by *something*, and the only honest
936/// source for that is the files in it. Before this existed the answer was assumed:
937/// any directory was scanned as Parquet, so a directory of `.json.gz` was opened by
938/// seeking to the end of each file for a `PAR1` that was never going to be there.
939#[derive(Debug, Clone, PartialEq, Eq)]
940pub enum DirectoryFormat {
941 /// Every data file directly in the directory reads as this one format, and these are
942 /// the files. Sorted, because a concatenation's row order is its file order.
943 One(crate::FileFormat, Vec<PathBuf>),
944 /// The files name more than one format. The commonest of them is the table — a
945 /// directory of a thousand CSVs and one stray JSON is a directory of CSVs — and the
946 /// rest are counted by format so the read can say what it passed over. Parquet wins a
947 /// tie, because it is the format a directory of data files is most likely to be about
948 /// and the one every other route here reads in place.
949 Mixed {
950 format: crate::FileFormat,
951 files: Vec<PathBuf>,
952 passed_over: Vec<(crate::FileFormat, usize)>,
953 },
954 /// The directory settles nothing by itself: it holds no readable data file directly,
955 /// or it has subdirectories and so may hold its data below. A hive dataset looks like
956 /// this — its files are a level down, under `key=value`.
957 Deeper,
958}
959
960/// What [`DirectoryFormat`] the data files directly in `dir` amount to.
961///
962/// One level only, and no file is opened: this reads names, exactly as the rest of
963/// this module does. A directory of two hundred thousand files costs one listing
964/// and nothing per file beyond it — the entry's own type comes back with the name, so
965/// there is no `stat` to bound.
966///
967/// Deliberately *not* bounded by [`MAX_ENTRIES_PER_DIR`], unlike every listing in this
968/// module. The files this returns are not a menu to show, they are the table to read:
969/// stopping at five thousand of a directory's six thousand CSVs would open it with a row
970/// count, a schema union and every aggregate quietly computed over a subset, and the
971/// `take` running before the sort would drop whichever file the directory read
972/// happened to return last. The Parquet route this mirrors hands the directory to a scan
973/// that enumerates it, with no cap either.
974pub fn directory_format(dir: &Path) -> DirectoryFormat {
975 let Ok(iter) = std::fs::read_dir(dir) else {
976 return DirectoryFormat::Deeper;
977 };
978
979 // Every format the directory names, with the files of each. A directory of one format
980 // takes the only entry; a directory of several takes the commonest and reports the
981 // rest, which is what stops one stray file deciding a directory cannot be read.
982 let mut by_format: Vec<(crate::FileFormat, Vec<PathBuf>)> = Vec::new();
983 // Files whose names say nothing, kept in case their bytes do. See below.
984 let mut nameless: Vec<PathBuf> = Vec::new();
985 let mut partitioned = false;
986 for entry in iter.flatten() {
987 let path = entry.path();
988 let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
989 continue;
990 };
991 // A table format's own files are not the table's, and a dotfile is nobody's.
992 // The same test every other route makes, so they agree on what is data.
993 if is_bookkeeping(name) {
994 continue;
995 }
996 // The type the directory read already returned, rather than a `stat` per
997 // entry: on a share that is a round trip per entry, and the question is only
998 // whether this is a file. A symlink still gets the stat, because `d_type`
999 // cannot say what is on the other end of one.
1000 let is_file = match entry.file_type() {
1001 Ok(kind) if kind.is_symlink() => is_regular_file(&path),
1002 Ok(kind) => kind.is_file(),
1003 Err(_) => is_regular_file(&path),
1004 };
1005 if !is_file {
1006 // Only a partition, not any subdirectory: a directory of CSVs with some
1007 // unrelated directory beside them is still a directory of CSVs, and handing
1008 // that to a scanner that walks trees is the very thing this function exists
1009 // to stop.
1010 partitioned |= is_partition_dir(&path);
1011 continue;
1012 }
1013 let Some(found) = data_format(&path) else {
1014 // A name that says nothing. Kept rather than dropped, because the bytes may
1015 // still say what it is — see below, where they are asked.
1016 if path.extension().is_none() {
1017 nameless.push(path);
1018 }
1019 continue;
1020 };
1021 match by_format.iter_mut().find(|(f, _)| *f == found) {
1022 Some((_, of_that_format)) => of_that_format.push(path),
1023 None => by_format.push((found, vec![path])),
1024 }
1025 }
1026
1027 // Files with no extension, in a directory whose names settled nothing. Spark and GBIF
1028 // both write part files this way; `occurrence.parquet/000001` is read by its
1029 // directory's name, and the same files under a directory named anything else were not
1030 // data at all as far as datui was concerned — the directory was `dir` and its files
1031 // were not listed.
1032 //
1033 // Only when the names have nothing to say. A directory of Parquet with a `LICENSE` in
1034 // it is a directory of Parquet, and opening the `LICENSE` to find out is a read per
1035 // file for an answer already given.
1036 //
1037 // A spread rather than every one, for the reason `sample_footers` takes a spread:
1038 // the cost is one open per file, and a directory written by one job holds one kind of
1039 // thing. They have to agree — a directory where the ends disagree is not one table by
1040 // any reading — and then all of them are taken as that format, because a scan that
1041 // reads what it can and says what it could not is what happens to the odd one out.
1042 if by_format.is_empty() && !nameless.is_empty() {
1043 nameless.sort();
1044 let mut picks = vec![0, nameless.len() / 2, nameless.len() - 1];
1045 picks.dedup();
1046 let sniffed: Vec<crate::FileFormat> = picks
1047 .iter()
1048 .filter_map(|i| nameless.get(*i))
1049 .filter_map(|f| sniff_format(f))
1050 .collect();
1051 if sniffed.len() == picks.len()
1052 && let Some(found) = sniffed.first().copied()
1053 && sniffed.iter().all(|f| *f == found)
1054 {
1055 by_format.push((found, nameless));
1056 }
1057 }
1058
1059 // Text is data only where nothing else is: a README beside Parquet is not a
1060 // candidate, as the listing does not count it.
1061 if by_format.iter().any(|(f, _)| !f.is_lines()) {
1062 by_format.retain(|(f, _)| !f.is_lines());
1063 }
1064
1065 // A Hugging Face dataset's own JSON files, beside its shards.
1066 if by_format
1067 .iter()
1068 .any(|(f, _)| *f == crate::FileFormat::Arrow)
1069 {
1070 for (format, files) in &mut by_format {
1071 if *format == crate::FileFormat::Json {
1072 files.retain(|f| {
1073 !f.file_name()
1074 .and_then(|n| n.to_str())
1075 .is_some_and(is_hugging_face_metadata)
1076 });
1077 }
1078 }
1079 by_format.retain(|(_, files)| !files.is_empty());
1080 }
1081
1082 // Decided after the whole listing rather than at the first entry that could settle
1083 // it, so the answer does not depend on the order a directory read happens to
1084 // return. One `key=value` below and the directory stops being the whole story: a hive
1085 // dataset's data is down there, whatever strays are lying at the top.
1086 if partitioned {
1087 return DirectoryFormat::Deeper;
1088 }
1089
1090 // The one order every route ranks a directory's formats by, so the reader this picks
1091 // is the format the label names.
1092 by_format.sort_by(|a, b| rank_formats((a.0.name(), a.1.len()), (b.0.name(), b.1.len())));
1093 // A directory holding model weights is the model. Its config and tokenizer JSON
1094 // sit beside the shards and often outnumber them, which does not make it a table
1095 // of JSON; the JSON is what the read passes over.
1096 if is_model_directory(by_format.iter().map(|(f, _)| f.name()))
1097 && let Some(at) = by_format.iter().position(|(f, _)| is_weights(f.name()))
1098 {
1099 let weights = by_format.remove(at);
1100 by_format.insert(0, weights);
1101 }
1102 let mut by_format = by_format.into_iter();
1103 let Some((format, mut files)) = by_format.next() else {
1104 return DirectoryFormat::Deeper;
1105 };
1106 files.sort();
1107 let passed_over: Vec<(crate::FileFormat, usize)> =
1108 by_format.map(|(f, of_that)| (f, of_that.len())).collect();
1109 if passed_over.is_empty() {
1110 DirectoryFormat::One(format, files)
1111 } else {
1112 DirectoryFormat::Mixed {
1113 format,
1114 files,
1115 passed_over,
1116 }
1117 }
1118}
1119
1120/// How far down a hive root is followed looking for the files it partitions.
1121///
1122/// A dataset partitioned by year, month, day and hour is four; past this the directory
1123/// is something other than a hive dataset, and guessing further costs a directory
1124/// read per level on a share.
1125const MAX_HIVE_DEPTH: usize = 16;
1126
1127/// What the files under a hive root's `key=value` partitions actually are.
1128///
1129/// A hive root holds no data itself, so [`directory_format`] can only say `Deeper` about
1130/// one. This follows a single spine down — the same one path through the tree a hive
1131/// scan reads its schema from — and reports what it finds at the bottom.
1132///
1133/// One spine, and the first partition at each level, so a dataset of ten thousand
1134/// partitions costs what one of two costs. That makes it a sample: a tree whose
1135/// partitions disagree is reported as whatever the first one holds. The alternative
1136/// is walking the dataset to answer a question asked before it is opened.
1137pub fn hive_leaf_format(dir: &Path) -> DirectoryFormat {
1138 let mut at = dir.to_path_buf();
1139 for _ in 0..MAX_HIVE_DEPTH {
1140 match directory_format(&at) {
1141 // Nothing here settles it. Follow the partitions down, if there are any.
1142 DirectoryFormat::Deeper => match first_partition(&at) {
1143 Some(next) => at = next,
1144 None => return DirectoryFormat::Deeper,
1145 },
1146 settled => return settled,
1147 }
1148 }
1149 DirectoryFormat::Deeper
1150}
1151
1152/// The first `key=value` subdirectory of `dir`, by name.
1153///
1154/// By name rather than in directory order: two runs asking what a dataset holds must
1155/// not look at different partitions and give different answers.
1156fn first_partition(dir: &Path) -> Option<PathBuf> {
1157 let iter = std::fs::read_dir(dir).ok()?;
1158 iter.flatten()
1159 .take(MAX_ENTRIES_PER_DIR)
1160 .map(|entry| entry.path())
1161 .filter(|path| is_partition_dir(path) && path.is_dir())
1162 .min()
1163}
1164
1165/// Whether a directory name is a hive partition (`year=2024`).
1166fn is_partition_dir(path: &Path) -> bool {
1167 path.file_name()
1168 .and_then(|n| n.to_str())
1169 .is_some_and(is_partition_name)
1170}
1171
1172/// Classify a directory without walking it.
1173///
1174/// Reads one listing, bounded by [`MAX_ENTRIES_PER_DIR`] rather than by a probe of the
1175/// first few entries. A probe makes the answer depend on the order the filesystem hands
1176/// entries back: a directory of eight Parquet files followed by `_metadata.json` answered
1177/// `multi` locally, where a bucket listing the same directory sorts the JSON first and
1178/// answered `dir`.
1179///
1180/// It costs more than the probe did: a plain directory row is enriched with nothing, so
1181/// its listing is read for this alone, and a directory of five thousand entries is read
1182/// whole where eight used to settle it — six hundred times the entries, for the worst
1183/// row, and `look_into_batch` walks a batch of sixteen of them one at a time.
1184///
1185/// That includes rows on a network mount: `network_check` gates listing a directory you
1186/// have browsed into, not classifying the rows of one. It is one `getdents` walk, with
1187/// a `stat` only for a symlink, since `d_type` cannot say what is on the far end of one
1188/// — so a directory of symlinks is the expensive case. `home_open_selected` makes the
1189/// call on the thread reading keys; every other caller is on a worker.
1190///
1191/// Capping it lower again would put the order-dependence back exactly where the
1192/// directories are biggest.
1193pub fn classify_directory(path: &Path) -> EntryKind {
1194 look_at_directory(path).0
1195}
1196
1197/// The kind *and* what the listing found, from one read of it.
1198///
1199/// Two answers to two questions. The kind decides what `Enter` does with the directory;
1200/// the count says what is in it, and the row's label is written from that — so a label
1201/// that is wrong about the first is still true about the second.
1202pub fn look_at_directory(path: &Path) -> (EntryKind, Holds) {
1203 let mut holds = Holds::default();
1204 // Before anything is counted: a lake table's data files genuinely do agree on a
1205 // schema, so every rule below says "one table" and is right about the schema and
1206 // wrong about the rows.
1207 //
1208 // And before the listing, which it does not need: three `join` tests answer it, and
1209 // counting would walk up to `MAX_ENTRIES_PER_DIR` entries of every table in a
1210 // warehouse, on every pass, for a `holds` line beside a table whose files `enrich`
1211 // then refuses to read. A prefix in a bucket does carry one, because the listing it
1212 // is counted from had already been paid for.
1213 if let Some(lake) = lake_table(path) {
1214 return (lake, holds);
1215 }
1216 let Ok(iter) = std::fs::read_dir(path) else {
1217 return (EntryKind::Directory, holds);
1218 };
1219
1220 let mut partitions = 0usize;
1221 let mut data_files = 0usize;
1222 let mut seen = 0usize;
1223 // As `scan_dir_bounded` does, so the tally agrees with the rows inside.
1224 let mut sniffs_left = if crate::home::is_remote_path(path) {
1225 0
1226 } else {
1227 MAX_SNIFFS_PER_DIR
1228 };
1229 let mut counts: Vec<(crate::FileFormat, usize)> = Vec::new();
1230 // A multi-file dataset is homogeneous by definition; a directory that merely
1231 // contains two different spreadsheets is not one. Compared as formats rather than
1232 // as extensions, so `.ipc` beside `.arrow` is one kind of thing and not two.
1233 let mut format: Option<crate::FileFormat> = None;
1234 let mut mixed_formats = false;
1235 // Counted as data until the listing is done; see `is_hugging_face_metadata`.
1236 let mut hugging_face: Vec<String> = Vec::new();
1237
1238 // Bounded where the entries come from rather than after they are counted: a
1239 // Hadoop-style output directory is a `.crc` per data file, and skipping those before
1240 // the count would let the walk run to twice the cap. The cap is a cost bound and not
1241 // a correctness one either way — a directory past it is decided by whichever entries
1242 // came back first, whether they were data or a writer's own.
1243 // One past the cap, so "there is more" is known without paying to process it —
1244 // the same shape `scan_dir_bounded` uses, and the reason a directory of exactly five
1245 // thousand entries is a total rather than a floor.
1246 for (entries, entry) in iter.flatten().take(MAX_ENTRIES_PER_DIR + 1).enumerate() {
1247 if entries >= MAX_ENTRIES_PER_DIR {
1248 holds.truncated = true;
1249 break;
1250 }
1251 let entry_path = entry.path();
1252 let name = entry.file_name();
1253 // The markers and job files tools leave beside their output, by the one test
1254 // every route makes.
1255 let name = name.to_string_lossy().into_owned();
1256 if is_bookkeeping(&name) {
1257 holds.skipped += 1;
1258 // The first few by name, not the first few the filesystem returned: a line
1259 // in the pane that reads differently on two runs of the same directory is the
1260 // order-dependence this module just spent a release removing. Past
1261 // `MAX_ENTRIES_PER_DIR` it is the first few by name *of what was read*, and
1262 // where the walk stopped is the filesystem's order again — which the `+` on
1263 // every count beside them says.
1264 if holds.skipped_names.last().is_none_or(|last| &name < last)
1265 || holds.skipped_names.len() < SKIPPED_NAMES_SHOWN
1266 {
1267 holds.skipped_names.push(name);
1268 holds.skipped_names.sort();
1269 holds.skipped_names.truncate(SKIPPED_NAMES_SHOWN);
1270 }
1271 continue;
1272 }
1273 // The type the directory read already returned, rather than a `stat` per entry:
1274 // this walks the whole listing now, and on a share every stat is a round trip.
1275 // A symlink still gets one, because `d_type` cannot say what is on the far end.
1276 //
1277 // A regular file rather than "not a directory", the test `directory_format`
1278 // makes: a FIFO named `a.csv` blocks whoever opens it until a writer appears, and
1279 // a broken symlink named `b.csv` opens as nothing. Counting either as data offers
1280 // a directory that cannot be read.
1281 let followed = |path: &Path| {
1282 // One `stat`, not two: what is on the far end is one question, and asking
1283 // it twice is a second round trip on a share.
1284 std::fs::metadata(path).map_or((false, false), |m| (m.is_dir(), m.is_file()))
1285 };
1286 let (is_dir, is_file) = match entry.file_type() {
1287 Ok(kind) if kind.is_symlink() => followed(&entry_path),
1288 Ok(kind) => (kind.is_dir(), kind.is_file()),
1289 Err(_) => followed(&entry_path),
1290 };
1291 if is_dir {
1292 holds.directories += 1;
1293 if is_partition_dir(&entry_path) {
1294 partitions += 1;
1295 }
1296 } else if let Some(found) = data_format(&entry_path)
1297 // Text by its name, unless its bytes say more: below.
1298 .filter(|f| !f.is_lines())
1299 // A sharded checkpoint's index is counted as the JSON it is, so the label
1300 // counts the shards; the read still takes it, for the metadata it carries.
1301 .map(
1302 |found| match crate::model_files::is_safetensors_index(&entry_path) {
1303 true => crate::FileFormat::Json,
1304 false => found,
1305 },
1306 )
1307 // Data by where it sits rather than by its name: see [`is_data_file`].
1308 .or_else(|| {
1309 is_parquet_key(&directory_and_name(&entry_path))
1310 .then_some(crate::FileFormat::Parquet)
1311 })
1312 .or_else(|| {
1313 (is_file && sniffs_left > 0 && worth_sniffing(&entry_path)).then(|| {
1314 sniffs_left -= 1;
1315 sniff_format(&entry_path)
1316 })?
1317 })
1318 .or_else(|| data_format(&entry_path))
1319 .filter(|_| is_file)
1320 {
1321 if found == crate::FileFormat::Json && is_hugging_face_metadata(&name) {
1322 hugging_face.push(name.clone());
1323 }
1324 data_files += 1;
1325 match counts.iter_mut().find(|(f, _)| *f == found) {
1326 Some((_, n)) => *n += 1,
1327 None => counts.push((found, 1)),
1328 }
1329 match format {
1330 None => format = Some(found),
1331 Some(first) if first != found => mixed_formats = true,
1332 Some(_) => {}
1333 }
1334 } else if is_file && has_no_extension(&entry_path) {
1335 holds.unnamed += 1;
1336 } else {
1337 // Everything else in the listing: a file with no reader, and a name with
1338 // nothing behind it — a FIFO, a socket, a broken symlink. Named like data
1339 // or not, none of them can be read.
1340 holds.not_read += 1;
1341 }
1342 seen += 1;
1343 }
1344
1345 // A Hugging Face dataset's own JSON files are its writer's, like `_SUCCESS`.
1346 if !hugging_face.is_empty() && counts.iter().any(|(f, _)| *f == crate::FileFormat::Arrow) {
1347 let n = hugging_face.len();
1348 for (format, count) in &mut counts {
1349 if *format == crate::FileFormat::Json {
1350 *count -= n;
1351 }
1352 }
1353 counts.retain(|(_, count)| *count > 0);
1354 data_files -= n;
1355 seen -= n;
1356 holds.skipped += n;
1357 holds.skipped_names.extend(hugging_face);
1358 holds.skipped_names.sort();
1359 holds.skipped_names.truncate(SKIPPED_NAMES_SHOWN);
1360 mixed_formats = counts.len() > 1;
1361 format = counts.first().map(|(f, _)| *f);
1362 }
1363
1364 // Text is data only where nothing else is: a README beside Parquet is a file
1365 // nothing reads as the directory's table, as it is to the cloud route.
1366 if counts.iter().any(|(f, _)| !f.is_lines())
1367 && let Some(at) = counts.iter().position(|(f, _)| f.is_lines())
1368 {
1369 let (_, n) = counts.remove(at);
1370 data_files -= n;
1371 holds.not_read += n;
1372 mixed_formats = counts.len() > 1;
1373 format = counts.first().map(|(f, _)| *f);
1374 }
1375
1376 holds.partitions = partitions;
1377 order_formats(&mut counts);
1378 holds.formats = counts
1379 .into_iter()
1380 .map(|(f, n)| (f.name().to_string(), n))
1381 .collect();
1382
1383 // Unchanged from before the probe went, deliberately. Reading the whole listing
1384 // makes one `notes=old` among twenty ordinary subdirectories a hive root every time
1385 // rather than only when it came back first, and two attempts at a majority to rule
1386 // that out each refused a real hive root instead — against everything present, one
1387 // with a README beside it; against the other directories, one with a `scripts/` and a
1388 // `docs/`. Refusing a dataset is the worse direction, and a rule per case is what
1389 // #275 exists to stop. The label stops deciding what `Enter` does in phase 3, and
1390 // the question goes with it.
1391 // Deterministic now rather than occasional, which is the cost of the whole listing:
1392 // a source tree with a `cfg=debug/` in it reads `hive` on every pass, and `enrich`
1393 // then walks it to depth four looking for footers. Left alone all the same — see
1394 // above for the two majorities that refused real hive roots instead.
1395 if partitions > 0 && partitions >= data_files {
1396 return (EntryKind::Hive, holds);
1397 }
1398
1399 // Require a format that can actually be read as many files. Without this the home
1400 // screen offers a directory of `.tsv` or `.xlsx` as one dataset and the open refuses
1401 // it — the same "offered but unreadable" the one vocabulary exists to stop, one
1402 // layer up.
1403 let readable_as_one = format.is_some_and(crate::FileFormat::reads_many_files);
1404 // Require homogeneity *and* that data is what this directory is mostly for.
1405 // Without the majority test, any directory with a couple of stray CSVs in it would
1406 // be offered as a dataset, which is worse than useless: it hides the directory.
1407 let homogeneous = data_files > 1 && !mixed_formats && readable_as_one;
1408 let mostly_data = data_files * 2 >= seen;
1409 // A model directory opens as the model: its shards as one table, the JSON beside
1410 // them left out. One file of weights is a model too.
1411 let model = partitions == 0 && is_model_directory(counts_names(&holds));
1412 let kind = if (homogeneous && mostly_data) || model {
1413 EntryKind::MultiFile
1414 } else {
1415 // Everything else — including a directory holding a single data file — is a
1416 // place to look inside, not a dataset in its own right.
1417 EntryKind::Directory
1418 };
1419 (kind, holds)
1420}
1421
1422/// Entries under `metadata/` to look at before giving up on Iceberg. A table with a
1423/// long history has thousands, and the newest are not first in any order a directory
1424/// read promises — but `vN.metadata.json` is written on the first commit and never
1425/// removed, so one is always there to find.
1426const ICEBERG_METADATA_PROBE: usize = 64;
1427
1428/// Whether `path` is the root of a lake table, and which.
1429///
1430/// Marker directory names are convention knowledge, which the one-table rule
1431/// deliberately keeps out: inferring a dataset from filenames is a list that is never
1432/// finished. These three are a different thing — a declared format with a specified
1433/// layout, where the marker is part of the spec.
1434///
1435/// Named directly rather than found by walking the listing: three `join` tests answer it
1436/// whatever the directory holds, where a walk pays for every entry of a table with a
1437/// hundred thousand data files to find one name it already knows.
1438fn lake_table(path: &Path) -> Option<EntryKind> {
1439 if path.join("_delta_log").is_dir() {
1440 return Some(EntryKind::Delta);
1441 }
1442 if path.join(".hoodie").is_dir() {
1443 return Some(EntryKind::Hudi);
1444 }
1445 // Iceberg's marker is a plain name, so it takes the whole shape: metadata beside
1446 // data, and a metadata file actually in it. `metadata/` alone is a directory anybody
1447 // may have.
1448 let metadata = path.join("metadata");
1449 if path.join("data").is_dir()
1450 && metadata.is_dir()
1451 && std::fs::read_dir(&metadata).is_ok_and(|entries| {
1452 entries
1453 .flatten()
1454 .take(ICEBERG_METADATA_PROBE)
1455 .any(|e| e.file_name().to_string_lossy().ends_with(".metadata.json"))
1456 })
1457 {
1458 return Some(EntryKind::Iceberg);
1459 }
1460 None
1461}
1462
1463/// List one directory level, classified. Never recurses.
1464///
1465/// Errors are swallowed deliberately: an unreadable or unmounted directory yields an
1466/// empty listing rather than failing the home screen, and the caller reports
1467/// availability separately.
1468pub fn scan_dir(dir: &Path) -> Vec<Entry> {
1469 scan_dir_bounded(dir).entries
1470}
1471
1472/// What one directory listing produced, and whether it saw all of it.
1473#[derive(Debug, Clone, Default)]
1474pub struct Scan {
1475 pub entries: Vec<Entry>,
1476 /// The directory held more than `MAX_ENTRIES_PER_DIR`; `entries` is a prefix of
1477 /// it. Worth saying out loud: a listing that silently stops at five thousand
1478 /// looks identical to a directory that simply has five thousand things in it.
1479 pub truncated: bool,
1480}
1481
1482/// List one directory, doing a bounded amount of work regardless of what is in it.
1483///
1484/// The cost is one `read_dir` and a `stat` per entry, bounded by
1485/// [`MAX_ENTRIES_PER_DIR`], and nothing per subdirectory at all.
1486///
1487/// **Nothing here is classified.** Telling a hive dataset from a plain directory means
1488/// reading the directory, which is a round trip apiece on a share — so no listing pays
1489/// for it, however small. Every subdirectory comes back [`EntryKind::Unknown`], which
1490/// claims nothing, and is looked into later from the viewport, a batch at a time, by
1491/// whoever is actually reading the rows.
1492///
1493/// That is what makes a row's label a fact about the row. Classifying the first
1494/// sixty-four subdirectories and calling every identical one after them a plain directory
1495/// made it a fact about position instead; classifying them only when a listing is small
1496/// enough moved the arbitrariness rather than removing it, since two directories holding
1497/// the same subdirectories would still disagree about what to call them.
1498pub fn scan_dir_bounded(dir: &Path) -> Scan {
1499 scan_dir_progressive(dir, |_| {})
1500}
1501
1502/// [`scan_dir_bounded`], naming the files it looks inside by `formats` too: a file
1503/// whose first bytes carry a spec's magic is listed as that spec's.
1504pub fn scan_dir_specs(dir: &Path, formats: &crate::formats::Registry) -> Scan {
1505 scan_dir_with(dir, formats, |_| {})
1506}
1507
1508/// How often a listing still being read shows what it has so far.
1509const LISTING_PROGRESS_EVERY: std::time::Duration = std::time::Duration::from_millis(250);
1510
1511/// [`scan_dir_bounded`], handing `progress` the rows read so far, sorted, every
1512/// [`LISTING_PROGRESS_EVERY`] while the read goes on. A directory a share takes seconds
1513/// to list shows its first rows as they arrive rather than a spinner until the last.
1514pub fn scan_dir_progressive(dir: &Path, progress: impl FnMut(&[Entry])) -> Scan {
1515 scan_dir_with(dir, &crate::formats::Registry::default(), progress)
1516}
1517
1518fn scan_dir_with(
1519 dir: &Path,
1520 formats: &crate::formats::Registry,
1521 mut progress: impl FnMut(&[Entry]),
1522) -> Scan {
1523 let Ok(iter) = std::fs::read_dir(dir) else {
1524 return Scan::default();
1525 };
1526 let mut shown = std::time::Instant::now();
1527
1528 let mut entries = Vec::new();
1529 let mut seen = 0usize;
1530 let mut truncated = false;
1531 // Files with no extension are looked at, a few bytes each, so a Spark part file
1532 // is listed as the data it is while a LICENSE stays out of the way. Never on a
1533 // share, where each open is a round trip and one that may not come back.
1534 let mut sniffs_left = if crate::home::is_remote_path(dir) {
1535 0
1536 } else {
1537 MAX_SNIFFS_PER_DIR
1538 };
1539
1540 // One past the cap: enough to know more exists without paying to process it.
1541 for dir_entry in iter.flatten().take(MAX_ENTRIES_PER_DIR + 1) {
1542 seen += 1;
1543 if seen > MAX_ENTRIES_PER_DIR {
1544 truncated = true;
1545 break;
1546 }
1547
1548 let path = dir_entry.path();
1549 let name = dir_entry.file_name();
1550 if name.to_string_lossy().starts_with('.') {
1551 continue;
1552 }
1553
1554 let Ok(meta) = dir_entry.metadata() else {
1555 continue;
1556 };
1557
1558 let mut spec = None;
1559 let kind = if meta.is_dir() {
1560 EntryKind::Unknown
1561 } else if meta.is_file() && is_data_file(&path) {
1562 EntryKind::File
1563 } else if meta.is_file() && sniffs_left > 0 && worth_sniffing(&path) {
1564 sniffs_left -= 1;
1565 match sniff_listed(&path, formats) {
1566 Some(Sniffed::Format(_)) => EntryKind::File,
1567 Some(Sniffed::Spec(found)) => {
1568 spec = Some(found);
1569 EntryKind::File
1570 }
1571 None => EntryKind::Other,
1572 }
1573 } else if meta.is_file() {
1574 EntryKind::Other
1575 } else {
1576 // Not a directory or a regular file. A FIFO named `x.parquet` is a
1577 // listing entry datui must never offer to open.
1578 continue;
1579 };
1580
1581 let mut entry = Entry::new(path, kind).with_fs_metadata(&meta);
1582 if let Some(spec) = spec {
1583 name_spec_file(&mut entry, &spec);
1584 }
1585 entries.push(entry);
1586 if shown.elapsed() >= LISTING_PROGRESS_EVERY {
1587 let mut so_far = entries.clone();
1588 sort_entries(&mut so_far);
1589 progress(&so_far);
1590 shown = std::time::Instant::now();
1591 }
1592 }
1593
1594 sort_entries(&mut entries);
1595 Scan { entries, truncated }
1596}
1597
1598/// Datasets first, then directories; each group alphabetical.
1599///
1600/// Recency is a better sort for recents, but a directory listing is a place you
1601/// scan by name, so name order wins here.
1602///
1603/// A row nothing has looked into yet sorts with the directories, although
1604/// [`EntryKind::is_dataset`] offers it as openable. In a fresh listing that is every
1605/// subdirectory, so what this amounts to there is files first and directories after —
1606/// and it is the one ordering a directory can be given before anything is known about
1607/// it, since it is where the row lands if the directory turns out to be a plain one.
1608///
1609/// Which is the point: a kind arriving later never moves the row, because a row that
1610/// moves out from under the cursor while you are scrolling is worse than a label that
1611/// is late.
1612pub(crate) fn sort_entries(entries: &mut [Entry]) {
1613 entries.sort_by(|a, b| {
1614 // Data, then directories, then what datui cannot read.
1615 let group = |k: EntryKind| match k {
1616 k if k.is_known_dataset() => 0,
1617 EntryKind::Other => 2,
1618 _ => 1,
1619 };
1620 group(a.kind).cmp(&group(b.kind)).then_with(|| {
1621 a.name
1622 .to_ascii_lowercase()
1623 .cmp(&b.name.to_ascii_lowercase())
1624 })
1625 });
1626}
1627
1628/// How far below a directory the footer walk goes. A hive dataset partitioned by year,
1629/// month, day and hour is four; past this the files belong to something else.
1630const MAX_WALK_DEPTH: u8 = 4;
1631
1632/// Upper bound on Parquet footers read to size a multi-file or hive dataset.
1633///
1634/// Two partitions is cheap; five thousand is not, and a home screen that stalls on
1635/// the biggest dataset is worse than one that admits it does not know. Past this
1636/// bound the count is left blank rather than reported as a partial total.
1637const MAX_FOOTERS_PER_DATASET: usize = 64;
1638
1639/// Fill in row and column counts for a dataset, from Parquet footers only.
1640///
1641/// Handles a single file, and sums a bounded number of files for hive and multi-file
1642/// datasets. Anything not backed by Parquet keeps `None`, which the UI shows as an
1643/// honest blank.
1644pub fn enrich(entry: &mut Entry) {
1645 enrich_as(entry, &crate::schema_union::ReadAs::default())
1646}
1647
1648/// As [`enrich`], reading each file the way the open that follows will read it.
1649///
1650/// The rule that decides whether a directory's files are one table reads the names at the
1651/// front of them, and where those names are is a reader setting. A pass that used its own
1652/// answers would judge a directory by a reading nobody is going to make — which is how
1653/// `datui --no-header directory/` came to open the home screen for a directory the flag
1654/// reads perfectly as one table.
1655pub fn enrich_as(entry: &mut Entry, as_read: &crate::schema_union::ReadAs) {
1656 enrich_with(entry, as_read, None)
1657}
1658
1659/// As [`enrich_as`], taking a dataset's measure from the shape an open kept of it in
1660/// `remembered`, where its files are as they were then, rather than from a sample of
1661/// its footers.
1662pub fn enrich_with(
1663 entry: &mut Entry,
1664 as_read: &crate::schema_union::ReadAs,
1665 remembered: Option<&crate::cache::CacheManager>,
1666) {
1667 match entry.kind {
1668 EntryKind::File => {
1669 enrich_parquet(entry);
1670 enrich_tables(entry);
1671 enrich_arrow(entry);
1672 }
1673 EntryKind::Hive | EntryKind::MultiFile => enrich_dataset(entry, as_read, remembered),
1674 // Nothing to read for a plain directory, and nothing that *may* be read for
1675 // one that has not been looked at. Nor for a lake table: summing the footers
1676 // under one counts tombstoned rows, every rewritten version and both sides of
1677 // a compaction, which is the whole reason it is not offered as a dataset.
1678 EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => {}
1679 EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => {}
1680 }
1681}
1682
1683/// Sum footers across a bounded set of Parquet files under `entry`.
1684fn enrich_dataset(
1685 entry: &mut Entry,
1686 as_read: &crate::schema_union::ReadAs,
1687 remembered: Option<&crate::cache::CacheManager>,
1688) {
1689 // A directory of JSON is not described by the Parquet under it. The walk below
1690 // recurses — it has to, because that is what opening the directory reads — so for a
1691 // directory whose own files are a format this cannot count, every number it produced
1692 // belonged to something the row does not name: `6 json` reported the sixty-one
1693 // columns of the Parquet in its subdirectories.
1694 //
1695 // A directory of Parquet with more Parquet beneath it is the opposite case and keeps
1696 // the walk. The counts are a promise about what `Enter` gives, and `Enter` reads
1697 // the subtree; measuring only the top would promise three files and open
1698 // twenty-three, and would ask `is_one_table` about three files while unioning all
1699 // twenty-three. The `holds` line names the directory that explains the difference.
1700
1701 // The partition layout comes from directory names, so it is knowable even for a
1702 // dataset far too large to count the rows of — which is exactly the dataset whose
1703 // shape you most want described before opening it.
1704 if entry.kind == EntryKind::Hive {
1705 entry.cost.partitions = partition_layout(&entry.path);
1706 }
1707
1708 // Whether the footers below are this directory's own shape, or something else's. A
1709 // directory's own format is counted exactly, so this is exact for one.
1710 //
1711 // Not asked of a hive root at all. Its own files are strays beside the partitions —
1712 // a `schema.json` or a `manifest.csv` left at the top — so its counted format is
1713 // not its data's, and one such file would blank the whole dataset. Its data is down
1714 // in the partitions, where the format can only be sampled, and one spine tells the
1715 // two cases apart in neither direction: a CSV tree with a stray `snapshot.parquet`
1716 // in the sampled partition and a Parquet tree with a stray `notes.csv` in it both
1717 // come back `NotOneTable`. A stray Parquet in a CSV tree is still counted as the
1718 // dataset's, which #275 phase 4 settles by making the tree readable in its own
1719 // format.
1720 let reads_as_parquet = entry.kind == EntryKind::Hive
1721 || match entry.holds.one_format() {
1722 // No single format to object with, so nothing to object. No row reaches
1723 // this today — a directory of more than one format is a `Directory` and
1724 // `enrich` leaves those alone — so it is a default, and the safe one:
1725 // leaving the counts off a directory is a mistake opening it undoes.
1726 None => true,
1727 // A name this build cannot read back is not Parquet as far as anything here
1728 // knows. Leaving the counts off a directory is the mistake that can be undone
1729 // by opening it; giving it another format's numbers is not.
1730 Some(name) => crate::FileFormat::from_name(name) == Some(crate::FileFormat::Parquet),
1731 };
1732 if !reads_as_parquet {
1733 entry.size = None;
1734 judge_by_names(entry, as_read);
1735 return;
1736 }
1737
1738 // The stat'ed size of a dataset directory is its own inode: a couple of hundred
1739 // bytes that have nothing to do with the terabyte inside it. Dropped up front and
1740 // restored only if the files are actually totalled, so no path out of here can
1741 // leave it behind to be read as an answer.
1742 entry.size = None;
1743
1744 let files = parquet_files_under(&entry.path);
1745 // Past the budget the footers are not read here, but an open that read them all
1746 // kept them: listing the dataset again says whether they still describe it.
1747 if files.len() > MAX_FOOTERS_PER_DATASET
1748 && let Some((listed, footers)) = remembered
1749 .and_then(|cache| crate::dataset_files::remembered_footers(&entry.path, cache))
1750 {
1751 measure_from_footers(entry, &listed, &footers);
1752 return;
1753 }
1754 if files.is_empty() || files.len() > MAX_FOOTERS_PER_DATASET {
1755 // Whether these are one table is still worth asking, and it does not need
1756 // every footer: three files spread across the directory answer it. Without this a
1757 // directory large enough to be past the counting limit would skip the check
1758 // entirely, which is backwards — the more tables it holds, the more a union of
1759 // them costs.
1760 let sampled = sample_footers(&files);
1761 let names: Vec<Vec<String>> = sampled.iter().map(column_names).collect();
1762 if entry.kind == EntryKind::MultiFile && !files_nest(&names) {
1763 // Whether the directory is one table is asked of everything under it, because
1764 // that is what opening it would union. What it *holds* is the files the
1765 // label counts — the ones directly inside — and a downgraded row is never
1766 // opened as one table, so a *count* spanning the subtree would be a width
1767 // nothing produces. Three more footers, on a directory being downgraded, to
1768 // say `2 parquet` and mean those two.
1769 let own_files = direct_children(&files, &entry.path);
1770 let own = sample_footers(&own_files);
1771 // The names, though, are every one sampled under it, the same as the arm
1772 // below: they are the home screen's search index, and a directory is found by
1773 // a column that looking inside it will reach. Narrowing these to the direct
1774 // children made a big directory unfindable by a column a small one is found
1775 // by.
1776 entry.columns = union_of(&names);
1777 // A floor only when a footer was left unread. The directory is past the
1778 // counting budget, but its *own* files may be three of the seventy — and
1779 // then `5+ cols` claims a sample that did not happen.
1780 entry.cols_sampled = own.len() < own_files.len();
1781 // The columns a reader sees, from the schema rather than by splitting leaf
1782 // paths on a dot: a column named `user.id` and a struct `user` with a field
1783 // `id` are not the same thing, and a string cannot tell them apart.
1784 let top = union_of(&own.iter().map(top_level_names).collect::<Vec<_>>());
1785 downgrade_to_directory(entry, (!top.is_empty()).then_some(top.len()));
1786 return;
1787 }
1788 // Still worth knowing the shape, even when the row count is out of reach.
1789 if let Some(meta) = sampled.first() {
1790 // Three files rather than the first, because a directory written over time
1791 // keeps its newest columns in its last file — and the first is where a
1792 // dataset that grew is narrowest. Still a sample and not a total: the
1793 // count beside it is already `?`.
1794 entry.columns = union_of(&names);
1795 let top = union_of(&sampled.iter().map(top_level_names).collect::<Vec<_>>());
1796 entry.cols = Some(top.len() + partition_columns_beyond(entry, &top));
1797 entry.cols_sampled = true;
1798 // From one file, so it describes how the dataset is written rather
1799 // than its total: codec and row-group sizing are a property of the
1800 // writer and are uniform in practice.
1801 physical_facts(meta, &mut entry.cost);
1802 entry.cost.uncompressed = None;
1803 }
1804 return;
1805 }
1806
1807 let mut rows = 0usize;
1808 let mut bytes = 0u64;
1809 // Every column any file has, in the order they first appear — not the first
1810 // file's. A dataset whose columns grew over time reported the shape it was born
1811 // with: Bitcoin transactions, whose `inputs` gained `address` and then
1812 // `txinwitness`, answered no to "which of these has `txinwitness`?".
1813 let mut columns: Vec<String> = Vec::new();
1814 let mut seen_columns = std::collections::HashSet::new();
1815 // The columns a reader sees, unioned the same way. Kept beside the leaves rather
1816 // than derived from them, because a leaf path cannot say whether its dots are
1817 // nesting or part of a name. See [`top_level_names`].
1818 let mut top_level: Vec<String> = Vec::new();
1819 let mut seen_top_level = std::collections::HashSet::new();
1820 let mut per_file: Vec<Vec<String>> = Vec::with_capacity(files.len());
1821 // The width and the size, restricted to the directory's own files. A directory the
1822 // footers downgrade is never opened as one table, so a *count* spanning the subtree
1823 // would be a width nothing produces — and the label beside it counts only what is
1824 // inside. The column names stay the subtree's: they are the search index, not the
1825 // label.
1826 let mut own_bytes = 0u64;
1827 let mut own_top_level: Vec<String> = Vec::new();
1828 let mut own_seen_top = std::collections::HashSet::new();
1829 let mut cost = Cost::default();
1830 let mut uncompressed = 0u64;
1831 let mut row_groups = 0usize;
1832 for file in &files {
1833 let Some(meta) = crate::parquet_footer::read_parquet_metadata(file) else {
1834 return; // A file we cannot read makes the total a guess; report nothing.
1835 };
1836 rows += meta.num_rows;
1837 let names = column_names(&meta);
1838 for name in &names {
1839 if seen_columns.insert(name.clone()) {
1840 columns.push(name.clone());
1841 }
1842 }
1843 for name in top_level_names(&meta) {
1844 if seen_top_level.insert(name.clone()) {
1845 top_level.push(name);
1846 }
1847 }
1848 // The columns a reader sees, not the leaves the footer names: see
1849 // [`crate::schema_union::top_level_columns`].
1850 per_file.push(crate::schema_union::top_level_columns(&names));
1851 // And the same again for this directory's own files, which is what a downgraded
1852 // row is labelled from: `2 parquet` must mean those two.
1853 // One stat, feeding both totals: on a share each is a round trip, and a directory
1854 // of sixty-four files directly inside would have paid twice for every one.
1855 let file_bytes = std::fs::metadata(file).map(|m| m.len()).unwrap_or(0);
1856 bytes += file_bytes;
1857 if file.parent() == Some(entry.path.as_path()) {
1858 own_bytes += file_bytes;
1859 for name in top_level_names(&meta) {
1860 if own_seen_top.insert(name.clone()) {
1861 own_top_level.push(name);
1862 }
1863 }
1864 }
1865 let mut per_file = Cost::default();
1866 physical_facts(&meta, &mut per_file);
1867 uncompressed += per_file.uncompressed.unwrap_or(0);
1868 row_groups += per_file.row_groups.unwrap_or(0);
1869 if cost.codec.is_none() {
1870 cost.codec = per_file.codec;
1871 }
1872 }
1873 // The footers are read by now, so whether these files are one table is known
1874 // rather than guessed. A directory of separate tables is a place to look inside: its
1875 // row count is the sum of unrelated things, its column count belongs to whichever
1876 // file happened to be read first, and opening it unions tables that share nothing.
1877 //
1878 // Only `multi` is reconsidered. A `key=value` layout says what the writer meant,
1879 // and a hive directory's files hold the same table by construction.
1880 if entry.kind == EntryKind::MultiFile && !crate::schema_union::is_nested(&per_file) {
1881 // Its own files' bytes, not the subtree's. The label counts what is directly
1882 // inside and so do the columns beside it; a size summed over a different set of
1883 // files is a third number on one row measured against neither of the other two.
1884 entry.size = Some(own_bytes);
1885 // Nothing here is one table's shape, but the names are what the directory holds,
1886 // and searching the home screen by column should still find the directory that
1887 // has one. The count is the directory's own files, which is what the label names.
1888 // The column *names* are every one under it: they are the home screen's search
1889 // index, and "which of these has a `txinwitness`?" is answered by the directory
1890 // that has one anywhere, which is where looking inside will find it.
1891 entry.columns = columns;
1892 entry.cols_sampled = false;
1893 downgrade_to_directory(
1894 entry,
1895 (!own_top_level.is_empty()).then_some(own_top_level.len()),
1896 );
1897 return;
1898 }
1899
1900 entry.rows = Some(rows);
1901 entry.cols = Some(top_level.len() + partition_columns_beyond(entry, &top_level));
1902 entry.size = Some(bytes);
1903 entry.columns = columns;
1904 cost.uncompressed = (uncompressed > 0).then_some(uncompressed);
1905 cost.row_groups = (row_groups > 0).then_some(row_groups);
1906 cost.partitions = entry.cost.partitions.take();
1907 entry.cost = cost;
1908}
1909
1910/// Partition keys the files do not carry themselves. The open hoists them in as
1911/// columns, so a hive table's width counts them: `12 × 4`, not the `12 × 2` its footers
1912/// say.
1913fn partition_columns_beyond(entry: &Entry, top_level: &[String]) -> usize {
1914 entry.cost.partitions.as_ref().map_or(0, |layout| {
1915 layout
1916 .keys
1917 .iter()
1918 .filter(|key| !top_level.contains(key))
1919 .count()
1920 })
1921}
1922
1923/// The files of `dir` itself, out of a walk that went below it.
1924fn direct_children(files: &[PathBuf], dir: &Path) -> Vec<PathBuf> {
1925 files
1926 .iter()
1927 .filter(|f| f.parent() == Some(dir))
1928 .cloned()
1929 .collect()
1930}
1931
1932/// The footers at the ends and the middle of a directory too large to read every one of.
1933///
1934/// The ends and the middle, because keys and filenames sort: a directory written table by
1935/// table can easily start with several files of the same table, so its head answers
1936/// nothing. The last file earns its place twice over — in a directory written over time
1937/// it is the newest, which is where a column added last year is.
1938fn sample_footers(files: &[PathBuf]) -> Vec<crate::parquet_footer::Footer> {
1939 if files.is_empty() {
1940 return Vec::new();
1941 }
1942 let mut picks = vec![0, files.len() / 2, files.len() - 1];
1943 picks.dedup();
1944 picks
1945 .iter()
1946 .filter_map(|i| files.get(*i))
1947 .filter_map(|file| crate::parquet_footer::read_parquet_metadata(file))
1948 .collect()
1949}
1950
1951/// Whether a spread of a directory's files agree on a schema.
1952///
1953/// Fewer than two readable footers decide nothing, and the directory keeps the kind its
1954/// names suggested.
1955fn files_nest(sampled: &[Vec<String>]) -> bool {
1956 let per_file: Vec<Vec<String>> = sampled
1957 .iter()
1958 .map(|names| crate::schema_union::top_level_columns(names))
1959 .collect();
1960 per_file.len() < 2 || crate::schema_union::is_nested(&per_file)
1961}
1962
1963/// Ask a directory with no footers whether its files are one table, by the names at the
1964/// front of them.
1965///
1966/// The same rule as [`files_nest`] on the same evidence — the column names — from the
1967/// only place a CSV or an NDJSON file keeps them. Without this a directory of forty
1968/// unrelated CSVs was labelled `40 csv`, `Enter` promised one table because nothing had
1969/// looked, and the read then refused it: the permissive rule with the strict reader,
1970/// which is the pairing #275 exists to stop. Parquet has had the test since phase 3;
1971/// this is the rest of the formats catching up.
1972///
1973/// Silence is optimism, as it is for an unreadable footer: too few files, a format whose
1974/// schema costs a whole read, or a file that would not parse all leave the directory as
1975/// its names suggested. That is only safe because the read behind it unions by name and
1976/// widens types rather than failing — see `DataTableState::union_of_files`.
1977fn judge_by_names(entry: &mut Entry, as_read: &crate::schema_union::ReadAs) {
1978 if entry.kind != EntryKind::MultiFile {
1979 return;
1980 }
1981 let Some(format) = entry
1982 .holds
1983 .one_format()
1984 .and_then(crate::FileFormat::from_name)
1985 else {
1986 return;
1987 };
1988 // The directory's own files, which is what the label counts and what the open reads.
1989 // A `MultiFile` directory is flat by construction — a `key=value` below it would have
1990 // made it `Hive` — so there is no subtree to walk for these.
1991 //
1992 // `One` and nothing else. `one_format` above already returned for a directory of more
1993 // than one format, and `look_at_directory` only calls a directory `MultiFile` when
1994 // its formats agree, so `Mixed` cannot arrive here — matching it as well read as
1995 // coverage this does not have. A directory of forty disjoint CSVs beside one stray
1996 // `.json` is a `Directory` before it reaches this, and goes inside for that reason
1997 // rather than for this one.
1998 let DirectoryFormat::One(_, files) = directory_format(&entry.path) else {
1999 return;
2000 };
2001 // Read the way the open that follows will read it: where the header is decides
2002 // what these names are, and a verdict reached by another reading is about a directory
2003 // nobody is going to open.
2004 let sampled = crate::schema_union::sample_files(&files, format, as_read);
2005 if sampled.nests == Some(false) {
2006 // The columns the sample found, so searching the home screen by column still
2007 // finds the directory that has one — the same thing the Parquet path keeps when
2008 // it downgrades. From the spread that was read rather than from every file: a
2009 // directory of forty thousand CSVs must cost what a directory of four costs, and
2010 // this runs on the thread that opens a path named on the command line.
2011 // `cols_sampled` is what says the count is a floor.
2012 let cols = (!sampled.columns.is_empty()).then_some(sampled.columns.len());
2013 // A floor only when there were files the sample did not open. A directory of two
2014 // or three had every one read, and `N+ cols` on that row claims a hedge the
2015 // count does not need — the Parquet path next door works this out the same way.
2016 entry.cols_sampled = sampled.read < files.len();
2017 entry.columns = sampled.columns;
2018 downgrade_to_directory(entry, cols);
2019 }
2020}
2021
2022/// A directory whose files turned out to be separate tables is a place to look inside.
2023///
2024/// Its row count would be the sum of unrelated things, so it is not reported. The column
2025/// count is: the union of what the directory's files hold is a true answer to "what is in
2026/// here" even when "how many rows" has none, so a directory of fifteen tables reads
2027/// `15 parquet · 72 columns` and no row count. Passed in rather than derived from
2028/// `columns`, which names leaves: see [`top_level_names`] for why a leaf path cannot be
2029/// split back into the columns a reader sees.
2030fn downgrade_to_directory(entry: &mut Entry, cols: Option<usize>) {
2031 entry.kind = EntryKind::Directory;
2032 entry.rows = None;
2033 entry.cols = cols;
2034 entry.cost = Cost {
2035 partitions: entry.cost.partitions.take(),
2036 ..Cost::default()
2037 };
2038}
2039
2040/// The columns a reader sees: the schema's own top-level fields.
2041///
2042/// Not the leaves a footer names, and not those leaves split on a dot either. Leaves
2043/// counted directly double for a directory whose writer changed — the same nested column
2044/// written by parquet-mr and by Arrow gives `inputs.list.element.address` in one file and
2045/// `inputs.bag.array_element.address` in the other, and a union of leaf paths holds both.
2046/// Splitting the dotted path fixes that and breaks something else: a column literally
2047/// named `user.id` is one column, and so is a struct `user` with a field `id`, and the
2048/// string cannot tell them apart.
2049///
2050/// The schema knows. `fields()` is the root's own children, which is what a struct counts
2051/// as here, what `schema_preview` lists in the details pane, and what the table shows.
2052fn top_level_names(meta: &crate::parquet_footer::Footer) -> Vec<String> {
2053 meta.schema_descr
2054 .fields()
2055 .iter()
2056 .map(|field| field.name().to_string())
2057 .collect()
2058}
2059
2060/// Every column name any of the files has, in the order they first appear.
2061fn union_of(per_file: &[Vec<String>]) -> Vec<String> {
2062 let mut seen = std::collections::HashSet::new();
2063 per_file
2064 .iter()
2065 .flatten()
2066 .filter(|name| seen.insert(name.as_str()))
2067 .cloned()
2068 .collect()
2069}
2070
2071/// The Parquet files under `dir`, sorted, as far down as a dataset goes: one past the
2072/// budget when there are more, which is what says there are too many to count. The
2073/// open's own walk, stopped once it has seen enough.
2074fn parquet_files_under(dir: &Path) -> Vec<PathBuf> {
2075 let mut files = crate::dataset_files::LocalFiles::new(dir)
2076 .first_files(MAX_WALK_DEPTH as usize + 1, MAX_FOOTERS_PER_DATASET);
2077 // The walk takes a directory entry's own type; a file this reads must be one.
2078 files.retain(|p| is_regular_file(p));
2079 files
2080}
2081
2082/// Measure a dataset from every file's footer as an open kept them: its rows, width,
2083/// size and row groups, and whether its files are one table, with none read here.
2084fn measure_from_footers(
2085 entry: &mut Entry,
2086 files: &[crate::dataset_files::DatasetFile],
2087 footers: &[Option<crate::schema_union::FileFooter>],
2088) {
2089 let per_file: Vec<Vec<String>> = footers
2090 .iter()
2091 .flatten()
2092 .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
2093 .collect();
2094 let columns = union_of(&per_file);
2095 if entry.kind == EntryKind::MultiFile && !crate::schema_union::is_nested(&per_file) {
2096 // As a read of every footer judges it: the label counts the directory's own
2097 // files, and so do the width and the size beside it.
2098 let own: Vec<usize> = files
2099 .iter()
2100 .enumerate()
2101 .filter(|(_, f)| Path::new(&f.key).parent() == Some(entry.path.as_path()))
2102 .map(|(i, _)| i)
2103 .collect();
2104 entry.size = Some(own.iter().map(|&i| files[i].size).sum());
2105 let own_columns = union_of(
2106 &own.iter()
2107 .filter_map(|&i| footers[i].as_ref())
2108 .map(|f| f.schema.iter_names().map(|n| n.to_string()).collect())
2109 .collect::<Vec<_>>(),
2110 );
2111 entry.columns = columns;
2112 entry.cols_sampled = false;
2113 downgrade_to_directory(
2114 entry,
2115 (!own_columns.is_empty()).then_some(own_columns.len()),
2116 );
2117 return;
2118 }
2119 let footers: Vec<&crate::schema_union::FileFooter> = footers.iter().flatten().collect();
2120 let uncompressed: u64 = footers
2121 .iter()
2122 .flat_map(|f| &f.column_bytes)
2123 .map(|(_, bytes)| *bytes as u64)
2124 .sum();
2125 let row_groups: usize = footers.iter().map(|f| f.row_group_rows.len()).sum();
2126 entry.rows = Some(footers.iter().map(|f| f.rows()).sum());
2127 entry.cols = Some(columns.len() + partition_columns_beyond(entry, &columns));
2128 entry.size = Some(files.iter().map(|f| f.size).sum());
2129 entry.columns = columns;
2130 entry.cols_sampled = false;
2131 entry.cost = Cost {
2132 uncompressed: (uncompressed > 0).then_some(uncompressed),
2133 row_groups: (row_groups > 0).then_some(row_groups),
2134 partitions: entry.cost.partitions.take(),
2135 ..Cost::default()
2136 };
2137}
2138
2139/// Fill in row and column counts for a Parquet file from its footer.
2140///
2141/// Free in the sense that matters: no column data is read. Non-Parquet formats have
2142/// no equivalent — a CSV's row count cannot be known without scanning it — so those
2143/// entries keep `None`, and the UI shows the absence honestly rather than guessing.
2144pub fn enrich_parquet(entry: &mut Entry) {
2145 if entry.kind != EntryKind::File {
2146 return;
2147 }
2148 if !is_parquet_path(&entry.path) {
2149 return;
2150 }
2151 if !is_regular_file(&entry.path) {
2152 return;
2153 }
2154 if let Some(meta) = crate::parquet_footer::read_parquet_metadata(&entry.path) {
2155 entry.rows = Some(meta.num_rows);
2156 entry.columns = column_names(&meta);
2157 // The columns a reader sees, as a directory's row reports them: `schema_descr`
2158 // names the leaves, so a file with one struct of three fields counted four and
2159 // then listed two in the pane beside it. See [`top_level_names`].
2160 entry.cols = Some(top_level_names(&meta).len());
2161 physical_facts(&meta, &mut entry.cost);
2162 }
2163}
2164
2165/// A file of tables' tables (a SQLite database's schema, a NumPy archive's directory):
2166/// how many of its own, whether Enter opens one of them, and the columns of the one when
2167/// there is one. A file whose name says a format its bytes must say (a `.db` file that
2168/// is not SQLite) is one datui cannot open.
2169pub fn enrich_tables(entry: &mut Entry) {
2170 if entry.kind != EntryKind::File || entry.table.is_some() {
2171 return;
2172 }
2173 let named = data_format(&entry.path);
2174 if !is_regular_file(&entry.path) {
2175 return;
2176 }
2177 let Some(format) = crate::members::holder(&entry.path) else {
2178 if named.is_some_and(|f| f.holds_tables() && crate::readers::of(f).bytes_decide) {
2179 entry.kind = EntryKind::Other;
2180 }
2181 return;
2182 };
2183 let Ok(tables) = crate::members::tables(&entry.path, format) else {
2184 return;
2185 };
2186 let own: Vec<&crate::sqlite::Table> = tables.iter().filter(|t| !t.internal).collect();
2187 entry.cost.tables = Some(own.len());
2188 entry.cost.opens_one = format.opens_one_table();
2189 if let [one] = own.as_slice()
2190 && !one.columns.is_empty()
2191 {
2192 entry.columns = one.columns.iter().map(|(name, _)| name.clone()).collect();
2193 entry.cols = Some(entry.columns.len());
2194 }
2195}
2196
2197/// The rows of a file of tables' listing on the home screen: a database's tables and
2198/// views, or an archive's arrays, by name, as a directory lists its files, SQLite's own
2199/// marked to be hidden, each at its path inside the file.
2200pub fn database_rows(file: &Path) -> Vec<Entry> {
2201 let Some(format) = crate::members::holder(file) else {
2202 return Vec::new();
2203 };
2204 let Ok(mut tables) = crate::members::tables(file, format) else {
2205 return Vec::new();
2206 };
2207 // A database's tables by name; an archive's arrays in the order they were saved.
2208 if format
2209 .descriptor()
2210 .tables
2211 .as_ref()
2212 .is_some_and(|t| t.by_name)
2213 {
2214 tables.sort_by_cached_key(|t| t.name.to_lowercase());
2215 }
2216 let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
2217 tables
2218 .into_iter()
2219 .map(|table| table_entry(file, format, table, modified))
2220 .collect()
2221}
2222
2223/// The rows of a Hugging Face cache directory's splits, each at its path inside the
2224/// directory (`cache/test`): opened, it is the directory read with `--table`. Empty
2225/// for any other directory, and for a cache of one split, which its door opens.
2226pub fn split_rows(dir: &Path) -> Vec<Entry> {
2227 let splits = crate::hf_splits::cache_splits(dir);
2228 if splits.len() < 2 {
2229 return Vec::new();
2230 }
2231 splits
2232 .into_iter()
2233 .map(|split| split_entry(dir, split))
2234 .collect()
2235}
2236
2237/// The row of a split named by its path inside its cache directory, as a recent is
2238/// listed: `None` when the path names no split of one.
2239pub fn split_row(path: &Path) -> Option<Entry> {
2240 let (dir, split) = crate::hf_splits::split_place(path)?;
2241 Some(split_entry(&dir, split))
2242}
2243
2244fn split_entry(dir: &Path, split: String) -> Entry {
2245 let mut entry = Entry::new(dir.join(&split), EntryKind::File);
2246 entry.name = split;
2247 entry.table = Some(TableOf {
2248 format: Some(crate::FileFormat::Arrow),
2249 kind: "split".to_string(),
2250 internal: false,
2251 });
2252 entry
2253}
2254
2255/// The rows of a file a format spec reads as several variants, one a variant, each at
2256/// its path inside the file (`day.itch/add`): opened, it is the file read with
2257/// `--table`. Empty for any other file.
2258pub fn variant_rows(file: &Path, formats: &crate::formats::Registry) -> Vec<Entry> {
2259 let Some((spec, tables)) = crate::members::variants(file, formats) else {
2260 return Vec::new();
2261 };
2262 let modified = std::fs::metadata(file).and_then(|m| m.modified()).ok();
2263 tables
2264 .into_iter()
2265 .map(|table| variant_entry(file, &spec, table, modified))
2266 .collect()
2267}
2268
2269/// The row of a variant named by its path inside its file (`day.itch/add`), as a
2270/// recent is listed: `None` when the path names no variant of such a file.
2271pub fn variant_row(path: &Path, formats: &crate::formats::Registry) -> Option<Entry> {
2272 let (file, name) = crate::members::split_variant(path, formats)?;
2273 let (spec, tables) = crate::members::variants(&file, formats)?;
2274 let table = tables.into_iter().find(|t| t.name == name)?;
2275 let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
2276 let mut entry = variant_entry(&file, &spec, table, modified);
2277 entry.path = path.to_path_buf();
2278 Some(entry)
2279}
2280
2281fn variant_entry(
2282 file: &Path,
2283 spec: &str,
2284 table: crate::sqlite::Table,
2285 modified: Option<std::time::SystemTime>,
2286) -> Entry {
2287 let mut entry = Entry::new(crate::members::place(file, &table.name), EntryKind::File);
2288 entry.name = table.name;
2289 entry.modified = modified;
2290 entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
2291 entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
2292 entry.format_spec = Some(spec.to_string());
2293 entry.table = Some(TableOf {
2294 format: None,
2295 kind: table.kind,
2296 internal: false,
2297 });
2298 entry
2299}
2300
2301/// The row of a table inside a file of tables named by its path (`app.db/users`), as a
2302/// recent is listed: `None` when the path names no table of such a file.
2303pub fn table_row(path: &Path) -> Option<Entry> {
2304 let (file, name) = crate::members::split(path)?;
2305 let format = crate::members::holder(&file)?;
2306 let table = crate::members::tables(&file, format)
2307 .ok()?
2308 .into_iter()
2309 .find(|t| t.name == name)?;
2310 let modified = std::fs::metadata(&file).and_then(|m| m.modified()).ok();
2311 let mut entry = table_entry(&file, format, table, modified);
2312 entry.path = path.to_path_buf();
2313 Some(entry)
2314}
2315
2316fn table_entry(
2317 file: &Path,
2318 format: crate::FileFormat,
2319 table: crate::sqlite::Table,
2320 modified: Option<std::time::SystemTime>,
2321) -> Entry {
2322 let mut entry = Entry::new(crate::members::place(file, &table.name), EntryKind::File);
2323 entry.name = table.name;
2324 entry.modified = modified;
2325 entry.columns = table.columns.into_iter().map(|(name, _)| name).collect();
2326 entry.cols = (!entry.columns.is_empty()).then_some(entry.columns.len());
2327 entry.table = Some(TableOf {
2328 format: Some(format),
2329 kind: table.kind,
2330 internal: table.internal,
2331 });
2332 entry
2333}
2334
2335/// The first bytes of `path`, as many as fit in `buf`. A short read is the whole file.
2336fn read_head<'a>(path: &Path, buf: &'a mut [u8]) -> Option<&'a [u8]> {
2337 use std::io::Read;
2338 let mut file = std::fs::File::open(path).ok()?;
2339 let mut filled = 0;
2340 loop {
2341 match file.read(&mut buf[filled..]) {
2342 Ok(0) => break,
2343 Ok(n) => filled += n,
2344 Err(_) => return None,
2345 }
2346 if filled == buf.len() {
2347 break;
2348 }
2349 }
2350 Some(&buf[..filled])
2351}
2352
2353/// Whether an Arrow file is an IPC stream: an IPC file starts `ARROW1`, a stream with
2354/// its schema message. Eight bytes, so a listing can say which will be converted.
2355fn enrich_arrow(entry: &mut Entry) {
2356 if entry.kind != EntryKind::File
2357 || data_format(&entry.path) != Some(crate::FileFormat::Arrow)
2358 || crate::CompressionFormat::from_extension(&entry.path).is_some()
2359 || !is_regular_file(&entry.path)
2360 {
2361 return;
2362 }
2363 let mut head = [0u8; 8];
2364 if let Some(head) = read_head(&entry.path, &mut head) {
2365 entry.cost.ipc_stream = !head.starts_with(b"ARROW1");
2366 }
2367}
2368
2369/// Pull layout and compression out of a footer that has already been read.
2370///
2371/// Every one of these was being parsed and thrown away. They are the difference
2372/// between knowing how big a file is and knowing what reading it will do.
2373pub fn physical_facts(meta: &crate::parquet_footer::Footer, cost: &mut Cost) {
2374 if meta.row_groups.is_empty() {
2375 return;
2376 }
2377 cost.row_groups = Some(meta.row_groups.len());
2378
2379 let mut uncompressed: u64 = 0;
2380 let mut codecs: Vec<String> = Vec::new();
2381 for rg in &meta.row_groups {
2382 uncompressed = uncompressed.saturating_add(rg.total_byte_size() as u64);
2383 for cc in rg.parquet_columns() {
2384 let codec = format!("{:?}", cc.compression()).to_lowercase();
2385 if !codecs.contains(&codec) {
2386 codecs.push(codec);
2387 }
2388 }
2389 }
2390 if uncompressed > 0 {
2391 cost.uncompressed = Some(uncompressed);
2392 }
2393 // A file usually uses one codec throughout. When it does not, say so rather than
2394 // picking one and implying uniformity that is not there.
2395 cost.codec = match codecs.len() {
2396 0 => None,
2397 1 => Some(codecs.remove(0)),
2398 n => Some(format!("mixed ({n})")),
2399 };
2400}
2401
2402/// Outermost directories to look at when describing a hive dataset's partitioning.
2403///
2404/// Enough to name the keys and show the shape of the first one; bounded because a
2405/// dataset partitioned by day over a decade has thousands, and counting all of them
2406/// to print "3,653" is not worth a second of anyone's time on a network share.
2407const MAX_PARTITION_DIRS: usize = 512;
2408
2409/// Describe how a hive dataset is partitioned, from directory names alone.
2410pub fn partition_layout(dir: &Path) -> Option<Partitions> {
2411 let iter = std::fs::read_dir(dir).ok()?;
2412 let mut values: Vec<String> = Vec::new();
2413 let mut keys: Vec<String> = Vec::new();
2414 let mut count = 0usize;
2415 let mut more = false;
2416
2417 for entry in iter.flatten() {
2418 if count >= MAX_PARTITION_DIRS {
2419 more = true;
2420 break;
2421 }
2422 let name = entry.file_name().to_string_lossy().into_owned();
2423 let Some((key, value)) = name.split_once('=') else {
2424 continue;
2425 };
2426 if !entry.path().is_dir() {
2427 continue;
2428 }
2429 if keys.is_empty() {
2430 keys.push(key.to_string());
2431 // Only the first partition directory is descended into, for the nested
2432 // key names. One is representative, and a hive dataset that disagrees
2433 // with itself about its own schema is not a dataset datui can help with.
2434 keys.extend(nested_keys(&entry.path()));
2435 }
2436 values.push(value.to_string());
2437 count += 1;
2438 }
2439
2440 if keys.is_empty() {
2441 return None;
2442 }
2443 values.sort();
2444 values.dedup();
2445 Some(Partitions {
2446 keys,
2447 first_key_values: values,
2448 count,
2449 more,
2450 })
2451}
2452
2453/// Partition keys below `dir`, following the first child at each level.
2454fn nested_keys(dir: &Path) -> Vec<String> {
2455 let mut keys = Vec::new();
2456 let mut current = dir.to_path_buf();
2457 // Bounded: a hive path deeper than this is pathological, and each level costs a
2458 // directory read.
2459 for _ in 0..6 {
2460 let Ok(iter) = std::fs::read_dir(¤t) else {
2461 break;
2462 };
2463 let Some(child) = iter
2464 .flatten()
2465 .find(|e| e.file_name().to_string_lossy().contains('=') && e.path().is_dir())
2466 else {
2467 break;
2468 };
2469 let name = child.file_name().to_string_lossy().into_owned();
2470 let Some((key, _)) = name.split_once('=') else {
2471 break;
2472 };
2473 keys.push(key.to_string());
2474 current = child.path();
2475 }
2476 keys
2477}
2478
2479/// Render a byte count compactly for a listing (`340 MB`).
2480pub fn format_size(bytes: u64) -> String {
2481 const UNITS: &[&str] = &["B", "KB", "MB", "GB", "TB"];
2482 let mut value = bytes as f64;
2483 let mut unit = 0;
2484 while value >= 1024.0 && unit < UNITS.len() - 1 {
2485 value /= 1024.0;
2486 unit += 1;
2487 }
2488 if unit == 0 {
2489 format!("{} {}", bytes, UNITS[0])
2490 } else if value >= 100.0 {
2491 format!("{:.0} {}", value, UNITS[unit])
2492 } else {
2493 format!("{:.1} {}", value, UNITS[unit])
2494 }
2495}
2496
2497/// Render a row count compactly (`2.4M`).
2498pub fn format_rows(rows: usize) -> String {
2499 let r = rows as f64;
2500 if rows >= 1_000_000_000 {
2501 format!("{:.1}B", r / 1e9)
2502 } else if rows >= 1_000_000 {
2503 format!("{:.1}M", r / 1e6)
2504 } else if rows >= 10_000 {
2505 format!("{:.0}k", r / 1e3)
2506 } else if rows >= 1_000 {
2507 // Below ten thousand the exact count fits and rounding actively misleads:
2508 // 3,653 daily observations is ten years of data, and "4k" is not.
2509 let mut out = String::new();
2510 let digits = rows.to_string();
2511 for (i, c) in digits.chars().enumerate() {
2512 if i > 0 && (digits.len() - i).is_multiple_of(3) {
2513 out.push(',');
2514 }
2515 out.push(c);
2516 }
2517 out
2518 } else {
2519 rows.to_string()
2520 }
2521}
2522
2523/// Render "how long ago" compactly (`2d`, `3h`).
2524pub fn format_age(t: std::time::SystemTime) -> String {
2525 let Ok(elapsed) = t.elapsed() else {
2526 return String::new();
2527 };
2528 let secs = elapsed.as_secs();
2529 if secs < 60 {
2530 "now".to_string()
2531 } else if secs < 3600 {
2532 format!("{}m", secs / 60)
2533 } else if secs < 86_400 {
2534 format!("{}h", secs / 3600)
2535 } else if secs < 86_400 * 365 {
2536 format!("{}d", secs / 86_400)
2537 } else {
2538 format!("{}y", secs / (86_400 * 365))
2539 }
2540}
2541
2542/// Column name and type, for the home screen's preview pane.
2543pub type SchemaPreview = Vec<(String, polars::prelude::DataType)>;
2544
2545/// The preview of a table of a file of tables: a row inside a database or an archive, or
2546/// a file of tables (or a NumPy array file) as it opens. `None` when the entry is none of
2547/// these, `Some(None)` when it is and has nothing to show.
2548fn table_preview(entry: &Entry) -> Option<Option<SchemaPreview>> {
2549 let (file, format, name) = match &entry.table {
2550 Some(table) => match (crate::members::split(&entry.path), table.format) {
2551 (Some((file, _)), Some(format)) => (file, format, Some(entry.name.as_str())),
2552 _ => return Some(None),
2553 },
2554 None if is_regular_file(&entry.path) => {
2555 let format = crate::members::holder(&entry.path).or_else(|| {
2556 data_format(&entry.path)
2557 .filter(|f| f.holds_tables() && crate::readers::of(*f).table_schema.is_some())
2558 })?;
2559 (entry.path.clone(), format, None)
2560 }
2561 None => return None,
2562 };
2563 Some(
2564 crate::readers::of(format)
2565 .table_schema
2566 .and_then(|schema| schema(&file, name)),
2567 )
2568}
2569
2570/// Find the first Parquet file at or under `dir`, without walking the whole tree.
2571///
2572/// Bounded on both breadth and depth so a hive dataset with thousands of partitions
2573/// costs the same as one with three.
2574fn first_parquet_under(dir: &Path, depth: u8) -> Option<PathBuf> {
2575 if depth > MAX_WALK_DEPTH {
2576 return None;
2577 }
2578 let mut subdirs = Vec::new();
2579 for entry in std::fs::read_dir(dir).ok()?.flatten().take(64) {
2580 let path = entry.path();
2581 if path.is_dir() {
2582 subdirs.push(path);
2583 } else if is_parquet_path(&path) && is_regular_file(&path) {
2584 return Some(path);
2585 }
2586 }
2587 subdirs.sort();
2588 subdirs
2589 .into_iter()
2590 .take(4)
2591 .find_map(|d| first_parquet_under(&d, depth + 1))
2592}
2593
2594/// Column names from a Parquet footer.
2595pub fn column_names(meta: &crate::parquet_footer::Footer) -> Vec<String> {
2596 meta.schema_descr
2597 .columns()
2598 .iter()
2599 .map(|c| c.path_in_schema.join("."))
2600 .collect()
2601}
2602
2603/// Whether `path` is a regular file that is safe to open.
2604///
2605/// Opening a FIFO blocks until a writer appears — indefinitely, for a named pipe
2606/// nobody is writing to — and opening a device or a socket does something stranger
2607/// still. A directory listing happily reports any of these with a `.parquet` name,
2608/// so every read here is gated on the kind first. `symlink_metadata` follows nothing
2609/// and `metadata` only stats, so neither can block the way an open can.
2610fn is_regular_file(path: &Path) -> bool {
2611 std::fs::metadata(path)
2612 .map(|m| m.file_type().is_file())
2613 .unwrap_or(false)
2614}
2615
2616/// Read a dataset's column names and types without reading any data.
2617///
2618/// Parquet only — a CSV's schema cannot be known without scanning it, and doing that
2619/// for every row the cursor passes over would defeat the point of a preview. Returns
2620/// `None` for anything else, and the UI says so rather than guessing.
2621pub fn schema_preview(entry: &Entry) -> Option<SchemaPreview> {
2622 // `SerReader` is what brings `ParquetReader::new` into scope.
2623 use polars::prelude::{ParquetReader, Schema, SchemaExt, SerReader};
2624
2625 let file_path = match entry.kind {
2626 EntryKind::File => {
2627 if let Some(preview) = table_preview(entry) {
2628 return preview;
2629 }
2630 if !is_parquet_path(&entry.path) {
2631 return None;
2632 }
2633 entry.path.clone()
2634 }
2635 EntryKind::Hive | EntryKind::MultiFile => first_parquet_under(&entry.path, 0)?,
2636 EntryKind::Directory | EntryKind::Unknown | EntryKind::Other => return None,
2637 // One data file's schema is not the table's: Iceberg field IDs and Delta
2638 // column mapping both mean a renamed column reads as two.
2639 EntryKind::Delta | EntryKind::Iceberg | EntryKind::Hudi => return None,
2640 };
2641
2642 if !is_regular_file(&file_path) {
2643 return None;
2644 }
2645 let file = std::fs::File::open(&file_path).ok()?;
2646 let mut reader = ParquetReader::new(file);
2647 let arrow_schema = reader.schema().ok()?;
2648 let schema = Schema::from_arrow_schema(arrow_schema.as_ref());
2649 let mut preview: SchemaPreview = Vec::new();
2650 // A hive table opens with its partition keys hoisted to the front, so the pane lists
2651 // them there too, typed from the one path already in hand the way the scan infers
2652 // them.
2653 if entry.kind == EntryKind::Hive
2654 && let Ok(below) = file_path.strip_prefix(&entry.path)
2655 {
2656 for part in below.parent().into_iter().flat_map(Path::components) {
2657 let part = part.as_os_str().to_string_lossy();
2658 if let Some((key, value)) = part.split_once('=')
2659 && !key.is_empty()
2660 && schema.get(key).is_none()
2661 {
2662 // The scan's own inference, dates and booleans included.
2663 let dtype = if value.is_empty() || value == "__HIVE_DEFAULT_PARTITION__" {
2664 polars::prelude::DataType::String
2665 } else {
2666 polars::io::csv::read::schema_inference::infer_field_schema(value, true, false)
2667 };
2668 preview.push((key.to_string(), dtype));
2669 }
2670 }
2671 }
2672 preview.extend(
2673 schema
2674 .iter()
2675 .map(|(name, dtype)| (name.to_string(), dtype.clone())),
2676 );
2677 Some(preview)
2678}
2679
2680#[cfg(test)]
2681mod classification_tests {
2682 use super::*;
2683 use polars::prelude::*;
2684
2685 /// A file row's read follows its format and how it is stored, and a remote file
2686 /// other than a Parquet object or a model file is downloaded first. Directories say nothing.
2687 #[test]
2688 fn how_a_row_is_read() {
2689 use crate::ReadMode::*;
2690 let how = |path: &str| how_read(&Entry::for_test(Path::new(path), path));
2691 let at = |path: &str, mode, download| {
2692 assert_eq!(how(path), Some(HowRead { mode, download }), "{path}");
2693 };
2694 at("/d/a.parquet", Lazy, false);
2695 at("/d/a.csv", Lazy, false);
2696 at("/d/a.csv.gz", Decompressed, false);
2697 at("/d/a.json", InMemory, false);
2698 at("/d/a.gpx", Converted, false);
2699 at("/d/a.arrow", Lazy, false);
2700 at("s3://b/a.parquet", Lazy, false);
2701 at("s3://b/a.csv", Lazy, true);
2702 at("gs://b/a.json", InMemory, true);
2703 at("https://example.com/a.parquet", Lazy, true);
2704 at("s3://b/m.safetensors", InMemory, false);
2705 at("https://example.com/m.gguf", InMemory, false);
2706 assert_eq!(how("/d/a.parquet.gz"), None, "does not open");
2707 assert_eq!(how("/d/README"), None);
2708
2709 let mut stream = Entry::for_test(Path::new("/d/x.arrow"), "x.arrow");
2710 stream.cost.ipc_stream = true;
2711 assert_eq!(how_read(&stream).map(|h| h.mode), Some(Converted));
2712 let mut spec = Entry::for_test(Path::new("/d/day.l2.zst"), "day.l2.zst");
2713 spec.format_spec = Some("acme.l2feed".into());
2714 assert_eq!(how_read(&spec).map(|h| h.mode), Some(Decompressed));
2715 at("/d/shop.db", Lazy, false);
2716 at("s3://b/shop.sqlite", Lazy, true);
2717 let mut table = Entry::for_test(Path::new("/d/shop.db/orders"), "orders");
2718 table.table = Some(TableOf {
2719 format: Some(crate::FileFormat::Sqlite),
2720 kind: "table".into(),
2721 internal: false,
2722 });
2723 assert_eq!(how_read(&table).map(|h| h.mode), Some(Lazy));
2724 assert_eq!(how_read(&Entry::directory(Path::new("/d/x"))), None);
2725 }
2726
2727 /// An Arrow file is told a stream by its first bytes when it is measured.
2728 #[test]
2729 fn measuring_an_arrow_file_tells_a_stream() {
2730 let dir = tempfile::tempdir().unwrap();
2731 let file = dir.path().join("file.arrow");
2732 std::fs::write(&file, b"ARROW1\0\0rest").unwrap();
2733 let stream = dir.path().join("stream.arrow");
2734 std::fs::write(&stream, b"\xff\xff\xff\xff\x10\x01\0\0").unwrap();
2735 for (path, is_stream) in [(file, false), (stream, true)] {
2736 let mut entry = Entry::for_test(&path, "x.arrow");
2737 enrich(&mut entry);
2738 assert_eq!(entry.cost.ipc_stream, is_stream, "{}", path.display());
2739 }
2740 }
2741
2742 /// Every extension the home screen offers has a reader behind it, and every
2743 /// extension a reader knows is offered. The two lists had drifted: `.psv` and
2744 /// `.xlsb` opened but were invisible, and `.txt` was listed and then refused.
2745 #[test]
2746 fn what_is_offered_and_what_opens_are_one_list() {
2747 for ext in [
2748 "parquet", "csv", "tsv", "psv", "json", "jsonl", "ndjson", "arrow", "arrows", "ipc",
2749 "feather", "avro", "orc", "xls", "xlsx", "xlsm", "xlsb",
2750 ] {
2751 let named = PathBuf::from(format!("sales.{ext}"));
2752 assert!(
2753 data_format(&named).is_some(),
2754 ".{ext} opens, so the home screen must offer it"
2755 );
2756 }
2757 // A README is text, read as lines; beside data it is not the directory's
2758 // table (`a_file_datui_does_not_read_does_not_disqualify_a_directory`).
2759 assert_eq!(
2760 data_format(Path::new("README.txt")),
2761 Some(crate::FileFormat::Text)
2762 );
2763 assert!(data_format(Path::new("notes")).is_none());
2764 }
2765
2766 /// A format's name is not an extension, and the one place that stores a name has
2767 /// to read it back with the inverse of what wrote it. `excel` is a name no
2768 /// extension spells, so parsing it as one answers `None` — and `None` there means
2769 /// "not Parquet", which leaves a directory's counts off rather than filling them from
2770 /// whatever Parquet is under it.
2771 #[test]
2772 fn a_format_name_round_trips_only_through_from_name() {
2773 use crate::FileFormat;
2774 for format in FileFormat::ALL {
2775 assert_eq!(
2776 FileFormat::from_name(format.name()),
2777 Some(format),
2778 "{} is a name",
2779 format.name()
2780 );
2781 }
2782 assert_eq!(FileFormat::from_extension("excel"), None);
2783 assert_eq!(FileFormat::from_name("xlsx"), None);
2784 }
2785
2786 /// A compression suffix is how a file is stored, not what it holds, on both routes.
2787 #[test]
2788 fn a_compressed_name_reads_as_the_format_under_it() {
2789 assert_eq!(
2790 data_format(Path::new("sales.csv.gz")),
2791 Some(crate::FileFormat::Csv)
2792 );
2793 assert_eq!(
2794 data_format(Path::new("events.json.zst")),
2795 Some(crate::FileFormat::Json)
2796 );
2797 }
2798
2799 /// `.ipc`, `.arrow` and `.feather` are one format under three names, so a directory
2800 /// holding two of them is one kind of thing rather than a mixture.
2801 #[test]
2802 fn one_format_under_several_names_is_not_a_mixture() {
2803 let dir = tempfile::tempdir().unwrap();
2804 std::fs::write(dir.path().join("a.arrow"), b"x").unwrap();
2805 std::fs::write(dir.path().join("b.ipc"), b"x").unwrap();
2806 assert_eq!(classify_directory(dir.path()), EntryKind::MultiFile);
2807 }
2808
2809 /// A README is neither a marker nor data. Locally it counts toward the majority
2810 /// and does not disqualify the directory; the cloud route counted it as data and
2811 /// answered `dir` where the local one said `multi`.
2812 #[test]
2813 fn a_file_datui_does_not_read_does_not_disqualify_a_directory() {
2814 let dir = tempfile::tempdir().unwrap();
2815 write(dir.path(), "a.parquet", &["id"]);
2816 write(dir.path(), "b.parquet", &["id"]);
2817 std::fs::write(dir.path().join("README.txt"), b"notes").unwrap();
2818
2819 #[cfg(feature = "cloud")]
2820 let objects: Vec<(String, u64)> = [
2821 ("out/a.parquet", 100u64),
2822 ("out/b.parquet", 100),
2823 ("out/README.txt", 12),
2824 ]
2825 .iter()
2826 .map(|(k, s)| ((*k).to_string(), *s))
2827 .collect();
2828
2829 #[cfg(feature = "cloud")]
2830 assert_eq!(
2831 classify_directory(dir.path()),
2832 crate::cloud_browse::look_at_listing("out/", &[], &objects).0,
2833 "the two routes answer the same directory alike"
2834 );
2835 assert_eq!(classify_directory(dir.path()), EntryKind::MultiFile);
2836 }
2837
2838 /// A label says what is inside, so it is true whatever `Enter` then does. The same
2839 /// directory of three tables reads `3 parquet` and is one to look inside.
2840 #[test]
2841 fn a_label_counts_what_is_there_rather_than_naming_a_decision() {
2842 let dir = tempfile::tempdir().unwrap();
2843 for name in ["a.parquet", "b.parquet", "c.parquet"] {
2844 write(dir.path(), name, &["id"]);
2845 }
2846 std::fs::create_dir_all(dir.path().join("archive")).unwrap();
2847 std::fs::write(dir.path().join("notes.csv"), b"x").unwrap();
2848 std::fs::write(dir.path().join("_SUCCESS"), b"").unwrap();
2849 std::fs::write(dir.path().join(".part.crc"), b"").unwrap();
2850
2851 let entry = measured(dir.path());
2852 assert_eq!(entry.label(), "mixed", "two formats is two formats");
2853 assert_eq!(
2854 entry.holds.line(true).as_deref(),
2855 Some("3 parquet · 1 csv · 1 directory"),
2856 "and the pane says what the label boiled down"
2857 );
2858 assert_eq!(entry.holds.data_files(), 4);
2859 assert_eq!(entry.holds.directories, 1);
2860 }
2861
2862 /// One format, and the count is the files.
2863 #[test]
2864 fn a_directory_of_one_format_is_labelled_by_it() {
2865 let dir = tempfile::tempdir().unwrap();
2866 for i in 0..12 {
2867 write(dir.path(), &format!("part-{i:05}.parquet"), &["id", "ts"]);
2868 }
2869 let entry = measured(dir.path());
2870 assert_eq!(entry.label(), "12 parquet");
2871 assert_eq!(entry.holds.line(true).as_deref(), Some("12 parquet"));
2872
2873 // A directory with nothing in it datui reads is a place to look inside. The
2874 // files are counted, but the pane's line is about what can be opened, and
2875 // "20 not read" read as a fault in a directory with nothing wrong in it.
2876 let plain = tempfile::tempdir().unwrap();
2877 for i in 0..20 {
2878 std::fs::write(plain.path().join(format!("note{i}.md")), b"x").unwrap();
2879 }
2880 let plain = measured(plain.path());
2881 assert_eq!(plain.label(), "dir");
2882 assert_eq!(plain.holds.not_read, 20);
2883 assert_eq!(plain.holds.line(true), None);
2884 }
2885
2886 /// A row nothing has looked into has only its kind to go on, and a hive root or a
2887 /// lake table is named by the thing it is rather than counted.
2888 #[test]
2889 fn a_kind_that_names_itself_keeps_its_name() {
2890 let dir = tempfile::tempdir().unwrap();
2891 std::fs::create_dir_all(dir.path().join("year=2024")).unwrap();
2892 std::fs::create_dir_all(dir.path().join("year=2025")).unwrap();
2893 assert_eq!(measured(dir.path()).label(), "hive");
2894
2895 let lake = tempfile::tempdir().unwrap();
2896 std::fs::create_dir_all(lake.path().join("_delta_log")).unwrap();
2897 assert_eq!(measured(lake.path()).label(), "delta");
2898
2899 let unlooked = Entry::new(PathBuf::from("/nowhere"), EntryKind::Unknown);
2900 assert_eq!(unlooked.label(), "");
2901 }
2902
2903 /// A directory of exactly the cap is a total, not a floor. `5000+` claims there is
2904 /// more; saying so about a directory that was read whole is a lie in the direction
2905 /// nobody can check.
2906 #[test]
2907 fn a_directory_read_whole_does_not_claim_there_is_more() {
2908 let dir = tempfile::tempdir().unwrap();
2909 for i in 0..MAX_ENTRIES_PER_DIR {
2910 std::fs::write(dir.path().join(format!("f{i:05}.csv")), b"x").unwrap();
2911 }
2912 let holds = look_at_directory(dir.path()).1;
2913 assert!(!holds.truncated, "every entry was read");
2914 assert_eq!(holds.label(), format!("{MAX_ENTRIES_PER_DIR} csv"));
2915
2916 std::fs::write(dir.path().join("one-more.csv"), b"x").unwrap();
2917 let holds = look_at_directory(dir.path()).1;
2918 assert!(holds.truncated, "and now there is more than was read");
2919 assert!(holds.label().contains('+'));
2920 }
2921
2922 /// A lake table is answered by three `join` tests and costs no listing. Counting
2923 /// one would walk every table in a warehouse on every pass, for a line beside a
2924 /// table whose files `enrich` then refuses to read anyway.
2925 #[test]
2926 fn a_lake_table_is_not_counted() {
2927 let dir = tempfile::tempdir().unwrap();
2928 std::fs::create_dir_all(dir.path().join("_delta_log")).unwrap();
2929 write(dir.path(), "part-00000.parquet", &["id"]);
2930 write(dir.path(), "part-00001.parquet", &["id"]);
2931
2932 let (kind, holds) = look_at_directory(dir.path());
2933 assert_eq!(kind, EntryKind::Delta);
2934 assert!(holds.is_empty(), "and its label is the format's own name");
2935 let entry = measured(dir.path());
2936 assert_eq!(entry.label(), "delta");
2937 }
2938
2939 /// A listing cut short cannot say there is no data in a directory, only that it found
2940 /// none among the entries it read. `mixed` needs no such qualifier — more files
2941 /// cannot unmake it — and `dir` does, because they can.
2942 #[test]
2943 fn a_cut_short_listing_does_not_claim_a_directory_is_empty() {
2944 let seen = Holds {
2945 skipped: 5000,
2946 truncated: true,
2947 ..Default::default()
2948 };
2949 assert_eq!(seen.label(), "dir+");
2950
2951 let whole = Holds {
2952 skipped: 3,
2953 ..Default::default()
2954 };
2955 assert_eq!(whole.label(), "dir");
2956
2957 // And a listing cut short before it found anything at all still says so: it is
2958 // not an empty tally, or the row falls back to its kind and reads `dir`.
2959 let nothing_yet = Holds {
2960 truncated: true,
2961 ..Default::default()
2962 };
2963 assert!(!nothing_yet.is_empty());
2964 assert_eq!(nothing_yet.label(), "dir+");
2965 assert!(Holds::default().is_empty());
2966
2967 let mixed = Holds {
2968 formats: vec![("parquet".to_string(), 3), ("csv".to_string(), 2)],
2969 truncated: true,
2970 ..Default::default()
2971 };
2972 assert_eq!(mixed.label(), "mixed", "more files cannot unmake it");
2973 }
2974
2975 /// A name cut to fit keeps both ends. A Hadoop output directory's `.crc` files are
2976 /// named for the file they check, and the head and the tail are what say so.
2977 #[test]
2978 fn a_long_name_keeps_both_ends() {
2979 let name = ".part-00000-8f3a91c2-7b4d-4e19-a6f0-c1d2e3f4a5b6-c000.snappy.parquet.crc";
2980 let line = shorten(name, 24);
2981 assert!(line.starts_with(".part-00000"), "the head: {line}");
2982 // The tail, as much of it as the ellipsis leaves: it takes three characters of
2983 // the twenty-four in the ASCII glyph set and one in the Unicode one.
2984 assert!(line.ends_with(".crc"), "and the tail: {line}");
2985 assert!(!line.contains("8f3a91c2"), "the middle goes: {line}");
2986 assert!(line.chars().count() <= 24, "{line}");
2987 }
2988
2989 /// The same directory read twice reads the same. Skipped names come back in whatever
2990 /// order the filesystem holds them, so the pane takes the first few *by name*.
2991 #[test]
2992 fn what_a_directory_holds_reads_the_same_twice() {
2993 let dir = tempfile::tempdir().unwrap();
2994 write(dir.path(), "a.parquet", &["id"]);
2995 for marker in [
2996 "_SUCCESS",
2997 "_committed_9",
2998 "_committed_1",
2999 ".crc",
3000 "_started_4",
3001 ] {
3002 std::fs::write(dir.path().join(marker), b"").unwrap();
3003 }
3004 let first = look_at_directory(dir.path()).1;
3005 for _ in 0..8 {
3006 assert_eq!(look_at_directory(dir.path()).1, first);
3007 }
3008 assert_eq!(
3009 first.skipped_names,
3010 vec![".crc", "_SUCCESS", "_committed_1", "_committed_9"],
3011 "the first four by name, of five"
3012 );
3013 assert_eq!(first.skipped, 5);
3014 }
3015
3016 /// The label counts what is directly inside; the numbers beside it are a promise
3017 /// about what `Enter` gives, and `Enter` reads the subtree. Measuring only the top
3018 /// would promise three files and open twenty-three — and would ask `is_one_table`
3019 /// about three files while unioning all twenty-three, which is the union the
3020 /// downgrade exists to prevent. The `holds` line names the directory that explains
3021 /// it.
3022 #[test]
3023 fn a_directories_numbers_are_what_opening_it_gives() {
3024 let dir = tempfile::tempdir().unwrap();
3025 for name in ["a.parquet", "b.parquet", "c.parquet"] {
3026 write(dir.path(), name, &["id", "legacy"]);
3027 }
3028 let archive = dir.path().join("archive");
3029 std::fs::create_dir_all(&archive).unwrap();
3030 for i in 0..20 {
3031 write(&archive, &format!("old-{i}.parquet"), &["id", "legacy"]);
3032 }
3033
3034 let entry = measured(dir.path());
3035 assert_eq!(
3036 entry.label(),
3037 "3 parquet",
3038 "three files are directly inside"
3039 );
3040 assert_eq!(
3041 entry.holds.line(true).as_deref(),
3042 Some("3 parquet · 1 directory")
3043 );
3044 assert_eq!(entry.rows, Some(23), "and opening it reads all of them");
3045 }
3046
3047 /// A directory past the counting budget whose *own* files were all read says an exact
3048 /// width. The budget is about the subtree; three files at the top are three
3049 /// footers, and `5+ cols` claims a sample that did not happen.
3050 #[test]
3051 fn a_width_is_a_floor_only_when_a_footer_went_unread() {
3052 let dir = tempfile::tempdir().unwrap();
3053 for name in ["a.parquet", "b.parquet", "c.parquet"] {
3054 write(dir.path(), name, &["id", "ts"]);
3055 }
3056 let archive = dir.path().join("archive");
3057 std::fs::create_dir_all(&archive).unwrap();
3058 for i in 0..MAX_FOOTERS_PER_DATASET + 6 {
3059 write(
3060 &archive,
3061 &format!("old-{i:03}.parquet"),
3062 &["wholly", "different"],
3063 );
3064 }
3065
3066 let entry = measured(dir.path());
3067 assert_eq!(entry.kind, EntryKind::Directory, "not one table");
3068 assert_eq!(entry.label(), "3 parquet");
3069 assert_eq!(entry.cols, Some(2), "id and ts");
3070 assert!(
3071 !entry.cols_sampled,
3072 "all three of its own footers were read"
3073 );
3074 }
3075
3076 #[test]
3077 fn a_big_directory_is_still_found_by_a_column_one_level_down() {
3078 let dir = tempfile::tempdir().unwrap();
3079 for name in ["a.parquet", "b.parquet", "c.parquet"] {
3080 write(dir.path(), name, &["id", "ts"]);
3081 }
3082 let archive = dir.path().join("archive");
3083 std::fs::create_dir_all(&archive).unwrap();
3084 for i in 0..MAX_FOOTERS_PER_DATASET + 6 {
3085 write(
3086 &archive,
3087 &format!("old-{i:03}.parquet"),
3088 &["wholly", "different"],
3089 );
3090 }
3091
3092 let entry = measured(dir.path());
3093 assert_eq!(entry.kind, EntryKind::Directory);
3094 // The label and the width are the three files directly inside.
3095 assert_eq!(entry.label(), "3 parquet");
3096 assert_eq!(entry.cols, Some(2), "id and ts");
3097 // The names are not: they are the home screen's search index, and `wholly` has
3098 // to reach the directory that holds one whether the directory was small enough to
3099 // read every footer or, as here, too big and sampled instead. Narrowing these
3100 // to the directory's own files made the answer depend on the directory's size.
3101 assert!(
3102 entry.columns.contains(&"wholly".to_string()),
3103 "{:?}",
3104 entry.columns
3105 );
3106 assert!(entry.columns.contains(&"id".to_string()));
3107 }
3108
3109 #[test]
3110 fn a_width_over_a_directories_own_files_is_a_floor_when_there_are_too_many() {
3111 let dir = tempfile::tempdir().unwrap();
3112 // Past the footer budget with the directory's *own* files, and no two of them one
3113 // table, so the downgrade samples its own files as well and says so. Every file
3114 // gets its own column, because which three get sampled is `read_dir` order.
3115 for i in 0..MAX_FOOTERS_PER_DATASET + 6 {
3116 write(
3117 dir.path(),
3118 &format!("f-{i:03}.parquet"),
3119 &[&format!("c{i}")],
3120 );
3121 }
3122
3123 let entry = measured(dir.path());
3124 assert_eq!(entry.kind, EntryKind::Directory, "not one table");
3125 assert_eq!(entry.label(), "70 parquet");
3126 assert!(
3127 entry.cols_sampled,
3128 "three of seventy footers were read, so the width is a floor"
3129 );
3130 }
3131
3132 #[test]
3133 fn a_directory_read_as_one_table_is_sized_by_everything_under_it() {
3134 let dir = tempfile::tempdir().unwrap();
3135 write(dir.path(), "a.parquet", &["id", "ts"]);
3136 write(dir.path(), "b.parquet", &["id", "ts"]);
3137 let more = dir.path().join("more");
3138 std::fs::create_dir_all(&more).unwrap();
3139 write(&more, "c.parquet", &["id", "ts"]);
3140
3141 let entry = measured(dir.path());
3142 assert_eq!(entry.kind, EntryKind::MultiFile, "one table");
3143 // `Enter` unions the subtree, so the size and the rows beside it are the
3144 // subtree's — the opposite of a downgraded directory, whose numbers are its own
3145 // files because it is never opened as one table.
3146 let all: u64 = [
3147 dir.path().join("a.parquet"),
3148 dir.path().join("b.parquet"),
3149 more.join("c.parquet"),
3150 ]
3151 .iter()
3152 .map(|p| std::fs::metadata(p).unwrap().len())
3153 .sum();
3154 assert_eq!(entry.size, Some(all));
3155 assert_eq!(entry.rows, Some(3));
3156 }
3157
3158 #[test]
3159 fn nothing_counted_is_the_only_thing_holds_calls_empty() {
3160 // `is_empty` stops a peek's answer reaching a row and keeps a `Holds` out of
3161 // the cache, so anything it calls empty is thrown away. Asked of each field on
3162 // its own, because the contract is the function's and not its callers': both
3163 // routes happen to set `directories` beside `partitions` and `skipped` beside
3164 // `skipped_names` today, which is exactly the kind of agreement that stops
3165 // holding one refactor later.
3166 assert!(Holds::default().is_empty());
3167 let one = |f: fn(&mut Holds)| {
3168 let mut h = Holds::default();
3169 f(&mut h);
3170 h
3171 };
3172 for (what, holds) in [
3173 ("a data file", one(|h| h.formats.push(("csv".into(), 1)))),
3174 ("a directory", one(|h| h.directories = 1)),
3175 ("a partition", one(|h| h.partitions = 1)),
3176 ("a file it cannot read", one(|h| h.not_read = 1)),
3177 ("a writer's own file", one(|h| h.skipped = 1)),
3178 (
3179 "the name of one",
3180 one(|h| h.skipped_names.push("_SUCCESS".into())),
3181 ),
3182 ("a listing cut short", one(|h| h.truncated = true)),
3183 ] {
3184 assert!(!holds.is_empty(), "{what} is something to say");
3185 }
3186 }
3187
3188 #[test]
3189 fn formats_that_tie_are_ordered_by_name_whatever_order_they_arrived_in() {
3190 use crate::FileFormat;
3191 // Given in the order that is wrong on both counts, so neither clause of the
3192 // comparison can be the one doing nothing. Without the tie-break a directory of
3193 // two CSV and two JSON reads `2 csv · 2 json` on one pass and `2 json · 2 csv`
3194 // on the next, which is the `read_dir` order this release exists to remove.
3195 let mut counts = vec![
3196 (FileFormat::Json, 2),
3197 (FileFormat::Csv, 2),
3198 (FileFormat::Parquet, 5),
3199 ];
3200 order_formats(&mut counts);
3201 assert_eq!(
3202 counts,
3203 vec![
3204 (FileFormat::Parquet, 5),
3205 (FileFormat::Csv, 2),
3206 (FileFormat::Json, 2)
3207 ]
3208 );
3209 }
3210
3211 /// The label and the read name the same format, including on a tie.
3212 ///
3213 /// They agreed on the common case and not on a tie: the label sorted equal counts
3214 /// by name and the read put Parquet first, so a directory of two CSV and two Parquet
3215 /// was labelled `2 csv · 2 parquet` and opened as Parquet. One order now, and this
3216 /// is the case that tells the two orders apart.
3217 #[test]
3218 fn the_label_and_the_read_pick_the_same_format_on_a_tie() {
3219 use crate::FileFormat;
3220 let tmp = tempfile::TempDir::new().unwrap();
3221 for name in ["a.csv", "b.csv", "c.parquet", "d.parquet"] {
3222 std::fs::write(tmp.path().join(name), b"x").unwrap();
3223 }
3224
3225 let (_, holds) = look_at_directory(tmp.path());
3226 assert_eq!(
3227 holds.formats.first().map(|(f, n)| (f.as_str(), *n)),
3228 Some(("parquet", 2)),
3229 "the label names Parquet first: {:?}",
3230 holds.formats
3231 );
3232
3233 match directory_format(tmp.path()) {
3234 DirectoryFormat::Mixed { format, .. } => assert_eq!(
3235 format,
3236 FileFormat::Parquet,
3237 "and so does the reader the open picks"
3238 ),
3239 other => panic!("a directory of two formats is mixed, got {other:?}"),
3240 }
3241 }
3242
3243 /// A Hugging Face dataset saved to disk is one shard and two JSON files that
3244 /// describe it: a dataset of Arrow, labelled and read as one, the JSON its writer's
3245 /// own. Beside no Arrow, the same names are data.
3246 #[test]
3247 fn a_hugging_face_dataset_is_its_shards() {
3248 use crate::FileFormat;
3249 let tmp = tempfile::TempDir::new().unwrap();
3250 for name in [
3251 "data-00000-of-00002.arrow",
3252 "data-00001-of-00002.arrow",
3253 "dataset_info.json",
3254 "state.json",
3255 ] {
3256 std::fs::write(tmp.path().join(name), b"x").unwrap();
3257 }
3258 let (kind, holds) = look_at_directory(tmp.path());
3259 assert_eq!(kind, EntryKind::MultiFile);
3260 assert_eq!(holds.formats, [("arrow".to_string(), 2)]);
3261 assert_eq!(holds.skipped, 2);
3262 assert_eq!(holds.skipped_names, ["dataset_info.json", "state.json"]);
3263 match directory_format(tmp.path()) {
3264 DirectoryFormat::One(FileFormat::Arrow, files) => assert_eq!(files.len(), 2),
3265 other => panic!("the shards are the dataset, got {other:?}"),
3266 }
3267
3268 std::fs::remove_file(tmp.path().join("data-00001-of-00002.arrow")).unwrap();
3269 assert!(matches!(
3270 directory_format(tmp.path()),
3271 DirectoryFormat::One(FileFormat::Arrow, _)
3272 ));
3273
3274 let json = tempfile::TempDir::new().unwrap();
3275 for name in ["state.json", "other.json"] {
3276 std::fs::write(json.path().join(name), b"{}").unwrap();
3277 }
3278 assert_eq!(
3279 look_at_directory(json.path()).1.formats,
3280 [("json".to_string(), 2)]
3281 );
3282 }
3283
3284 #[test]
3285 fn partitions_carry_a_directory_only_while_they_are_the_most_of_it() {
3286 // The boundary the local rule turns on, and the twin of the cloud route's
3287 // `partitions_carry_a_prefix_only_while_they_are_the_most_of_it`. Every directory
3288 // on disk goes through this one.
3289 let laid_out = |strays: usize| {
3290 let dir = tempfile::tempdir().unwrap();
3291 for year in ["year=2024", "year=2025"] {
3292 let part = dir.path().join(year);
3293 std::fs::create_dir_all(&part).unwrap();
3294 write(&part, "data.parquet", &["id"]);
3295 }
3296 for i in 0..strays {
3297 write(dir.path(), &format!("stray-{i}.parquet"), &["id"]);
3298 }
3299 classify_directory(dir.path())
3300 };
3301 assert_eq!(
3302 laid_out(2),
3303 EntryKind::Hive,
3304 "two partitions against two files beside them"
3305 );
3306 assert_ne!(
3307 laid_out(3),
3308 EntryKind::Hive,
3309 "one more file than partitions is a directory that holds a key=value"
3310 );
3311 }
3312
3313 #[test]
3314 fn a_folder_marker_is_bookkeeping_even_beside_a_partition() {
3315 // Legacy s3n and EMR write a zero-byte `<name>_$folder$` object beside every
3316 // prefix. Where the prefix is a partition the marker carries the `=` too, so a
3317 // partition test that only looks for one calls the marker data and the pane
3318 // reports one unreadable file per partition.
3319 assert!(is_bookkeeping("year=2024_$folder$"));
3320 assert!(is_bookkeeping("alpha_$folder$"));
3321 assert!(!is_bookkeeping("year=2024"), "the partition itself is data");
3322 assert!(
3323 !is_bookkeeping("_date=2024-01-01"),
3324 "Spark partitions on internal columns"
3325 );
3326 }
3327
3328 /// And the files under it are what the one-table test is asked about, since they
3329 /// are what the union would hold.
3330 #[test]
3331 fn a_table_hidden_under_a_directory_still_downgrades_it() {
3332 let dir = tempfile::tempdir().unwrap();
3333 for name in ["a.parquet", "b.parquet"] {
3334 write(dir.path(), name, &["id", "ts"]);
3335 }
3336 let archive = dir.path().join("archive");
3337 std::fs::create_dir_all(&archive).unwrap();
3338 write(
3339 &archive,
3340 "other.parquet",
3341 &["wholly", "different", "columns"],
3342 );
3343
3344 let entry = measured(dir.path());
3345 assert_eq!(
3346 entry.kind,
3347 EntryKind::Directory,
3348 "a union over these is not one table"
3349 );
3350 assert_eq!(entry.rows, None);
3351 // And the width beside `2 parquet` is those two files. The check is asked of
3352 // everything under the directory, because that is what opening it would union;
3353 // a downgraded row is never opened as one, so reporting the subtree's union
3354 // would be a set of columns nothing produces.
3355 assert_eq!(entry.label(), "2 parquet");
3356 assert_eq!(entry.cols, Some(2), "id and ts");
3357 // The names are every column under the directory, because they are what the home
3358 // screen searches: the directory does hold a `wholly`, one level down.
3359 assert!(entry.columns.contains(&"wholly".to_string()));
3360 assert!(entry.columns.contains(&"id".to_string()));
3361
3362 // And the size is those two files, not the subtree's: three numbers on one row
3363 // measured over three different sets of files is no row at all.
3364 let own: u64 = ["a.parquet", "b.parquet"]
3365 .iter()
3366 .map(|n| std::fs::metadata(dir.path().join(n)).unwrap().len())
3367 .sum();
3368 assert_eq!(entry.size, Some(own));
3369 }
3370
3371 /// A hive tree of CSV is still laid out, whatever its rows cannot say. The layout
3372 /// is directory names — no footers, no opens — and it is the thing you most want
3373 /// before opening a dataset too large to count.
3374 #[test]
3375 fn a_hive_tree_of_another_format_is_still_laid_out() {
3376 let dir = tempfile::tempdir().unwrap();
3377 for year in ["year=2024", "year=2025"] {
3378 let part = dir.path().join(year);
3379 std::fs::create_dir_all(&part).unwrap();
3380 std::fs::write(part.join("data.csv"), b"id\n1\n").unwrap();
3381 }
3382 // A stray data file at the root, which is a hive root's ordinary furniture.
3383 std::fs::write(dir.path().join("summary.csv"), b"id\n1\n").unwrap();
3384
3385 let entry = measured(dir.path());
3386 assert_eq!(entry.kind, EntryKind::Hive);
3387 assert!(entry.cost.partitions.is_some(), "the layout is named");
3388 assert_eq!(entry.rows, None, "and nothing is invented about its rows");
3389 }
3390
3391 /// A hive root's own files are strays beside the partitions — a `schema.json` or a
3392 /// `manifest.csv` left at the top — so its counted format is not its data's, and
3393 /// asking it would blank the whole dataset for one such file.
3394 #[test]
3395 fn a_hive_dataset_is_described_despite_a_stray_file_at_its_root() {
3396 let dir = tempfile::tempdir().unwrap();
3397 for year in ["year=2024", "year=2025"] {
3398 let part = dir.path().join(year);
3399 std::fs::create_dir_all(&part).unwrap();
3400 write(&part, "data.parquet", &["id"]);
3401 }
3402 std::fs::write(dir.path().join("schema.json"), b"{}").unwrap();
3403
3404 let entry = measured(dir.path());
3405 assert_eq!(entry.kind, EntryKind::Hive);
3406 assert_eq!(entry.holds.one_format(), Some("json"), "its own only file");
3407 assert_eq!(entry.rows, Some(2), "and the dataset is still counted");
3408 assert_eq!(entry.cols, Some(2), "`id` and the partition column `year`");
3409 }
3410
3411 /// A hive dataset is described whatever odd file is lying in a partition. One
3412 /// spine cannot tell a Parquet tree with a stray CSV in it from a CSV tree with a
3413 /// stray Parquet, and blanking a dataset that opens perfectly is the worse of the
3414 /// two mistakes.
3415 #[test]
3416 fn a_hive_dataset_is_described_despite_a_stray_file() {
3417 let dir = tempfile::tempdir().unwrap();
3418 for year in ["year=2024", "year=2025"] {
3419 let part = dir.path().join(year);
3420 std::fs::create_dir_all(&part).unwrap();
3421 write(&part, "data.parquet", &["id"]);
3422 }
3423 // Somebody's notes, dropped in beside the data.
3424 std::fs::write(dir.path().join("year=2024/notes.csv"), b"x").unwrap();
3425
3426 let entry = measured(dir.path());
3427 assert_eq!(entry.kind, EntryKind::Hive);
3428 assert_eq!(entry.rows, Some(2), "the dataset is still counted");
3429 assert!(
3430 entry.cost.partitions.is_some(),
3431 "and its layout still named"
3432 );
3433 }
3434
3435 /// A directory is described by its own files, not by what is under them. The footer
3436 /// walk recurses, which is right for a hive root and wrong for a directory of JSON
3437 /// that happens to have Parquet in a subdirectory.
3438 #[test]
3439 fn a_directory_is_not_described_by_files_it_does_not_name() {
3440 let dir = tempfile::tempdir().unwrap();
3441 for name in ["a.json", "b.json", "c.json"] {
3442 std::fs::write(dir.path().join(name), b"{}").unwrap();
3443 }
3444 let under = dir.path().join("derived");
3445 std::fs::create_dir_all(&under).unwrap();
3446 write(&under, "one.parquet", &["id", "ts", "amount"]);
3447 write(&under, "two.parquet", &["id", "ts", "amount"]);
3448
3449 let entry = measured(dir.path());
3450 assert_eq!(entry.label(), "3 json");
3451 assert_eq!(
3452 entry.cols, None,
3453 "the Parquet below it is not this directory's shape"
3454 );
3455 assert_eq!(entry.rows, None);
3456 assert!(entry.columns.is_empty());
3457 }
3458
3459 /// The gate's default, for a dataset row that counted nothing.
3460 ///
3461 /// **Nothing produces this row today.** `enrich` only reaches the gate for `Hive`
3462 /// and `MultiFile`; `look_at_directory` cannot answer `MultiFile` without counting
3463 /// a format, a hive root skips the gate outright, `CLASSIFIER_VERSION` 4 refuses a
3464 /// cached kind that arrives without a tally, and the cloud `(all files)` row is
3465 /// built from a listing and never measured. So this constructs the row by hand, and
3466 /// it pins a default rather than a path.
3467 ///
3468 /// It is worth pinning because the default is the arguable one. Turning it away
3469 /// would blank the size, the width and the row count of any such row the moment one
3470 /// appeared, and leaving the counts off a directory is a mistake opening it undoes —
3471 /// giving it another format's numbers is not.
3472 #[test]
3473 fn a_dataset_row_that_counted_nothing_is_still_described() {
3474 let dir = tempfile::tempdir().unwrap();
3475 write(dir.path(), "a.parquet", &["id", "ts"]);
3476 write(dir.path(), "b.parquet", &["id", "ts"]);
3477
3478 let mut entry = Entry {
3479 kind: EntryKind::MultiFile,
3480 ..Entry::for_test(dir.path(), "data")
3481 };
3482 assert!(entry.holds.one_format().is_none(), "nothing counted");
3483
3484 enrich(&mut entry);
3485 assert!(entry.size.is_some(), "the footers were read");
3486 assert_eq!(entry.rows, Some(2));
3487 assert_eq!(entry.cols, Some(2), "id and ts");
3488 }
3489
3490 /// A Parquet file whose name begins with `_` is still a Parquet file. It does not
3491 /// count toward what the directory around it holds — that is what `is_bookkeeping` is
3492 /// for — but the listing shows it, `Enter` opens it, and the row beside it must say
3493 /// how many rows it has rather than nothing at all.
3494 #[test]
3495 fn a_parquet_file_named_like_a_writers_file_is_still_measured() {
3496 let dir = tempfile::tempdir().unwrap();
3497 write(dir.path(), "_2024_sales.parquet", &["id", "amount"]);
3498
3499 let mut entry = Entry {
3500 path: dir.path().join("_2024_sales.parquet"),
3501 kind: EntryKind::File,
3502 name: "_2024_sales.parquet".into(),
3503 size: None,
3504 modified: None,
3505 rows: None,
3506 cols: None,
3507 cols_sampled: false,
3508 columns: Vec::new(),
3509 cost: Cost::default(),
3510 holds: Default::default(),
3511 opens_whole_directory: false,
3512 format_spec: None,
3513 table: None,
3514 };
3515 enrich(&mut entry);
3516 assert_eq!(entry.rows, Some(1), "its footer was read");
3517 assert_eq!(entry.cols, Some(2));
3518 assert!(schema_preview(&entry).is_some(), "and the pane shows it");
3519
3520 // And it still does not make the directory around it a dataset.
3521 assert!(is_bookkeeping("_2024_sales.parquet"));
3522 assert_eq!(classify_directory(dir.path()), EntryKind::Directory);
3523 }
3524
3525 /// A hive table's pane lists its partition keys typed the way the scan types them:
3526 /// a date is a date and `true` a boolean, not text.
3527 #[test]
3528 fn a_hive_preview_types_its_keys_as_the_scan_does() {
3529 use polars::prelude::DataType;
3530 let dir = tempfile::tempdir().unwrap();
3531 let leaf = dir.path().join("day=2024-01-02/flag=true/n=3/x=1.5");
3532 std::fs::create_dir_all(&leaf).unwrap();
3533 write(&leaf, "part.parquet", &["id"]);
3534 let mut entry = Entry::directory(dir.path());
3535 entry.kind = EntryKind::Hive;
3536 let preview = schema_preview(&entry).expect("a footer to read");
3537 let types: Vec<(&str, &DataType)> = preview.iter().map(|(n, t)| (n.as_str(), t)).collect();
3538 assert_eq!(
3539 types[..4],
3540 [
3541 ("day", &DataType::Date),
3542 ("flag", &DataType::Boolean),
3543 ("n", &DataType::Int64),
3544 ("x", &DataType::Float64),
3545 ]
3546 );
3547 assert_eq!(types[4].0, "id");
3548 }
3549
3550 /// Part files with no extension inside a `.parquet` directory are data by where they
3551 /// sit. The cloud route has always counted them; the local one said `dir`.
3552 #[cfg(feature = "cloud")]
3553 #[test]
3554 fn extensionless_part_files_are_data_on_both_routes() {
3555 let dir = tempfile::tempdir().unwrap();
3556 let table = dir.path().join("occurrence.parquet");
3557 std::fs::create_dir_all(&table).unwrap();
3558 std::fs::write(table.join("000001"), b"PAR1").unwrap();
3559 std::fs::write(table.join("000002"), b"PAR1").unwrap();
3560
3561 let objects: Vec<(String, u64)> = [
3562 "gbif/occurrence.parquet/000001",
3563 "gbif/occurrence.parquet/000002",
3564 ]
3565 .iter()
3566 .map(|k| ((*k).to_string(), 10u64))
3567 .collect();
3568
3569 assert_eq!(
3570 classify_directory(&table),
3571 crate::cloud_browse::look_at_listing("gbif/occurrence.parquet/", &[], &objects).0,
3572 "the two routes answer the same directory alike"
3573 );
3574 assert_eq!(classify_directory(&table), EntryKind::MultiFile);
3575 }
3576
3577 /// And a directory offered as a dataset is one whose files can be counted. The same
3578 /// name test decides both, or the row promises a dataset and shows `?` rows and an
3579 /// empty schema for the rest of the session.
3580 #[test]
3581 fn extensionless_part_files_are_measured_not_just_offered() {
3582 let dir = tempfile::tempdir().unwrap();
3583 let table = dir.path().join("occurrence.parquet");
3584 std::fs::create_dir_all(&table).unwrap();
3585 // Named as GBIF and Spark leave them: no extension, inside a `.parquet`
3586 // directory.
3587 write(&table, "000001", &["id", "species"]);
3588 write(&table, "000002", &["id", "species"]);
3589
3590 let entry = measured(&table);
3591 assert_eq!(entry.kind, EntryKind::MultiFile);
3592 assert_eq!(entry.rows, Some(2), "both footers were read");
3593 assert_eq!(entry.cols, Some(2));
3594 assert!(
3595 schema_preview(&entry).is_some(),
3596 "and the schema pane shows what those footers said, rather than asking \
3597 for a full read of files already read"
3598 );
3599
3600 // → goes inside a directory labelled `multi`, so the listing has to show the
3601 // files the label was counted from — and each is a Parquet file in its own right.
3602 let mut listed = scan_dir(&table);
3603 assert_eq!(
3604 listed.iter().map(|e| e.name.as_str()).collect::<Vec<_>>(),
3605 vec!["000001", "000002"],
3606 "the directory the label promises is not an empty listing"
3607 );
3608 let part = listed.first_mut().expect("a part file is listed");
3609 enrich(part);
3610 assert_eq!(part.rows, Some(1), "a part file counts its own rows");
3611 assert_eq!(part.cols, Some(2));
3612 }
3613
3614 /// One `key=value` prefix among files datui does not read is a hive root on both
3615 /// routes. It is not much of one — but the local route has always said so, and the
3616 /// cloud route disagreeing was the divergence. Pinned rather than left to be
3617 /// rediscovered: #275 phase 3 takes the consequence off the label.
3618 #[cfg(feature = "cloud")]
3619 #[test]
3620 fn one_partition_beside_files_datui_cannot_read_answers_alike() {
3621 let dir = tempfile::tempdir().unwrap();
3622 std::fs::create_dir_all(dir.path().join("notes=old")).unwrap();
3623 for note in ["README.md", "LICENSE", "logo.png"] {
3624 std::fs::write(dir.path().join(note), b"x").unwrap();
3625 }
3626 let objects: Vec<(String, u64)> = ["out/README.md", "out/LICENSE", "out/logo.png"]
3627 .iter()
3628 .map(|k| ((*k).to_string(), 12u64))
3629 .collect();
3630
3631 assert_eq!(
3632 classify_directory(dir.path()),
3633 crate::cloud_browse::look_at_listing("out/", &["out/notes=old/".to_string()], &objects)
3634 .0,
3635 "the two routes answer the same directory alike"
3636 );
3637 }
3638
3639 /// A partition is a partition whatever it starts with. Spark and Hive partition on
3640 /// internal columns — `_date=2024-01-01`, `_c0=…` — and reading those as a writer's
3641 /// own files loses the whole dataset.
3642 #[test]
3643 fn a_partition_named_like_a_writers_file_is_still_a_partition() {
3644 let dir = tempfile::tempdir().unwrap();
3645 let mut directories = Vec::new();
3646 for day in ["2024-01-01", "2024-01-02", "2024-01-03"] {
3647 std::fs::create_dir_all(dir.path().join(format!("_date={day}"))).unwrap();
3648 directories.push(format!("events/_date={day}/"));
3649 }
3650
3651 #[cfg(feature = "cloud")]
3652 assert_eq!(
3653 classify_directory(dir.path()),
3654 crate::cloud_browse::look_at_listing("events/", &directories, &[]).0,
3655 "the two routes answer the same directory alike"
3656 );
3657 assert_eq!(classify_directory(dir.path()), EntryKind::Hive);
3658 assert!(!is_bookkeeping("_date=2024-01-01"));
3659 assert!(is_bookkeeping("_temporary"));
3660 }
3661
3662 /// A prefix a writer made for itself is not a directory somebody put data in, on
3663 /// either route. `_temporary/` counted toward the majority in a bucket and not
3664 /// locally, so the same directory came back two different kinds.
3665 #[cfg(feature = "cloud")]
3666 #[test]
3667 fn a_writers_own_directory_is_skipped_on_both_routes() {
3668 let dir = tempfile::tempdir().unwrap();
3669 write(dir.path(), "part-00000.parquet", &["id"]);
3670 write(dir.path(), "part-00001.parquet", &["id"]);
3671 std::fs::create_dir_all(dir.path().join("_temporary")).unwrap();
3672 std::fs::create_dir_all(dir.path().join("notes")).unwrap();
3673 std::fs::create_dir_all(dir.path().join("archive")).unwrap();
3674
3675 let local = classify_directory(dir.path());
3676 let directories: Vec<String> = ["out/_temporary/", "out/notes/", "out/archive/"]
3677 .iter()
3678 .map(|f| (*f).to_string())
3679 .collect();
3680 let objects: Vec<(String, u64)> = [
3681 ("out/part-00000.parquet", 100u64),
3682 ("out/part-00001.parquet", 100),
3683 ]
3684 .iter()
3685 .map(|(k, s)| ((*k).to_string(), *s))
3686 .collect();
3687 let cloud = crate::cloud_browse::look_at_listing("out/", &directories, &objects).0;
3688
3689 assert_eq!(
3690 local, cloud,
3691 "the two routes answer the same directory alike"
3692 );
3693 assert_eq!(local, EntryKind::MultiFile);
3694 }
3695
3696 /// Every entry is in exactly one count, including the ones with nothing behind
3697 /// them. A FIFO and a broken symlink named like data are not data and are not a
3698 /// writer's own; without a count they were in nothing, and the pane said `1 csv`
3699 /// about a directory of three entries.
3700 #[cfg(unix)]
3701 #[test]
3702 fn every_entry_is_in_one_count() {
3703 let dir = tempfile::tempdir().unwrap();
3704 std::fs::write(dir.path().join("real.csv"), b"id\n1\n").unwrap();
3705 std::os::unix::fs::symlink(dir.path().join("gone"), dir.path().join("broken.csv")).unwrap();
3706 std::fs::write(dir.path().join("notes.md"), b"x").unwrap();
3707 std::fs::write(dir.path().join("_SUCCESS"), b"").unwrap();
3708
3709 let holds = look_at_directory(dir.path()).1;
3710 assert_eq!(holds.data_files(), 1);
3711 assert_eq!(holds.not_read, 2, "the note and the broken link");
3712 assert_eq!(holds.skipped, 1);
3713 assert_eq!(holds.line(true).as_deref(), Some("1 csv"));
3714 }
3715
3716 /// Named like data and impossible to read: a FIFO blocks whoever opens it until a
3717 /// writer appears, and a broken symlink opens as nothing. `directory_format` has
3718 /// always skipped both; the listing now agrees.
3719 #[cfg(unix)]
3720 #[test]
3721 fn a_name_with_nothing_behind_it_is_not_a_data_file() {
3722 let dir = tempfile::tempdir().unwrap();
3723 std::os::unix::fs::symlink(dir.path().join("gone.csv"), dir.path().join("a.csv")).unwrap();
3724 std::os::unix::fs::symlink(dir.path().join("gone.csv"), dir.path().join("b.csv")).unwrap();
3725 assert_eq!(
3726 classify_directory(dir.path()),
3727 EntryKind::Directory,
3728 "two broken symlinks are not a dataset"
3729 );
3730 }
3731
3732 /// A checkpoint directory is the model: its shards are the table, the config and
3733 /// tokenizer JSON beside them are passed over however many there are, and the label
3734 /// names the weights rather than calling the directory mixed.
3735 #[test]
3736 fn a_model_directory_is_its_weights() {
3737 let dir = tempfile::tempdir().unwrap();
3738 for name in [
3739 "model-00001-of-00002.safetensors",
3740 "model-00002-of-00002.safetensors",
3741 "config.json",
3742 "generation_config.json",
3743 "tokenizer.json",
3744 "tokenizer_config.json",
3745 "model.safetensors.index.json",
3746 ] {
3747 std::fs::write(dir.path().join(name), b"x").unwrap();
3748 }
3749 let DirectoryFormat::Mixed {
3750 format,
3751 files,
3752 passed_over,
3753 } = directory_format(dir.path())
3754 else {
3755 panic!("weights and JSON are two formats");
3756 };
3757 assert_eq!(format, crate::FileFormat::Safetensors);
3758 assert_eq!(files.len(), 3, "the shards, and the index for its metadata");
3759 assert_eq!(passed_over, [(crate::FileFormat::Json, 4)]);
3760 let (kind, holds) = look_at_directory(dir.path());
3761 assert_eq!(kind, EntryKind::MultiFile, "opened as one");
3762 assert_eq!(holds.label(), "2 safetensors", "the shards, not the index");
3763
3764 // The index is the model too, named by what it is rather than its extension.
3765 assert_eq!(
3766 data_format(Path::new("model.safetensors.index.json")),
3767 Some(crate::FileFormat::Safetensors)
3768 );
3769 // Weights beside another table format are not a model directory.
3770 std::fs::write(dir.path().join("data.parquet"), b"x").unwrap();
3771 let (kind, holds) = look_at_directory(dir.path());
3772 assert_eq!(
3773 (kind, holds.label().as_str()),
3774 (EntryKind::Directory, "mixed")
3775 );
3776 }
3777
3778 /// A model or MIDI file is known by its first bytes under any name.
3779 #[test]
3780 fn signed_files_are_sniffed_by_their_first_bytes() {
3781 let dir = tempfile::tempdir().unwrap();
3782 let gguf = dir.path().join("weights");
3783 std::fs::write(&gguf, b"GGUF\x03\x00\x00\x00").unwrap();
3784 let st = dir.path().join("checkpoint.bin");
3785 let mut bytes = 2u64.to_le_bytes().to_vec();
3786 bytes.extend_from_slice(b"{}");
3787 std::fs::write(&st, &bytes).unwrap();
3788 let text = dir.path().join("notes");
3789 std::fs::write(&text, b"just some text").unwrap();
3790 assert_eq!(sniff_format(&gguf), Some(crate::FileFormat::Gguf));
3791 let opened = |path: &Path| crate::readers::sniff_open(path, None);
3792 assert_eq!(opened(&st), Some(crate::FileFormat::Safetensors));
3793 assert_eq!(opened(&text), None);
3794 let midi = dir.path().join("song.bin");
3795 std::fs::write(&midi, b"MThd\0\0\0\x06\0\0\0\x01\0\x60").unwrap();
3796 assert_eq!(opened(&midi), Some(crate::FileFormat::Midi));
3797 }
3798
3799 /// A directory is offered as one dataset only when its format can be read as many
3800 /// files. `.tsv`, `.psv` and Excel have a single-file reader and nothing that takes
3801 /// a list, so offering them puts the refusal one keystroke later instead of not
3802 /// making the promise.
3803 #[test]
3804 fn a_format_that_cannot_be_read_as_many_is_not_offered_as_one() {
3805 for ext in ["tsv", "psv", "xlsx", "xlsb"] {
3806 let dir = tempfile::tempdir().unwrap();
3807 std::fs::write(dir.path().join(format!("a.{ext}")), b"x").unwrap();
3808 std::fs::write(dir.path().join(format!("b.{ext}")), b"x").unwrap();
3809 assert_eq!(
3810 classify_directory(dir.path()),
3811 EntryKind::Directory,
3812 "a directory of .{ext} has no reader that takes a list"
3813 );
3814 }
3815 // The ones that do are unaffected — every arm the multi-path open handles.
3816 for ext in [
3817 "parquet", "csv", "json", "jsonl", "ndjson", "arrow", "arrows", "ipc", "feather",
3818 "avro", "orc",
3819 ] {
3820 let dir = tempfile::tempdir().unwrap();
3821 std::fs::write(dir.path().join(format!("a.{ext}")), b"x").unwrap();
3822 std::fs::write(dir.path().join(format!("b.{ext}")), b"x").unwrap();
3823 assert_eq!(
3824 classify_directory(dir.path()),
3825 EntryKind::MultiFile,
3826 ".{ext} reads as many files"
3827 );
3828 }
3829 }
3830
3831 /// The readdir-order bug: eight Parquet files and a ninth entry that is a writer's
3832 /// own file. A probe of the first eight entries never saw the JSON and said `multi`;
3833 /// a bucket listing sorts `_metadata.json` first and said `dir`. Same directory, two
3834 /// answers, decided by the order the filesystem happened to return.
3835 #[cfg(feature = "cloud")]
3836 #[test]
3837 fn a_writers_own_file_is_skipped_whatever_order_it_is_listed_in() {
3838 let dir = tempfile::tempdir().unwrap();
3839 for part in 0..8 {
3840 write(dir.path(), &format!("{part}.parquet"), &["season"]);
3841 }
3842 std::fs::write(dir.path().join("_metadata.json"), b"{}").unwrap();
3843
3844 let local = classify_directory(dir.path());
3845 // The same directory as a bucket lists it: lexicographic, so the JSON comes
3846 // first.
3847 let mut keys: Vec<(String, u64)> = vec![("jolpica/2000/_metadata.json".into(), 2)];
3848 for part in 0..8 {
3849 keys.push((format!("jolpica/2000/{part}.parquet"), 100));
3850 }
3851 keys.sort();
3852 let cloud = crate::cloud_browse::look_at_listing("jolpica/2000/", &[], &keys).0;
3853
3854 assert_eq!(
3855 local, cloud,
3856 "the two routes answer the same directory alike"
3857 );
3858 assert_eq!(local, EntryKind::MultiFile);
3859 }
3860
3861 /// The files a job leaves beside its output are skipped on every route, not just
3862 /// the two names each route happened to know.
3863 #[test]
3864 fn job_files_are_skipped_on_every_route() {
3865 let dir = tempfile::tempdir().unwrap();
3866 write(dir.path(), "part-00000.parquet", &["id"]);
3867 write(dir.path(), "part-00001.parquet", &["id"]);
3868 for marker in [
3869 "_SUCCESS",
3870 "_committed_1727",
3871 "_committed_1728",
3872 "_started_1727",
3873 ".part.crc",
3874 ] {
3875 std::fs::write(dir.path().join(marker), b"").unwrap();
3876 }
3877 assert_eq!(
3878 classify_directory(dir.path()),
3879 EntryKind::MultiFile,
3880 "five markers beside two data files do not outvote them"
3881 );
3882
3883 #[cfg(feature = "cloud")]
3884 let keys: Vec<(String, u64)> = [
3885 ("out/_SUCCESS", 0u64),
3886 ("out/_committed_1727", 12),
3887 ("out/_committed_1728", 12),
3888 ("out/_started_1727", 12),
3889 ("out/.part.crc", 8),
3890 ("out/part-00000.parquet", 100),
3891 ("out/part-00001.parquet", 100),
3892 ]
3893 .iter()
3894 .map(|(k, s)| ((*k).to_string(), *s))
3895 .collect();
3896 #[cfg(feature = "cloud")]
3897 assert_eq!(
3898 crate::cloud_browse::look_at_listing("out/", &[], &keys).0,
3899 EntryKind::MultiFile,
3900 "and the same in a bucket"
3901 );
3902 }
3903
3904 /// Write `columns` as a one-row Parquet file named `name` under `dir`.
3905 fn write(dir: &Path, name: &str, columns: &[&str]) {
3906 let mut frame = DataFrame::new(
3907 1,
3908 columns
3909 .iter()
3910 .map(|c| Column::new((*c).into(), &[1i32]))
3911 .collect::<Vec<_>>(),
3912 )
3913 .unwrap();
3914 let file = std::fs::File::create(dir.join(name)).unwrap();
3915 ParquetWriter::new(file).finish(&mut frame).unwrap();
3916 }
3917
3918 /// A one-row Parquet file with a struct column, so the leaves and the columns a
3919 /// reader sees are genuinely different things rather than dots in a name.
3920 fn write_nested(dir: &Path, name: &str, struct_name: &str, fields: &[&str]) {
3921 let inner = DataFrame::new(
3922 1,
3923 fields
3924 .iter()
3925 .map(|f| Column::new((*f).into(), &[1i32]))
3926 .collect::<Vec<_>>(),
3927 )
3928 .unwrap();
3929 let nested = inner
3930 .into_struct(struct_name.into())
3931 .into_series()
3932 .into_column();
3933 let mut frame = DataFrame::new(1, vec![Column::new("id".into(), &[1i32]), nested]).unwrap();
3934 let file = std::fs::File::create(dir.join(name)).unwrap();
3935 ParquetWriter::new(file).finish(&mut frame).unwrap();
3936 }
3937
3938 fn measured(dir: &Path) -> Entry {
3939 let (kind, holds) = look_at_directory(dir);
3940 let mut entry = Entry {
3941 path: dir.to_path_buf(),
3942 kind,
3943 name: dir.file_name().unwrap().to_string_lossy().into_owned(),
3944 size: None,
3945 modified: None,
3946 rows: None,
3947 cols: None,
3948 cols_sampled: false,
3949 columns: Vec::new(),
3950 cost: Cost::default(),
3951 holds,
3952 opens_whole_directory: false,
3953 format_spec: None,
3954 table: None,
3955 };
3956 enrich(&mut entry);
3957 entry
3958 }
3959
3960 /// The shape that prompted this: one Parquet file per table, sharing an extension
3961 /// and nothing else. Named for what it is rather than what it is called, because
3962 /// the filenames are exactly what cannot decide it.
3963 /// A record cached before the rename still says how many subdirectories it saw.
3964 #[test]
3965 fn holds_written_as_folders_still_reads() {
3966 let old: Holds = serde_json::from_str(r#"{"folders":3,"partitions":2}"#).unwrap();
3967 assert_eq!((old.directories, old.partitions), (3, 2));
3968 let new = serde_json::to_string(&old).unwrap();
3969 assert!(new.contains(r#""directories":3"#), "{new}");
3970 }
3971
3972 #[test]
3973 fn a_directory_of_separate_tables_is_not_a_dataset() {
3974 let dir = tempfile::tempdir().unwrap();
3975 write(
3976 dir.path(),
3977 "circuits.parquet",
3978 &["circuit_id", "lat", "lng"],
3979 );
3980 write(
3981 dir.path(),
3982 "drivers.parquet",
3983 &["driver_id", "code", "nationality"],
3984 );
3985 write(
3986 dir.path(),
3987 "laps.parquet",
3988 &["lap", "position", "time_millis"],
3989 );
3990
3991 assert_eq!(
3992 classify_directory(dir.path()),
3993 EntryKind::MultiFile,
3994 "the filenames alone still say multi"
3995 );
3996 let entry = measured(dir.path());
3997 assert_eq!(
3998 entry.kind,
3999 EntryKind::Directory,
4000 "reading the footers says otherwise"
4001 );
4002 assert_eq!(
4003 entry.rows, None,
4004 "a sum across separate tables is not a row count"
4005 );
4006 assert_eq!(
4007 entry.cols,
4008 Some(9),
4009 "the union of what the directory holds is still a true answer to what is in it"
4010 );
4011 assert_eq!(entry.label(), "3 parquet", "and the label counts the files");
4012 }
4013
4014 /// The rows of one table split across files, which is what `multi` is for.
4015 /// The directories the old threshold took as one table and nesting does not.
4016 ///
4017 /// Two files that each bring a column the other lacks — a renamed column is the
4018 /// everyday case — scored two thirds against a bar of a half, so they opened as one
4019 /// table and the union carried both spellings with nulls under each. Nothing datui
4020 /// can see tells that apart from two tables that share most of their columns, which
4021 /// is why the number moved rather than the question.
4022 ///
4023 /// The directory is not refused. It is a place to look inside, and the row inside it
4024 /// opens the union anyway.
4025 #[test]
4026 fn a_directory_whose_files_each_bring_a_column_is_a_place_to_look_inside() {
4027 let dir = tempfile::tempdir().unwrap();
4028 write(dir.path(), "old.parquet", &["id", "ts", "amount"]);
4029 write(dir.path(), "new.parquet", &["id", "ts", "amt"]);
4030
4031 assert_eq!(
4032 classify_directory(dir.path()),
4033 EntryKind::MultiFile,
4034 "the names alone still say two Parquet files"
4035 );
4036 let entry = measured(dir.path());
4037 assert_eq!(
4038 entry.kind,
4039 EntryKind::Directory,
4040 "and the footers say neither file's columns are in the other's"
4041 );
4042 assert_eq!(entry.label(), "2 parquet", "which the label still reports");
4043 assert_eq!(entry.rows, None, "a sum over two tables is not a number");
4044 }
4045
4046 #[test]
4047 fn a_directory_of_one_table_stays_a_dataset() {
4048 let dir = tempfile::tempdir().unwrap();
4049 for part in 0..3 {
4050 write(
4051 dir.path(),
4052 &format!("part-0000{part}.parquet"),
4053 &["id", "ts", "amount"],
4054 );
4055 }
4056
4057 let entry = measured(dir.path());
4058 assert_eq!(entry.kind, EntryKind::MultiFile);
4059 assert_eq!(entry.rows, Some(3));
4060 assert_eq!(entry.cols, Some(3));
4061 }
4062
4063 /// A dataset whose columns changed over time is still one dataset. This is the
4064 /// case a rule about shared columns gets wrong: the older files have a third of
4065 /// what the newest one does.
4066 #[test]
4067 fn a_dataset_that_gained_columns_stays_a_dataset() {
4068 let dir = tempfile::tempdir().unwrap();
4069 write(dir.path(), "2009.parquet", &["id", "ts"]);
4070 write(dir.path(), "2015.parquet", &["id", "ts", "fee"]);
4071 write(
4072 dir.path(),
4073 "2025.parquet",
4074 &["id", "ts", "fee", "witness", "address", "value"],
4075 );
4076
4077 let entry = measured(dir.path());
4078 assert_eq!(entry.kind, EntryKind::MultiFile);
4079 assert_eq!(entry.rows, Some(3));
4080 assert_eq!(
4081 entry.columns,
4082 vec!["id", "ts", "fee", "witness", "address", "value"],
4083 "every column any file has, in the order they first appear — not the \
4084 2009 shape"
4085 );
4086 assert_eq!(entry.cols, Some(6), "and the count is of those");
4087 }
4088
4089 /// The same, for a hive tree: the row is the dataset's columns, not one
4090 /// partition's.
4091 #[test]
4092 fn a_hive_dataset_that_gained_columns_reports_all_of_them() {
4093 let dir = tempfile::tempdir().unwrap();
4094 for (part, columns) in [
4095 ("year=2009", &["id", "ts"][..]),
4096 ("year=2025", &["id", "ts", "address"][..]),
4097 ] {
4098 let sub = dir.path().join(part);
4099 std::fs::create_dir_all(&sub).unwrap();
4100 write(&sub, "part-0.parquet", columns);
4101 }
4102
4103 let entry = measured(dir.path());
4104 assert_eq!(entry.kind, EntryKind::Hive);
4105 assert_eq!(entry.columns, vec!["id", "ts", "address"]);
4106 assert_eq!(entry.cols, Some(4), "and the partition column `year`");
4107 }
4108
4109 /// Past the counting limit the columns come from a spread of the directory rather
4110 /// than its head, because a directory written over time is narrowest at the start.
4111 #[test]
4112 fn a_directory_too_large_to_count_still_reports_the_columns_it_gained() {
4113 let dir = tempfile::tempdir().unwrap();
4114 for part in 0..MAX_FOOTERS_PER_DATASET + 1 {
4115 let mut columns = vec!["id".to_string(), "ts".to_string()];
4116 if part > MAX_FOOTERS_PER_DATASET / 2 {
4117 columns.push("address".to_string());
4118 }
4119 let refs: Vec<&str> = columns.iter().map(String::as_str).collect();
4120 write(dir.path(), &format!("part-{part:03}.parquet"), &refs);
4121 }
4122
4123 let entry = measured(dir.path());
4124 assert_eq!(entry.kind, EntryKind::MultiFile, "still one table");
4125 assert_eq!(entry.rows, None, "too many files to count");
4126 assert!(
4127 entry.columns.contains(&"address".to_string()),
4128 "the column the dataset gained is in the row: {:?}",
4129 entry.columns
4130 );
4131 }
4132
4133 /// A lake table's data files agree on a schema, so the one-table rule says `multi`
4134 /// and is right about the schema and wrong about the rows: the files a delete
4135 /// tombstoned are still on disk, every rewritten version is here together, and
4136 /// compaction leaves both sides in place.
4137 #[test]
4138 fn a_lake_table_is_not_a_directory_of_parquet_files() {
4139 for (marker, expected) in [
4140 ("_delta_log", EntryKind::Delta),
4141 (".hoodie", EntryKind::Hudi),
4142 ] {
4143 let dir = tempfile::tempdir().unwrap();
4144 write(dir.path(), "part-0.parquet", &["id", "amount"]);
4145 write(dir.path(), "part-1.parquet", &["id", "amount"]);
4146 write(dir.path(), "part-2.parquet", &["id", "amount"]);
4147 let log = dir.path().join(marker);
4148 std::fs::create_dir_all(&log).unwrap();
4149 std::fs::write(log.join("00000000000000000000.json"), b"{}").unwrap();
4150
4151 assert_eq!(
4152 classify_directory(dir.path()),
4153 expected,
4154 "{marker} says what this directory is"
4155 );
4156 let entry = measured(dir.path());
4157 assert_eq!(entry.kind, expected);
4158 assert_eq!(
4159 entry.rows, None,
4160 "and no row count is claimed for it: summing the footers would count \
4161 the rows the log says are gone"
4162 );
4163 assert!(!entry.kind.is_dataset(), "it does not open as one table");
4164 }
4165 }
4166
4167 /// Iceberg's marker is a plain name, so it takes the whole shape rather than the
4168 /// name alone.
4169 #[test]
4170 fn an_iceberg_root_is_metadata_beside_data() {
4171 let iceberg = tempfile::tempdir().unwrap();
4172 let data = iceberg.path().join("data");
4173 let metadata = iceberg.path().join("metadata");
4174 std::fs::create_dir_all(&data).unwrap();
4175 std::fs::create_dir_all(&metadata).unwrap();
4176 write(&data, "00000-0-abc.parquet", &["id", "amount"]);
4177 write(&data, "00001-0-def.parquet", &["id", "amount"]);
4178 std::fs::write(metadata.join("v2.metadata.json"), b"{}").unwrap();
4179 std::fs::write(metadata.join("snap-1.avro"), b"x").unwrap();
4180 assert_eq!(classify_directory(iceberg.path()), EntryKind::Iceberg);
4181
4182 // A directory that merely has those names is not a table.
4183 let plain = tempfile::tempdir().unwrap();
4184 std::fs::create_dir_all(plain.path().join("data")).unwrap();
4185 std::fs::create_dir_all(plain.path().join("metadata")).unwrap();
4186 std::fs::write(plain.path().join("metadata/notes.txt"), b"x").unwrap();
4187 assert_eq!(
4188 classify_directory(plain.path()),
4189 EntryKind::Directory,
4190 "no *.metadata.json, so no Iceberg table"
4191 );
4192
4193 let no_data = tempfile::tempdir().unwrap();
4194 let metadata = no_data.path().join("metadata");
4195 std::fs::create_dir_all(&metadata).unwrap();
4196 std::fs::write(metadata.join("v1.metadata.json"), b"{}").unwrap();
4197 write(no_data.path(), "part-0.parquet", &["id"]);
4198 write(no_data.path(), "part-1.parquet", &["id"]);
4199 assert_eq!(
4200 classify_directory(no_data.path()),
4201 EntryKind::MultiFile,
4202 "metadata with no data/ beside it is somebody's directory, not a table root"
4203 );
4204 }
4205
4206 /// A single file counts its columns the same way a directory does, and both count
4207 /// what opening it shows.
4208 ///
4209 /// `enrich_parquet` read `schema_descr.columns()`, which is the leaf list — so a file
4210 /// with one struct of two fields said `columns 3` above a schema list of two, and a
4211 /// directory holding only that file said something different again.
4212 #[test]
4213 fn a_file_and_a_directory_of_it_count_the_same_columns() {
4214 let dir = tempfile::tempdir().unwrap();
4215 write_nested(dir.path(), "one.parquet", "inputs", &["address", "value"]);
4216
4217 let mut file = Entry::new(dir.path().join("one.parquet"), EntryKind::File);
4218 enrich(&mut file);
4219 assert_eq!(
4220 file.cols,
4221 Some(2),
4222 "`id` and `inputs`, which is what opening it shows: {:?}",
4223 file.columns
4224 );
4225 assert!(
4226 file.columns.iter().any(|c| c == "inputs.address"),
4227 "the leaves are still searchable: {:?}",
4228 file.columns
4229 );
4230
4231 write_nested(dir.path(), "two.parquet", "inputs", &["address", "value"]);
4232 let directory = measured(dir.path());
4233 assert_eq!(directory.kind, EntryKind::MultiFile);
4234 assert_eq!(
4235 directory.cols, file.cols,
4236 "and a directory of them says the same number"
4237 );
4238 }
4239
4240 /// Dots in a column's own name are not nesting, and are not counted as if they were.
4241 ///
4242 /// The obvious fix for the leaf problem — split the dotted path and count the roots —
4243 /// gets this wrong: `user.id` and `user.name` written by a flattening export are two
4244 /// columns, not one. The schema says which is which; the string cannot.
4245 #[test]
4246 fn a_dotted_column_name_is_its_own_column() {
4247 let dir = tempfile::tempdir().unwrap();
4248 write(dir.path(), "flat.parquet", &["id", "user.id", "user.name"]);
4249
4250 let mut file = Entry::new(dir.path().join("flat.parquet"), EntryKind::File);
4251 enrich(&mut file);
4252 assert_eq!(file.cols, Some(3), "three columns: {:?}", file.columns);
4253 }
4254
4255 /// A directory whose files encode the same nested column differently counts it once.
4256 ///
4257 /// The union is over leaf paths, and the same nested column written by parquet-mr and
4258 /// by Arrow gives different leaves — so the row reported roughly twice the width of a
4259 /// directory `is_one_table` had just called one dataset. Counted from each file's own
4260 /// root fields, the two spellings are one `inputs` whatever the leaves under it are.
4261 #[test]
4262 fn a_writer_change_does_not_double_the_column_count() {
4263 let dir = tempfile::tempdir().unwrap();
4264 write_nested(dir.path(), "old.parquet", "inputs", &["address"]);
4265 // The same column, one field wider, as a later writer left it.
4266 write_nested(dir.path(), "new.parquet", "inputs", &["address", "value"]);
4267
4268 let entry = measured(dir.path());
4269 assert_eq!(entry.kind, EntryKind::MultiFile, "still one table");
4270 assert_eq!(
4271 entry.cols,
4272 Some(2),
4273 "one `inputs`, not one per shape of it: {:?}",
4274 entry.columns
4275 );
4276 assert!(
4277 entry.columns.len() > 2,
4278 "while every leaf stays searchable: {:?}",
4279 entry.columns
4280 );
4281 }
4282
4283 /// A count read from a spread of a directory rather than all of it says it is a
4284 /// floor.
4285 #[test]
4286 fn a_sampled_column_count_says_it_is_a_floor() {
4287 let dir = tempfile::tempdir().unwrap();
4288 for part in 0..MAX_FOOTERS_PER_DATASET * 2 {
4289 write(
4290 dir.path(),
4291 &format!("part-{part:04}.parquet"),
4292 &["id", "ts"],
4293 );
4294 }
4295 let entry = measured(dir.path());
4296 assert_eq!(entry.rows, None, "too many files to count");
4297 assert!(entry.cols.is_some(), "but the width is still worth having");
4298 assert!(
4299 entry.cols_sampled,
4300 "and it is marked as the floor it is, not presented as a total"
4301 );
4302
4303 // A directory small enough to read every footer of claims no such thing.
4304 let small = tempfile::tempdir().unwrap();
4305 write(small.path(), "a.parquet", &["id", "ts"]);
4306 write(small.path(), "b.parquet", &["id", "ts"]);
4307 assert!(!measured(small.path()).cols_sampled);
4308 }
4309
4310 /// The files a directory offers come back in order, whatever order it was
4311 /// written in.
4312 ///
4313 /// Every caller reads order as meaning something — `sample_footers` takes the ends
4314 /// and the middle, and the union of the columns is built in the order the files
4315 /// appear. Unsorted, "the last file" was whichever one the filesystem happened to
4316 /// return last, which on the filesystems that return creation order is the one
4317 /// written first as often as not.
4318 #[test]
4319 fn the_files_a_directory_offers_come_back_in_order() {
4320 let dir = tempfile::tempdir().unwrap();
4321 for name in ["c.parquet", "a.parquet", "d.parquet", "b.parquet"] {
4322 write(dir.path(), name, &["id"]);
4323 }
4324 let files = parquet_files_under(dir.path());
4325 let names: Vec<String> = files
4326 .iter()
4327 .map(|p| p.file_name().unwrap().to_string_lossy().into_owned())
4328 .collect();
4329 assert_eq!(
4330 names,
4331 vec!["a.parquet", "b.parquet", "c.parquet", "d.parquet"],
4332 "sorted, not in the order the directory was written"
4333 );
4334 }
4335
4336 /// A directory past the budget still says so, and the files it keeps are the
4337 /// directory's first rather than the listing's.
4338 ///
4339 /// The ordering itself is `the_files_a_directory_offers_come_back_in_order`'s to
4340 /// prove: a directory read may return sorted entries of its own accord, so an
4341 /// assertion here about order could hold for the wrong reason. What this pins is
4342 /// *which* files survive the cap, and that the cap still says "too many to count".
4343 #[test]
4344 fn a_directory_past_the_budget_keeps_the_directories_first_files() {
4345 let dir = tempfile::tempdir().unwrap();
4346 for part in 0..MAX_FOOTERS_PER_DATASET * 3 {
4347 write(dir.path(), &format!("part-{part:04}.parquet"), &["id"]);
4348 }
4349 let files = parquet_files_under(dir.path());
4350
4351 assert_eq!(
4352 files.len(),
4353 MAX_FOOTERS_PER_DATASET + 1,
4354 "one past the budget, which is what says there are too many to count"
4355 );
4356 let names: Vec<String> = files
4357 .iter()
4358 .map(|p| p.file_name().unwrap().to_string_lossy().into_owned())
4359 .collect();
4360 let expected: Vec<String> = (0..=MAX_FOOTERS_PER_DATASET)
4361 .map(|part| format!("part-{part:04}.parquet"))
4362 .collect();
4363 // Not "sorted", which a directory read may be of its own accord, but the
4364 // directory's own first sixty-five. Sorting after truncating gives sixty-five
4365 // sorted names from wherever the read began, which is a different set.
4366 assert_eq!(
4367 names, expected,
4368 "the directory's first files, not the listing's"
4369 );
4370 }
4371
4372 /// The log is named rather than looked for, so a table's own data files cannot
4373 /// crowd it out of the listing however many of them there are.
4374 #[test]
4375 fn a_lake_table_is_recognized_among_its_data_files() {
4376 let dir = tempfile::tempdir().unwrap();
4377 for part in 0..32 {
4378 write(dir.path(), &format!("part-{part:03}.parquet"), &["id"]);
4379 }
4380 std::fs::create_dir_all(dir.path().join("_delta_log")).unwrap();
4381 assert_eq!(classify_directory(dir.path()), EntryKind::Delta);
4382 }
4383
4384 /// Past the counting limit the row count is out of reach, but whether the directory
4385 /// is one table is not — and a directory of a hundred tables is exactly where reading
4386 /// them as one costs most.
4387 #[test]
4388 fn a_directory_too_large_to_count_is_still_checked() {
4389 let dir = tempfile::tempdir().unwrap();
4390 for table in 0..MAX_FOOTERS_PER_DATASET + 1 {
4391 write(
4392 dir.path(),
4393 &format!("table_{table:03}.parquet"),
4394 &[&format!("{table}_id"), &format!("{table}_value")],
4395 );
4396 }
4397
4398 let entry = measured(dir.path());
4399 assert_eq!(entry.kind, EntryKind::Directory);
4400 assert_eq!(entry.rows, None, "too many files to count either way");
4401 }
4402
4403 /// The same directory size, but one table split across it.
4404 #[test]
4405 fn a_large_directory_of_one_table_stays_a_dataset() {
4406 let dir = tempfile::tempdir().unwrap();
4407 for part in 0..MAX_FOOTERS_PER_DATASET + 1 {
4408 write(
4409 dir.path(),
4410 &format!("part-{part:05}.parquet"),
4411 &["id", "ts"],
4412 );
4413 }
4414
4415 let entry = measured(dir.path());
4416 assert_eq!(entry.kind, EntryKind::MultiFile);
4417 }
4418
4419 /// Searching the home screen by column should still find a directory that holds one,
4420 /// even once the directory is no longer offered as a single table.
4421 #[test]
4422 fn a_downgraded_directory_keeps_every_column_its_files_have() {
4423 let dir = tempfile::tempdir().unwrap();
4424 write(
4425 dir.path(),
4426 "circuits.parquet",
4427 &["circuit_id", "lat", "lng"],
4428 );
4429 write(
4430 dir.path(),
4431 "drivers.parquet",
4432 &["driver_id", "code", "nationality"],
4433 );
4434 // And one a level down, so "every column its files have" is a claim about more
4435 // than the directory's own: the row's width is its own files, its names are
4436 // everything under it, and a fixture with no subdirectory cannot tell those
4437 // apart.
4438 let seasons = dir.path().join("seasons");
4439 std::fs::create_dir_all(&seasons).unwrap();
4440 write(&seasons, "2024.parquet", &["season_year", "round"]);
4441
4442 let entry = measured(dir.path());
4443 assert_eq!(entry.kind, EntryKind::Directory);
4444 for column in [
4445 "circuit_id",
4446 "lat",
4447 "lng",
4448 "driver_id",
4449 "code",
4450 "nationality",
4451 "season_year",
4452 "round",
4453 ] {
4454 assert!(
4455 entry.columns.iter().any(|c| c == column),
4456 "{column} in {:?}",
4457 entry.columns
4458 );
4459 }
4460 }
4461
4462 #[test]
4463 fn a_name_no_reader_takes_is_refused_before_opening() {
4464 let refused = |name: &str| unreadable_by_name(std::path::Path::new(name));
4465 assert!(refused("gs://b/ml/onnx/pipeline_rf.onnx"));
4466 assert!(refused("model.onnx.gz"));
4467 assert!(refused("README.md"));
4468 for readable in [
4469 "a.csv",
4470 "a.CSV",
4471 "a.csv.gz",
4472 "a.parquet",
4473 "a.xlsx",
4474 "data.gz",
4475 "part-0000",
4476 ] {
4477 assert!(!refused(readable), "{readable}");
4478 }
4479 }
4480}