Skip to main content

datui_lib/
notes.rs

1//! What datui noticed about a dataset while doing what it was already doing. A note
2//! never costs its own request or scan (all come from footers the schema and count
3//! needed), is never styled as an alarm, and states its basis, so "in 1 of 3 files" is
4//! never read as a claim about files not looked at. One claim per note, and one
5//! function (`out_of`) deciding the only ratio any note states.
6
7use crate::formats::schema_union::{
8    ColumnDrift, ColumnRange, DatasetSchema, SchemaOrigin, SkippedFiles,
9};
10use crate::numfmt::group_chrome;
11use polars::prelude::{DataType, PlSmallStr};
12
13/// One thing datui noticed.
14#[derive(Debug, Clone, PartialEq, Eq)]
15pub struct Note {
16    /// The one line shown in the panel.
17    pub summary: String,
18    /// What the note is based on, so its reach is never overstated.
19    pub scope: String,
20    /// The column datui can offer to read as text, when this note is about one and that
21    /// would work; `None` otherwise. A name, so the panel never parses prose.
22    pub read_as_text: Option<PlSmallStr>,
23    /// How many files this note counts as passed over, when it says a mixed directory was
24    /// read as its commonest format; `None` otherwise. A number, so [`merged`] can subtract
25    /// it from the footer pass's tally without parsing prose.
26    pub passed_over: Option<usize>,
27}
28
29/// Whether a dataset's schema came from a sample of its files.
30fn sampled(dataset: &DatasetSchema) -> bool {
31    matches!(dataset.origin, SchemaOrigin::FooterSample { .. })
32}
33
34/// How many, in the noun that is true of what datui looked at: files when it read
35/// every one, footers when it read only a sample of them.
36fn how_many(dataset: &DatasetSchema, n: usize) -> String {
37    let noun = match (sampled(dataset), n) {
38        (false, 1) => "file",
39        (false, _) => "files",
40        (true, 1) => "footer",
41        (true, _) => "footers",
42    };
43    format!("{} {noun}", group_chrome(n))
44}
45
46/// The denominator of the dataset's one ratio: files read and parsed. "files" only when
47/// every file was read and parsed; otherwise saying so would claim more than was seen.
48fn out_of(dataset: &DatasetSchema) -> String {
49    let readable = dataset.files.saturating_sub(dataset.unreadable.len());
50    match (sampled(dataset), dataset.unreadable.is_empty()) {
51        (false, true) => how_many(dataset, readable),
52        (false, false) => format!("the {} that could be read", how_many(dataset, readable)),
53        (true, true) => format!("the {} read", how_many(dataset, readable)),
54        (true, false) => format!("the {} that could be read", how_many(dataset, readable)),
55    }
56}
57
58/// Names for a set of types that tell them apart: names escalate in detail (to
59/// `Debug`) until distinct, since structs all print `struct[1]` and enums
60/// `Enum([...])`.
61fn distinct_names(types: &[&DataType]) -> Vec<String> {
62    let collides = |names: &[String]| {
63        names
64            .iter()
65            .enumerate()
66            .any(|(i, name)| names[i + 1..].contains(name))
67    };
68    // `Display` is how the Schema tab names a type, and it carries the unit, the
69    // precision and the time zone that the table header's one word drops.
70    let shown: Vec<String> = types.iter().map(|t| format!("{t}")).collect();
71    if !collides(&shown) {
72        return shown;
73    }
74    // Two structs are both `struct[1]`, and only the fields tell them apart.
75    let spelled: Vec<String> = types.iter().map(|t| format!("{t:?}")).collect();
76    if !collides(&spelled) {
77        return spelled;
78    }
79    // Enums and global Categoricals can collide even in full form: number only those that
80    // collide.
81    let mut seen: Vec<&String> = Vec::new();
82    spelled
83        .iter()
84        .map(|name| {
85            if spelled.iter().filter(|other| *other == name).count() > 1 {
86                seen.push(name);
87                format!("{name} #{}", seen.iter().filter(|s| **s == name).count())
88            } else {
89                name.clone()
90            }
91        })
92        .collect()
93}
94
95/// What the footers said, as notes. Empty when every file agrees, which is the common
96/// case and the one where there is nothing to say.
97pub fn from_dataset(dataset: &DatasetSchema) -> Vec<Note> {
98    let scope = format!("in {}", dataset.origin);
99    let readable = dataset.files.saturating_sub(dataset.unreadable.len());
100    let denominator = out_of(dataset);
101    let mut notes = Vec::new();
102
103    for column in dataset.drifting() {
104        // The read type is named once, so notes about one column name it alike.
105        let mut types: Vec<&DataType> = vec![&column.dtype];
106        types.extend(column.conflicting_types.iter());
107        let names = distinct_names(&types);
108        let (chosen, others) = names.split_first().expect("the chosen type is first");
109
110        // Missing, conflicting and narrower are each true on its own, each said on its own.
111        notes.extend(absence_note(
112            column,
113            readable,
114            &denominator,
115            dataset.column_ranges.get(&column.name),
116            &scope,
117        ));
118        notes.extend(conflict_note(column, dataset, chosen, others, &scope));
119        notes.extend(widening_note(column, chosen, &scope));
120    }
121
122    for column in &dataset.read_as_text {
123        notes.push(text_note(column, &scope));
124    }
125
126    notes.extend(empty_files_note(dataset, &scope));
127    notes.extend(row_group_note(dataset, &scope));
128    notes.extend(small_files_note(dataset, &scope));
129    notes.extend(partition_layout_note(dataset));
130    notes.extend(skipped_files_note(dataset));
131
132    if !dataset.unreadable.is_empty() {
133        notes.push(Note {
134            summary: format!(
135                "{} unreadable, left out",
136                how_many(dataset, dataset.unreadable.len())
137            ),
138            scope: scope.clone(),
139            read_as_text: None,
140            passed_over: None,
141        });
142    }
143
144    notes
145}
146
147/// Rows a filter or sort leaves out because their files do not hold the named column
148/// in its read type: the one note about the view. Phrased about the files (a footer
149/// fact), not as "the view is N rows shorter", which an earlier filter may have made
150/// false; so it also appears where those rows were already gone.
151pub fn left_out_note(
152    column: &ColumnDrift,
153    dataset: &DatasetSchema,
154    rows: usize,
155    filtered: bool,
156    sorted: bool,
157) -> Note {
158    let what = match (filtered, sorted) {
159        (true, true) => "filter and sort",
160        (true, false) => "filter",
161        // Called only for a column the view names, so it names it one way or the other.
162        _ => "sort",
163    };
164    let there = if rows == 1 {
165        "1 row".to_string()
166    } else {
167        format!("{} rows", group_chrome(rows))
168    };
169    Note {
170        summary: format!(
171            "{}: {there} in {} left out of the {what}",
172            column.name,
173            how_many(dataset, column.conflicting_files)
174        ),
175        scope: format!("in {}", dataset.origin),
176        read_as_text: None,
177        passed_over: None,
178    }
179}
180
181/// Files holding no rows (an empty day's partition, a header-only file): not a fault,
182/// but directories then overcount days. Counts without dividing. Says "file" even when
183/// sampled (a footer records rows, it holds none); the scope line says what was read.
184fn empty_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
185    let empty = dataset.empty_files;
186    if empty == 0 {
187        return None;
188    }
189    let (count, verb) = if empty == 1 {
190        ("1 file".to_string(), "holds")
191    } else {
192        (format!("{} files", group_chrome(empty)), "hold")
193    };
194    Some(Note {
195        summary: format!("{count} {verb} no rows"),
196        scope: scope.to_string(),
197        read_as_text: None,
198        passed_over: None,
199    })
200}
201
202/// Row groups big enough that a page reads much more than the page: a row group is
203/// fetched whole, so over a network scrolling waits on it, and the user cannot fix it
204/// here. Fires past one noticeable download, on the median group size over every
205/// footer read (each group once, not row-weighted). States the group's size, not a
206/// page's cost (pages skip binary columns, see `binary_stub_exprs`).
207fn row_group_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
208    /// Sixty-four mebibytes, the size a page in that range costs to reach.
209    const BIG: usize = 64 * 1024 * 1024;
210    let median = dataset.median_row_group_bytes?;
211    if median <= BIG {
212        return None;
213    }
214    Some(Note {
215        summary: format!(
216            "median row group {}, each read whole",
217            crate::numfmt::bytes(median as u64)
218        ),
219        scope: scope.to_string(),
220        read_as_text: None,
221        passed_over: None,
222    })
223}
224
225/// A dataset of very many, very small files: says a footer was read for every one
226/// before any row (footers are file suffixes, so finding never costs more than
227/// reading). Both conditions must hold; the fix is upstream. The count is every listed
228/// file, the median size over footers opened, which the sentence names.
229fn small_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
230    /// Past this many files the footer pass is a job of its own. Above a year of
231    /// hourly partitions, which is an ordinary shape and not a complaint.
232    const MANY: usize = 10_000;
233    /// Below this a file is small by any warehouse's standard, where the figure aimed
234    /// at is hundreds of megabytes.
235    const SMALL: usize = 1024 * 1024;
236    let files = dataset.origin.total_files();
237    let median = dataset.median_file_bytes?;
238    // A zero median means sizes are unknown, not small.
239    if median == 0 || files <= MANY || median >= SMALL {
240        return None;
241    }
242    let read = dataset.files;
243    // "opened for its footer" (an unparsable footer was still opened), joined by a
244    // semicolon: only the count, not the size, causes one read per file.
245    let footers = if read == files {
246        "every footer".to_string()
247    } else {
248        format!("{} footers", group_chrome(read))
249    };
250    Some(Note {
251        summary: format!(
252            "{} files, median {}; {footers} read before any row",
253            group_chrome(files),
254            crate::numfmt::bytes(median as u64)
255        ),
256        scope: scope.to_string(),
257        read_as_text: None,
258        passed_over: None,
259    })
260}
261
262/// Directories that do not all partition by the same keys. States the shape only:
263/// the consequence (null column or failed open) depends on which path sorts first,
264/// which a note cannot see; the user guide explains. Read from every file name, so it
265/// has its own scope line and covers all of a too-large dataset.
266fn partition_layout_note(dataset: &DatasetSchema) -> Option<Note> {
267    /// Layouts named before the rest are counted rather than spelled. A note is one
268    /// sentence, and a dataset with a hundred layouts would otherwise make it a page.
269    const NAMED: usize = 2;
270    if dataset.partition_layouts.len() < 2 {
271        return None;
272    }
273    let (named, rest) = dataset
274        .partition_layouts
275        .split_at(dataset.partition_layouts.len().min(NAMED));
276    let mut clauses: Vec<String> = named
277        .iter()
278        .map(|(keys, files)| format!("{} by {}", how_many_files(*files), keys.join("/")))
279        .collect();
280    let (dropped_ways, dropped_files) = dataset.partition_layouts_dropped;
281    let ways = rest.len() + dropped_ways;
282    let files: usize = rest.iter().map(|(_, files)| files).sum::<usize>() + dropped_files;
283    if ways > 0 {
284        clauses.push(format!(
285            "{} by {} other {}",
286            how_many_files(files),
287            group_chrome(ways),
288            if ways == 1 { "way" } else { "ways" }
289        ));
290    }
291    Some(Note {
292        summary: format!("mixed partition keys: {}", clauses.join(", ")),
293        scope: format!(
294            "in the names of {} files",
295            group_chrome(dataset.listed_files)
296        ),
297        read_as_text: None,
298        passed_over: None,
299    })
300}
301
302/// `n files`, or `1 file`. Plain files, because the caller counted every one of them.
303fn how_many_files(n: usize) -> String {
304    format!(
305        "{} {}",
306        group_chrome(n),
307        if n == 1 { "file" } else { "files" }
308    )
309}
310
311/// A column read as text from every file on request, replacing its conflict note: a
312/// filter or sort on it now compares text (`n > 5` keeps `"sixty"`, drops `"10"`).
313fn text_note(column: &PlSmallStr, scope: &str) -> Note {
314    Note {
315        summary: format!("{column} read as text: filter and sort compare text"),
316        scope: scope.to_string(),
317        read_as_text: None,
318        passed_over: None,
319    }
320}
321
322/// Files beside the data that are not Parquet, so not in the table: only those someone
323/// might have meant as data (not `_SUCCESS`, `.crc`, `_metadata`, which every job
324/// leaves). Counts them beside what they accompany, as far as the listing saw.
325fn skipped_files_note(dataset: &DatasetSchema) -> Option<Note> {
326    note_about_skipped(dataset.skipped)
327}
328
329/// [`skipped_files_note`]'s note from the tally alone, so [`merged`] can rebuild it
330/// with the open's files removed.
331fn note_about_skipped(skipped: SkippedFiles) -> Option<Note> {
332    let SkippedFiles {
333        bookkeeping,
334        not_parquet,
335        empty,
336    } = skipped;
337    if not_parquet == 0 && empty == 0 {
338        return None;
339    }
340    let files = |n: usize| {
341        if n == 1 {
342            "1 file".to_string()
343        } else {
344            format!("{} files", group_chrome(n))
345        }
346    };
347    // An object with nothing in it is a write that stopped, and saying so is the point;
348    // the rest is counted beside it so the total is the directory's, not a selection.
349    let mut said = Vec::new();
350    if empty > 0 {
351        let what = if empty == 1 { "file" } else { "files" };
352        said.push(format!("{} empty {what}", group_chrome(empty)));
353    }
354    if not_parquet > 0 {
355        said.push(format!("{} not Parquet", files(not_parquet)));
356    }
357    if bookkeeping > 0 {
358        let what = if bookkeeping == 1 { "file" } else { "files" };
359        said.push(format!(
360            "{} writer bookkeeping {what}",
361            group_chrome(bookkeeping)
362        ));
363    }
364    Some(Note {
365        summary: format!("skipped: {}", said.join(", ")),
366        scope: "in this directory's listing".to_string(),
367        read_as_text: None,
368        passed_over: None,
369    })
370}
371
372/// The most names a note lists before it cuts the rest with an ellipsis.
373const NAMES_SHOWN: usize = 3;
374
375/// `names`, the first few of them, joined, and an ellipsis for the rest.
376pub fn some_names<S: AsRef<str>>(names: &[S]) -> String {
377    let mut said: Vec<&str> = names.iter().take(NAMES_SHOWN).map(AsRef::as_ref).collect();
378    let ellipsis = crate::glyphs::get().ellipsis;
379    if names.len() > NAMES_SHOWN {
380        said.push(ellipsis);
381    }
382    said.join(", ")
383}
384
385/// The files a read of several passed over because they hold no header: empty, blank,
386/// or nothing but NUL padding.
387pub fn no_header(files: &[&std::path::Path]) -> Option<Note> {
388    if files.is_empty() {
389        return None;
390    }
391    let names: Vec<String> = files
392        .iter()
393        .map(|f| {
394            f.file_name().map_or_else(
395                || f.display().to_string(),
396                |n| n.to_string_lossy().into_owned(),
397            )
398        })
399        .collect();
400    let what = if files.len() == 1 { "file" } else { "files" };
401    Some(Note {
402        summary: format!(
403            "{} {what} with no header skipped: {}",
404            files.len(),
405            some_names(&names)
406        ),
407        scope: "empty, blank, or only NUL padding".to_string(),
408        read_as_text: None,
409        passed_over: None,
410    })
411}
412
413/// What the open has to say before any footer: which data files this read passed
414/// over, and whether a lake table is read as its plain files. Choices, not defects:
415/// datui never refuses a read asked for, and says what it did instead.
416pub fn from_the_open(
417    left_out: &[(crate::FileFormat, usize)],
418    lake: Option<&str>,
419    files_differ: crate::formats::schema_union::Disagreement,
420    names_look_like_data: bool,
421) -> Vec<Note> {
422    let mut notes = Vec::new();
423    // What stacking the files required, as found. Footerless formats have no per-column
424    // tally (Parquet gets the exact version from its footers); the scope is a spread of
425    // three files.
426    let scope = || "in a spread of this directory's files".to_string();
427    if files_differ.columns {
428        notes.push(Note {
429            summary: "columns differ across files: a missing column reads null".to_string(),
430            scope: scope(),
431            read_as_text: None,
432            passed_over: None,
433        });
434    }
435    if files_differ.headerless {
436        notes.push(Note {
437            summary: concat!(
438                "no header row? first row read as names: ",
439                "H on Schema, or --no-header, reads it as data"
440            )
441            .to_string(),
442            scope: scope(),
443            read_as_text: None,
444            passed_over: None,
445        });
446    }
447    // The same shape in one file: every column name a number, which a header almost
448    // never is and a first row of data often is.
449    if names_look_like_data && !files_differ.headerless {
450        notes.push(Note {
451            summary: "column names look like data: H on Schema reads them as a row".to_string(),
452            scope: "from the column names".to_string(),
453            read_as_text: None,
454            passed_over: None,
455        });
456    }
457    if files_differ.types {
458        notes.push(Note {
459            summary: "a column's type differs across files: read as the wider type".to_string(),
460            scope: scope(),
461            read_as_text: None,
462            passed_over: None,
463        });
464    }
465    if let Some(format) = lake {
466        notes.push(Note {
467            // The strongest sentence: here a number on screen is not about the table (deleted rows,
468            // old versions and compaction leftovers all counted).
469            summary: format!(
470                "{format} table's files, not the table: deleted rows and old versions counted"
471            ),
472            scope: format!("in this {format} table's directory"),
473            read_as_text: None,
474            passed_over: None,
475        });
476    }
477    if !left_out.is_empty() {
478        let said: Vec<String> = left_out
479            .iter()
480            .map(|(format, n)| format!("{n} {}", format.name()))
481            .collect();
482        notes.push(Note {
483            summary: format!(
484                "mixed formats, read as the commonest: {} not read",
485                said.join(", ")
486            ),
487            scope: "in this directory's listing".to_string(),
488            read_as_text: None,
489            // Carried so [`merged`] can take these files back out of the footer
490            // pass's tally, which walks the same directory and counts them again.
491            passed_over: Some(left_out.iter().map(|(_, n)| n).sum()),
492        });
493    }
494    notes
495}
496
497/// The `cache-*.arrow` files `map()` wrote beside a Hugging Face cache's splits, which
498/// the read left out: their columns are the mapping's, not the split's.
499pub fn map_caches(count: usize) -> Option<Note> {
500    let files = if count == 1 { "file" } else { "files" };
501    (count > 0).then(|| Note {
502        summary: format!("{count} cache {files} written by map() not read"),
503        scope: "in this directory's listing".to_string(),
504        read_as_text: None,
505        passed_over: None,
506    })
507}
508
509/// Every note the panel shows (the open's, the footers', the view's), with the one fact
510/// the first two both report said once: a mixed directory's skipped files appear in the
511/// open's note and the footer walk's. The open's sentence is kept; the walk's is rebuilt
512/// without those files, keeping skips only it saw.
513pub fn merged(
514    open: &[Note],
515    dataset: &[Note],
516    view: &[Note],
517    schema: Option<&DatasetSchema>,
518) -> Vec<Note> {
519    // The walk's note as is and without the open's files; matched as a whole note, and
520    // left alone if it is not one this module wrote.
521    let rebuilt: Option<(Note, Option<Note>)> = match (
522        open.iter().find_map(|n| n.passed_over),
523        schema.map(|s| s.skipped),
524    ) {
525        (Some(covered), Some(skipped)) => note_about_skipped(skipped).map(|full| {
526            let remaining = SkippedFiles {
527                // Saturating: the walk counts more than the open, but "0 files are not Parquet" must
528                // stay unwritable.
529                not_parquet: skipped.not_parquet.saturating_sub(covered),
530                ..skipped
531            };
532            (full, note_about_skipped(remaining))
533        }),
534        _ => None,
535    };
536    let mut out: Vec<Note> = open.to_vec();
537    for note in dataset {
538        match &rebuilt {
539            Some((full, reduced)) if note == full => out.extend(reduced.clone()),
540            _ => out.push(note.clone()),
541        }
542    }
543    out.extend(view.iter().cloned());
544    out
545}
546
547/// A column that some files were written without. The one note that states a ratio,
548/// because "some" is only meaningful against a total.
549fn absence_note(
550    column: &ColumnDrift,
551    readable: usize,
552    denominator: &str,
553    range: Option<&ColumnRange>,
554    scope: &str,
555) -> Option<Note> {
556    if column.present_in == 0 || column.present_in >= readable {
557        return None;
558    }
559    // Where, as well as how many. A count says a column is unusual; a partition says
560    // where to look, and for a field a feed started sending it says when.
561    let where_it_is = match range {
562        Some(ColumnRange::Only(partition)) => format!(", only {partition}"),
563        Some(ColumnRange::NoneBefore(partition)) => format!(", none before {partition}"),
564        None => String::new(),
565    };
566    Some(Note {
567        summary: format!(
568            "{} is in {} of {}{}; absent from the rest, not null",
569            column.name,
570            group_chrome(column.present_in),
571            denominator,
572            where_it_is
573        ),
574        scope: scope.to_string(),
575        read_as_text: None,
576        passed_over: None,
577    })
578}
579
580/// A column whose files disagree on type beyond widening; counts files without
581/// dividing.
582fn conflict_note(
583    column: &ColumnDrift,
584    dataset: &DatasetSchema,
585    chosen: &str,
586    others: &[String],
587    scope: &str,
588) -> Option<Note> {
589    if column.conflicting_files == 0 {
590        return None;
591    }
592    Some(Note {
593        summary: format!(
594            "{} is {} in {}; read as {} and not read there",
595            column.name,
596            others.join(" or "),
597            how_many(dataset, column.conflicting_files),
598            chosen
599        ),
600        scope: scope.to_string(),
601        // The offer only where it works (a list column cannot be text); `lenient_scan` asks the
602        // same question, so they agree.
603        read_as_text: column.can_read_as_text().then(|| column.name.clone()),
604        passed_over: None,
605    })
606}
607
608/// A column stored in several types that the scan reads into one. Says only that: not
609/// "width" (units, grown structs and untyped files land here too), nor "without loss"
610/// (huge integers as floats, ms datetimes past 2262 as ns).
611fn widening_note(column: &ColumnDrift, chosen: &str, scope: &str) -> Option<Note> {
612    if !column.widened {
613        return None;
614    }
615    Some(Note {
616        summary: format!(
617            "{} is stored as more than one type; read as {chosen}",
618            column.name
619        ),
620        scope: scope.to_string(),
621        read_as_text: None,
622        passed_over: None,
623    })
624}
625
626#[cfg(test)]
627mod tests;