Skip to main content

datui_lib/
notes.rs

1//! What datui noticed about a dataset while doing what it was already doing.
2//!
3//! A note never costs a request or a scan of its own: every one here is read off the
4//! footers the schema and the row count already needed. Notes are never alarming — no
5//! pop-up, no error styling — and every one says what it is based on, so "in 1 of 3
6//! files" is never mistaken for a claim about files datui has not looked at.
7//!
8//! A note is one sentence and the line it rests on. Review after review found false
9//! or empty statements here, and every one of them was in prose that went beyond the
10//! sentence: a denominator over the wrong population, an explanatory line that
11//! contradicted the note above it, a type named by a word two types share. What is
12//! left is what can be checked: one claim per note, and one function deciding the only
13//! ratio any of them states.
14
15use crate::numfmt::group_chrome;
16use crate::schema_union::{ColumnDrift, ColumnRange, DatasetSchema, SchemaOrigin, SkippedFiles};
17use polars::prelude::{DataType, PlSmallStr};
18
19/// One thing datui noticed.
20#[derive(Debug, Clone, PartialEq, Eq)]
21pub struct Note {
22    /// The one line shown in the panel.
23    pub summary: String,
24    /// What the note is based on, so its reach is never overstated.
25    pub scope: String,
26    /// The column datui can offer to read as text, when this note is about one and
27    /// reading it that way would work. `None` for every other note.
28    ///
29    /// The name rather than the prose, because the panel has to act on it: matching a
30    /// column out of a sentence is the kind of thing this module exists to avoid.
31    pub read_as_text: Option<PlSmallStr>,
32    /// How many files this note already counts as passed over by the read, when it is
33    /// the one that says a mixed directory was read as its commonest format. `None`
34    /// for every other note.
35    ///
36    /// The number rather than the prose, for the same reason as `read_as_text`:
37    /// [`merged`] has to take these files back out of the footer pass's tally, and
38    /// reading a count out of a sentence is the kind of thing this module exists to
39    /// avoid.
40    pub passed_over: Option<usize>,
41}
42
43/// Whether a dataset's schema came from a sample of its files.
44fn sampled(dataset: &DatasetSchema) -> bool {
45    matches!(dataset.origin, SchemaOrigin::FooterSample { .. })
46}
47
48/// How many, in the noun that is true of what datui looked at: files when it read
49/// every one, footers when it read only a sample of them.
50fn how_many(dataset: &DatasetSchema, n: usize) -> String {
51    let noun = match (sampled(dataset), n) {
52        (false, 1) => "file",
53        (false, _) => "files",
54        (true, 1) => "footer",
55        (true, _) => "footers",
56    };
57    format!("{} {noun}", group_chrome(n))
58}
59
60/// What the dataset's one ratio is out of: the files datui read and could parse.
61///
62/// This is the only denominator in the module. Where datui read every file and all of
63/// them parsed they are simply "files"; where it sampled, or where a footer would not
64/// parse, saying "files" would claim more than it looked at.
65fn out_of(dataset: &DatasetSchema) -> String {
66    let readable = dataset.files.saturating_sub(dataset.unreadable.len());
67    match (sampled(dataset), dataset.unreadable.is_empty()) {
68        (false, true) => how_many(dataset, readable),
69        (false, false) => format!("the {} that could be read", how_many(dataset, readable)),
70        (true, true) => format!("the {} read", how_many(dataset, readable)),
71        (true, false) => format!("the {} that could be read", how_many(dataset, readable)),
72    }
73}
74
75/// Names for a set of types that tell them apart.
76///
77/// Several types print alike at any given level of detail: every `Struct` is
78/// `struct[1]` however its fields differ, and every `Enum` is `Enum([...])` however its
79/// categories do. A note reading "s is struct[1] in 1 file; read as struct[1]" says
80/// nothing, so the names escalate until they tell the types apart. Polars' `Display`
81/// collapses too — a struct is `struct[1]` — so the full form is its `Debug`.
82fn distinct_names(types: &[&DataType]) -> Vec<String> {
83    let collides = |names: &[String]| {
84        names
85            .iter()
86            .enumerate()
87            .any(|(i, name)| names[i + 1..].contains(name))
88    };
89    // `Display` is how the Schema tab names a type, and it carries the unit, the
90    // precision and the time zone that the table header's one word drops.
91    let shown: Vec<String> = types.iter().map(|t| format!("{t}")).collect();
92    if !collides(&shown) {
93        return shown;
94    }
95    // Two structs are both `struct[1]`, and only the fields tell them apart.
96    let spelled: Vec<String> = types.iter().map(|t| format!("{t:?}")).collect();
97    if !collides(&spelled) {
98        return spelled;
99    }
100    // Polars prints every `Enum` as `Enum([...])` and every global `Categorical` as
101    // `Categorical`, whatever their categories, so even the full form can collide. Numbering them says less than naming them, but "the first
102    // and the second" is at least two things rather than one said twice. Only the ones
103    // that actually collide are numbered; a name that was already unique keeps it.
104    let mut seen: Vec<&String> = Vec::new();
105    spelled
106        .iter()
107        .map(|name| {
108            if spelled.iter().filter(|other| *other == name).count() > 1 {
109                seen.push(name);
110                format!("{name} #{}", seen.iter().filter(|s| **s == name).count())
111            } else {
112                name.clone()
113            }
114        })
115        .collect()
116}
117
118/// What the footers said, as notes. Empty when every file agrees, which is the common
119/// case and the one where there is nothing to say.
120pub fn from_dataset(dataset: &DatasetSchema) -> Vec<Note> {
121    let scope = format!("in {}", dataset.origin);
122    let readable = dataset.files.saturating_sub(dataset.unreadable.len());
123    let denominator = out_of(dataset);
124    let mut notes = Vec::new();
125
126    for column in dataset.drifting() {
127        // The type the column is read as is named once, so two notes about the same
128        // column cannot name it two different ways: whatever it takes to tell the
129        // conflicting types apart is what the widening note calls it too.
130        let mut types: Vec<&DataType> = vec![&column.dtype];
131        types.extend(column.conflicting_types.iter());
132        let names = distinct_names(&types);
133        let (chosen, others) = names.split_first().expect("the chosen type is first");
134
135        // A column can be missing from some files, stored differently in others, and
136        // stored in a narrower type among the rest. Each is true on its own, so each
137        // is said on its own: nothing here suppresses anything else.
138        notes.extend(absence_note(
139            column,
140            readable,
141            &denominator,
142            dataset.column_ranges.get(&column.name),
143            &scope,
144        ));
145        notes.extend(conflict_note(column, dataset, chosen, others, &scope));
146        notes.extend(widening_note(column, chosen, &scope));
147    }
148
149    for column in &dataset.read_as_text {
150        notes.push(text_note(column, &scope));
151    }
152
153    notes.extend(empty_files_note(dataset, &scope));
154    notes.extend(row_group_note(dataset, &scope));
155    notes.extend(small_files_note(dataset, &scope));
156    notes.extend(partition_layout_note(dataset));
157    notes.extend(skipped_files_note(dataset));
158
159    if !dataset.unreadable.is_empty() {
160        notes.push(Note {
161            summary: format!(
162                "{} unreadable, left out",
163                how_many(dataset, dataset.unreadable.len())
164            ),
165            scope: scope.clone(),
166            read_as_text: None,
167            passed_over: None,
168        });
169    }
170
171    notes
172}
173
174/// Rows a filter or sort leaves out, because the column it names is not read from the
175/// files those rows came from.
176///
177/// The only note here about the view rather than the dataset, and the only one datui
178/// writes in answer to something the user just did.
179///
180/// It is phrased about the files, not about the view, and that is the whole care of
181/// it. "2 rows are left out" reads as a claim that the view is two rows shorter, which
182/// is not true when a filter had already dropped one of them; how many rows are in the
183/// files that hold the column in another type is a fact of the footers, true whatever
184/// else the view is doing. Both counts here are of that kind.
185///
186/// The cost of saying it that way is that the note also appears where those rows had
187/// already gone — a filter that excluded them, then a sort on the column. It is still
188/// true there, and the alternative is a count of what the view actually lost, which
189/// cannot be had without collecting the frame twice.
190pub fn left_out_note(
191    column: &ColumnDrift,
192    dataset: &DatasetSchema,
193    rows: usize,
194    filtered: bool,
195    sorted: bool,
196) -> Note {
197    let what = match (filtered, sorted) {
198        (true, true) => "filter and sort",
199        (true, false) => "filter",
200        // Called only for a column the view names, so it names it one way or the other.
201        _ => "sort",
202    };
203    let there = if rows == 1 {
204        "1 row".to_string()
205    } else {
206        format!("{} rows", group_chrome(rows))
207    };
208    Note {
209        summary: format!(
210            "{}: {there} in {} left out of the {what}",
211            column.name,
212            how_many(dataset, column.conflicting_files)
213        ),
214        scope: format!("in {}", dataset.origin),
215        read_as_text: None,
216        passed_over: None,
217    }
218}
219
220/// Files that hold no rows at all.
221///
222/// A partition written for a day nothing happened, or a job that produced a header and no
223/// data. Worth saying because the dataset then has fewer days of data than it has
224/// directories, and a reader counting directories would get the wrong answer — but it is
225/// not a fault, and a pipeline that writes a file per day will have some.
226///
227/// Counts without dividing: how many of the files datui read hold nothing is a fact
228/// about those files, and the scope line says which files those were.
229///
230/// The one counting note that says "file" even where the schema came from a sample,
231/// rather than the "footer" [`how_many`] would give it. A footer does not hold rows —
232/// it records how many the file holds — so the substitution that keeps the other notes
233/// honest makes this one a category slip. It costs nothing here: there is no ratio to
234/// overstate, and the scope line already says only a sample was read, so "1 file holds
235/// no rows · in 20,000 of 500,000 footers (sample)" claims nothing about the other
236/// 480,000.
237fn empty_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
238    let empty = dataset.empty_files;
239    if empty == 0 {
240        return None;
241    }
242    let (count, verb) = if empty == 1 {
243        ("1 file".to_string(), "holds")
244    } else {
245        (format!("{} files", group_chrome(empty)), "hold")
246    };
247    Some(Note {
248        summary: format!("{count} {verb} no rows"),
249        scope: scope.to_string(),
250        read_as_text: None,
251        passed_over: None,
252    })
253}
254
255/// Row groups big enough that reading a page means reading a lot more than the page.
256///
257/// A row group is what a reader fetches: a hundred rows anywhere inside one costs the
258/// whole of it. Over a network that is the difference between a page arriving and a
259/// page arriving after sixty-four megabytes do, and there is nothing the user can do
260/// about it from here — which is exactly why it is worth saying rather than leaving
261/// them to wonder why scrolling is slow.
262///
263/// The threshold is the size at which one row group is a noticeable download on an
264/// ordinary connection; below it, nobody needs telling. States the middle size rather
265/// than the largest: one row group of a gigabyte among thousands of small ones is a
266/// different dataset from one where every row group is a gigabyte, and only the second
267/// is worth a note.
268///
269/// Says "the middle row group is", not "row groups are, apiece" — the middle of
270/// `[1 MiB, 100 MiB, 100 MiB]` is 100 MiB and one of those row groups is not.
271///
272/// And says the size, not what a page costs to fetch. They are not the same number:
273/// a page projects away binary columns (see `binary_stub_exprs`), so their chunks are
274/// never downloaded, while this size counts every chunk in the group. The figure is
275/// the row group's; what follows it is why a row group's size is the one that matters.
276///
277/// The middle is over every row group of every footer read, each counting once. Not
278/// weighted by rows, though a page is likelier to land in a group that holds more of
279/// them: a dataset of one file of ten thousand small groups beside a hundred files of
280/// one huge group each is called small by this and would be called large by that.
281/// Counting groups is the statistic that matches the sentence — how big a row group
282/// is, of the row groups there are.
283fn row_group_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
284    /// Sixty-four mebibytes, the size a page in that range costs to reach.
285    const BIG: usize = 64 * 1024 * 1024;
286    let median = dataset.median_row_group_bytes?;
287    if median <= BIG {
288        return None;
289    }
290    Some(Note {
291        summary: format!(
292            "median row group {}, each read whole",
293            crate::widgets::info::format_bytes(median as u64)
294        ),
295        scope: scope.to_string(),
296        read_as_text: None,
297        passed_over: None,
298    })
299}
300
301/// A dataset of very many files, each holding very little.
302///
303/// Says what happened, not what it cost. The first draft of this note said finding the
304/// files costs more than reading them, and review measured it: a thousand two hundred
305/// files took nine milliseconds to find and eight hundred to read. It cannot be true
306/// over a network either — a footer read is a *suffix* of the file, so it can never
307/// move more bytes than reading the file does. What is true, and is the thing the user
308/// waited for, is that a footer was read for every one of these files before a single
309/// row was.
310///
311/// Both halves have to hold. Small files on their own are a normal day's partitions,
312/// and a few large ones cost nothing to open. It is very many *and* very small that
313/// makes the opening a job of its own — and one nothing here can fix, since the remedy
314/// is upstream in whatever writes them.
315///
316/// The count is every file the listing found; the middle size is over the footers
317/// datui opened, which the sentence names. The two are different populations where the
318/// dataset was too large to open every footer, and saying both numbers is what keeps
319/// the middle from reading as a fact about all of them.
320fn small_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
321    /// Past this many files the footer pass is a job of its own. Above a year of
322    /// hourly partitions, which is an ordinary shape and not a complaint.
323    const MANY: usize = 10_000;
324    /// Below this a file is small by any warehouse's standard, where the figure aimed
325    /// at is hundreds of megabytes.
326    const SMALL: usize = 1024 * 1024;
327    let files = dataset.origin.total_files();
328    let median = dataset.median_file_bytes?;
329    // A file of no bytes is not a Parquet file, so a middle of zero means datui does
330    // not know the sizes rather than that they are small — and "the middle one is 0 B"
331    // would be a claim about a dataset that cannot exist.
332    if median == 0 || files <= MANY || median >= SMALL {
333        return None;
334    }
335    let read = dataset.files;
336    // "opened for its footer", not "a footer was read": a footer that would not parse
337    // was still opened for, and the note beside this one says three of them were.
338    //
339    // And a semicolon, not "so": the footer pass is one per file whatever the files
340    // hold, so only the count leads to it. Joining the two with "so" would make the
341    // size look like half the reason.
342    let footers = if read == files {
343        "every footer".to_string()
344    } else {
345        format!("{} footers", group_chrome(read))
346    };
347    Some(Note {
348        summary: format!(
349            "{} files, median {}; {footers} read before any row",
350            group_chrome(files),
351            crate::widgets::info::format_bytes(median as u64)
352        ),
353        scope: scope.to_string(),
354        read_as_text: None,
355        passed_over: None,
356    })
357}
358
359/// Directories that do not all partition by the same keys.
360///
361/// Says the shape and stops there. What it *costs* is not something this note can see.
362/// The scan reads its partition columns off one branch of the tree, and which branch
363/// that is comes back from the filesystem in whatever order it likes. Usually every
364/// file under the other key then fails and the dataset does not open at all.
365///
366/// But not always, and what decides it is not the branch — it is which file name sorts
367/// first. The scan hands Polars its paths sorted, and Polars takes the hive schema from
368/// the first of them: put one unpartitioned file at the root and whether it sorts above
369/// `date=` decides whether the dataset opens with the column null or fails to open. A
370/// file called `data.parquet` does; one called `loose.parquet` does not. Nothing a note
371/// can see, and about as good a reason as there could be for a note not to say what
372/// something costs.
373///
374/// Two rounds were spent on sentences that picked one of those and stated it as the
375/// consequence. The user guide has room to set them out; a note has one sentence, and
376/// the sentence true of every such dataset is the shape itself.
377///
378/// Read off the names of every file, which is the one thing the listing knows that
379/// reading a file cannot tell you — so it has a scope line of its own, and on a dataset
380/// too large to open every footer this note still saw all of it.
381fn partition_layout_note(dataset: &DatasetSchema) -> Option<Note> {
382    /// Layouts named before the rest are counted rather than spelled. A note is one
383    /// sentence, and a dataset with a hundred layouts would otherwise make it a page.
384    const NAMED: usize = 2;
385    if dataset.partition_layouts.len() < 2 {
386        return None;
387    }
388    let (named, rest) = dataset
389        .partition_layouts
390        .split_at(dataset.partition_layouts.len().min(NAMED));
391    let mut clauses: Vec<String> = named
392        .iter()
393        .map(|(keys, files)| format!("{} by {}", how_many_files(*files), keys.join("/")))
394        .collect();
395    let (dropped_ways, dropped_files) = dataset.partition_layouts_dropped;
396    let ways = rest.len() + dropped_ways;
397    let files: usize = rest.iter().map(|(_, files)| files).sum::<usize>() + dropped_files;
398    if ways > 0 {
399        clauses.push(format!(
400            "{} by {} other {}",
401            how_many_files(files),
402            group_chrome(ways),
403            if ways == 1 { "way" } else { "ways" }
404        ));
405    }
406    Some(Note {
407        summary: format!("mixed partition keys: {}", clauses.join(", ")),
408        scope: format!(
409            "in the names of {} files",
410            group_chrome(dataset.listed_files)
411        ),
412        read_as_text: None,
413        passed_over: None,
414    })
415}
416
417/// `n files`, or `1 file`. Plain files, because the caller counted every one of them.
418fn how_many_files(n: usize) -> String {
419    format!(
420        "{} {}",
421        group_chrome(n),
422        if n == 1 { "file" } else { "files" }
423    )
424}
425
426/// A column being read as text from every file, because it was asked for that way.
427///
428/// Stands in for the conflict note it replaced, and says the one thing that changes
429/// about the column beyond what is now visible in it: a filter or sort on it compares
430/// text. `n > 5` written for a number keeps `"sixty"` and drops `"10"`, and a view
431/// that quietly did that with nothing on screen to say so would be a view the user
432/// reads wrongly.
433fn text_note(column: &PlSmallStr, scope: &str) -> Note {
434    Note {
435        summary: format!("{column} read as text: filter and sort compare text"),
436        scope: scope.to_string(),
437        read_as_text: None,
438        passed_over: None,
439    }
440}
441
442/// Files in the directory that are not Parquet, and so are not in the table.
443///
444/// Only the ones somebody might have meant as data. A writer leaves `_SUCCESS`, `.crc`
445/// and `_metadata` beside what it wrote, and saying so on every directory a job produced
446/// would put an accent on the Info key for the most ordinary thing a directory can
447/// contain — the same reason the small-files note waits for a threshold rather than
448/// firing on every directory with more than one file in it. Where the note does fire it
449/// counts them beside what it is about, so the numbers are the directory's rather than a
450/// selection from it — as far as the listing saw, which is not the same as all of it:
451/// a subtree it could not read, or one below the depth it stops at, is in neither.
452fn skipped_files_note(dataset: &DatasetSchema) -> Option<Note> {
453    note_about_skipped(dataset.skipped)
454}
455
456/// The note [`skipped_files_note`] writes, from the tally alone.
457///
458/// Split out so [`merged`] can rebuild it with the files the open already reported
459/// taken back out, without reading a count out of the note's own sentence.
460fn note_about_skipped(skipped: SkippedFiles) -> Option<Note> {
461    let SkippedFiles {
462        bookkeeping,
463        not_parquet,
464        empty,
465    } = skipped;
466    if not_parquet == 0 && empty == 0 {
467        return None;
468    }
469    let files = |n: usize| {
470        if n == 1 {
471            "1 file".to_string()
472        } else {
473            format!("{} files", group_chrome(n))
474        }
475    };
476    // An object with nothing in it is a write that stopped, and saying so is the point;
477    // the rest is counted beside it so the total is the directory's, not a selection.
478    let mut said = Vec::new();
479    if empty > 0 {
480        let what = if empty == 1 { "file" } else { "files" };
481        said.push(format!("{} empty {what}", group_chrome(empty)));
482    }
483    if not_parquet > 0 {
484        said.push(format!("{} not Parquet", files(not_parquet)));
485    }
486    if bookkeeping > 0 {
487        let what = if bookkeeping == 1 { "file" } else { "files" };
488        said.push(format!(
489            "{} writer bookkeeping {what}",
490            group_chrome(bookkeeping)
491        ));
492    }
493    Some(Note {
494        summary: format!("skipped: {}", said.join(", ")),
495        scope: "in this directory's listing".to_string(),
496        read_as_text: None,
497        passed_over: None,
498    })
499}
500
501/// The most names a note lists before it cuts the rest with an ellipsis.
502const NAMES_SHOWN: usize = 3;
503
504/// `names`, the first few of them, joined, and an ellipsis for the rest.
505pub fn some_names<S: AsRef<str>>(names: &[S]) -> String {
506    let mut said: Vec<&str> = names.iter().take(NAMES_SHOWN).map(AsRef::as_ref).collect();
507    let ellipsis = crate::glyphs::get().ellipsis;
508    if names.len() > NAMES_SHOWN {
509        said.push(ellipsis);
510    }
511    said.join(", ")
512}
513
514/// The files a read of several passed over because they hold no header: empty, blank,
515/// or nothing but NUL padding.
516pub fn no_header(files: &[&std::path::Path]) -> Option<Note> {
517    if files.is_empty() {
518        return None;
519    }
520    let names: Vec<String> = files
521        .iter()
522        .map(|f| {
523            f.file_name().map_or_else(
524                || f.display().to_string(),
525                |n| n.to_string_lossy().into_owned(),
526            )
527        })
528        .collect();
529    let what = if files.len() == 1 { "file" } else { "files" };
530    Some(Note {
531        summary: format!(
532            "{} {what} with no header skipped: {}",
533            files.len(),
534            some_names(&names)
535        ),
536        scope: "empty, blank, or only NUL padding".to_string(),
537        read_as_text: None,
538        passed_over: None,
539    })
540}
541
542/// What the open itself has to say, before a footer has been read.
543///
544/// Two facts, both decided by the route that opened the directory rather than by anything
545/// in the data: which of the directory's data files this read passed over, and whether
546/// the directory is a lake table being read as its plain files. Neither is a defect in
547/// the data — they are what datui chose to do, and #275's rule is that datui never
548/// refuses a read the user asked for and always says what it did instead.
549pub fn from_the_open(
550    left_out: &[(crate::FileFormat, usize)],
551    lake: Option<&str>,
552    files_differ: crate::schema_union::Disagreement,
553    names_look_like_data: bool,
554) -> Vec<Note> {
555    let mut notes = Vec::new();
556    // What the read had to do to stack them, in the words of what it actually found.
557    // These formats carry no footer, so there is no per-column tally behind either
558    // sentence; a Parquet dataset gets the exact version instead — which columns, in
559    // how many files, and where — from footers it had to read anyway.
560    //
561    // The scope says a spread, because that is what was looked at: three files, the
562    // ends and the middle, whatever the directory's size.
563    let scope = || "in a spread of this directory's files".to_string();
564    if files_differ.columns {
565        notes.push(Note {
566            summary: "columns differ across files: a missing column reads null".to_string(),
567            scope: scope(),
568            read_as_text: None,
569            passed_over: None,
570        });
571    }
572    if files_differ.headerless {
573        notes.push(Note {
574            summary: concat!(
575                "no header row? first row read as names: ",
576                "H on Schema, or --no-header, reads it as data"
577            )
578            .to_string(),
579            scope: scope(),
580            read_as_text: None,
581            passed_over: None,
582        });
583    }
584    // The same shape in one file: every column name a number, which a header almost
585    // never is and a first row of data often is.
586    if names_look_like_data && !files_differ.headerless {
587        notes.push(Note {
588            summary: "column names look like data: H on Schema reads them as a row".to_string(),
589            scope: "from the column names".to_string(),
590            read_as_text: None,
591            passed_over: None,
592        });
593    }
594    if files_differ.types {
595        notes.push(Note {
596            summary: "a column's type differs across files: read as the wider type".to_string(),
597            scope: scope(),
598            read_as_text: None,
599            passed_over: None,
600        });
601    }
602    if let Some(format) = lake {
603        notes.push(Note {
604            // The strongest sentence the panel has, because it is the one place a
605            // number on screen is not a number about the table. A delete leaves its
606            // rows on disk, an update leaves the version it replaced, and compaction
607            // leaves both sides — all of them counted here.
608            summary: format!(
609                "{format} table's files, not the table: deleted rows and old versions counted"
610            ),
611            scope: format!("in this {format} table's directory"),
612            read_as_text: None,
613            passed_over: None,
614        });
615    }
616    if !left_out.is_empty() {
617        let said: Vec<String> = left_out
618            .iter()
619            .map(|(format, n)| format!("{n} {}", format.name()))
620            .collect();
621        notes.push(Note {
622            summary: format!(
623                "mixed formats, read as the commonest: {} not read",
624                said.join(", ")
625            ),
626            scope: "in this directory's listing".to_string(),
627            read_as_text: None,
628            // Carried so [`merged`] can take these files back out of the footer
629            // pass's tally, which walks the same directory and counts them again.
630            passed_over: Some(left_out.iter().map(|(_, n)| n).sum()),
631        });
632    }
633    notes
634}
635
636/// The `cache-*.arrow` files `map()` wrote beside a Hugging Face cache's splits, which
637/// the read left out: their columns are the mapping's, not the split's.
638pub fn map_caches(count: usize) -> Option<Note> {
639    let files = if count == 1 { "file" } else { "files" };
640    (count > 0).then(|| Note {
641        summary: format!("{count} cache {files} written by map() not read"),
642        scope: "in this directory's listing".to_string(),
643        read_as_text: None,
644        passed_over: None,
645    })
646}
647
648/// Every note the panel shows — what the open did, what the footers said, then what
649/// the view leaves out — with the one fact the first two both report said once.
650///
651/// A directory of mixed formats read as its commonest is reported by the open ("read
652/// as the commonest; 1 csv not read"), and then the footer pass walks the same
653/// directory and counts the same files among what it passed ("in the directory, 1
654/// file is not Parquet"). The open's sentence names the formats and says why they are
655/// not in the table, so it is the one kept; the footer note is rebuilt with those
656/// files taken back out, which keeps every skip only the walk can see — a stray in a
657/// partition below the top level, a name no reader claims, an empty object, a
658/// writer's bookkeeping. The two tallies never meet anywhere else: the open's is
659/// settled before a footer is read, and the walk's arrives with the dataset, so the
660/// caller that holds both halves hands them here.
661pub fn merged(
662    open: &[Note],
663    dataset: &[Note],
664    view: &[Note],
665    schema: Option<&DatasetSchema>,
666) -> Vec<Note> {
667    // What the walk said as it stands, and with the open's files taken out. Matched
668    // as a whole note rather than by its prose, and if the dataset does not hold
669    // exactly that note — a shape this module did not write — nothing is touched.
670    let rebuilt: Option<(Note, Option<Note>)> = match (
671        open.iter().find_map(|n| n.passed_over),
672        schema.map(|s| s.skipped),
673    ) {
674        (Some(covered), Some(skipped)) => note_about_skipped(skipped).map(|full| {
675            let remaining = SkippedFiles {
676                // Saturating: the open counts one level of recognized names, the
677                // walk everything it saw, so the walk's tally is never smaller —
678                // but a false "0 files are not Parquet" must stay unwritable.
679                not_parquet: skipped.not_parquet.saturating_sub(covered),
680                ..skipped
681            };
682            (full, note_about_skipped(remaining))
683        }),
684        _ => None,
685    };
686    let mut out: Vec<Note> = open.to_vec();
687    for note in dataset {
688        match &rebuilt {
689            Some((full, reduced)) if note == full => out.extend(reduced.clone()),
690            _ => out.push(note.clone()),
691        }
692    }
693    out.extend(view.iter().cloned());
694    out
695}
696
697/// A column that some files were written without. The one note that states a ratio,
698/// because "some" is only meaningful against a total.
699fn absence_note(
700    column: &ColumnDrift,
701    readable: usize,
702    denominator: &str,
703    range: Option<&ColumnRange>,
704    scope: &str,
705) -> Option<Note> {
706    if column.present_in == 0 || column.present_in >= readable {
707        return None;
708    }
709    // Where, as well as how many. A count says a column is unusual; a partition says
710    // where to look, and for a field a feed started sending it says when.
711    let where_it_is = match range {
712        Some(ColumnRange::Only(partition)) => format!(", only {partition}"),
713        Some(ColumnRange::NoneBefore(partition)) => format!(", none before {partition}"),
714        None => String::new(),
715    };
716    Some(Note {
717        summary: format!(
718            "{} is in {} of {}{}; absent from the rest, not null",
719            column.name,
720            group_chrome(column.present_in),
721            denominator,
722            where_it_is
723        ),
724        scope: scope.to_string(),
725        read_as_text: None,
726        passed_over: None,
727    })
728}
729
730/// A column whose files disagree on its type beyond what widening can settle.
731///
732/// Counts without dividing: how many files hold it in a type that lost is a fact about
733/// those files, and needs no total to be true.
734fn conflict_note(
735    column: &ColumnDrift,
736    dataset: &DatasetSchema,
737    chosen: &str,
738    others: &[String],
739    scope: &str,
740) -> Option<Note> {
741    if column.conflicting_files == 0 {
742        return None;
743    }
744    Some(Note {
745        summary: format!(
746            "{} is {} in {}; read as {} and not read there",
747            column.name,
748            others.join(" or "),
749            how_many(dataset, column.conflicting_files),
750            chosen
751        ),
752        scope: scope.to_string(),
753        // The offer, and only where it would work: a column one file holds as a list
754        // cannot be shown as text at all, and an offer that did nothing would be worse
755        // than none. `lenient_scan` asks the same question again, so the two cannot
756        // disagree about which columns are on offer.
757        read_as_text: column.can_read_as_text().then(|| column.name.clone()),
758        passed_over: None,
759    })
760}
761
762/// A column the files store in more than one type, where the scan reads them all into
763/// one.
764///
765/// Says only that: not "width", since a datetime unit, a struct that gained a field and
766/// a file that never typed the column all land here, and not "without loss", since a
767/// very large integer read as a float, or a millisecond datetime read as nanoseconds
768/// past the year 2262, is not exact.
769fn widening_note(column: &ColumnDrift, chosen: &str, scope: &str) -> Option<Note> {
770    if !column.widened {
771        return None;
772    }
773    Some(Note {
774        summary: format!(
775            "{} is stored as more than one type; read as {chosen}",
776            column.name
777        ),
778        scope: scope.to_string(),
779        read_as_text: None,
780        passed_over: None,
781    })
782}
783
784#[cfg(test)]
785mod tests {
786    use super::*;
787
788    use crate::schema_union::{FileFooter, union_file_schemas};
789    use polars::prelude::{DataType, Field, Schema, TimeUnit, TimeZone};
790    use std::sync::Arc;
791
792    /// rustfmt joins a `\`-continued literal back onto one line with its indentation
793    /// inside, which put eighteen spaces into the middle of two of these sentences.
794    #[test]
795    fn the_notes_from_an_open_have_no_holes_in_them() {
796        let notes = from_the_open(
797            &[(crate::FileFormat::Json, 1)],
798            Some("Delta"),
799            crate::schema_union::Disagreement {
800                columns: true,
801                types: true,
802                headerless: true,
803            },
804            false,
805        );
806        assert!(notes.len() >= 3, "{notes:?}");
807        for note in &notes {
808            assert!(!note.summary.contains("  "), "{:?}", note.summary);
809        }
810    }
811
812    /// A dataset with nothing but a skip tally, as the footer walk hands one over.
813    fn walked(skipped: SkippedFiles) -> DatasetSchema {
814        union_file_schemas(&[], SchemaOrigin::AllFooters(0)).with_skipped(skipped)
815    }
816
817    fn agree() -> crate::schema_union::Disagreement {
818        crate::schema_union::Disagreement {
819            columns: false,
820            types: false,
821            headerless: false,
822        }
823    }
824
825    /// The open and the footer walk both count what a mixed directory's read passed
826    /// over, and on screen that was the same fact twice, one wording above the other.
827    /// The open's sentence names the formats and says why, so it is the one kept.
828    #[test]
829    fn a_mixed_directory_is_not_reported_twice() {
830        let open = from_the_open(&[(crate::FileFormat::Csv, 1)], None, agree(), false);
831        let dataset = walked(SkippedFiles {
832            bookkeeping: 0,
833            not_parquet: 1,
834            empty: 0,
835        });
836        let said: Vec<String> = merged(&open, &from_dataset(&dataset), &[], Some(&dataset))
837            .into_iter()
838            .map(|n| n.summary)
839            .collect();
840        assert_eq!(
841            said,
842            ["mixed formats, read as the commonest: 1 csv not read"],
843            "one fact, said once, in the open's words"
844        );
845    }
846
847    /// What only the walk saw stays counted: a name no reader claims, an empty
848    /// object and a writer's bookkeeping are not among the formats the open named,
849    /// so taking the open's files out must not take these with them. The view's
850    /// notes still follow, untouched.
851    #[test]
852    fn what_only_the_walk_saw_stays_counted() {
853        let open = from_the_open(&[(crate::FileFormat::Csv, 1)], None, agree(), false);
854        let dataset = walked(SkippedFiles {
855            bookkeeping: 2,
856            not_parquet: 2,
857            empty: 1,
858        });
859        let view = [text_note(&PlSmallStr::from("n"), "in 3 files")];
860        let said: Vec<String> = merged(&open, &from_dataset(&dataset), &view, Some(&dataset))
861            .into_iter()
862            .map(|n| n.summary)
863            .collect();
864        assert_eq!(
865            said,
866            [
867                "mixed formats, read as the commonest: 1 csv not read".to_string(),
868                "skipped: 1 empty file, 1 file not Parquet, 2 writer bookkeeping files".to_string(),
869                "n read as text: filter and sort compare text".to_string(),
870            ],
871            "the walk's own findings and the view's notes survive the merge"
872        );
873    }
874
875    /// A tally the open never spoke to is left exactly as the walk wrote it: a hive
876    /// dataset's strays live below the top level, where the open's one-level look
877    /// never reaches, and its note is the only thing that reports them.
878    #[test]
879    fn a_walk_only_tally_is_left_alone() {
880        let open = from_the_open(&[], Some("Delta"), agree(), false);
881        let dataset = walked(SkippedFiles {
882            bookkeeping: 0,
883            not_parquet: 1,
884            empty: 0,
885        });
886        let said: Vec<String> = merged(&open, &from_dataset(&dataset), &[], Some(&dataset))
887            .into_iter()
888            .map(|n| n.summary)
889            .collect();
890        assert_eq!(said.len(), 2, "{said:?}");
891        assert!(
892            said[1] == "skipped: 1 file not Parquet",
893            "different facts do not merge: {said:?}"
894        );
895    }
896
897    fn file(columns: &[(&str, DataType)], rows: usize) -> Option<FileFooter> {
898        let mut schema = Schema::with_capacity(columns.len());
899        for (name, dtype) in columns {
900            schema.with_column((*name).into(), dtype.clone());
901        }
902        Some(FileFooter {
903            schema: Arc::new(schema),
904            row_group_rows: vec![rows],
905            file_bytes: 0,
906            row_group_bytes: Vec::new(),
907            column_bytes: Vec::new(),
908        })
909    }
910
911    /// One dataset shape, and the notes it should produce.
912    #[derive(Default)]
913    struct Shape {
914        what: &'static str,
915        files: Vec<Option<FileFooter>>,
916        /// `Some(total)` when the footers stand in for a larger dataset.
917        sampled: Option<usize>,
918        /// The file names, where the shape is about how they are laid out rather than
919        /// about what is in them. Empty for a shape that has nothing to say about it.
920        paths: Vec<&'static str>,
921        expected: Vec<&'static str>,
922    }
923
924    fn dataset_for(shape: &Shape) -> DatasetSchema {
925        let origin = match shape.sampled {
926            Some(total) => SchemaOrigin::FooterSample {
927                read: shape.files.len(),
928                total,
929            },
930            None => SchemaOrigin::AllFooters(shape.files.len()),
931        };
932        let dataset = union_file_schemas(&shape.files, origin);
933        if shape.paths.is_empty() {
934            return dataset;
935        }
936        let paths: Vec<String> = shape.paths.iter().map(|p| p.to_string()).collect();
937        dataset.with_partition_layouts("d", &paths)
938    }
939
940    fn notes_for(shape: &Shape) -> Vec<String> {
941        from_dataset(&dataset_for(shape))
942            .into_iter()
943            .map(|n| n.summary)
944            .collect()
945    }
946
947    /// Every shape of disagreement, and every line each one produces.
948    ///
949    /// Round after round of review found a wrong denominator in this module, each in a case
950    /// the previous fix had not considered. The wording layer now states one ratio and
951    /// counts everything else without dividing, and this lists the shapes end to end so
952    /// the next wrong one shows up as a diff. Deliberately a transcript rather than a
953    /// set of properties: the failure mode was plausible-looking prose, and a property
954    /// would have to encode the same reasoning that kept going wrong.
955    #[test]
956    fn every_shape_of_disagreement_reads_the_way_it_should() {
957        let i32 = DataType::Int32;
958        let i64 = DataType::Int64;
959        let str = DataType::String;
960        let with_n = |a: DataType, b: DataType| {
961            vec![
962                file(&[("id", i64.clone()), ("n", a)], 50),
963                file(&[("id", i64.clone()), ("n", b)], 50),
964            ]
965        };
966
967        let cases = vec![
968            // --- directories that disagree about what they are partitioned by ---
969            Shape {
970                what: "one directory under another key",
971                files: vec![
972                    file(&[("id", DataType::Int64)], 1),
973                    file(&[("id", DataType::Int64)], 1),
974                ],
975                paths: vec!["d/date=1/a.parquet", "d/dt=2/b.parquet"],
976                expected: vec!["mixed partition keys: 1 file by date, 1 file by dt"],
977                ..Shape::default()
978            },
979            Shape {
980                what: "directories that agree",
981                files: vec![
982                    file(&[("id", DataType::Int64)], 1),
983                    file(&[("id", DataType::Int64)], 1),
984                ],
985                paths: vec!["d/date=1/a.parquet", "d/date=2/b.parquet"],
986                expected: vec![],
987                ..Shape::default()
988            },
989            // --- where a column that is not in every file sits ---
990            Shape {
991                what: "a column only one partition has",
992                files: vec![
993                    file(&[("id", DataType::Int64)], 1),
994                    file(&[("id", DataType::Int64), ("oops", DataType::String)], 1),
995                    file(&[("id", DataType::Int64)], 1),
996                ],
997                paths: vec![
998                    "d/date=2024-03-01/a.parquet",
999                    "d/date=2024-03-02/b.parquet",
1000                    "d/date=2024-03-03/c.parquet",
1001                ],
1002                expected: vec![
1003                    "oops is in 1 of 3 files, only date=2024-03-02; absent from \
1004                     the rest, not null",
1005                ],
1006                ..Shape::default()
1007            },
1008            Shape {
1009                what: "a column the feed started sending",
1010                files: vec![
1011                    file(&[("id", DataType::Int64)], 1),
1012                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1013                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1014                ],
1015                paths: vec![
1016                    "d/date=2010-07-17/a.parquet",
1017                    "d/date=2010-07-18/b.parquet",
1018                    "d/date=2010-07-19/c.parquet",
1019                ],
1020                expected: vec![
1021                    "fee is in 2 of 3 files, none before date=2010-07-18; absent from \
1022                     the rest, not null",
1023                ],
1024                ..Shape::default()
1025            },
1026            Shape {
1027                what: "a column in some files but no pattern to where",
1028                files: vec![
1029                    file(&[("id", DataType::Int64), ("odd", DataType::Int64)], 1),
1030                    file(&[("id", DataType::Int64)], 1),
1031                    file(&[("id", DataType::Int64), ("odd", DataType::Int64)], 1),
1032                ],
1033                paths: vec![
1034                    "d/date=1/a.parquet",
1035                    "d/date=2/b.parquet",
1036                    "d/date=3/c.parquet",
1037                ],
1038                expected: vec!["odd is in 2 of 3 files; absent from the rest, not null"],
1039                ..Shape::default()
1040            },
1041            Shape {
1042                what: "a column starting where the listing and the reader disagree",
1043                files: vec![
1044                    file(&[("id", DataType::Int64)], 1),
1045                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1046                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1047                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1048                ],
1049                // Sorted bytewise, `part=10` comes second. Saying "from part=10 on"
1050                // would tell a reader that parts 2 and 3 are without it, and they are
1051                // not: there is no honest way to say where this one starts.
1052                paths: vec![
1053                    "d/part=1/a.parquet",
1054                    "d/part=10/b.parquet",
1055                    "d/part=2/c.parquet",
1056                    "d/part=3/d.parquet",
1057                ],
1058                expected: vec!["fee is in 3 of 4 files; absent from the rest, not null"],
1059                ..Shape::default()
1060            },
1061            Shape {
1062                what: "a column starting at a month spelled two ways",
1063                files: vec![
1064                    file(&[("id", DataType::Int64)], 1),
1065                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1066                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1067                ],
1068                // `m=03` is March and so is `m=3` — a backfill beside a job. March
1069                // lacks the column, so "none before m=3" says it starts at a month
1070                // whose directory does not have it.
1071                paths: vec!["d/m=03/a.parquet", "d/m=3/b.parquet", "d/m=4/c.parquet"],
1072                expected: vec!["fee is in 2 of 3 files; absent from the rest, not null"],
1073                ..Shape::default()
1074            },
1075            Shape {
1076                what: "a column of a dataset whose directories name their keys in two orders",
1077                files: vec![
1078                    file(&[("id", DataType::Int64)], 1),
1079                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1080                ],
1081                // Hive matches partition columns by name, so these two directories are
1082                // one partition written both ways round. Saying the column begins at the
1083                // second would be saying it begins where the first is.
1084                paths: vec!["d/m=03/y=2024/a.parquet", "d/y=2024/m=03/b.parquet"],
1085                // And nothing else says so: the layouts note compares which keys a
1086                // directory uses, not the order it writes them in, so these two agree.
1087                // This note staying quiet is the only thing between a reader and a
1088                // sentence about a place that is written down twice.
1089                expected: vec!["fee is in 1 of 2 files; absent from the rest, not null"],
1090                ..Shape::default()
1091            },
1092            Shape {
1093                what: "a column two files of one partition have",
1094                files: vec![
1095                    file(&[("id", DataType::Int64)], 1),
1096                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1097                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1098                ],
1099                // The ordinary shape: a partition holds more than one file. Counting
1100                // its partition twice would make it look like two, and the note that
1101                // names it would quietly stop appearing.
1102                paths: vec![
1103                    "d/date=2024-03-01/a.parquet",
1104                    "d/date=2024-03-02/b.parquet",
1105                    "d/date=2024-03-02/c.parquet",
1106                ],
1107                expected: vec![
1108                    "fee is in 2 of 3 files, only date=2024-03-02; absent from the \
1109                     rest, not null",
1110                ],
1111                ..Shape::default()
1112            },
1113            Shape {
1114                what: "a column starting halfway through a partition",
1115                files: vec![
1116                    file(&[("id", DataType::Int64)], 1),
1117                    file(&[("id", DataType::Int64)], 1),
1118                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1119                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1120                ],
1121                // `date=2024-01-02` holds one file with `fee` and one without, so it is
1122                // not a date the column begins at.
1123                paths: vec![
1124                    "d/date=2024-01-01/a.parquet",
1125                    "d/date=2024-01-02/b.parquet",
1126                    "d/date=2024-01-02/c.parquet",
1127                    "d/date=2024-01-03/d.parquet",
1128                ],
1129                expected: vec!["fee is in 2 of 4 files; absent from the rest, not null"],
1130                ..Shape::default()
1131            },
1132            Shape {
1133                what: "a column whose partition holds the file without it",
1134                files: vec![
1135                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1136                    file(&[("id", DataType::Int64)], 1),
1137                ],
1138                // The file without it is at `y=2024/m=03`, which is under `y=2024`.
1139                // "only y=2024" would send a reader to a directory that holds it.
1140                paths: vec!["d/y=2024/a.parquet", "d/y=2024/m=03/b.parquet"],
1141                expected: vec![
1142                    "fee is in 1 of 2 files; absent from the rest, not null",
1143                    // Ragged depth is a disagreement in its own right, and says so.
1144                    "mixed partition keys: 1 file by m/y, 1 file by y",
1145                ],
1146                ..Shape::default()
1147            },
1148            Shape {
1149                what: "a column of a dataset partitioned more than one level deep",
1150                files: vec![
1151                    file(&[("id", DataType::Int64)], 1),
1152                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1153                ],
1154                paths: vec!["d/y=2024/m=02/a.parquet", "d/y=2024/m=03/b.parquet"],
1155                expected: vec![
1156                    "fee is in 1 of 2 files, only y=2024/m=03; absent from the rest, \
1157                     not null",
1158                ],
1159                ..Shape::default()
1160            },
1161            Shape {
1162                what: "a column in a file that sits under no partition at all",
1163                files: vec![
1164                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1165                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1166                    file(&[("id", DataType::Int64)], 1),
1167                ],
1168                // One file at the root beside the partition directories. Half the files
1169                // that have `fee` are not under `date=2024-01-02`, so there is no
1170                // "only" to be had — and saying it anyway sends a reader to the
1171                // wrong directory, which is worse than the count on its own.
1172                paths: vec![
1173                    "d/aaa.parquet",
1174                    "d/date=2024-01-02/b.parquet",
1175                    "d/date=2024-01-03/c.parquet",
1176                ],
1177                expected: vec!["fee is in 2 of 3 files; absent from the rest, not null"],
1178                ..Shape::default()
1179            },
1180            Shape {
1181                what: "a column whose files are all in the one partition anyway",
1182                files: vec![
1183                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1184                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1185                    file(&[("id", DataType::Int64)], 1),
1186                ],
1187                // Every file is under `date=2024-03-02`, the one without it included.
1188                // "only" earns its place by contrast with the files that are not,
1189                // and there are none: the phrase would say nothing while sounding as
1190                // though the missing file were somewhere else.
1191                paths: vec![
1192                    "d/date=2024-03-02/a.parquet",
1193                    "d/date=2024-03-02/b.parquet",
1194                    "d/date=2024-03-02/c.parquet",
1195                ],
1196                expected: vec!["fee is in 2 of 3 files; absent from the rest, not null"],
1197                ..Shape::default()
1198            },
1199            Shape {
1200                what: "a column of a dataset with a footer that would not parse",
1201                files: vec![
1202                    file(&[("id", DataType::Int64)], 1),
1203                    None,
1204                    file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1205                ],
1206                // A file whose footer would not read counts as missing nothing, which
1207                // makes it look like a file that has the column. Anchoring on it would
1208                // put a partition datui never opened into a sentence whose own scope
1209                // says it read two files.
1210                paths: vec!["d/x=1/a.parquet", "d/x=2/b.parquet", "d/x=3/c.parquet"],
1211                expected: vec![
1212                    "fee is in 1 of the 2 files that could be read; absent from the \
1213                     rest, not null",
1214                    "1 file unreadable, left out",
1215                ],
1216                ..Shape::default()
1217            },
1218            Shape {
1219                what: "a column of a dataset whose footers were sampled",
1220                files: vec![
1221                    file(&[("id", DataType::Int64)], 1),
1222                    file(&[("id", DataType::Int64), ("oops", DataType::String)], 1),
1223                ],
1224                // A file whose footer was not read looks like a file missing nothing,
1225                // so where a column begins cannot be told from the two that were.
1226                sampled: Some(6541),
1227                paths: vec!["d/date=2024-03-01/a.parquet", "d/date=2024-03-02/b.parquet"],
1228                expected: vec![
1229                    "oops is in 1 of the 2 footers read; absent from the rest, not null",
1230                ],
1231            },
1232            // --- files that hold nothing at all ---
1233            Shape {
1234                what: "one empty file",
1235                files: vec![
1236                    file(&[("id", DataType::Int64)], 0),
1237                    file(&[("id", DataType::Int64)], 5),
1238                ],
1239                sampled: None,
1240                paths: Vec::new(),
1241                expected: vec!["1 file holds no rows"],
1242            },
1243            Shape {
1244                what: "several empty files",
1245                files: vec![
1246                    file(&[("id", DataType::Int64)], 0),
1247                    file(&[("id", DataType::Int64)], 0),
1248                    file(&[("id", DataType::Int64)], 5),
1249                ],
1250                sampled: None,
1251                paths: Vec::new(),
1252                expected: vec!["2 files hold no rows"],
1253            },
1254            Shape {
1255                what: "an empty file among sampled footers",
1256                files: vec![
1257                    file(&[("id", DataType::Int64)], 0),
1258                    file(&[("id", DataType::Int64)], 5),
1259                ],
1260                sampled: Some(900),
1261                paths: Vec::new(),
1262                // "file", not "footer": a footer does not hold rows. The scope line
1263                // is what says datui looked at two of nine hundred, and the note
1264                // claims nothing about the other 898.
1265                expected: vec!["1 file holds no rows"],
1266            },
1267            Shape {
1268                what: "an empty file and a column only the other has",
1269                files: vec![
1270                    file(&[("id", DataType::Int64)], 0),
1271                    file(&[("id", DataType::Int64), ("x", DataType::String)], 5),
1272                ],
1273                sampled: None,
1274                paths: Vec::new(),
1275                expected: vec![
1276                    "x is in 1 of 2 files; absent from the rest, not null",
1277                    "1 file holds no rows",
1278                ],
1279            },
1280            // --- a column that only some files have: the one note with a ratio ---
1281            Shape {
1282                what: "absent",
1283                files: vec![
1284                    file(&[("id", i64.clone())], 1),
1285                    file(&[("id", i64.clone()), ("x", str.clone())], 1),
1286                ],
1287                sampled: None,
1288                paths: Vec::new(),
1289                expected: vec!["x is in 1 of 2 files; absent from the rest, not null"],
1290            },
1291            Shape {
1292                what: "absent, sampled",
1293                files: vec![
1294                    file(&[("id", i64.clone())], 1),
1295                    file(&[("id", i64.clone()), ("x", str.clone())], 1),
1296                ],
1297                sampled: Some(200_000),
1298                paths: Vec::new(),
1299                expected: vec!["x is in 1 of the 2 footers read; absent from the rest, not null"],
1300            },
1301            Shape {
1302                what: "absent, with an unreadable footer",
1303                files: vec![
1304                    file(&[("id", i64.clone())], 1),
1305                    None,
1306                    file(&[("id", i64.clone()), ("x", str.clone())], 1),
1307                ],
1308                sampled: None,
1309                paths: Vec::new(),
1310                expected: vec![
1311                    "x is in 1 of the 2 files that could be read; absent from the rest, not null",
1312                    "1 file unreadable, left out",
1313                ],
1314            },
1315            Shape {
1316                what: "absent, sampled, with an unreadable footer",
1317                files: vec![
1318                    file(&[("id", i64.clone())], 1),
1319                    None,
1320                    file(&[("id", i64.clone()), ("x", str.clone())], 1),
1321                ],
1322                sampled: Some(200_000),
1323                paths: Vec::new(),
1324                expected: vec![
1325                    "x is in 1 of the 2 footers that could be read; absent from the rest, not null",
1326                    "1 footer unreadable, left out",
1327                ],
1328            },
1329            // --- a type the files disagree about: counted, never divided ---
1330            Shape {
1331                what: "conflicting",
1332                files: vec![
1333                    file(&[("n", str.clone())], 10),
1334                    file(&[("n", i64.clone())], 90),
1335                ],
1336                sampled: None,
1337                paths: Vec::new(),
1338                expected: vec!["n is str in 1 file; read as i64 and not read there"],
1339            },
1340            Shape {
1341                what: "conflicting, sampled",
1342                files: vec![
1343                    file(&[("n", str.clone())], 10),
1344                    file(&[("n", i64.clone())], 90),
1345                ],
1346                sampled: Some(200_000),
1347                paths: Vec::new(),
1348                expected: vec!["n is str in 1 footer; read as i64 and not read there"],
1349            },
1350            Shape {
1351                what: "conflicting, with an unreadable footer",
1352                files: vec![
1353                    file(&[("n", str.clone())], 10),
1354                    None,
1355                    file(&[("n", i64.clone())], 90),
1356                ],
1357                sampled: None,
1358                paths: Vec::new(),
1359                expected: vec![
1360                    "n is str in 1 file; read as i64 and not read there",
1361                    "1 file unreadable, left out",
1362                ],
1363            },
1364            Shape {
1365                what: "conflicting, sampled, with an unreadable footer",
1366                files: vec![
1367                    file(&[("n", str.clone())], 10),
1368                    None,
1369                    file(&[("n", i64.clone())], 90),
1370                ],
1371                sampled: Some(200_000),
1372                paths: Vec::new(),
1373                expected: vec![
1374                    "n is str in 1 footer; read as i64 and not read there",
1375                    "1 footer unreadable, left out",
1376                ],
1377            },
1378            // --- widening, which settles without loss ---
1379            Shape {
1380                what: "widened",
1381                files: with_n(i32.clone(), i64.clone()),
1382                sampled: None,
1383                paths: Vec::new(),
1384                expected: vec!["n is stored as more than one type; read as i64"],
1385            },
1386            // --- and the combinations, each saying all of what is true ---
1387            Shape {
1388                what: "absent and conflicting",
1389                files: vec![
1390                    file(&[("n", str.clone())], 10),
1391                    file(&[("n", i64.clone())], 90),
1392                    file(&[("id", i64.clone())], 5),
1393                ],
1394                sampled: None,
1395                paths: Vec::new(),
1396                // Schema order: the newest file's columns lead, so `id` comes first.
1397                expected: vec![
1398                    "id is in 1 of 3 files; absent from the rest, not null",
1399                    "n is in 2 of 3 files; absent from the rest, not null",
1400                    "n is str in 1 file; read as i64 and not read there",
1401                ],
1402            },
1403            Shape {
1404                what: "absent and widened",
1405                files: vec![
1406                    file(&[("id", i64.clone())], 1),
1407                    file(&[("id", i64.clone()), ("n", i32.clone())], 50),
1408                    file(&[("id", i64.clone()), ("n", i64.clone())], 50),
1409                ],
1410                sampled: None,
1411                paths: Vec::new(),
1412                expected: vec![
1413                    "n is in 2 of 3 files; absent from the rest, not null",
1414                    "n is stored as more than one type; read as i64",
1415                ],
1416            },
1417            Shape {
1418                what: "conflicting and widened",
1419                files: vec![
1420                    file(&[("n", i32.clone())], 50),
1421                    file(&[("n", i64.clone())], 50),
1422                    file(&[("n", str.clone())], 5),
1423                ],
1424                sampled: None,
1425                paths: Vec::new(),
1426                expected: vec![
1427                    "n is str in 1 file; read as i64 and not read there",
1428                    "n is stored as more than one type; read as i64",
1429                ],
1430            },
1431            Shape {
1432                what: "a chosen type no file stores",
1433                files: vec![
1434                    file(&[("n", i32.clone())], 50),
1435                    file(&[("n", DataType::Float32)], 50),
1436                    file(&[("n", str.clone())], 5),
1437                ],
1438                sampled: None,
1439                paths: Vec::new(),
1440                expected: vec![
1441                    "n is str in 1 file; read as f64 and not read there",
1442                    "n is stored as more than one type; read as f64",
1443                ],
1444            },
1445            Shape {
1446                what: "two types the table spells the same way",
1447                files: vec![
1448                    file(
1449                        &[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
1450                        10,
1451                    ),
1452                    file(
1453                        &[(
1454                            "t",
1455                            DataType::Datetime(TimeUnit::Nanoseconds, Some(TimeZone::UTC)),
1456                        )],
1457                        90,
1458                    ),
1459                ],
1460                sampled: None,
1461                paths: Vec::new(),
1462                // `datetime in 1 file; read as datetime` would say nothing, so the
1463                // note spells the types out where the short words collide.
1464                expected: vec![
1465                    "t is datetime[ns] in 1 file; read as datetime[ns, UTC] and not read there",
1466                ],
1467            },
1468            Shape {
1469                what: "two conflicting types the table spells the same way",
1470                files: vec![
1471                    file(
1472                        &[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
1473                        10,
1474                    ),
1475                    file(
1476                        &[(
1477                            "t",
1478                            DataType::Datetime(TimeUnit::Nanoseconds, Some(TimeZone::UTC)),
1479                        )],
1480                        10,
1481                    ),
1482                    file(&[("t", str.clone())], 90),
1483                ],
1484                sampled: None,
1485                paths: Vec::new(),
1486                // The two that lost share a word as much as either shares one with the
1487                // winner, so all three are spelled out.
1488                expected: vec![
1489                    "t is datetime[ns] or datetime[ns, UTC] in 2 files; read as str and not read there",
1490                ],
1491            },
1492            Shape {
1493                what: "two structs, which Display also spells the same way",
1494                files: vec![
1495                    file(
1496                        &[(
1497                            "s",
1498                            DataType::Struct(vec![Field::new("a".into(), i64.clone())]),
1499                        )],
1500                        90,
1501                    ),
1502                    file(
1503                        &[(
1504                            "s",
1505                            DataType::Struct(vec![Field::new("a".into(), str.clone())]),
1506                        )],
1507                        10,
1508                    ),
1509                ],
1510                sampled: None,
1511                paths: Vec::new(),
1512                // `struct[1]` for both, so only the fields tell them apart.
1513                expected: vec![
1514                    "s is Struct({'a': String}) in 1 file; \
1515                     read as Struct({'a': Int64}) and not read there",
1516                ],
1517            },
1518            Shape {
1519                what: "a chosen type whose word another type shares",
1520                files: vec![
1521                    file(
1522                        &[("t", DataType::Datetime(TimeUnit::Milliseconds, None))],
1523                        50,
1524                    ),
1525                    file(
1526                        &[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
1527                        50,
1528                    ),
1529                    file(&[("t", str.clone())], 5),
1530                ],
1531                sampled: None,
1532                paths: Vec::new(),
1533                // Both notes name the winner the same way, and the way the Schema tab
1534                // does: "read as datetime" would drop the unit that is the point.
1535                expected: vec![
1536                    "t is str in 1 file; read as datetime[ns] and not read there",
1537                    "t is stored as more than one type; read as datetime[ns]",
1538                ],
1539            },
1540            Shape {
1541                what: "widened between two datetime units",
1542                files: vec![
1543                    file(
1544                        &[("t", DataType::Datetime(TimeUnit::Milliseconds, None))],
1545                        50,
1546                    ),
1547                    file(
1548                        &[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
1549                        50,
1550                    ),
1551                ],
1552                sampled: None,
1553                paths: Vec::new(),
1554                // Which unit won is the whole content of the note, and `datetime`
1555                // alone would not carry it.
1556                expected: vec!["t is stored as more than one type; read as datetime[ns]"],
1557            },
1558            Shape {
1559                what: "a struct that both widens and conflicts",
1560                files: vec![
1561                    file(
1562                        &[(
1563                            "s",
1564                            DataType::Struct(vec![Field::new("a".into(), i32.clone())]),
1565                        )],
1566                        50,
1567                    ),
1568                    file(
1569                        &[(
1570                            "s",
1571                            DataType::Struct(vec![Field::new("a".into(), i64.clone())]),
1572                        )],
1573                        50,
1574                    ),
1575                    file(
1576                        &[(
1577                            "s",
1578                            DataType::Struct(vec![Field::new("a".into(), str.clone())]),
1579                        )],
1580                        5,
1581                    ),
1582                ],
1583                sampled: None,
1584                paths: Vec::new(),
1585                // Both notes name the winner the same way. Naming it separately let
1586                // one say `Struct({'a': Int64})` and the other `struct[1]`.
1587                expected: vec![
1588                    "s is Struct({'a': String}) in 1 file; \
1589                     read as Struct({'a': Int64}) and not read there",
1590                    "s is stored as more than one type; read as Struct({'a': Int64})",
1591                ],
1592            },
1593            Shape {
1594                what: "a decimal, whose precision the header word drops",
1595                files: vec![
1596                    file(&[("d", DataType::Decimal(38, 2))], 90),
1597                    file(&[("d", str.clone())], 10),
1598                ],
1599                sampled: None,
1600                paths: Vec::new(),
1601                expected: vec!["d is str in 1 file; read as decimal[38,2] and not read there"],
1602            },
1603            Shape {
1604                what: "absent, conflicting and widened",
1605                files: vec![
1606                    file(&[("id", i64.clone())], 1),
1607                    file(&[("id", i64.clone()), ("n", i32.clone())], 50),
1608                    file(&[("id", i64.clone()), ("n", i64.clone())], 50),
1609                    file(&[("id", i64.clone()), ("n", str.clone())], 5),
1610                ],
1611                sampled: None,
1612                paths: Vec::new(),
1613                expected: vec![
1614                    "n is in 3 of 4 files; absent from the rest, not null",
1615                    "n is str in 1 file; read as i64 and not read there",
1616                    "n is stored as more than one type; read as i64",
1617                ],
1618            },
1619        ];
1620
1621        for shape in &cases {
1622            assert_eq!(notes_for(shape), shape.expected, "{}", shape.what);
1623        }
1624
1625        // A column some file holds in another type always draws a note of its own.
1626        //
1627        // The table's view notes lean on this: `DataTableState::has_notes` answers
1628        // whether the Notes tab is on offer from the dataset's notes alone, which is
1629        // only sound if a note about what a sort leaves out can never be the only one
1630        // there. Every shape above that has a conflicting column is a case of it.
1631        for shape in &cases {
1632            let dataset = dataset_for(shape);
1633            let conflicting: Vec<&str> = dataset
1634                .columns
1635                .iter()
1636                .filter(|column| column.conflicting_files > 0)
1637                .map(|column| column.name.as_str())
1638                .collect();
1639            for name in conflicting {
1640                // The conflict note itself, not merely some note naming the column:
1641                // an absence or widening note about the same column would satisfy a
1642                // looser test while the one that matters had been deleted.
1643                assert!(
1644                    from_dataset(&dataset).iter().any(|note| {
1645                        note.summary.starts_with(&format!("{name} is "))
1646                            && note.summary.ends_with("and not read there")
1647                    }),
1648                    "{}: {name} conflicts, so it says so on its own account",
1649                    shape.what
1650                );
1651            }
1652        }
1653    }
1654
1655    /// The last resort, when a type's own `Debug` does not tell it from another's.
1656    ///
1657    /// Polars prints every `Enum` as `Enum([...])` and every global `Categorical` as
1658    /// `Categorical`, whatever their categories, so two of either collide through the
1659    /// short word, through `Display` and through `Debug` alike. Those are awkward to
1660    /// build here, so this drives the same branch with names that collide outright.
1661    /// Numbering says less than naming would, but it is two things rather than one
1662    /// thing said twice.
1663    #[test]
1664    fn types_that_print_alike_all_the_way_down_are_numbered() {
1665        let names = distinct_names(&[&DataType::Int64, &DataType::Int64]);
1666        assert_eq!(names, ["Int64 #1", "Int64 #2"]);
1667
1668        // A name that never collided keeps it.
1669        let mixed = distinct_names(&[&DataType::Int64, &DataType::Int64, &DataType::String]);
1670        // Once any pair collides every name comes from `Debug`, so `str` is `String`
1671        // here; only the colliding pair carries a number.
1672        assert_eq!(mixed, ["Int64 #1", "Int64 #2", "String"]);
1673    }
1674
1675    #[test]
1676    fn a_uniform_dataset_has_nothing_to_say() {
1677        let shape = Shape {
1678            what: "uniform",
1679            files: vec![
1680                file(&[("id", DataType::Int64)], 1),
1681                file(&[("id", DataType::Int64)], 1),
1682            ],
1683            sampled: None,
1684            paths: Vec::new(),
1685            expected: vec![],
1686        };
1687        assert!(notes_for(&shape).is_empty());
1688    }
1689
1690    /// The summary and the scope line sit one above the other, so a reader takes them
1691    /// together. Nothing in a summary may claim more than the scope allows.
1692    #[test]
1693    fn no_summary_claims_more_than_its_scope() {
1694        let files = [
1695            file(&[("id", DataType::Int64)], 1),
1696            None,
1697            file(&[("id", DataType::Int64), ("x", DataType::String)], 1),
1698        ];
1699        for (origin, expected_scope) in [
1700            (SchemaOrigin::AllFooters(3), "in all 3 footers"),
1701            (
1702                SchemaOrigin::FooterSample {
1703                    read: 3,
1704                    total: 200_000,
1705                },
1706                "in 3 of 200,000 footers (sample)",
1707            ),
1708        ] {
1709            let dataset = union_file_schemas(&files, origin);
1710            let notes = from_dataset(&dataset);
1711            assert_eq!(notes.len(), 2, "an absence note and an unreadable one");
1712            for note in notes {
1713                assert_eq!(note.scope, expected_scope);
1714                // The only ratio in the module is over what could be read, and it never
1715                // claims the whole dataset.
1716                assert!(
1717                    !note.summary.contains("of 3 files"),
1718                    "one of those three said nothing: {}",
1719                    note.summary
1720                );
1721            }
1722        }
1723    }
1724
1725    /// Two notes about one column must each say which files they mean, rather than
1726    /// both saying "the others" about different sets.
1727    #[test]
1728    fn the_notes_about_one_column_do_not_talk_past_each_other() {
1729        let files = [
1730            file(&[("n", DataType::String)], 10),
1731            file(&[("n", DataType::Int64)], 90),
1732            file(&[("id", DataType::Int64)], 5),
1733        ];
1734        let dataset = union_file_schemas(&files, SchemaOrigin::AllFooters(3));
1735        let about_n: Vec<String> = from_dataset(&dataset)
1736            .into_iter()
1737            .map(|note| note.summary)
1738            .filter(|summary| summary.starts_with('n'))
1739            .collect();
1740        assert_eq!(
1741            about_n,
1742            [
1743                "n is in 2 of 3 files; absent from the rest, not null",
1744                "n is str in 1 file; read as i64 and not read there",
1745            ]
1746        );
1747    }
1748}