datui_lib/notes.rs
1//! What datui noticed about a dataset while doing what it was already doing.
2//!
3//! A note never costs a request or a scan of its own: every one here is read off the
4//! footers the schema and the row count already needed. Notes are never alarming — no
5//! pop-up, no error styling — and every one says what it is based on, so "in 1 of 3
6//! files" is never mistaken for a claim about files datui has not looked at.
7//!
8//! A note is one sentence and the line it rests on. Review after review found false
9//! or empty statements here, and every one of them was in prose that went beyond the
10//! sentence: a denominator over the wrong population, an explanatory line that
11//! contradicted the note above it, a type named by a word two types share. What is
12//! left is what can be checked: one claim per note, and one function deciding the only
13//! ratio any of them states.
14
15use crate::numfmt::group_chrome;
16use crate::schema_union::{ColumnDrift, ColumnRange, DatasetSchema, SchemaOrigin, SkippedFiles};
17use polars::prelude::{DataType, PlSmallStr};
18
19/// One thing datui noticed.
20#[derive(Debug, Clone, PartialEq, Eq)]
21pub struct Note {
22 /// The one line shown in the panel.
23 pub summary: String,
24 /// What the note is based on, so its reach is never overstated.
25 pub scope: String,
26 /// The column datui can offer to read as text, when this note is about one and
27 /// reading it that way would work. `None` for every other note.
28 ///
29 /// The name rather than the prose, because the panel has to act on it: matching a
30 /// column out of a sentence is the kind of thing this module exists to avoid.
31 pub read_as_text: Option<PlSmallStr>,
32 /// How many files this note already counts as passed over by the read, when it is
33 /// the one that says a mixed directory was read as its commonest format. `None`
34 /// for every other note.
35 ///
36 /// The number rather than the prose, for the same reason as `read_as_text`:
37 /// [`merged`] has to take these files back out of the footer pass's tally, and
38 /// reading a count out of a sentence is the kind of thing this module exists to
39 /// avoid.
40 pub passed_over: Option<usize>,
41}
42
43/// Whether a dataset's schema came from a sample of its files.
44fn sampled(dataset: &DatasetSchema) -> bool {
45 matches!(dataset.origin, SchemaOrigin::FooterSample { .. })
46}
47
48/// How many, in the noun that is true of what datui looked at: files when it read
49/// every one, footers when it read only a sample of them.
50fn how_many(dataset: &DatasetSchema, n: usize) -> String {
51 let noun = match (sampled(dataset), n) {
52 (false, 1) => "file",
53 (false, _) => "files",
54 (true, 1) => "footer",
55 (true, _) => "footers",
56 };
57 format!("{} {noun}", group_chrome(n))
58}
59
60/// What the dataset's one ratio is out of: the files datui read and could parse.
61///
62/// This is the only denominator in the module. Where datui read every file and all of
63/// them parsed they are simply "files"; where it sampled, or where a footer would not
64/// parse, saying "files" would claim more than it looked at.
65fn out_of(dataset: &DatasetSchema) -> String {
66 let readable = dataset.files.saturating_sub(dataset.unreadable.len());
67 match (sampled(dataset), dataset.unreadable.is_empty()) {
68 (false, true) => how_many(dataset, readable),
69 (false, false) => format!("the {} that could be read", how_many(dataset, readable)),
70 (true, true) => format!("the {} read", how_many(dataset, readable)),
71 (true, false) => format!("the {} that could be read", how_many(dataset, readable)),
72 }
73}
74
75/// Names for a set of types that tell them apart.
76///
77/// Several types print alike at any given level of detail: every `Struct` is
78/// `struct[1]` however its fields differ, and every `Enum` is `Enum([...])` however its
79/// categories do. A note reading "s is struct[1] in 1 file; read as struct[1]" says
80/// nothing, so the names escalate until they tell the types apart. Polars' `Display`
81/// collapses too — a struct is `struct[1]` — so the full form is its `Debug`.
82fn distinct_names(types: &[&DataType]) -> Vec<String> {
83 let collides = |names: &[String]| {
84 names
85 .iter()
86 .enumerate()
87 .any(|(i, name)| names[i + 1..].contains(name))
88 };
89 // `Display` is how the Schema tab names a type, and it carries the unit, the
90 // precision and the time zone that the table header's one word drops.
91 let shown: Vec<String> = types.iter().map(|t| format!("{t}")).collect();
92 if !collides(&shown) {
93 return shown;
94 }
95 // Two structs are both `struct[1]`, and only the fields tell them apart.
96 let spelled: Vec<String> = types.iter().map(|t| format!("{t:?}")).collect();
97 if !collides(&spelled) {
98 return spelled;
99 }
100 // Polars prints every `Enum` as `Enum([...])` and every global `Categorical` as
101 // `Categorical`, whatever their categories, so even the full form can collide. Numbering them says less than naming them, but "the first
102 // and the second" is at least two things rather than one said twice. Only the ones
103 // that actually collide are numbered; a name that was already unique keeps it.
104 let mut seen: Vec<&String> = Vec::new();
105 spelled
106 .iter()
107 .map(|name| {
108 if spelled.iter().filter(|other| *other == name).count() > 1 {
109 seen.push(name);
110 format!("{name} #{}", seen.iter().filter(|s| **s == name).count())
111 } else {
112 name.clone()
113 }
114 })
115 .collect()
116}
117
118/// What the footers said, as notes. Empty when every file agrees, which is the common
119/// case and the one where there is nothing to say.
120pub fn from_dataset(dataset: &DatasetSchema) -> Vec<Note> {
121 let scope = format!("in {}", dataset.origin);
122 let readable = dataset.files.saturating_sub(dataset.unreadable.len());
123 let denominator = out_of(dataset);
124 let mut notes = Vec::new();
125
126 for column in dataset.drifting() {
127 // The type the column is read as is named once, so two notes about the same
128 // column cannot name it two different ways: whatever it takes to tell the
129 // conflicting types apart is what the widening note calls it too.
130 let mut types: Vec<&DataType> = vec![&column.dtype];
131 types.extend(column.conflicting_types.iter());
132 let names = distinct_names(&types);
133 let (chosen, others) = names.split_first().expect("the chosen type is first");
134
135 // A column can be missing from some files, stored differently in others, and
136 // stored in a narrower type among the rest. Each is true on its own, so each
137 // is said on its own: nothing here suppresses anything else.
138 notes.extend(absence_note(
139 column,
140 readable,
141 &denominator,
142 dataset.column_ranges.get(&column.name),
143 &scope,
144 ));
145 notes.extend(conflict_note(column, dataset, chosen, others, &scope));
146 notes.extend(widening_note(column, chosen, &scope));
147 }
148
149 for column in &dataset.read_as_text {
150 notes.push(text_note(column, &scope));
151 }
152
153 notes.extend(empty_files_note(dataset, &scope));
154 notes.extend(row_group_note(dataset, &scope));
155 notes.extend(small_files_note(dataset, &scope));
156 notes.extend(partition_layout_note(dataset));
157 notes.extend(skipped_files_note(dataset));
158
159 if !dataset.unreadable.is_empty() {
160 notes.push(Note {
161 summary: format!(
162 "{} unreadable, left out",
163 how_many(dataset, dataset.unreadable.len())
164 ),
165 scope: scope.clone(),
166 read_as_text: None,
167 passed_over: None,
168 });
169 }
170
171 notes
172}
173
174/// Rows a filter or sort leaves out, because the column it names is not read from the
175/// files those rows came from.
176///
177/// The only note here about the view rather than the dataset, and the only one datui
178/// writes in answer to something the user just did.
179///
180/// It is phrased about the files, not about the view, and that is the whole care of
181/// it. "2 rows are left out" reads as a claim that the view is two rows shorter, which
182/// is not true when a filter had already dropped one of them; how many rows are in the
183/// files that hold the column in another type is a fact of the footers, true whatever
184/// else the view is doing. Both counts here are of that kind.
185///
186/// The cost of saying it that way is that the note also appears where those rows had
187/// already gone — a filter that excluded them, then a sort on the column. It is still
188/// true there, and the alternative is a count of what the view actually lost, which
189/// cannot be had without collecting the frame twice.
190pub fn left_out_note(
191 column: &ColumnDrift,
192 dataset: &DatasetSchema,
193 rows: usize,
194 filtered: bool,
195 sorted: bool,
196) -> Note {
197 let what = match (filtered, sorted) {
198 (true, true) => "filter and sort",
199 (true, false) => "filter",
200 // Called only for a column the view names, so it names it one way or the other.
201 _ => "sort",
202 };
203 let there = if rows == 1 {
204 "1 row".to_string()
205 } else {
206 format!("{} rows", group_chrome(rows))
207 };
208 Note {
209 summary: format!(
210 "{}: {there} in {} left out of the {what}",
211 column.name,
212 how_many(dataset, column.conflicting_files)
213 ),
214 scope: format!("in {}", dataset.origin),
215 read_as_text: None,
216 passed_over: None,
217 }
218}
219
220/// Files that hold no rows at all.
221///
222/// A partition written for a day nothing happened, or a job that produced a header and no
223/// data. Worth saying because the dataset then has fewer days of data than it has
224/// directories, and a reader counting directories would get the wrong answer — but it is
225/// not a fault, and a pipeline that writes a file per day will have some.
226///
227/// Counts without dividing: how many of the files datui read hold nothing is a fact
228/// about those files, and the scope line says which files those were.
229///
230/// The one counting note that says "file" even where the schema came from a sample,
231/// rather than the "footer" [`how_many`] would give it. A footer does not hold rows —
232/// it records how many the file holds — so the substitution that keeps the other notes
233/// honest makes this one a category slip. It costs nothing here: there is no ratio to
234/// overstate, and the scope line already says only a sample was read, so "1 file holds
235/// no rows · in 20,000 of 500,000 footers (sample)" claims nothing about the other
236/// 480,000.
237fn empty_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
238 let empty = dataset.empty_files;
239 if empty == 0 {
240 return None;
241 }
242 let (count, verb) = if empty == 1 {
243 ("1 file".to_string(), "holds")
244 } else {
245 (format!("{} files", group_chrome(empty)), "hold")
246 };
247 Some(Note {
248 summary: format!("{count} {verb} no rows"),
249 scope: scope.to_string(),
250 read_as_text: None,
251 passed_over: None,
252 })
253}
254
255/// Row groups big enough that reading a page means reading a lot more than the page.
256///
257/// A row group is what a reader fetches: a hundred rows anywhere inside one costs the
258/// whole of it. Over a network that is the difference between a page arriving and a
259/// page arriving after sixty-four megabytes do, and there is nothing the user can do
260/// about it from here — which is exactly why it is worth saying rather than leaving
261/// them to wonder why scrolling is slow.
262///
263/// The threshold is the size at which one row group is a noticeable download on an
264/// ordinary connection; below it, nobody needs telling. States the middle size rather
265/// than the largest: one row group of a gigabyte among thousands of small ones is a
266/// different dataset from one where every row group is a gigabyte, and only the second
267/// is worth a note.
268///
269/// Says "the middle row group is", not "row groups are, apiece" — the middle of
270/// `[1 MiB, 100 MiB, 100 MiB]` is 100 MiB and one of those row groups is not.
271///
272/// And says the size, not what a page costs to fetch. They are not the same number:
273/// a page projects away binary columns (see `binary_stub_exprs`), so their chunks are
274/// never downloaded, while this size counts every chunk in the group. The figure is
275/// the row group's; what follows it is why a row group's size is the one that matters.
276///
277/// The middle is over every row group of every footer read, each counting once. Not
278/// weighted by rows, though a page is likelier to land in a group that holds more of
279/// them: a dataset of one file of ten thousand small groups beside a hundred files of
280/// one huge group each is called small by this and would be called large by that.
281/// Counting groups is the statistic that matches the sentence — how big a row group
282/// is, of the row groups there are.
283fn row_group_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
284 /// Sixty-four mebibytes, the size a page in that range costs to reach.
285 const BIG: usize = 64 * 1024 * 1024;
286 let median = dataset.median_row_group_bytes?;
287 if median <= BIG {
288 return None;
289 }
290 Some(Note {
291 summary: format!(
292 "median row group {}, each read whole",
293 crate::widgets::info::format_bytes(median as u64)
294 ),
295 scope: scope.to_string(),
296 read_as_text: None,
297 passed_over: None,
298 })
299}
300
301/// A dataset of very many files, each holding very little.
302///
303/// Says what happened, not what it cost. The first draft of this note said finding the
304/// files costs more than reading them, and review measured it: a thousand two hundred
305/// files took nine milliseconds to find and eight hundred to read. It cannot be true
306/// over a network either — a footer read is a *suffix* of the file, so it can never
307/// move more bytes than reading the file does. What is true, and is the thing the user
308/// waited for, is that a footer was read for every one of these files before a single
309/// row was.
310///
311/// Both halves have to hold. Small files on their own are a normal day's partitions,
312/// and a few large ones cost nothing to open. It is very many *and* very small that
313/// makes the opening a job of its own — and one nothing here can fix, since the remedy
314/// is upstream in whatever writes them.
315///
316/// The count is every file the listing found; the middle size is over the footers
317/// datui opened, which the sentence names. The two are different populations where the
318/// dataset was too large to open every footer, and saying both numbers is what keeps
319/// the middle from reading as a fact about all of them.
320fn small_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
321 /// Past this many files the footer pass is a job of its own. Above a year of
322 /// hourly partitions, which is an ordinary shape and not a complaint.
323 const MANY: usize = 10_000;
324 /// Below this a file is small by any warehouse's standard, where the figure aimed
325 /// at is hundreds of megabytes.
326 const SMALL: usize = 1024 * 1024;
327 let files = dataset.origin.total_files();
328 let median = dataset.median_file_bytes?;
329 // A file of no bytes is not a Parquet file, so a middle of zero means datui does
330 // not know the sizes rather than that they are small — and "the middle one is 0 B"
331 // would be a claim about a dataset that cannot exist.
332 if median == 0 || files <= MANY || median >= SMALL {
333 return None;
334 }
335 let read = dataset.files;
336 // "opened for its footer", not "a footer was read": a footer that would not parse
337 // was still opened for, and the note beside this one says three of them were.
338 //
339 // And a semicolon, not "so": the footer pass is one per file whatever the files
340 // hold, so only the count leads to it. Joining the two with "so" would make the
341 // size look like half the reason.
342 let footers = if read == files {
343 "every footer".to_string()
344 } else {
345 format!("{} footers", group_chrome(read))
346 };
347 Some(Note {
348 summary: format!(
349 "{} files, median {}; {footers} read before any row",
350 group_chrome(files),
351 crate::widgets::info::format_bytes(median as u64)
352 ),
353 scope: scope.to_string(),
354 read_as_text: None,
355 passed_over: None,
356 })
357}
358
359/// Directories that do not all partition by the same keys.
360///
361/// Says the shape and stops there. What it *costs* is not something this note can see.
362/// The scan reads its partition columns off one branch of the tree, and which branch
363/// that is comes back from the filesystem in whatever order it likes. Usually every
364/// file under the other key then fails and the dataset does not open at all.
365///
366/// But not always, and what decides it is not the branch — it is which file name sorts
367/// first. The scan hands Polars its paths sorted, and Polars takes the hive schema from
368/// the first of them: put one unpartitioned file at the root and whether it sorts above
369/// `date=` decides whether the dataset opens with the column null or fails to open. A
370/// file called `data.parquet` does; one called `loose.parquet` does not. Nothing a note
371/// can see, and about as good a reason as there could be for a note not to say what
372/// something costs.
373///
374/// Two rounds were spent on sentences that picked one of those and stated it as the
375/// consequence. The user guide has room to set them out; a note has one sentence, and
376/// the sentence true of every such dataset is the shape itself.
377///
378/// Read off the names of every file, which is the one thing the listing knows that
379/// reading a file cannot tell you — so it has a scope line of its own, and on a dataset
380/// too large to open every footer this note still saw all of it.
381fn partition_layout_note(dataset: &DatasetSchema) -> Option<Note> {
382 /// Layouts named before the rest are counted rather than spelled. A note is one
383 /// sentence, and a dataset with a hundred layouts would otherwise make it a page.
384 const NAMED: usize = 2;
385 if dataset.partition_layouts.len() < 2 {
386 return None;
387 }
388 let (named, rest) = dataset
389 .partition_layouts
390 .split_at(dataset.partition_layouts.len().min(NAMED));
391 let mut clauses: Vec<String> = named
392 .iter()
393 .map(|(keys, files)| format!("{} by {}", how_many_files(*files), keys.join("/")))
394 .collect();
395 let (dropped_ways, dropped_files) = dataset.partition_layouts_dropped;
396 let ways = rest.len() + dropped_ways;
397 let files: usize = rest.iter().map(|(_, files)| files).sum::<usize>() + dropped_files;
398 if ways > 0 {
399 clauses.push(format!(
400 "{} by {} other {}",
401 how_many_files(files),
402 group_chrome(ways),
403 if ways == 1 { "way" } else { "ways" }
404 ));
405 }
406 Some(Note {
407 summary: format!("mixed partition keys: {}", clauses.join(", ")),
408 scope: format!(
409 "in the names of {} files",
410 group_chrome(dataset.listed_files)
411 ),
412 read_as_text: None,
413 passed_over: None,
414 })
415}
416
417/// `n files`, or `1 file`. Plain files, because the caller counted every one of them.
418fn how_many_files(n: usize) -> String {
419 format!(
420 "{} {}",
421 group_chrome(n),
422 if n == 1 { "file" } else { "files" }
423 )
424}
425
426/// A column being read as text from every file, because it was asked for that way.
427///
428/// Stands in for the conflict note it replaced, and says the one thing that changes
429/// about the column beyond what is now visible in it: a filter or sort on it compares
430/// text. `n > 5` written for a number keeps `"sixty"` and drops `"10"`, and a view
431/// that quietly did that with nothing on screen to say so would be a view the user
432/// reads wrongly.
433fn text_note(column: &PlSmallStr, scope: &str) -> Note {
434 Note {
435 summary: format!("{column} read as text: filter and sort compare text"),
436 scope: scope.to_string(),
437 read_as_text: None,
438 passed_over: None,
439 }
440}
441
442/// Files in the directory that are not Parquet, and so are not in the table.
443///
444/// Only the ones somebody might have meant as data. A writer leaves `_SUCCESS`, `.crc`
445/// and `_metadata` beside what it wrote, and saying so on every directory a job produced
446/// would put an accent on the Info key for the most ordinary thing a directory can
447/// contain — the same reason the small-files note waits for a threshold rather than
448/// firing on every directory with more than one file in it. Where the note does fire it
449/// counts them beside what it is about, so the numbers are the directory's rather than a
450/// selection from it — as far as the listing saw, which is not the same as all of it:
451/// a subtree it could not read, or one below the depth it stops at, is in neither.
452fn skipped_files_note(dataset: &DatasetSchema) -> Option<Note> {
453 note_about_skipped(dataset.skipped)
454}
455
456/// The note [`skipped_files_note`] writes, from the tally alone.
457///
458/// Split out so [`merged`] can rebuild it with the files the open already reported
459/// taken back out, without reading a count out of the note's own sentence.
460fn note_about_skipped(skipped: SkippedFiles) -> Option<Note> {
461 let SkippedFiles {
462 bookkeeping,
463 not_parquet,
464 empty,
465 } = skipped;
466 if not_parquet == 0 && empty == 0 {
467 return None;
468 }
469 let files = |n: usize| {
470 if n == 1 {
471 "1 file".to_string()
472 } else {
473 format!("{} files", group_chrome(n))
474 }
475 };
476 // An object with nothing in it is a write that stopped, and saying so is the point;
477 // the rest is counted beside it so the total is the directory's, not a selection.
478 let mut said = Vec::new();
479 if empty > 0 {
480 let what = if empty == 1 { "file" } else { "files" };
481 said.push(format!("{} empty {what}", group_chrome(empty)));
482 }
483 if not_parquet > 0 {
484 said.push(format!("{} not Parquet", files(not_parquet)));
485 }
486 if bookkeeping > 0 {
487 let what = if bookkeeping == 1 { "file" } else { "files" };
488 said.push(format!(
489 "{} writer bookkeeping {what}",
490 group_chrome(bookkeeping)
491 ));
492 }
493 Some(Note {
494 summary: format!("skipped: {}", said.join(", ")),
495 scope: "in this directory's listing".to_string(),
496 read_as_text: None,
497 passed_over: None,
498 })
499}
500
501/// The most names a note lists before it cuts the rest with an ellipsis.
502const NAMES_SHOWN: usize = 3;
503
504/// `names`, the first few of them, joined, and an ellipsis for the rest.
505pub fn some_names<S: AsRef<str>>(names: &[S]) -> String {
506 let mut said: Vec<&str> = names.iter().take(NAMES_SHOWN).map(AsRef::as_ref).collect();
507 let ellipsis = crate::glyphs::get().ellipsis;
508 if names.len() > NAMES_SHOWN {
509 said.push(ellipsis);
510 }
511 said.join(", ")
512}
513
514/// The files a read of several passed over because they hold no header: empty, blank,
515/// or nothing but NUL padding.
516pub fn no_header(files: &[&std::path::Path]) -> Option<Note> {
517 if files.is_empty() {
518 return None;
519 }
520 let names: Vec<String> = files
521 .iter()
522 .map(|f| {
523 f.file_name().map_or_else(
524 || f.display().to_string(),
525 |n| n.to_string_lossy().into_owned(),
526 )
527 })
528 .collect();
529 let what = if files.len() == 1 { "file" } else { "files" };
530 Some(Note {
531 summary: format!(
532 "{} {what} with no header skipped: {}",
533 files.len(),
534 some_names(&names)
535 ),
536 scope: "empty, blank, or only NUL padding".to_string(),
537 read_as_text: None,
538 passed_over: None,
539 })
540}
541
542/// What the open itself has to say, before a footer has been read.
543///
544/// Two facts, both decided by the route that opened the directory rather than by anything
545/// in the data: which of the directory's data files this read passed over, and whether
546/// the directory is a lake table being read as its plain files. Neither is a defect in
547/// the data — they are what datui chose to do, and #275's rule is that datui never
548/// refuses a read the user asked for and always says what it did instead.
549pub fn from_the_open(
550 left_out: &[(crate::FileFormat, usize)],
551 lake: Option<&str>,
552 files_differ: crate::schema_union::Disagreement,
553 names_look_like_data: bool,
554) -> Vec<Note> {
555 let mut notes = Vec::new();
556 // What the read had to do to stack them, in the words of what it actually found.
557 // These formats carry no footer, so there is no per-column tally behind either
558 // sentence; a Parquet dataset gets the exact version instead — which columns, in
559 // how many files, and where — from footers it had to read anyway.
560 //
561 // The scope says a spread, because that is what was looked at: three files, the
562 // ends and the middle, whatever the directory's size.
563 let scope = || "in a spread of this directory's files".to_string();
564 if files_differ.columns {
565 notes.push(Note {
566 summary: "columns differ across files: a missing column reads null".to_string(),
567 scope: scope(),
568 read_as_text: None,
569 passed_over: None,
570 });
571 }
572 if files_differ.headerless {
573 notes.push(Note {
574 summary: concat!(
575 "no header row? first row read as names: ",
576 "H on Schema, or --no-header, reads it as data"
577 )
578 .to_string(),
579 scope: scope(),
580 read_as_text: None,
581 passed_over: None,
582 });
583 }
584 // The same shape in one file: every column name a number, which a header almost
585 // never is and a first row of data often is.
586 if names_look_like_data && !files_differ.headerless {
587 notes.push(Note {
588 summary: "column names look like data: H on Schema reads them as a row".to_string(),
589 scope: "from the column names".to_string(),
590 read_as_text: None,
591 passed_over: None,
592 });
593 }
594 if files_differ.types {
595 notes.push(Note {
596 summary: "a column's type differs across files: read as the wider type".to_string(),
597 scope: scope(),
598 read_as_text: None,
599 passed_over: None,
600 });
601 }
602 if let Some(format) = lake {
603 notes.push(Note {
604 // The strongest sentence the panel has, because it is the one place a
605 // number on screen is not a number about the table. A delete leaves its
606 // rows on disk, an update leaves the version it replaced, and compaction
607 // leaves both sides — all of them counted here.
608 summary: format!(
609 "{format} table's files, not the table: deleted rows and old versions counted"
610 ),
611 scope: format!("in this {format} table's directory"),
612 read_as_text: None,
613 passed_over: None,
614 });
615 }
616 if !left_out.is_empty() {
617 let said: Vec<String> = left_out
618 .iter()
619 .map(|(format, n)| format!("{n} {}", format.name()))
620 .collect();
621 notes.push(Note {
622 summary: format!(
623 "mixed formats, read as the commonest: {} not read",
624 said.join(", ")
625 ),
626 scope: "in this directory's listing".to_string(),
627 read_as_text: None,
628 // Carried so [`merged`] can take these files back out of the footer
629 // pass's tally, which walks the same directory and counts them again.
630 passed_over: Some(left_out.iter().map(|(_, n)| n).sum()),
631 });
632 }
633 notes
634}
635
636/// The `cache-*.arrow` files `map()` wrote beside a Hugging Face cache's splits, which
637/// the read left out: their columns are the mapping's, not the split's.
638pub fn map_caches(count: usize) -> Option<Note> {
639 let files = if count == 1 { "file" } else { "files" };
640 (count > 0).then(|| Note {
641 summary: format!("{count} cache {files} written by map() not read"),
642 scope: "in this directory's listing".to_string(),
643 read_as_text: None,
644 passed_over: None,
645 })
646}
647
648/// Every note the panel shows — what the open did, what the footers said, then what
649/// the view leaves out — with the one fact the first two both report said once.
650///
651/// A directory of mixed formats read as its commonest is reported by the open ("read
652/// as the commonest; 1 csv not read"), and then the footer pass walks the same
653/// directory and counts the same files among what it passed ("in the directory, 1
654/// file is not Parquet"). The open's sentence names the formats and says why they are
655/// not in the table, so it is the one kept; the footer note is rebuilt with those
656/// files taken back out, which keeps every skip only the walk can see — a stray in a
657/// partition below the top level, a name no reader claims, an empty object, a
658/// writer's bookkeeping. The two tallies never meet anywhere else: the open's is
659/// settled before a footer is read, and the walk's arrives with the dataset, so the
660/// caller that holds both halves hands them here.
661pub fn merged(
662 open: &[Note],
663 dataset: &[Note],
664 view: &[Note],
665 schema: Option<&DatasetSchema>,
666) -> Vec<Note> {
667 // What the walk said as it stands, and with the open's files taken out. Matched
668 // as a whole note rather than by its prose, and if the dataset does not hold
669 // exactly that note — a shape this module did not write — nothing is touched.
670 let rebuilt: Option<(Note, Option<Note>)> = match (
671 open.iter().find_map(|n| n.passed_over),
672 schema.map(|s| s.skipped),
673 ) {
674 (Some(covered), Some(skipped)) => note_about_skipped(skipped).map(|full| {
675 let remaining = SkippedFiles {
676 // Saturating: the open counts one level of recognized names, the
677 // walk everything it saw, so the walk's tally is never smaller —
678 // but a false "0 files are not Parquet" must stay unwritable.
679 not_parquet: skipped.not_parquet.saturating_sub(covered),
680 ..skipped
681 };
682 (full, note_about_skipped(remaining))
683 }),
684 _ => None,
685 };
686 let mut out: Vec<Note> = open.to_vec();
687 for note in dataset {
688 match &rebuilt {
689 Some((full, reduced)) if note == full => out.extend(reduced.clone()),
690 _ => out.push(note.clone()),
691 }
692 }
693 out.extend(view.iter().cloned());
694 out
695}
696
697/// A column that some files were written without. The one note that states a ratio,
698/// because "some" is only meaningful against a total.
699fn absence_note(
700 column: &ColumnDrift,
701 readable: usize,
702 denominator: &str,
703 range: Option<&ColumnRange>,
704 scope: &str,
705) -> Option<Note> {
706 if column.present_in == 0 || column.present_in >= readable {
707 return None;
708 }
709 // Where, as well as how many. A count says a column is unusual; a partition says
710 // where to look, and for a field a feed started sending it says when.
711 let where_it_is = match range {
712 Some(ColumnRange::Only(partition)) => format!(", only {partition}"),
713 Some(ColumnRange::NoneBefore(partition)) => format!(", none before {partition}"),
714 None => String::new(),
715 };
716 Some(Note {
717 summary: format!(
718 "{} is in {} of {}{}; absent from the rest, not null",
719 column.name,
720 group_chrome(column.present_in),
721 denominator,
722 where_it_is
723 ),
724 scope: scope.to_string(),
725 read_as_text: None,
726 passed_over: None,
727 })
728}
729
730/// A column whose files disagree on its type beyond what widening can settle.
731///
732/// Counts without dividing: how many files hold it in a type that lost is a fact about
733/// those files, and needs no total to be true.
734fn conflict_note(
735 column: &ColumnDrift,
736 dataset: &DatasetSchema,
737 chosen: &str,
738 others: &[String],
739 scope: &str,
740) -> Option<Note> {
741 if column.conflicting_files == 0 {
742 return None;
743 }
744 Some(Note {
745 summary: format!(
746 "{} is {} in {}; read as {} and not read there",
747 column.name,
748 others.join(" or "),
749 how_many(dataset, column.conflicting_files),
750 chosen
751 ),
752 scope: scope.to_string(),
753 // The offer, and only where it would work: a column one file holds as a list
754 // cannot be shown as text at all, and an offer that did nothing would be worse
755 // than none. `lenient_scan` asks the same question again, so the two cannot
756 // disagree about which columns are on offer.
757 read_as_text: column.can_read_as_text().then(|| column.name.clone()),
758 passed_over: None,
759 })
760}
761
762/// A column the files store in more than one type, where the scan reads them all into
763/// one.
764///
765/// Says only that: not "width", since a datetime unit, a struct that gained a field and
766/// a file that never typed the column all land here, and not "without loss", since a
767/// very large integer read as a float, or a millisecond datetime read as nanoseconds
768/// past the year 2262, is not exact.
769fn widening_note(column: &ColumnDrift, chosen: &str, scope: &str) -> Option<Note> {
770 if !column.widened {
771 return None;
772 }
773 Some(Note {
774 summary: format!(
775 "{} is stored as more than one type; read as {chosen}",
776 column.name
777 ),
778 scope: scope.to_string(),
779 read_as_text: None,
780 passed_over: None,
781 })
782}
783
784#[cfg(test)]
785mod tests {
786 use super::*;
787
788 use crate::schema_union::{FileFooter, union_file_schemas};
789 use polars::prelude::{DataType, Field, Schema, TimeUnit, TimeZone};
790 use std::sync::Arc;
791
792 /// rustfmt joins a `\`-continued literal back onto one line with its indentation
793 /// inside, which put eighteen spaces into the middle of two of these sentences.
794 #[test]
795 fn the_notes_from_an_open_have_no_holes_in_them() {
796 let notes = from_the_open(
797 &[(crate::FileFormat::Json, 1)],
798 Some("Delta"),
799 crate::schema_union::Disagreement {
800 columns: true,
801 types: true,
802 headerless: true,
803 },
804 false,
805 );
806 assert!(notes.len() >= 3, "{notes:?}");
807 for note in ¬es {
808 assert!(!note.summary.contains(" "), "{:?}", note.summary);
809 }
810 }
811
812 /// A dataset with nothing but a skip tally, as the footer walk hands one over.
813 fn walked(skipped: SkippedFiles) -> DatasetSchema {
814 union_file_schemas(&[], SchemaOrigin::AllFooters(0)).with_skipped(skipped)
815 }
816
817 fn agree() -> crate::schema_union::Disagreement {
818 crate::schema_union::Disagreement {
819 columns: false,
820 types: false,
821 headerless: false,
822 }
823 }
824
825 /// The open and the footer walk both count what a mixed directory's read passed
826 /// over, and on screen that was the same fact twice, one wording above the other.
827 /// The open's sentence names the formats and says why, so it is the one kept.
828 #[test]
829 fn a_mixed_directory_is_not_reported_twice() {
830 let open = from_the_open(&[(crate::FileFormat::Csv, 1)], None, agree(), false);
831 let dataset = walked(SkippedFiles {
832 bookkeeping: 0,
833 not_parquet: 1,
834 empty: 0,
835 });
836 let said: Vec<String> = merged(&open, &from_dataset(&dataset), &[], Some(&dataset))
837 .into_iter()
838 .map(|n| n.summary)
839 .collect();
840 assert_eq!(
841 said,
842 ["mixed formats, read as the commonest: 1 csv not read"],
843 "one fact, said once, in the open's words"
844 );
845 }
846
847 /// What only the walk saw stays counted: a name no reader claims, an empty
848 /// object and a writer's bookkeeping are not among the formats the open named,
849 /// so taking the open's files out must not take these with them. The view's
850 /// notes still follow, untouched.
851 #[test]
852 fn what_only_the_walk_saw_stays_counted() {
853 let open = from_the_open(&[(crate::FileFormat::Csv, 1)], None, agree(), false);
854 let dataset = walked(SkippedFiles {
855 bookkeeping: 2,
856 not_parquet: 2,
857 empty: 1,
858 });
859 let view = [text_note(&PlSmallStr::from("n"), "in 3 files")];
860 let said: Vec<String> = merged(&open, &from_dataset(&dataset), &view, Some(&dataset))
861 .into_iter()
862 .map(|n| n.summary)
863 .collect();
864 assert_eq!(
865 said,
866 [
867 "mixed formats, read as the commonest: 1 csv not read".to_string(),
868 "skipped: 1 empty file, 1 file not Parquet, 2 writer bookkeeping files".to_string(),
869 "n read as text: filter and sort compare text".to_string(),
870 ],
871 "the walk's own findings and the view's notes survive the merge"
872 );
873 }
874
875 /// A tally the open never spoke to is left exactly as the walk wrote it: a hive
876 /// dataset's strays live below the top level, where the open's one-level look
877 /// never reaches, and its note is the only thing that reports them.
878 #[test]
879 fn a_walk_only_tally_is_left_alone() {
880 let open = from_the_open(&[], Some("Delta"), agree(), false);
881 let dataset = walked(SkippedFiles {
882 bookkeeping: 0,
883 not_parquet: 1,
884 empty: 0,
885 });
886 let said: Vec<String> = merged(&open, &from_dataset(&dataset), &[], Some(&dataset))
887 .into_iter()
888 .map(|n| n.summary)
889 .collect();
890 assert_eq!(said.len(), 2, "{said:?}");
891 assert!(
892 said[1] == "skipped: 1 file not Parquet",
893 "different facts do not merge: {said:?}"
894 );
895 }
896
897 fn file(columns: &[(&str, DataType)], rows: usize) -> Option<FileFooter> {
898 let mut schema = Schema::with_capacity(columns.len());
899 for (name, dtype) in columns {
900 schema.with_column((*name).into(), dtype.clone());
901 }
902 Some(FileFooter {
903 schema: Arc::new(schema),
904 row_group_rows: vec![rows],
905 file_bytes: 0,
906 row_group_bytes: Vec::new(),
907 column_bytes: Vec::new(),
908 })
909 }
910
911 /// One dataset shape, and the notes it should produce.
912 #[derive(Default)]
913 struct Shape {
914 what: &'static str,
915 files: Vec<Option<FileFooter>>,
916 /// `Some(total)` when the footers stand in for a larger dataset.
917 sampled: Option<usize>,
918 /// The file names, where the shape is about how they are laid out rather than
919 /// about what is in them. Empty for a shape that has nothing to say about it.
920 paths: Vec<&'static str>,
921 expected: Vec<&'static str>,
922 }
923
924 fn dataset_for(shape: &Shape) -> DatasetSchema {
925 let origin = match shape.sampled {
926 Some(total) => SchemaOrigin::FooterSample {
927 read: shape.files.len(),
928 total,
929 },
930 None => SchemaOrigin::AllFooters(shape.files.len()),
931 };
932 let dataset = union_file_schemas(&shape.files, origin);
933 if shape.paths.is_empty() {
934 return dataset;
935 }
936 let paths: Vec<String> = shape.paths.iter().map(|p| p.to_string()).collect();
937 dataset.with_partition_layouts("d", &paths)
938 }
939
940 fn notes_for(shape: &Shape) -> Vec<String> {
941 from_dataset(&dataset_for(shape))
942 .into_iter()
943 .map(|n| n.summary)
944 .collect()
945 }
946
947 /// Every shape of disagreement, and every line each one produces.
948 ///
949 /// Round after round of review found a wrong denominator in this module, each in a case
950 /// the previous fix had not considered. The wording layer now states one ratio and
951 /// counts everything else without dividing, and this lists the shapes end to end so
952 /// the next wrong one shows up as a diff. Deliberately a transcript rather than a
953 /// set of properties: the failure mode was plausible-looking prose, and a property
954 /// would have to encode the same reasoning that kept going wrong.
955 #[test]
956 fn every_shape_of_disagreement_reads_the_way_it_should() {
957 let i32 = DataType::Int32;
958 let i64 = DataType::Int64;
959 let str = DataType::String;
960 let with_n = |a: DataType, b: DataType| {
961 vec![
962 file(&[("id", i64.clone()), ("n", a)], 50),
963 file(&[("id", i64.clone()), ("n", b)], 50),
964 ]
965 };
966
967 let cases = vec![
968 // --- directories that disagree about what they are partitioned by ---
969 Shape {
970 what: "one directory under another key",
971 files: vec![
972 file(&[("id", DataType::Int64)], 1),
973 file(&[("id", DataType::Int64)], 1),
974 ],
975 paths: vec!["d/date=1/a.parquet", "d/dt=2/b.parquet"],
976 expected: vec!["mixed partition keys: 1 file by date, 1 file by dt"],
977 ..Shape::default()
978 },
979 Shape {
980 what: "directories that agree",
981 files: vec![
982 file(&[("id", DataType::Int64)], 1),
983 file(&[("id", DataType::Int64)], 1),
984 ],
985 paths: vec!["d/date=1/a.parquet", "d/date=2/b.parquet"],
986 expected: vec![],
987 ..Shape::default()
988 },
989 // --- where a column that is not in every file sits ---
990 Shape {
991 what: "a column only one partition has",
992 files: vec![
993 file(&[("id", DataType::Int64)], 1),
994 file(&[("id", DataType::Int64), ("oops", DataType::String)], 1),
995 file(&[("id", DataType::Int64)], 1),
996 ],
997 paths: vec![
998 "d/date=2024-03-01/a.parquet",
999 "d/date=2024-03-02/b.parquet",
1000 "d/date=2024-03-03/c.parquet",
1001 ],
1002 expected: vec![
1003 "oops is in 1 of 3 files, only date=2024-03-02; absent from \
1004 the rest, not null",
1005 ],
1006 ..Shape::default()
1007 },
1008 Shape {
1009 what: "a column the feed started sending",
1010 files: vec![
1011 file(&[("id", DataType::Int64)], 1),
1012 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1013 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1014 ],
1015 paths: vec![
1016 "d/date=2010-07-17/a.parquet",
1017 "d/date=2010-07-18/b.parquet",
1018 "d/date=2010-07-19/c.parquet",
1019 ],
1020 expected: vec![
1021 "fee is in 2 of 3 files, none before date=2010-07-18; absent from \
1022 the rest, not null",
1023 ],
1024 ..Shape::default()
1025 },
1026 Shape {
1027 what: "a column in some files but no pattern to where",
1028 files: vec![
1029 file(&[("id", DataType::Int64), ("odd", DataType::Int64)], 1),
1030 file(&[("id", DataType::Int64)], 1),
1031 file(&[("id", DataType::Int64), ("odd", DataType::Int64)], 1),
1032 ],
1033 paths: vec![
1034 "d/date=1/a.parquet",
1035 "d/date=2/b.parquet",
1036 "d/date=3/c.parquet",
1037 ],
1038 expected: vec!["odd is in 2 of 3 files; absent from the rest, not null"],
1039 ..Shape::default()
1040 },
1041 Shape {
1042 what: "a column starting where the listing and the reader disagree",
1043 files: vec![
1044 file(&[("id", DataType::Int64)], 1),
1045 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1046 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1047 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1048 ],
1049 // Sorted bytewise, `part=10` comes second. Saying "from part=10 on"
1050 // would tell a reader that parts 2 and 3 are without it, and they are
1051 // not: there is no honest way to say where this one starts.
1052 paths: vec![
1053 "d/part=1/a.parquet",
1054 "d/part=10/b.parquet",
1055 "d/part=2/c.parquet",
1056 "d/part=3/d.parquet",
1057 ],
1058 expected: vec!["fee is in 3 of 4 files; absent from the rest, not null"],
1059 ..Shape::default()
1060 },
1061 Shape {
1062 what: "a column starting at a month spelled two ways",
1063 files: vec![
1064 file(&[("id", DataType::Int64)], 1),
1065 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1066 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1067 ],
1068 // `m=03` is March and so is `m=3` — a backfill beside a job. March
1069 // lacks the column, so "none before m=3" says it starts at a month
1070 // whose directory does not have it.
1071 paths: vec!["d/m=03/a.parquet", "d/m=3/b.parquet", "d/m=4/c.parquet"],
1072 expected: vec!["fee is in 2 of 3 files; absent from the rest, not null"],
1073 ..Shape::default()
1074 },
1075 Shape {
1076 what: "a column of a dataset whose directories name their keys in two orders",
1077 files: vec![
1078 file(&[("id", DataType::Int64)], 1),
1079 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1080 ],
1081 // Hive matches partition columns by name, so these two directories are
1082 // one partition written both ways round. Saying the column begins at the
1083 // second would be saying it begins where the first is.
1084 paths: vec!["d/m=03/y=2024/a.parquet", "d/y=2024/m=03/b.parquet"],
1085 // And nothing else says so: the layouts note compares which keys a
1086 // directory uses, not the order it writes them in, so these two agree.
1087 // This note staying quiet is the only thing between a reader and a
1088 // sentence about a place that is written down twice.
1089 expected: vec!["fee is in 1 of 2 files; absent from the rest, not null"],
1090 ..Shape::default()
1091 },
1092 Shape {
1093 what: "a column two files of one partition have",
1094 files: vec![
1095 file(&[("id", DataType::Int64)], 1),
1096 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1097 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1098 ],
1099 // The ordinary shape: a partition holds more than one file. Counting
1100 // its partition twice would make it look like two, and the note that
1101 // names it would quietly stop appearing.
1102 paths: vec![
1103 "d/date=2024-03-01/a.parquet",
1104 "d/date=2024-03-02/b.parquet",
1105 "d/date=2024-03-02/c.parquet",
1106 ],
1107 expected: vec![
1108 "fee is in 2 of 3 files, only date=2024-03-02; absent from the \
1109 rest, not null",
1110 ],
1111 ..Shape::default()
1112 },
1113 Shape {
1114 what: "a column starting halfway through a partition",
1115 files: vec![
1116 file(&[("id", DataType::Int64)], 1),
1117 file(&[("id", DataType::Int64)], 1),
1118 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1119 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1120 ],
1121 // `date=2024-01-02` holds one file with `fee` and one without, so it is
1122 // not a date the column begins at.
1123 paths: vec![
1124 "d/date=2024-01-01/a.parquet",
1125 "d/date=2024-01-02/b.parquet",
1126 "d/date=2024-01-02/c.parquet",
1127 "d/date=2024-01-03/d.parquet",
1128 ],
1129 expected: vec!["fee is in 2 of 4 files; absent from the rest, not null"],
1130 ..Shape::default()
1131 },
1132 Shape {
1133 what: "a column whose partition holds the file without it",
1134 files: vec![
1135 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1136 file(&[("id", DataType::Int64)], 1),
1137 ],
1138 // The file without it is at `y=2024/m=03`, which is under `y=2024`.
1139 // "only y=2024" would send a reader to a directory that holds it.
1140 paths: vec!["d/y=2024/a.parquet", "d/y=2024/m=03/b.parquet"],
1141 expected: vec![
1142 "fee is in 1 of 2 files; absent from the rest, not null",
1143 // Ragged depth is a disagreement in its own right, and says so.
1144 "mixed partition keys: 1 file by m/y, 1 file by y",
1145 ],
1146 ..Shape::default()
1147 },
1148 Shape {
1149 what: "a column of a dataset partitioned more than one level deep",
1150 files: vec![
1151 file(&[("id", DataType::Int64)], 1),
1152 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1153 ],
1154 paths: vec!["d/y=2024/m=02/a.parquet", "d/y=2024/m=03/b.parquet"],
1155 expected: vec![
1156 "fee is in 1 of 2 files, only y=2024/m=03; absent from the rest, \
1157 not null",
1158 ],
1159 ..Shape::default()
1160 },
1161 Shape {
1162 what: "a column in a file that sits under no partition at all",
1163 files: vec![
1164 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1165 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1166 file(&[("id", DataType::Int64)], 1),
1167 ],
1168 // One file at the root beside the partition directories. Half the files
1169 // that have `fee` are not under `date=2024-01-02`, so there is no
1170 // "only" to be had — and saying it anyway sends a reader to the
1171 // wrong directory, which is worse than the count on its own.
1172 paths: vec![
1173 "d/aaa.parquet",
1174 "d/date=2024-01-02/b.parquet",
1175 "d/date=2024-01-03/c.parquet",
1176 ],
1177 expected: vec!["fee is in 2 of 3 files; absent from the rest, not null"],
1178 ..Shape::default()
1179 },
1180 Shape {
1181 what: "a column whose files are all in the one partition anyway",
1182 files: vec![
1183 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1184 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1185 file(&[("id", DataType::Int64)], 1),
1186 ],
1187 // Every file is under `date=2024-03-02`, the one without it included.
1188 // "only" earns its place by contrast with the files that are not,
1189 // and there are none: the phrase would say nothing while sounding as
1190 // though the missing file were somewhere else.
1191 paths: vec![
1192 "d/date=2024-03-02/a.parquet",
1193 "d/date=2024-03-02/b.parquet",
1194 "d/date=2024-03-02/c.parquet",
1195 ],
1196 expected: vec!["fee is in 2 of 3 files; absent from the rest, not null"],
1197 ..Shape::default()
1198 },
1199 Shape {
1200 what: "a column of a dataset with a footer that would not parse",
1201 files: vec![
1202 file(&[("id", DataType::Int64)], 1),
1203 None,
1204 file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
1205 ],
1206 // A file whose footer would not read counts as missing nothing, which
1207 // makes it look like a file that has the column. Anchoring on it would
1208 // put a partition datui never opened into a sentence whose own scope
1209 // says it read two files.
1210 paths: vec!["d/x=1/a.parquet", "d/x=2/b.parquet", "d/x=3/c.parquet"],
1211 expected: vec![
1212 "fee is in 1 of the 2 files that could be read; absent from the \
1213 rest, not null",
1214 "1 file unreadable, left out",
1215 ],
1216 ..Shape::default()
1217 },
1218 Shape {
1219 what: "a column of a dataset whose footers were sampled",
1220 files: vec![
1221 file(&[("id", DataType::Int64)], 1),
1222 file(&[("id", DataType::Int64), ("oops", DataType::String)], 1),
1223 ],
1224 // A file whose footer was not read looks like a file missing nothing,
1225 // so where a column begins cannot be told from the two that were.
1226 sampled: Some(6541),
1227 paths: vec!["d/date=2024-03-01/a.parquet", "d/date=2024-03-02/b.parquet"],
1228 expected: vec![
1229 "oops is in 1 of the 2 footers read; absent from the rest, not null",
1230 ],
1231 },
1232 // --- files that hold nothing at all ---
1233 Shape {
1234 what: "one empty file",
1235 files: vec![
1236 file(&[("id", DataType::Int64)], 0),
1237 file(&[("id", DataType::Int64)], 5),
1238 ],
1239 sampled: None,
1240 paths: Vec::new(),
1241 expected: vec!["1 file holds no rows"],
1242 },
1243 Shape {
1244 what: "several empty files",
1245 files: vec![
1246 file(&[("id", DataType::Int64)], 0),
1247 file(&[("id", DataType::Int64)], 0),
1248 file(&[("id", DataType::Int64)], 5),
1249 ],
1250 sampled: None,
1251 paths: Vec::new(),
1252 expected: vec!["2 files hold no rows"],
1253 },
1254 Shape {
1255 what: "an empty file among sampled footers",
1256 files: vec![
1257 file(&[("id", DataType::Int64)], 0),
1258 file(&[("id", DataType::Int64)], 5),
1259 ],
1260 sampled: Some(900),
1261 paths: Vec::new(),
1262 // "file", not "footer": a footer does not hold rows. The scope line
1263 // is what says datui looked at two of nine hundred, and the note
1264 // claims nothing about the other 898.
1265 expected: vec!["1 file holds no rows"],
1266 },
1267 Shape {
1268 what: "an empty file and a column only the other has",
1269 files: vec![
1270 file(&[("id", DataType::Int64)], 0),
1271 file(&[("id", DataType::Int64), ("x", DataType::String)], 5),
1272 ],
1273 sampled: None,
1274 paths: Vec::new(),
1275 expected: vec![
1276 "x is in 1 of 2 files; absent from the rest, not null",
1277 "1 file holds no rows",
1278 ],
1279 },
1280 // --- a column that only some files have: the one note with a ratio ---
1281 Shape {
1282 what: "absent",
1283 files: vec![
1284 file(&[("id", i64.clone())], 1),
1285 file(&[("id", i64.clone()), ("x", str.clone())], 1),
1286 ],
1287 sampled: None,
1288 paths: Vec::new(),
1289 expected: vec!["x is in 1 of 2 files; absent from the rest, not null"],
1290 },
1291 Shape {
1292 what: "absent, sampled",
1293 files: vec![
1294 file(&[("id", i64.clone())], 1),
1295 file(&[("id", i64.clone()), ("x", str.clone())], 1),
1296 ],
1297 sampled: Some(200_000),
1298 paths: Vec::new(),
1299 expected: vec!["x is in 1 of the 2 footers read; absent from the rest, not null"],
1300 },
1301 Shape {
1302 what: "absent, with an unreadable footer",
1303 files: vec![
1304 file(&[("id", i64.clone())], 1),
1305 None,
1306 file(&[("id", i64.clone()), ("x", str.clone())], 1),
1307 ],
1308 sampled: None,
1309 paths: Vec::new(),
1310 expected: vec![
1311 "x is in 1 of the 2 files that could be read; absent from the rest, not null",
1312 "1 file unreadable, left out",
1313 ],
1314 },
1315 Shape {
1316 what: "absent, sampled, with an unreadable footer",
1317 files: vec![
1318 file(&[("id", i64.clone())], 1),
1319 None,
1320 file(&[("id", i64.clone()), ("x", str.clone())], 1),
1321 ],
1322 sampled: Some(200_000),
1323 paths: Vec::new(),
1324 expected: vec![
1325 "x is in 1 of the 2 footers that could be read; absent from the rest, not null",
1326 "1 footer unreadable, left out",
1327 ],
1328 },
1329 // --- a type the files disagree about: counted, never divided ---
1330 Shape {
1331 what: "conflicting",
1332 files: vec![
1333 file(&[("n", str.clone())], 10),
1334 file(&[("n", i64.clone())], 90),
1335 ],
1336 sampled: None,
1337 paths: Vec::new(),
1338 expected: vec!["n is str in 1 file; read as i64 and not read there"],
1339 },
1340 Shape {
1341 what: "conflicting, sampled",
1342 files: vec![
1343 file(&[("n", str.clone())], 10),
1344 file(&[("n", i64.clone())], 90),
1345 ],
1346 sampled: Some(200_000),
1347 paths: Vec::new(),
1348 expected: vec!["n is str in 1 footer; read as i64 and not read there"],
1349 },
1350 Shape {
1351 what: "conflicting, with an unreadable footer",
1352 files: vec![
1353 file(&[("n", str.clone())], 10),
1354 None,
1355 file(&[("n", i64.clone())], 90),
1356 ],
1357 sampled: None,
1358 paths: Vec::new(),
1359 expected: vec![
1360 "n is str in 1 file; read as i64 and not read there",
1361 "1 file unreadable, left out",
1362 ],
1363 },
1364 Shape {
1365 what: "conflicting, sampled, with an unreadable footer",
1366 files: vec![
1367 file(&[("n", str.clone())], 10),
1368 None,
1369 file(&[("n", i64.clone())], 90),
1370 ],
1371 sampled: Some(200_000),
1372 paths: Vec::new(),
1373 expected: vec![
1374 "n is str in 1 footer; read as i64 and not read there",
1375 "1 footer unreadable, left out",
1376 ],
1377 },
1378 // --- widening, which settles without loss ---
1379 Shape {
1380 what: "widened",
1381 files: with_n(i32.clone(), i64.clone()),
1382 sampled: None,
1383 paths: Vec::new(),
1384 expected: vec!["n is stored as more than one type; read as i64"],
1385 },
1386 // --- and the combinations, each saying all of what is true ---
1387 Shape {
1388 what: "absent and conflicting",
1389 files: vec![
1390 file(&[("n", str.clone())], 10),
1391 file(&[("n", i64.clone())], 90),
1392 file(&[("id", i64.clone())], 5),
1393 ],
1394 sampled: None,
1395 paths: Vec::new(),
1396 // Schema order: the newest file's columns lead, so `id` comes first.
1397 expected: vec![
1398 "id is in 1 of 3 files; absent from the rest, not null",
1399 "n is in 2 of 3 files; absent from the rest, not null",
1400 "n is str in 1 file; read as i64 and not read there",
1401 ],
1402 },
1403 Shape {
1404 what: "absent and widened",
1405 files: vec![
1406 file(&[("id", i64.clone())], 1),
1407 file(&[("id", i64.clone()), ("n", i32.clone())], 50),
1408 file(&[("id", i64.clone()), ("n", i64.clone())], 50),
1409 ],
1410 sampled: None,
1411 paths: Vec::new(),
1412 expected: vec![
1413 "n is in 2 of 3 files; absent from the rest, not null",
1414 "n is stored as more than one type; read as i64",
1415 ],
1416 },
1417 Shape {
1418 what: "conflicting and widened",
1419 files: vec![
1420 file(&[("n", i32.clone())], 50),
1421 file(&[("n", i64.clone())], 50),
1422 file(&[("n", str.clone())], 5),
1423 ],
1424 sampled: None,
1425 paths: Vec::new(),
1426 expected: vec![
1427 "n is str in 1 file; read as i64 and not read there",
1428 "n is stored as more than one type; read as i64",
1429 ],
1430 },
1431 Shape {
1432 what: "a chosen type no file stores",
1433 files: vec![
1434 file(&[("n", i32.clone())], 50),
1435 file(&[("n", DataType::Float32)], 50),
1436 file(&[("n", str.clone())], 5),
1437 ],
1438 sampled: None,
1439 paths: Vec::new(),
1440 expected: vec![
1441 "n is str in 1 file; read as f64 and not read there",
1442 "n is stored as more than one type; read as f64",
1443 ],
1444 },
1445 Shape {
1446 what: "two types the table spells the same way",
1447 files: vec![
1448 file(
1449 &[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
1450 10,
1451 ),
1452 file(
1453 &[(
1454 "t",
1455 DataType::Datetime(TimeUnit::Nanoseconds, Some(TimeZone::UTC)),
1456 )],
1457 90,
1458 ),
1459 ],
1460 sampled: None,
1461 paths: Vec::new(),
1462 // `datetime in 1 file; read as datetime` would say nothing, so the
1463 // note spells the types out where the short words collide.
1464 expected: vec![
1465 "t is datetime[ns] in 1 file; read as datetime[ns, UTC] and not read there",
1466 ],
1467 },
1468 Shape {
1469 what: "two conflicting types the table spells the same way",
1470 files: vec![
1471 file(
1472 &[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
1473 10,
1474 ),
1475 file(
1476 &[(
1477 "t",
1478 DataType::Datetime(TimeUnit::Nanoseconds, Some(TimeZone::UTC)),
1479 )],
1480 10,
1481 ),
1482 file(&[("t", str.clone())], 90),
1483 ],
1484 sampled: None,
1485 paths: Vec::new(),
1486 // The two that lost share a word as much as either shares one with the
1487 // winner, so all three are spelled out.
1488 expected: vec![
1489 "t is datetime[ns] or datetime[ns, UTC] in 2 files; read as str and not read there",
1490 ],
1491 },
1492 Shape {
1493 what: "two structs, which Display also spells the same way",
1494 files: vec![
1495 file(
1496 &[(
1497 "s",
1498 DataType::Struct(vec![Field::new("a".into(), i64.clone())]),
1499 )],
1500 90,
1501 ),
1502 file(
1503 &[(
1504 "s",
1505 DataType::Struct(vec![Field::new("a".into(), str.clone())]),
1506 )],
1507 10,
1508 ),
1509 ],
1510 sampled: None,
1511 paths: Vec::new(),
1512 // `struct[1]` for both, so only the fields tell them apart.
1513 expected: vec![
1514 "s is Struct({'a': String}) in 1 file; \
1515 read as Struct({'a': Int64}) and not read there",
1516 ],
1517 },
1518 Shape {
1519 what: "a chosen type whose word another type shares",
1520 files: vec![
1521 file(
1522 &[("t", DataType::Datetime(TimeUnit::Milliseconds, None))],
1523 50,
1524 ),
1525 file(
1526 &[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
1527 50,
1528 ),
1529 file(&[("t", str.clone())], 5),
1530 ],
1531 sampled: None,
1532 paths: Vec::new(),
1533 // Both notes name the winner the same way, and the way the Schema tab
1534 // does: "read as datetime" would drop the unit that is the point.
1535 expected: vec![
1536 "t is str in 1 file; read as datetime[ns] and not read there",
1537 "t is stored as more than one type; read as datetime[ns]",
1538 ],
1539 },
1540 Shape {
1541 what: "widened between two datetime units",
1542 files: vec![
1543 file(
1544 &[("t", DataType::Datetime(TimeUnit::Milliseconds, None))],
1545 50,
1546 ),
1547 file(
1548 &[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
1549 50,
1550 ),
1551 ],
1552 sampled: None,
1553 paths: Vec::new(),
1554 // Which unit won is the whole content of the note, and `datetime`
1555 // alone would not carry it.
1556 expected: vec!["t is stored as more than one type; read as datetime[ns]"],
1557 },
1558 Shape {
1559 what: "a struct that both widens and conflicts",
1560 files: vec![
1561 file(
1562 &[(
1563 "s",
1564 DataType::Struct(vec![Field::new("a".into(), i32.clone())]),
1565 )],
1566 50,
1567 ),
1568 file(
1569 &[(
1570 "s",
1571 DataType::Struct(vec![Field::new("a".into(), i64.clone())]),
1572 )],
1573 50,
1574 ),
1575 file(
1576 &[(
1577 "s",
1578 DataType::Struct(vec![Field::new("a".into(), str.clone())]),
1579 )],
1580 5,
1581 ),
1582 ],
1583 sampled: None,
1584 paths: Vec::new(),
1585 // Both notes name the winner the same way. Naming it separately let
1586 // one say `Struct({'a': Int64})` and the other `struct[1]`.
1587 expected: vec![
1588 "s is Struct({'a': String}) in 1 file; \
1589 read as Struct({'a': Int64}) and not read there",
1590 "s is stored as more than one type; read as Struct({'a': Int64})",
1591 ],
1592 },
1593 Shape {
1594 what: "a decimal, whose precision the header word drops",
1595 files: vec![
1596 file(&[("d", DataType::Decimal(38, 2))], 90),
1597 file(&[("d", str.clone())], 10),
1598 ],
1599 sampled: None,
1600 paths: Vec::new(),
1601 expected: vec!["d is str in 1 file; read as decimal[38,2] and not read there"],
1602 },
1603 Shape {
1604 what: "absent, conflicting and widened",
1605 files: vec![
1606 file(&[("id", i64.clone())], 1),
1607 file(&[("id", i64.clone()), ("n", i32.clone())], 50),
1608 file(&[("id", i64.clone()), ("n", i64.clone())], 50),
1609 file(&[("id", i64.clone()), ("n", str.clone())], 5),
1610 ],
1611 sampled: None,
1612 paths: Vec::new(),
1613 expected: vec![
1614 "n is in 3 of 4 files; absent from the rest, not null",
1615 "n is str in 1 file; read as i64 and not read there",
1616 "n is stored as more than one type; read as i64",
1617 ],
1618 },
1619 ];
1620
1621 for shape in &cases {
1622 assert_eq!(notes_for(shape), shape.expected, "{}", shape.what);
1623 }
1624
1625 // A column some file holds in another type always draws a note of its own.
1626 //
1627 // The table's view notes lean on this: `DataTableState::has_notes` answers
1628 // whether the Notes tab is on offer from the dataset's notes alone, which is
1629 // only sound if a note about what a sort leaves out can never be the only one
1630 // there. Every shape above that has a conflicting column is a case of it.
1631 for shape in &cases {
1632 let dataset = dataset_for(shape);
1633 let conflicting: Vec<&str> = dataset
1634 .columns
1635 .iter()
1636 .filter(|column| column.conflicting_files > 0)
1637 .map(|column| column.name.as_str())
1638 .collect();
1639 for name in conflicting {
1640 // The conflict note itself, not merely some note naming the column:
1641 // an absence or widening note about the same column would satisfy a
1642 // looser test while the one that matters had been deleted.
1643 assert!(
1644 from_dataset(&dataset).iter().any(|note| {
1645 note.summary.starts_with(&format!("{name} is "))
1646 && note.summary.ends_with("and not read there")
1647 }),
1648 "{}: {name} conflicts, so it says so on its own account",
1649 shape.what
1650 );
1651 }
1652 }
1653 }
1654
1655 /// The last resort, when a type's own `Debug` does not tell it from another's.
1656 ///
1657 /// Polars prints every `Enum` as `Enum([...])` and every global `Categorical` as
1658 /// `Categorical`, whatever their categories, so two of either collide through the
1659 /// short word, through `Display` and through `Debug` alike. Those are awkward to
1660 /// build here, so this drives the same branch with names that collide outright.
1661 /// Numbering says less than naming would, but it is two things rather than one
1662 /// thing said twice.
1663 #[test]
1664 fn types_that_print_alike_all_the_way_down_are_numbered() {
1665 let names = distinct_names(&[&DataType::Int64, &DataType::Int64]);
1666 assert_eq!(names, ["Int64 #1", "Int64 #2"]);
1667
1668 // A name that never collided keeps it.
1669 let mixed = distinct_names(&[&DataType::Int64, &DataType::Int64, &DataType::String]);
1670 // Once any pair collides every name comes from `Debug`, so `str` is `String`
1671 // here; only the colliding pair carries a number.
1672 assert_eq!(mixed, ["Int64 #1", "Int64 #2", "String"]);
1673 }
1674
1675 #[test]
1676 fn a_uniform_dataset_has_nothing_to_say() {
1677 let shape = Shape {
1678 what: "uniform",
1679 files: vec![
1680 file(&[("id", DataType::Int64)], 1),
1681 file(&[("id", DataType::Int64)], 1),
1682 ],
1683 sampled: None,
1684 paths: Vec::new(),
1685 expected: vec![],
1686 };
1687 assert!(notes_for(&shape).is_empty());
1688 }
1689
1690 /// The summary and the scope line sit one above the other, so a reader takes them
1691 /// together. Nothing in a summary may claim more than the scope allows.
1692 #[test]
1693 fn no_summary_claims_more_than_its_scope() {
1694 let files = [
1695 file(&[("id", DataType::Int64)], 1),
1696 None,
1697 file(&[("id", DataType::Int64), ("x", DataType::String)], 1),
1698 ];
1699 for (origin, expected_scope) in [
1700 (SchemaOrigin::AllFooters(3), "in all 3 footers"),
1701 (
1702 SchemaOrigin::FooterSample {
1703 read: 3,
1704 total: 200_000,
1705 },
1706 "in 3 of 200,000 footers (sample)",
1707 ),
1708 ] {
1709 let dataset = union_file_schemas(&files, origin);
1710 let notes = from_dataset(&dataset);
1711 assert_eq!(notes.len(), 2, "an absence note and an unreadable one");
1712 for note in notes {
1713 assert_eq!(note.scope, expected_scope);
1714 // The only ratio in the module is over what could be read, and it never
1715 // claims the whole dataset.
1716 assert!(
1717 !note.summary.contains("of 3 files"),
1718 "one of those three said nothing: {}",
1719 note.summary
1720 );
1721 }
1722 }
1723 }
1724
1725 /// Two notes about one column must each say which files they mean, rather than
1726 /// both saying "the others" about different sets.
1727 #[test]
1728 fn the_notes_about_one_column_do_not_talk_past_each_other() {
1729 let files = [
1730 file(&[("n", DataType::String)], 10),
1731 file(&[("n", DataType::Int64)], 90),
1732 file(&[("id", DataType::Int64)], 5),
1733 ];
1734 let dataset = union_file_schemas(&files, SchemaOrigin::AllFooters(3));
1735 let about_n: Vec<String> = from_dataset(&dataset)
1736 .into_iter()
1737 .map(|note| note.summary)
1738 .filter(|summary| summary.starts_with('n'))
1739 .collect();
1740 assert_eq!(
1741 about_n,
1742 [
1743 "n is in 2 of 3 files; absent from the rest, not null",
1744 "n is str in 1 file; read as i64 and not read there",
1745 ]
1746 );
1747 }
1748}