1use crate::formats::schema_union::{
8 ColumnDrift, ColumnRange, DatasetSchema, SchemaOrigin, SkippedFiles,
9};
10use crate::numfmt::group_chrome;
11use polars::prelude::{DataType, PlSmallStr};
12
13#[derive(Debug, Clone, PartialEq, Eq)]
15pub struct Note {
16 pub summary: String,
18 pub scope: String,
20 pub read_as_text: Option<PlSmallStr>,
23 pub passed_over: Option<usize>,
27}
28
29fn sampled(dataset: &DatasetSchema) -> bool {
31 matches!(dataset.origin, SchemaOrigin::FooterSample { .. })
32}
33
34fn how_many(dataset: &DatasetSchema, n: usize) -> String {
37 let noun = match (sampled(dataset), n) {
38 (false, 1) => "file",
39 (false, _) => "files",
40 (true, 1) => "footer",
41 (true, _) => "footers",
42 };
43 format!("{} {noun}", group_chrome(n))
44}
45
46fn out_of(dataset: &DatasetSchema) -> String {
49 let readable = dataset.files.saturating_sub(dataset.unreadable.len());
50 match (sampled(dataset), dataset.unreadable.is_empty()) {
51 (false, true) => how_many(dataset, readable),
52 (false, false) => format!("the {} that could be read", how_many(dataset, readable)),
53 (true, true) => format!("the {} read", how_many(dataset, readable)),
54 (true, false) => format!("the {} that could be read", how_many(dataset, readable)),
55 }
56}
57
58fn distinct_names(types: &[&DataType]) -> Vec<String> {
62 let collides = |names: &[String]| {
63 names
64 .iter()
65 .enumerate()
66 .any(|(i, name)| names[i + 1..].contains(name))
67 };
68 let shown: Vec<String> = types.iter().map(|t| format!("{t}")).collect();
71 if !collides(&shown) {
72 return shown;
73 }
74 let spelled: Vec<String> = types.iter().map(|t| format!("{t:?}")).collect();
76 if !collides(&spelled) {
77 return spelled;
78 }
79 let mut seen: Vec<&String> = Vec::new();
82 spelled
83 .iter()
84 .map(|name| {
85 if spelled.iter().filter(|other| *other == name).count() > 1 {
86 seen.push(name);
87 format!("{name} #{}", seen.iter().filter(|s| **s == name).count())
88 } else {
89 name.clone()
90 }
91 })
92 .collect()
93}
94
95pub fn from_dataset(dataset: &DatasetSchema) -> Vec<Note> {
98 let scope = format!("in {}", dataset.origin);
99 let readable = dataset.files.saturating_sub(dataset.unreadable.len());
100 let denominator = out_of(dataset);
101 let mut notes = Vec::new();
102
103 for column in dataset.drifting() {
104 let mut types: Vec<&DataType> = vec![&column.dtype];
106 types.extend(column.conflicting_types.iter());
107 let names = distinct_names(&types);
108 let (chosen, others) = names.split_first().expect("the chosen type is first");
109
110 notes.extend(absence_note(
112 column,
113 readable,
114 &denominator,
115 dataset.column_ranges.get(&column.name),
116 &scope,
117 ));
118 notes.extend(conflict_note(column, dataset, chosen, others, &scope));
119 notes.extend(widening_note(column, chosen, &scope));
120 }
121
122 for column in &dataset.read_as_text {
123 notes.push(text_note(column, &scope));
124 }
125
126 notes.extend(empty_files_note(dataset, &scope));
127 notes.extend(row_group_note(dataset, &scope));
128 notes.extend(small_files_note(dataset, &scope));
129 notes.extend(partition_layout_note(dataset));
130 notes.extend(skipped_files_note(dataset));
131
132 if !dataset.unreadable.is_empty() {
133 notes.push(Note {
134 summary: format!(
135 "{} unreadable, left out",
136 how_many(dataset, dataset.unreadable.len())
137 ),
138 scope: scope.clone(),
139 read_as_text: None,
140 passed_over: None,
141 });
142 }
143
144 notes
145}
146
147pub fn left_out_note(
152 column: &ColumnDrift,
153 dataset: &DatasetSchema,
154 rows: usize,
155 filtered: bool,
156 sorted: bool,
157) -> Note {
158 let what = match (filtered, sorted) {
159 (true, true) => "filter and sort",
160 (true, false) => "filter",
161 _ => "sort",
163 };
164 let there = if rows == 1 {
165 "1 row".to_string()
166 } else {
167 format!("{} rows", group_chrome(rows))
168 };
169 Note {
170 summary: format!(
171 "{}: {there} in {} left out of the {what}",
172 column.name,
173 how_many(dataset, column.conflicting_files)
174 ),
175 scope: format!("in {}", dataset.origin),
176 read_as_text: None,
177 passed_over: None,
178 }
179}
180
181fn empty_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
185 let empty = dataset.empty_files;
186 if empty == 0 {
187 return None;
188 }
189 let (count, verb) = if empty == 1 {
190 ("1 file".to_string(), "holds")
191 } else {
192 (format!("{} files", group_chrome(empty)), "hold")
193 };
194 Some(Note {
195 summary: format!("{count} {verb} no rows"),
196 scope: scope.to_string(),
197 read_as_text: None,
198 passed_over: None,
199 })
200}
201
202fn row_group_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
208 const BIG: usize = 64 * 1024 * 1024;
210 let median = dataset.median_row_group_bytes?;
211 if median <= BIG {
212 return None;
213 }
214 Some(Note {
215 summary: format!(
216 "median row group {}, each read whole",
217 crate::numfmt::bytes(median as u64)
218 ),
219 scope: scope.to_string(),
220 read_as_text: None,
221 passed_over: None,
222 })
223}
224
225fn small_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
230 const MANY: usize = 10_000;
233 const SMALL: usize = 1024 * 1024;
236 let files = dataset.origin.total_files();
237 let median = dataset.median_file_bytes?;
238 if median == 0 || files <= MANY || median >= SMALL {
240 return None;
241 }
242 let read = dataset.files;
243 let footers = if read == files {
246 "every footer".to_string()
247 } else {
248 format!("{} footers", group_chrome(read))
249 };
250 Some(Note {
251 summary: format!(
252 "{} files, median {}; {footers} read before any row",
253 group_chrome(files),
254 crate::numfmt::bytes(median as u64)
255 ),
256 scope: scope.to_string(),
257 read_as_text: None,
258 passed_over: None,
259 })
260}
261
262fn partition_layout_note(dataset: &DatasetSchema) -> Option<Note> {
267 const NAMED: usize = 2;
270 if dataset.partition_layouts.len() < 2 {
271 return None;
272 }
273 let (named, rest) = dataset
274 .partition_layouts
275 .split_at(dataset.partition_layouts.len().min(NAMED));
276 let mut clauses: Vec<String> = named
277 .iter()
278 .map(|(keys, files)| format!("{} by {}", how_many_files(*files), keys.join("/")))
279 .collect();
280 let (dropped_ways, dropped_files) = dataset.partition_layouts_dropped;
281 let ways = rest.len() + dropped_ways;
282 let files: usize = rest.iter().map(|(_, files)| files).sum::<usize>() + dropped_files;
283 if ways > 0 {
284 clauses.push(format!(
285 "{} by {} other {}",
286 how_many_files(files),
287 group_chrome(ways),
288 if ways == 1 { "way" } else { "ways" }
289 ));
290 }
291 Some(Note {
292 summary: format!("mixed partition keys: {}", clauses.join(", ")),
293 scope: format!(
294 "in the names of {} files",
295 group_chrome(dataset.listed_files)
296 ),
297 read_as_text: None,
298 passed_over: None,
299 })
300}
301
302fn how_many_files(n: usize) -> String {
304 format!(
305 "{} {}",
306 group_chrome(n),
307 if n == 1 { "file" } else { "files" }
308 )
309}
310
311fn text_note(column: &PlSmallStr, scope: &str) -> Note {
314 Note {
315 summary: format!("{column} read as text: filter and sort compare text"),
316 scope: scope.to_string(),
317 read_as_text: None,
318 passed_over: None,
319 }
320}
321
322fn skipped_files_note(dataset: &DatasetSchema) -> Option<Note> {
326 note_about_skipped(dataset.skipped)
327}
328
329fn note_about_skipped(skipped: SkippedFiles) -> Option<Note> {
332 let SkippedFiles {
333 bookkeeping,
334 not_parquet,
335 empty,
336 } = skipped;
337 if not_parquet == 0 && empty == 0 {
338 return None;
339 }
340 let files = |n: usize| {
341 if n == 1 {
342 "1 file".to_string()
343 } else {
344 format!("{} files", group_chrome(n))
345 }
346 };
347 let mut said = Vec::new();
350 if empty > 0 {
351 let what = if empty == 1 { "file" } else { "files" };
352 said.push(format!("{} empty {what}", group_chrome(empty)));
353 }
354 if not_parquet > 0 {
355 said.push(format!("{} not Parquet", files(not_parquet)));
356 }
357 if bookkeeping > 0 {
358 let what = if bookkeeping == 1 { "file" } else { "files" };
359 said.push(format!(
360 "{} writer bookkeeping {what}",
361 group_chrome(bookkeeping)
362 ));
363 }
364 Some(Note {
365 summary: format!("skipped: {}", said.join(", ")),
366 scope: "in this directory's listing".to_string(),
367 read_as_text: None,
368 passed_over: None,
369 })
370}
371
372const NAMES_SHOWN: usize = 3;
374
375pub fn some_names<S: AsRef<str>>(names: &[S]) -> String {
377 let mut said: Vec<&str> = names.iter().take(NAMES_SHOWN).map(AsRef::as_ref).collect();
378 let ellipsis = crate::glyphs::get().ellipsis;
379 if names.len() > NAMES_SHOWN {
380 said.push(ellipsis);
381 }
382 said.join(", ")
383}
384
385pub fn no_header(files: &[&std::path::Path]) -> Option<Note> {
388 if files.is_empty() {
389 return None;
390 }
391 let names: Vec<String> = files
392 .iter()
393 .map(|f| {
394 f.file_name().map_or_else(
395 || f.display().to_string(),
396 |n| n.to_string_lossy().into_owned(),
397 )
398 })
399 .collect();
400 let what = if files.len() == 1 { "file" } else { "files" };
401 Some(Note {
402 summary: format!(
403 "{} {what} with no header skipped: {}",
404 files.len(),
405 some_names(&names)
406 ),
407 scope: "empty, blank, or only NUL padding".to_string(),
408 read_as_text: None,
409 passed_over: None,
410 })
411}
412
413pub fn from_the_open(
417 left_out: &[(crate::FileFormat, usize)],
418 lake: Option<&str>,
419 files_differ: crate::formats::schema_union::Disagreement,
420 names_look_like_data: bool,
421) -> Vec<Note> {
422 let mut notes = Vec::new();
423 let scope = || "in a spread of this directory's files".to_string();
427 if files_differ.columns {
428 notes.push(Note {
429 summary: "columns differ across files: a missing column reads null".to_string(),
430 scope: scope(),
431 read_as_text: None,
432 passed_over: None,
433 });
434 }
435 if files_differ.headerless {
436 notes.push(Note {
437 summary: concat!(
438 "no header row? first row read as names: ",
439 "H on Schema, or --no-header, reads it as data"
440 )
441 .to_string(),
442 scope: scope(),
443 read_as_text: None,
444 passed_over: None,
445 });
446 }
447 if names_look_like_data && !files_differ.headerless {
450 notes.push(Note {
451 summary: "column names look like data: H on Schema reads them as a row".to_string(),
452 scope: "from the column names".to_string(),
453 read_as_text: None,
454 passed_over: None,
455 });
456 }
457 if files_differ.types {
458 notes.push(Note {
459 summary: "a column's type differs across files: read as the wider type".to_string(),
460 scope: scope(),
461 read_as_text: None,
462 passed_over: None,
463 });
464 }
465 if let Some(format) = lake {
466 notes.push(Note {
467 summary: format!(
470 "{format} table's files, not the table: deleted rows and old versions counted"
471 ),
472 scope: format!("in this {format} table's directory"),
473 read_as_text: None,
474 passed_over: None,
475 });
476 }
477 if !left_out.is_empty() {
478 let said: Vec<String> = left_out
479 .iter()
480 .map(|(format, n)| format!("{n} {}", format.name()))
481 .collect();
482 notes.push(Note {
483 summary: format!(
484 "mixed formats, read as the commonest: {} not read",
485 said.join(", ")
486 ),
487 scope: "in this directory's listing".to_string(),
488 read_as_text: None,
489 passed_over: Some(left_out.iter().map(|(_, n)| n).sum()),
492 });
493 }
494 notes
495}
496
497pub fn map_caches(count: usize) -> Option<Note> {
500 let files = if count == 1 { "file" } else { "files" };
501 (count > 0).then(|| Note {
502 summary: format!("{count} cache {files} written by map() not read"),
503 scope: "in this directory's listing".to_string(),
504 read_as_text: None,
505 passed_over: None,
506 })
507}
508
509pub fn merged(
514 open: &[Note],
515 dataset: &[Note],
516 view: &[Note],
517 schema: Option<&DatasetSchema>,
518) -> Vec<Note> {
519 let rebuilt: Option<(Note, Option<Note>)> = match (
522 open.iter().find_map(|n| n.passed_over),
523 schema.map(|s| s.skipped),
524 ) {
525 (Some(covered), Some(skipped)) => note_about_skipped(skipped).map(|full| {
526 let remaining = SkippedFiles {
527 not_parquet: skipped.not_parquet.saturating_sub(covered),
530 ..skipped
531 };
532 (full, note_about_skipped(remaining))
533 }),
534 _ => None,
535 };
536 let mut out: Vec<Note> = open.to_vec();
537 for note in dataset {
538 match &rebuilt {
539 Some((full, reduced)) if note == full => out.extend(reduced.clone()),
540 _ => out.push(note.clone()),
541 }
542 }
543 out.extend(view.iter().cloned());
544 out
545}
546
547fn absence_note(
550 column: &ColumnDrift,
551 readable: usize,
552 denominator: &str,
553 range: Option<&ColumnRange>,
554 scope: &str,
555) -> Option<Note> {
556 if column.present_in == 0 || column.present_in >= readable {
557 return None;
558 }
559 let where_it_is = match range {
562 Some(ColumnRange::Only(partition)) => format!(", only {partition}"),
563 Some(ColumnRange::NoneBefore(partition)) => format!(", none before {partition}"),
564 None => String::new(),
565 };
566 Some(Note {
567 summary: format!(
568 "{} is in {} of {}{}; absent from the rest, not null",
569 column.name,
570 group_chrome(column.present_in),
571 denominator,
572 where_it_is
573 ),
574 scope: scope.to_string(),
575 read_as_text: None,
576 passed_over: None,
577 })
578}
579
580fn conflict_note(
583 column: &ColumnDrift,
584 dataset: &DatasetSchema,
585 chosen: &str,
586 others: &[String],
587 scope: &str,
588) -> Option<Note> {
589 if column.conflicting_files == 0 {
590 return None;
591 }
592 Some(Note {
593 summary: format!(
594 "{} is {} in {}; read as {} and not read there",
595 column.name,
596 others.join(" or "),
597 how_many(dataset, column.conflicting_files),
598 chosen
599 ),
600 scope: scope.to_string(),
601 read_as_text: column.can_read_as_text().then(|| column.name.clone()),
604 passed_over: None,
605 })
606}
607
608fn widening_note(column: &ColumnDrift, chosen: &str, scope: &str) -> Option<Note> {
612 if !column.widened {
613 return None;
614 }
615 Some(Note {
616 summary: format!(
617 "{} is stored as more than one type; read as {chosen}",
618 column.name
619 ),
620 scope: scope.to_string(),
621 read_as_text: None,
622 passed_over: None,
623 })
624}
625
626#[cfg(test)]
627mod tests;