use crate::numfmt::group_chrome;
use crate::schema_union::{ColumnDrift, ColumnRange, DatasetSchema, SchemaOrigin, SkippedFiles};
use polars::prelude::{DataType, PlSmallStr};
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Note {
pub summary: String,
pub scope: String,
pub read_as_text: Option<PlSmallStr>,
pub passed_over: Option<usize>,
}
fn sampled(dataset: &DatasetSchema) -> bool {
matches!(dataset.origin, SchemaOrigin::FooterSample { .. })
}
fn how_many(dataset: &DatasetSchema, n: usize) -> String {
let noun = match (sampled(dataset), n) {
(false, 1) => "file",
(false, _) => "files",
(true, 1) => "footer",
(true, _) => "footers",
};
format!("{} {noun}", group_chrome(n))
}
fn out_of(dataset: &DatasetSchema) -> String {
let readable = dataset.files.saturating_sub(dataset.unreadable.len());
match (sampled(dataset), dataset.unreadable.is_empty()) {
(false, true) => how_many(dataset, readable),
(false, false) => format!("the {} that could be read", how_many(dataset, readable)),
(true, true) => format!("the {} read", how_many(dataset, readable)),
(true, false) => format!("the {} that could be read", how_many(dataset, readable)),
}
}
fn distinct_names(types: &[&DataType]) -> Vec<String> {
let collides = |names: &[String]| {
names
.iter()
.enumerate()
.any(|(i, name)| names[i + 1..].contains(name))
};
let shown: Vec<String> = types.iter().map(|t| format!("{t}")).collect();
if !collides(&shown) {
return shown;
}
let spelled: Vec<String> = types.iter().map(|t| format!("{t:?}")).collect();
if !collides(&spelled) {
return spelled;
}
let mut seen: Vec<&String> = Vec::new();
spelled
.iter()
.map(|name| {
if spelled.iter().filter(|other| *other == name).count() > 1 {
seen.push(name);
format!("{name} #{}", seen.iter().filter(|s| **s == name).count())
} else {
name.clone()
}
})
.collect()
}
pub fn from_dataset(dataset: &DatasetSchema) -> Vec<Note> {
let scope = format!("in {}", dataset.origin);
let readable = dataset.files.saturating_sub(dataset.unreadable.len());
let denominator = out_of(dataset);
let mut notes = Vec::new();
for column in dataset.drifting() {
let mut types: Vec<&DataType> = vec![&column.dtype];
types.extend(column.conflicting_types.iter());
let names = distinct_names(&types);
let (chosen, others) = names.split_first().expect("the chosen type is first");
notes.extend(absence_note(
column,
readable,
&denominator,
dataset.column_ranges.get(&column.name),
&scope,
));
notes.extend(conflict_note(column, dataset, chosen, others, &scope));
notes.extend(widening_note(column, chosen, &scope));
}
for column in &dataset.read_as_text {
notes.push(text_note(column, &scope));
}
notes.extend(empty_files_note(dataset, &scope));
notes.extend(row_group_note(dataset, &scope));
notes.extend(small_files_note(dataset, &scope));
notes.extend(partition_layout_note(dataset));
notes.extend(skipped_files_note(dataset));
if !dataset.unreadable.is_empty() {
notes.push(Note {
summary: format!(
"{} unreadable, left out",
how_many(dataset, dataset.unreadable.len())
),
scope: scope.clone(),
read_as_text: None,
passed_over: None,
});
}
notes
}
pub fn left_out_note(
column: &ColumnDrift,
dataset: &DatasetSchema,
rows: usize,
filtered: bool,
sorted: bool,
) -> Note {
let what = match (filtered, sorted) {
(true, true) => "filter and sort",
(true, false) => "filter",
_ => "sort",
};
let there = if rows == 1 {
"1 row".to_string()
} else {
format!("{} rows", group_chrome(rows))
};
Note {
summary: format!(
"{}: {there} in {} left out of the {what}",
column.name,
how_many(dataset, column.conflicting_files)
),
scope: format!("in {}", dataset.origin),
read_as_text: None,
passed_over: None,
}
}
fn empty_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
let empty = dataset.empty_files;
if empty == 0 {
return None;
}
let (count, verb) = if empty == 1 {
("1 file".to_string(), "holds")
} else {
(format!("{} files", group_chrome(empty)), "hold")
};
Some(Note {
summary: format!("{count} {verb} no rows"),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
fn row_group_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
const BIG: usize = 64 * 1024 * 1024;
let median = dataset.median_row_group_bytes?;
if median <= BIG {
return None;
}
Some(Note {
summary: format!(
"median row group {}, each read whole",
crate::widgets::info::format_bytes(median as u64)
),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
fn small_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
const MANY: usize = 10_000;
const SMALL: usize = 1024 * 1024;
let files = dataset.origin.total_files();
let median = dataset.median_file_bytes?;
if median == 0 || files <= MANY || median >= SMALL {
return None;
}
let read = dataset.files;
let footers = if read == files {
"every footer".to_string()
} else {
format!("{} footers", group_chrome(read))
};
Some(Note {
summary: format!(
"{} files, median {}; {footers} read before any row",
group_chrome(files),
crate::widgets::info::format_bytes(median as u64)
),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
fn partition_layout_note(dataset: &DatasetSchema) -> Option<Note> {
const NAMED: usize = 2;
if dataset.partition_layouts.len() < 2 {
return None;
}
let (named, rest) = dataset
.partition_layouts
.split_at(dataset.partition_layouts.len().min(NAMED));
let mut clauses: Vec<String> = named
.iter()
.map(|(keys, files)| format!("{} by {}", how_many_files(*files), keys.join("/")))
.collect();
let (dropped_ways, dropped_files) = dataset.partition_layouts_dropped;
let ways = rest.len() + dropped_ways;
let files: usize = rest.iter().map(|(_, files)| files).sum::<usize>() + dropped_files;
if ways > 0 {
clauses.push(format!(
"{} by {} other {}",
how_many_files(files),
group_chrome(ways),
if ways == 1 { "way" } else { "ways" }
));
}
Some(Note {
summary: format!("mixed partition keys: {}", clauses.join(", ")),
scope: format!(
"in the names of {} files",
group_chrome(dataset.listed_files)
),
read_as_text: None,
passed_over: None,
})
}
fn how_many_files(n: usize) -> String {
format!(
"{} {}",
group_chrome(n),
if n == 1 { "file" } else { "files" }
)
}
fn text_note(column: &PlSmallStr, scope: &str) -> Note {
Note {
summary: format!("{column} read as text: filter and sort compare text"),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
}
}
fn skipped_files_note(dataset: &DatasetSchema) -> Option<Note> {
note_about_skipped(dataset.skipped)
}
fn note_about_skipped(skipped: SkippedFiles) -> Option<Note> {
let SkippedFiles {
bookkeeping,
not_parquet,
empty,
} = skipped;
if not_parquet == 0 && empty == 0 {
return None;
}
let files = |n: usize| {
if n == 1 {
"1 file".to_string()
} else {
format!("{} files", group_chrome(n))
}
};
let mut said = Vec::new();
if empty > 0 {
let what = if empty == 1 { "file" } else { "files" };
said.push(format!("{} empty {what}", group_chrome(empty)));
}
if not_parquet > 0 {
said.push(format!("{} not Parquet", files(not_parquet)));
}
if bookkeeping > 0 {
let what = if bookkeeping == 1 { "file" } else { "files" };
said.push(format!(
"{} writer bookkeeping {what}",
group_chrome(bookkeeping)
));
}
Some(Note {
summary: format!("skipped: {}", said.join(", ")),
scope: "in this directory's listing".to_string(),
read_as_text: None,
passed_over: None,
})
}
const NAMES_SHOWN: usize = 3;
pub fn some_names<S: AsRef<str>>(names: &[S]) -> String {
let mut said: Vec<&str> = names.iter().take(NAMES_SHOWN).map(AsRef::as_ref).collect();
let ellipsis = crate::glyphs::get().ellipsis;
if names.len() > NAMES_SHOWN {
said.push(ellipsis);
}
said.join(", ")
}
pub fn no_header(files: &[&std::path::Path]) -> Option<Note> {
if files.is_empty() {
return None;
}
let names: Vec<String> = files
.iter()
.map(|f| {
f.file_name().map_or_else(
|| f.display().to_string(),
|n| n.to_string_lossy().into_owned(),
)
})
.collect();
let what = if files.len() == 1 { "file" } else { "files" };
Some(Note {
summary: format!(
"{} {what} with no header skipped: {}",
files.len(),
some_names(&names)
),
scope: "empty, blank, or only NUL padding".to_string(),
read_as_text: None,
passed_over: None,
})
}
pub fn from_the_open(
left_out: &[(crate::FileFormat, usize)],
lake: Option<&str>,
files_differ: crate::schema_union::Disagreement,
names_look_like_data: bool,
) -> Vec<Note> {
let mut notes = Vec::new();
let scope = || "in a spread of this directory's files".to_string();
if files_differ.columns {
notes.push(Note {
summary: "columns differ across files: a missing column reads null".to_string(),
scope: scope(),
read_as_text: None,
passed_over: None,
});
}
if files_differ.headerless {
notes.push(Note {
summary: concat!(
"no header row? first row read as names: ",
"H on Schema, or --no-header, reads it as data"
)
.to_string(),
scope: scope(),
read_as_text: None,
passed_over: None,
});
}
if names_look_like_data && !files_differ.headerless {
notes.push(Note {
summary: "column names look like data: H on Schema reads them as a row".to_string(),
scope: "from the column names".to_string(),
read_as_text: None,
passed_over: None,
});
}
if files_differ.types {
notes.push(Note {
summary: "a column's type differs across files: read as the wider type".to_string(),
scope: scope(),
read_as_text: None,
passed_over: None,
});
}
if let Some(format) = lake {
notes.push(Note {
summary: format!(
"{format} table's files, not the table: deleted rows and old versions counted"
),
scope: format!("in this {format} table's directory"),
read_as_text: None,
passed_over: None,
});
}
if !left_out.is_empty() {
let said: Vec<String> = left_out
.iter()
.map(|(format, n)| format!("{n} {}", format.name()))
.collect();
notes.push(Note {
summary: format!(
"mixed formats, read as the commonest: {} not read",
said.join(", ")
),
scope: "in this directory's listing".to_string(),
read_as_text: None,
passed_over: Some(left_out.iter().map(|(_, n)| n).sum()),
});
}
notes
}
pub fn map_caches(count: usize) -> Option<Note> {
let files = if count == 1 { "file" } else { "files" };
(count > 0).then(|| Note {
summary: format!("{count} cache {files} written by map() not read"),
scope: "in this directory's listing".to_string(),
read_as_text: None,
passed_over: None,
})
}
pub fn merged(
open: &[Note],
dataset: &[Note],
view: &[Note],
schema: Option<&DatasetSchema>,
) -> Vec<Note> {
let rebuilt: Option<(Note, Option<Note>)> = match (
open.iter().find_map(|n| n.passed_over),
schema.map(|s| s.skipped),
) {
(Some(covered), Some(skipped)) => note_about_skipped(skipped).map(|full| {
let remaining = SkippedFiles {
not_parquet: skipped.not_parquet.saturating_sub(covered),
..skipped
};
(full, note_about_skipped(remaining))
}),
_ => None,
};
let mut out: Vec<Note> = open.to_vec();
for note in dataset {
match &rebuilt {
Some((full, reduced)) if note == full => out.extend(reduced.clone()),
_ => out.push(note.clone()),
}
}
out.extend(view.iter().cloned());
out
}
fn absence_note(
column: &ColumnDrift,
readable: usize,
denominator: &str,
range: Option<&ColumnRange>,
scope: &str,
) -> Option<Note> {
if column.present_in == 0 || column.present_in >= readable {
return None;
}
let where_it_is = match range {
Some(ColumnRange::Only(partition)) => format!(", only {partition}"),
Some(ColumnRange::NoneBefore(partition)) => format!(", none before {partition}"),
None => String::new(),
};
Some(Note {
summary: format!(
"{} is in {} of {}{}; absent from the rest, not null",
column.name,
group_chrome(column.present_in),
denominator,
where_it_is
),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
fn conflict_note(
column: &ColumnDrift,
dataset: &DatasetSchema,
chosen: &str,
others: &[String],
scope: &str,
) -> Option<Note> {
if column.conflicting_files == 0 {
return None;
}
Some(Note {
summary: format!(
"{} is {} in {}; read as {} and not read there",
column.name,
others.join(" or "),
how_many(dataset, column.conflicting_files),
chosen
),
scope: scope.to_string(),
read_as_text: column.can_read_as_text().then(|| column.name.clone()),
passed_over: None,
})
}
fn widening_note(column: &ColumnDrift, chosen: &str, scope: &str) -> Option<Note> {
if !column.widened {
return None;
}
Some(Note {
summary: format!(
"{} is stored as more than one type; read as {chosen}",
column.name
),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
#[cfg(test)]
mod tests {
use super::*;
use crate::schema_union::{FileFooter, union_file_schemas};
use polars::prelude::{DataType, Field, Schema, TimeUnit, TimeZone};
use std::sync::Arc;
#[test]
fn the_notes_from_an_open_have_no_holes_in_them() {
let notes = from_the_open(
&[(crate::FileFormat::Json, 1)],
Some("Delta"),
crate::schema_union::Disagreement {
columns: true,
types: true,
headerless: true,
},
false,
);
assert!(notes.len() >= 3, "{notes:?}");
for note in ¬es {
assert!(!note.summary.contains(" "), "{:?}", note.summary);
}
}
fn walked(skipped: SkippedFiles) -> DatasetSchema {
union_file_schemas(&[], SchemaOrigin::AllFooters(0)).with_skipped(skipped)
}
fn agree() -> crate::schema_union::Disagreement {
crate::schema_union::Disagreement {
columns: false,
types: false,
headerless: false,
}
}
#[test]
fn a_mixed_directory_is_not_reported_twice() {
let open = from_the_open(&[(crate::FileFormat::Csv, 1)], None, agree(), false);
let dataset = walked(SkippedFiles {
bookkeeping: 0,
not_parquet: 1,
empty: 0,
});
let said: Vec<String> = merged(&open, &from_dataset(&dataset), &[], Some(&dataset))
.into_iter()
.map(|n| n.summary)
.collect();
assert_eq!(
said,
["mixed formats, read as the commonest: 1 csv not read"],
"one fact, said once, in the open's words"
);
}
#[test]
fn what_only_the_walk_saw_stays_counted() {
let open = from_the_open(&[(crate::FileFormat::Csv, 1)], None, agree(), false);
let dataset = walked(SkippedFiles {
bookkeeping: 2,
not_parquet: 2,
empty: 1,
});
let view = [text_note(&PlSmallStr::from("n"), "in 3 files")];
let said: Vec<String> = merged(&open, &from_dataset(&dataset), &view, Some(&dataset))
.into_iter()
.map(|n| n.summary)
.collect();
assert_eq!(
said,
[
"mixed formats, read as the commonest: 1 csv not read".to_string(),
"skipped: 1 empty file, 1 file not Parquet, 2 writer bookkeeping files".to_string(),
"n read as text: filter and sort compare text".to_string(),
],
"the walk's own findings and the view's notes survive the merge"
);
}
#[test]
fn a_walk_only_tally_is_left_alone() {
let open = from_the_open(&[], Some("Delta"), agree(), false);
let dataset = walked(SkippedFiles {
bookkeeping: 0,
not_parquet: 1,
empty: 0,
});
let said: Vec<String> = merged(&open, &from_dataset(&dataset), &[], Some(&dataset))
.into_iter()
.map(|n| n.summary)
.collect();
assert_eq!(said.len(), 2, "{said:?}");
assert!(
said[1] == "skipped: 1 file not Parquet",
"different facts do not merge: {said:?}"
);
}
fn file(columns: &[(&str, DataType)], rows: usize) -> Option<FileFooter> {
let mut schema = Schema::with_capacity(columns.len());
for (name, dtype) in columns {
schema.with_column((*name).into(), dtype.clone());
}
Some(FileFooter {
schema: Arc::new(schema),
row_group_rows: vec![rows],
file_bytes: 0,
row_group_bytes: Vec::new(),
column_bytes: Vec::new(),
})
}
#[derive(Default)]
struct Shape {
what: &'static str,
files: Vec<Option<FileFooter>>,
sampled: Option<usize>,
paths: Vec<&'static str>,
expected: Vec<&'static str>,
}
fn dataset_for(shape: &Shape) -> DatasetSchema {
let origin = match shape.sampled {
Some(total) => SchemaOrigin::FooterSample {
read: shape.files.len(),
total,
},
None => SchemaOrigin::AllFooters(shape.files.len()),
};
let dataset = union_file_schemas(&shape.files, origin);
if shape.paths.is_empty() {
return dataset;
}
let paths: Vec<String> = shape.paths.iter().map(|p| p.to_string()).collect();
dataset.with_partition_layouts("d", &paths)
}
fn notes_for(shape: &Shape) -> Vec<String> {
from_dataset(&dataset_for(shape))
.into_iter()
.map(|n| n.summary)
.collect()
}
#[test]
fn every_shape_of_disagreement_reads_the_way_it_should() {
let i32 = DataType::Int32;
let i64 = DataType::Int64;
let str = DataType::String;
let with_n = |a: DataType, b: DataType| {
vec![
file(&[("id", i64.clone()), ("n", a)], 50),
file(&[("id", i64.clone()), ("n", b)], 50),
]
};
let cases = vec![
Shape {
what: "one directory under another key",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64)], 1),
],
paths: vec!["d/date=1/a.parquet", "d/dt=2/b.parquet"],
expected: vec!["mixed partition keys: 1 file by date, 1 file by dt"],
..Shape::default()
},
Shape {
what: "directories that agree",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64)], 1),
],
paths: vec!["d/date=1/a.parquet", "d/date=2/b.parquet"],
expected: vec![],
..Shape::default()
},
Shape {
what: "a column only one partition has",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("oops", DataType::String)], 1),
file(&[("id", DataType::Int64)], 1),
],
paths: vec![
"d/date=2024-03-01/a.parquet",
"d/date=2024-03-02/b.parquet",
"d/date=2024-03-03/c.parquet",
],
expected: vec![
"oops is in 1 of 3 files, only date=2024-03-02; absent from \
the rest, not null",
],
..Shape::default()
},
Shape {
what: "a column the feed started sending",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
],
paths: vec![
"d/date=2010-07-17/a.parquet",
"d/date=2010-07-18/b.parquet",
"d/date=2010-07-19/c.parquet",
],
expected: vec![
"fee is in 2 of 3 files, none before date=2010-07-18; absent from \
the rest, not null",
],
..Shape::default()
},
Shape {
what: "a column in some files but no pattern to where",
files: vec![
file(&[("id", DataType::Int64), ("odd", DataType::Int64)], 1),
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("odd", DataType::Int64)], 1),
],
paths: vec![
"d/date=1/a.parquet",
"d/date=2/b.parquet",
"d/date=3/c.parquet",
],
expected: vec!["odd is in 2 of 3 files; absent from the rest, not null"],
..Shape::default()
},
Shape {
what: "a column starting where the listing and the reader disagree",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
],
paths: vec![
"d/part=1/a.parquet",
"d/part=10/b.parquet",
"d/part=2/c.parquet",
"d/part=3/d.parquet",
],
expected: vec!["fee is in 3 of 4 files; absent from the rest, not null"],
..Shape::default()
},
Shape {
what: "a column starting at a month spelled two ways",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
],
paths: vec!["d/m=03/a.parquet", "d/m=3/b.parquet", "d/m=4/c.parquet"],
expected: vec!["fee is in 2 of 3 files; absent from the rest, not null"],
..Shape::default()
},
Shape {
what: "a column of a dataset whose directories name their keys in two orders",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
],
paths: vec!["d/m=03/y=2024/a.parquet", "d/y=2024/m=03/b.parquet"],
expected: vec!["fee is in 1 of 2 files; absent from the rest, not null"],
..Shape::default()
},
Shape {
what: "a column two files of one partition have",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
],
paths: vec![
"d/date=2024-03-01/a.parquet",
"d/date=2024-03-02/b.parquet",
"d/date=2024-03-02/c.parquet",
],
expected: vec![
"fee is in 2 of 3 files, only date=2024-03-02; absent from the \
rest, not null",
],
..Shape::default()
},
Shape {
what: "a column starting halfway through a partition",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
],
paths: vec![
"d/date=2024-01-01/a.parquet",
"d/date=2024-01-02/b.parquet",
"d/date=2024-01-02/c.parquet",
"d/date=2024-01-03/d.parquet",
],
expected: vec!["fee is in 2 of 4 files; absent from the rest, not null"],
..Shape::default()
},
Shape {
what: "a column whose partition holds the file without it",
files: vec![
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64)], 1),
],
paths: vec!["d/y=2024/a.parquet", "d/y=2024/m=03/b.parquet"],
expected: vec![
"fee is in 1 of 2 files; absent from the rest, not null",
"mixed partition keys: 1 file by m/y, 1 file by y",
],
..Shape::default()
},
Shape {
what: "a column of a dataset partitioned more than one level deep",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
],
paths: vec!["d/y=2024/m=02/a.parquet", "d/y=2024/m=03/b.parquet"],
expected: vec![
"fee is in 1 of 2 files, only y=2024/m=03; absent from the rest, \
not null",
],
..Shape::default()
},
Shape {
what: "a column in a file that sits under no partition at all",
files: vec![
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64)], 1),
],
paths: vec![
"d/aaa.parquet",
"d/date=2024-01-02/b.parquet",
"d/date=2024-01-03/c.parquet",
],
expected: vec!["fee is in 2 of 3 files; absent from the rest, not null"],
..Shape::default()
},
Shape {
what: "a column whose files are all in the one partition anyway",
files: vec![
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
file(&[("id", DataType::Int64)], 1),
],
paths: vec![
"d/date=2024-03-02/a.parquet",
"d/date=2024-03-02/b.parquet",
"d/date=2024-03-02/c.parquet",
],
expected: vec!["fee is in 2 of 3 files; absent from the rest, not null"],
..Shape::default()
},
Shape {
what: "a column of a dataset with a footer that would not parse",
files: vec![
file(&[("id", DataType::Int64)], 1),
None,
file(&[("id", DataType::Int64), ("fee", DataType::Int64)], 1),
],
paths: vec!["d/x=1/a.parquet", "d/x=2/b.parquet", "d/x=3/c.parquet"],
expected: vec![
"fee is in 1 of the 2 files that could be read; absent from the \
rest, not null",
"1 file unreadable, left out",
],
..Shape::default()
},
Shape {
what: "a column of a dataset whose footers were sampled",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64), ("oops", DataType::String)], 1),
],
sampled: Some(6541),
paths: vec!["d/date=2024-03-01/a.parquet", "d/date=2024-03-02/b.parquet"],
expected: vec![
"oops is in 1 of the 2 footers read; absent from the rest, not null",
],
},
Shape {
what: "one empty file",
files: vec![
file(&[("id", DataType::Int64)], 0),
file(&[("id", DataType::Int64)], 5),
],
sampled: None,
paths: Vec::new(),
expected: vec!["1 file holds no rows"],
},
Shape {
what: "several empty files",
files: vec![
file(&[("id", DataType::Int64)], 0),
file(&[("id", DataType::Int64)], 0),
file(&[("id", DataType::Int64)], 5),
],
sampled: None,
paths: Vec::new(),
expected: vec!["2 files hold no rows"],
},
Shape {
what: "an empty file among sampled footers",
files: vec![
file(&[("id", DataType::Int64)], 0),
file(&[("id", DataType::Int64)], 5),
],
sampled: Some(900),
paths: Vec::new(),
expected: vec!["1 file holds no rows"],
},
Shape {
what: "an empty file and a column only the other has",
files: vec![
file(&[("id", DataType::Int64)], 0),
file(&[("id", DataType::Int64), ("x", DataType::String)], 5),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"x is in 1 of 2 files; absent from the rest, not null",
"1 file holds no rows",
],
},
Shape {
what: "absent",
files: vec![
file(&[("id", i64.clone())], 1),
file(&[("id", i64.clone()), ("x", str.clone())], 1),
],
sampled: None,
paths: Vec::new(),
expected: vec!["x is in 1 of 2 files; absent from the rest, not null"],
},
Shape {
what: "absent, sampled",
files: vec![
file(&[("id", i64.clone())], 1),
file(&[("id", i64.clone()), ("x", str.clone())], 1),
],
sampled: Some(200_000),
paths: Vec::new(),
expected: vec!["x is in 1 of the 2 footers read; absent from the rest, not null"],
},
Shape {
what: "absent, with an unreadable footer",
files: vec![
file(&[("id", i64.clone())], 1),
None,
file(&[("id", i64.clone()), ("x", str.clone())], 1),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"x is in 1 of the 2 files that could be read; absent from the rest, not null",
"1 file unreadable, left out",
],
},
Shape {
what: "absent, sampled, with an unreadable footer",
files: vec![
file(&[("id", i64.clone())], 1),
None,
file(&[("id", i64.clone()), ("x", str.clone())], 1),
],
sampled: Some(200_000),
paths: Vec::new(),
expected: vec![
"x is in 1 of the 2 footers that could be read; absent from the rest, not null",
"1 footer unreadable, left out",
],
},
Shape {
what: "conflicting",
files: vec![
file(&[("n", str.clone())], 10),
file(&[("n", i64.clone())], 90),
],
sampled: None,
paths: Vec::new(),
expected: vec!["n is str in 1 file; read as i64 and not read there"],
},
Shape {
what: "conflicting, sampled",
files: vec![
file(&[("n", str.clone())], 10),
file(&[("n", i64.clone())], 90),
],
sampled: Some(200_000),
paths: Vec::new(),
expected: vec!["n is str in 1 footer; read as i64 and not read there"],
},
Shape {
what: "conflicting, with an unreadable footer",
files: vec![
file(&[("n", str.clone())], 10),
None,
file(&[("n", i64.clone())], 90),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"n is str in 1 file; read as i64 and not read there",
"1 file unreadable, left out",
],
},
Shape {
what: "conflicting, sampled, with an unreadable footer",
files: vec![
file(&[("n", str.clone())], 10),
None,
file(&[("n", i64.clone())], 90),
],
sampled: Some(200_000),
paths: Vec::new(),
expected: vec![
"n is str in 1 footer; read as i64 and not read there",
"1 footer unreadable, left out",
],
},
Shape {
what: "widened",
files: with_n(i32.clone(), i64.clone()),
sampled: None,
paths: Vec::new(),
expected: vec!["n is stored as more than one type; read as i64"],
},
Shape {
what: "absent and conflicting",
files: vec![
file(&[("n", str.clone())], 10),
file(&[("n", i64.clone())], 90),
file(&[("id", i64.clone())], 5),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"id is in 1 of 3 files; absent from the rest, not null",
"n is in 2 of 3 files; absent from the rest, not null",
"n is str in 1 file; read as i64 and not read there",
],
},
Shape {
what: "absent and widened",
files: vec![
file(&[("id", i64.clone())], 1),
file(&[("id", i64.clone()), ("n", i32.clone())], 50),
file(&[("id", i64.clone()), ("n", i64.clone())], 50),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"n is in 2 of 3 files; absent from the rest, not null",
"n is stored as more than one type; read as i64",
],
},
Shape {
what: "conflicting and widened",
files: vec![
file(&[("n", i32.clone())], 50),
file(&[("n", i64.clone())], 50),
file(&[("n", str.clone())], 5),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"n is str in 1 file; read as i64 and not read there",
"n is stored as more than one type; read as i64",
],
},
Shape {
what: "a chosen type no file stores",
files: vec![
file(&[("n", i32.clone())], 50),
file(&[("n", DataType::Float32)], 50),
file(&[("n", str.clone())], 5),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"n is str in 1 file; read as f64 and not read there",
"n is stored as more than one type; read as f64",
],
},
Shape {
what: "two types the table spells the same way",
files: vec![
file(
&[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
10,
),
file(
&[(
"t",
DataType::Datetime(TimeUnit::Nanoseconds, Some(TimeZone::UTC)),
)],
90,
),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"t is datetime[ns] in 1 file; read as datetime[ns, UTC] and not read there",
],
},
Shape {
what: "two conflicting types the table spells the same way",
files: vec![
file(
&[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
10,
),
file(
&[(
"t",
DataType::Datetime(TimeUnit::Nanoseconds, Some(TimeZone::UTC)),
)],
10,
),
file(&[("t", str.clone())], 90),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"t is datetime[ns] or datetime[ns, UTC] in 2 files; read as str and not read there",
],
},
Shape {
what: "two structs, which Display also spells the same way",
files: vec![
file(
&[(
"s",
DataType::Struct(vec![Field::new("a".into(), i64.clone())]),
)],
90,
),
file(
&[(
"s",
DataType::Struct(vec![Field::new("a".into(), str.clone())]),
)],
10,
),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"s is Struct({'a': String}) in 1 file; \
read as Struct({'a': Int64}) and not read there",
],
},
Shape {
what: "a chosen type whose word another type shares",
files: vec![
file(
&[("t", DataType::Datetime(TimeUnit::Milliseconds, None))],
50,
),
file(
&[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
50,
),
file(&[("t", str.clone())], 5),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"t is str in 1 file; read as datetime[ns] and not read there",
"t is stored as more than one type; read as datetime[ns]",
],
},
Shape {
what: "widened between two datetime units",
files: vec![
file(
&[("t", DataType::Datetime(TimeUnit::Milliseconds, None))],
50,
),
file(
&[("t", DataType::Datetime(TimeUnit::Nanoseconds, None))],
50,
),
],
sampled: None,
paths: Vec::new(),
expected: vec!["t is stored as more than one type; read as datetime[ns]"],
},
Shape {
what: "a struct that both widens and conflicts",
files: vec![
file(
&[(
"s",
DataType::Struct(vec![Field::new("a".into(), i32.clone())]),
)],
50,
),
file(
&[(
"s",
DataType::Struct(vec![Field::new("a".into(), i64.clone())]),
)],
50,
),
file(
&[(
"s",
DataType::Struct(vec![Field::new("a".into(), str.clone())]),
)],
5,
),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"s is Struct({'a': String}) in 1 file; \
read as Struct({'a': Int64}) and not read there",
"s is stored as more than one type; read as Struct({'a': Int64})",
],
},
Shape {
what: "a decimal, whose precision the header word drops",
files: vec![
file(&[("d", DataType::Decimal(38, 2))], 90),
file(&[("d", str.clone())], 10),
],
sampled: None,
paths: Vec::new(),
expected: vec!["d is str in 1 file; read as decimal[38,2] and not read there"],
},
Shape {
what: "absent, conflicting and widened",
files: vec![
file(&[("id", i64.clone())], 1),
file(&[("id", i64.clone()), ("n", i32.clone())], 50),
file(&[("id", i64.clone()), ("n", i64.clone())], 50),
file(&[("id", i64.clone()), ("n", str.clone())], 5),
],
sampled: None,
paths: Vec::new(),
expected: vec![
"n is in 3 of 4 files; absent from the rest, not null",
"n is str in 1 file; read as i64 and not read there",
"n is stored as more than one type; read as i64",
],
},
];
for shape in &cases {
assert_eq!(notes_for(shape), shape.expected, "{}", shape.what);
}
for shape in &cases {
let dataset = dataset_for(shape);
let conflicting: Vec<&str> = dataset
.columns
.iter()
.filter(|column| column.conflicting_files > 0)
.map(|column| column.name.as_str())
.collect();
for name in conflicting {
assert!(
from_dataset(&dataset).iter().any(|note| {
note.summary.starts_with(&format!("{name} is "))
&& note.summary.ends_with("and not read there")
}),
"{}: {name} conflicts, so it says so on its own account",
shape.what
);
}
}
}
#[test]
fn types_that_print_alike_all_the_way_down_are_numbered() {
let names = distinct_names(&[&DataType::Int64, &DataType::Int64]);
assert_eq!(names, ["Int64 #1", "Int64 #2"]);
let mixed = distinct_names(&[&DataType::Int64, &DataType::Int64, &DataType::String]);
assert_eq!(mixed, ["Int64 #1", "Int64 #2", "String"]);
}
#[test]
fn a_uniform_dataset_has_nothing_to_say() {
let shape = Shape {
what: "uniform",
files: vec![
file(&[("id", DataType::Int64)], 1),
file(&[("id", DataType::Int64)], 1),
],
sampled: None,
paths: Vec::new(),
expected: vec![],
};
assert!(notes_for(&shape).is_empty());
}
#[test]
fn no_summary_claims_more_than_its_scope() {
let files = [
file(&[("id", DataType::Int64)], 1),
None,
file(&[("id", DataType::Int64), ("x", DataType::String)], 1),
];
for (origin, expected_scope) in [
(SchemaOrigin::AllFooters(3), "in all 3 footers"),
(
SchemaOrigin::FooterSample {
read: 3,
total: 200_000,
},
"in 3 of 200,000 footers (sample)",
),
] {
let dataset = union_file_schemas(&files, origin);
let notes = from_dataset(&dataset);
assert_eq!(notes.len(), 2, "an absence note and an unreadable one");
for note in notes {
assert_eq!(note.scope, expected_scope);
assert!(
!note.summary.contains("of 3 files"),
"one of those three said nothing: {}",
note.summary
);
}
}
}
#[test]
fn the_notes_about_one_column_do_not_talk_past_each_other() {
let files = [
file(&[("n", DataType::String)], 10),
file(&[("n", DataType::Int64)], 90),
file(&[("id", DataType::Int64)], 5),
];
let dataset = union_file_schemas(&files, SchemaOrigin::AllFooters(3));
let about_n: Vec<String> = from_dataset(&dataset)
.into_iter()
.map(|note| note.summary)
.filter(|summary| summary.starts_with('n'))
.collect();
assert_eq!(
about_n,
[
"n is in 2 of 3 files; absent from the rest, not null",
"n is str in 1 file; read as i64 and not read there",
]
);
}
}