use crate::formats::schema_union::{
ColumnDrift, ColumnRange, DatasetSchema, SchemaOrigin, SkippedFiles,
};
use crate::numfmt::group_chrome;
use polars::prelude::{DataType, PlSmallStr};
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Note {
pub summary: String,
pub scope: String,
pub read_as_text: Option<PlSmallStr>,
pub passed_over: Option<usize>,
}
fn sampled(dataset: &DatasetSchema) -> bool {
matches!(dataset.origin, SchemaOrigin::FooterSample { .. })
}
fn how_many(dataset: &DatasetSchema, n: usize) -> String {
let noun = match (sampled(dataset), n) {
(false, 1) => "file",
(false, _) => "files",
(true, 1) => "footer",
(true, _) => "footers",
};
format!("{} {noun}", group_chrome(n))
}
fn out_of(dataset: &DatasetSchema) -> String {
let readable = dataset.files.saturating_sub(dataset.unreadable.len());
match (sampled(dataset), dataset.unreadable.is_empty()) {
(false, true) => how_many(dataset, readable),
(false, false) => format!("the {} that could be read", how_many(dataset, readable)),
(true, true) => format!("the {} read", how_many(dataset, readable)),
(true, false) => format!("the {} that could be read", how_many(dataset, readable)),
}
}
fn distinct_names(types: &[&DataType]) -> Vec<String> {
let collides = |names: &[String]| {
names
.iter()
.enumerate()
.any(|(i, name)| names[i + 1..].contains(name))
};
let shown: Vec<String> = types.iter().map(|t| format!("{t}")).collect();
if !collides(&shown) {
return shown;
}
let spelled: Vec<String> = types.iter().map(|t| format!("{t:?}")).collect();
if !collides(&spelled) {
return spelled;
}
let mut seen: Vec<&String> = Vec::new();
spelled
.iter()
.map(|name| {
if spelled.iter().filter(|other| *other == name).count() > 1 {
seen.push(name);
format!("{name} #{}", seen.iter().filter(|s| **s == name).count())
} else {
name.clone()
}
})
.collect()
}
pub fn from_dataset(dataset: &DatasetSchema) -> Vec<Note> {
let scope = format!("in {}", dataset.origin);
let readable = dataset.files.saturating_sub(dataset.unreadable.len());
let denominator = out_of(dataset);
let mut notes = Vec::new();
for column in dataset.drifting() {
let mut types: Vec<&DataType> = vec![&column.dtype];
types.extend(column.conflicting_types.iter());
let names = distinct_names(&types);
let (chosen, others) = names.split_first().expect("the chosen type is first");
notes.extend(absence_note(
column,
readable,
&denominator,
dataset.column_ranges.get(&column.name),
&scope,
));
notes.extend(conflict_note(column, dataset, chosen, others, &scope));
notes.extend(widening_note(column, chosen, &scope));
}
for column in &dataset.read_as_text {
notes.push(text_note(column, &scope));
}
notes.extend(empty_files_note(dataset, &scope));
notes.extend(row_group_note(dataset, &scope));
notes.extend(small_files_note(dataset, &scope));
notes.extend(partition_layout_note(dataset));
notes.extend(skipped_files_note(dataset));
if !dataset.unreadable.is_empty() {
notes.push(Note {
summary: format!(
"{} unreadable, left out",
how_many(dataset, dataset.unreadable.len())
),
scope: scope.clone(),
read_as_text: None,
passed_over: None,
});
}
notes
}
pub fn left_out_note(
column: &ColumnDrift,
dataset: &DatasetSchema,
rows: usize,
filtered: bool,
sorted: bool,
) -> Note {
let what = match (filtered, sorted) {
(true, true) => "filter and sort",
(true, false) => "filter",
_ => "sort",
};
let there = if rows == 1 {
"1 row".to_string()
} else {
format!("{} rows", group_chrome(rows))
};
Note {
summary: format!(
"{}: {there} in {} left out of the {what}",
column.name,
how_many(dataset, column.conflicting_files)
),
scope: format!("in {}", dataset.origin),
read_as_text: None,
passed_over: None,
}
}
fn empty_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
let empty = dataset.empty_files;
if empty == 0 {
return None;
}
let (count, verb) = if empty == 1 {
("1 file".to_string(), "holds")
} else {
(format!("{} files", group_chrome(empty)), "hold")
};
Some(Note {
summary: format!("{count} {verb} no rows"),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
fn row_group_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
const BIG: usize = 64 * 1024 * 1024;
let median = dataset.median_row_group_bytes?;
if median <= BIG {
return None;
}
Some(Note {
summary: format!(
"median row group {}, each read whole",
crate::numfmt::bytes(median as u64)
),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
fn small_files_note(dataset: &DatasetSchema, scope: &str) -> Option<Note> {
const MANY: usize = 10_000;
const SMALL: usize = 1024 * 1024;
let files = dataset.origin.total_files();
let median = dataset.median_file_bytes?;
if median == 0 || files <= MANY || median >= SMALL {
return None;
}
let read = dataset.files;
let footers = if read == files {
"every footer".to_string()
} else {
format!("{} footers", group_chrome(read))
};
Some(Note {
summary: format!(
"{} files, median {}; {footers} read before any row",
group_chrome(files),
crate::numfmt::bytes(median as u64)
),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
fn partition_layout_note(dataset: &DatasetSchema) -> Option<Note> {
const NAMED: usize = 2;
if dataset.partition_layouts.len() < 2 {
return None;
}
let (named, rest) = dataset
.partition_layouts
.split_at(dataset.partition_layouts.len().min(NAMED));
let mut clauses: Vec<String> = named
.iter()
.map(|(keys, files)| format!("{} by {}", how_many_files(*files), keys.join("/")))
.collect();
let (dropped_ways, dropped_files) = dataset.partition_layouts_dropped;
let ways = rest.len() + dropped_ways;
let files: usize = rest.iter().map(|(_, files)| files).sum::<usize>() + dropped_files;
if ways > 0 {
clauses.push(format!(
"{} by {} other {}",
how_many_files(files),
group_chrome(ways),
if ways == 1 { "way" } else { "ways" }
));
}
Some(Note {
summary: format!("mixed partition keys: {}", clauses.join(", ")),
scope: format!(
"in the names of {} files",
group_chrome(dataset.listed_files)
),
read_as_text: None,
passed_over: None,
})
}
fn how_many_files(n: usize) -> String {
format!(
"{} {}",
group_chrome(n),
if n == 1 { "file" } else { "files" }
)
}
fn text_note(column: &PlSmallStr, scope: &str) -> Note {
Note {
summary: format!("{column} read as text: filter and sort compare text"),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
}
}
fn skipped_files_note(dataset: &DatasetSchema) -> Option<Note> {
note_about_skipped(dataset.skipped)
}
fn note_about_skipped(skipped: SkippedFiles) -> Option<Note> {
let SkippedFiles {
bookkeeping,
not_parquet,
empty,
} = skipped;
if not_parquet == 0 && empty == 0 {
return None;
}
let files = |n: usize| {
if n == 1 {
"1 file".to_string()
} else {
format!("{} files", group_chrome(n))
}
};
let mut said = Vec::new();
if empty > 0 {
let what = if empty == 1 { "file" } else { "files" };
said.push(format!("{} empty {what}", group_chrome(empty)));
}
if not_parquet > 0 {
said.push(format!("{} not Parquet", files(not_parquet)));
}
if bookkeeping > 0 {
let what = if bookkeeping == 1 { "file" } else { "files" };
said.push(format!(
"{} writer bookkeeping {what}",
group_chrome(bookkeeping)
));
}
Some(Note {
summary: format!("skipped: {}", said.join(", ")),
scope: "in this directory's listing".to_string(),
read_as_text: None,
passed_over: None,
})
}
const NAMES_SHOWN: usize = 3;
pub fn some_names<S: AsRef<str>>(names: &[S]) -> String {
let mut said: Vec<&str> = names.iter().take(NAMES_SHOWN).map(AsRef::as_ref).collect();
let ellipsis = crate::glyphs::get().ellipsis;
if names.len() > NAMES_SHOWN {
said.push(ellipsis);
}
said.join(", ")
}
pub fn no_header(files: &[&std::path::Path]) -> Option<Note> {
if files.is_empty() {
return None;
}
let names: Vec<String> = files
.iter()
.map(|f| {
f.file_name().map_or_else(
|| f.display().to_string(),
|n| n.to_string_lossy().into_owned(),
)
})
.collect();
let what = if files.len() == 1 { "file" } else { "files" };
Some(Note {
summary: format!(
"{} {what} with no header skipped: {}",
files.len(),
some_names(&names)
),
scope: "empty, blank, or only NUL padding".to_string(),
read_as_text: None,
passed_over: None,
})
}
pub fn from_the_open(
left_out: &[(crate::FileFormat, usize)],
lake: Option<&str>,
files_differ: crate::formats::schema_union::Disagreement,
names_look_like_data: bool,
) -> Vec<Note> {
let mut notes = Vec::new();
let scope = || "in a spread of this directory's files".to_string();
if files_differ.columns {
notes.push(Note {
summary: "columns differ across files: a missing column reads null".to_string(),
scope: scope(),
read_as_text: None,
passed_over: None,
});
}
if files_differ.headerless {
notes.push(Note {
summary: concat!(
"no header row? first row read as names: ",
"H on Schema, or --no-header, reads it as data"
)
.to_string(),
scope: scope(),
read_as_text: None,
passed_over: None,
});
}
if names_look_like_data && !files_differ.headerless {
notes.push(Note {
summary: "column names look like data: H on Schema reads them as a row".to_string(),
scope: "from the column names".to_string(),
read_as_text: None,
passed_over: None,
});
}
if files_differ.types {
notes.push(Note {
summary: "a column's type differs across files: read as the wider type".to_string(),
scope: scope(),
read_as_text: None,
passed_over: None,
});
}
if let Some(format) = lake {
notes.push(Note {
summary: format!(
"{format} table's files, not the table: deleted rows and old versions counted"
),
scope: format!("in this {format} table's directory"),
read_as_text: None,
passed_over: None,
});
}
if !left_out.is_empty() {
let said: Vec<String> = left_out
.iter()
.map(|(format, n)| format!("{n} {}", format.name()))
.collect();
notes.push(Note {
summary: format!(
"mixed formats, read as the commonest: {} not read",
said.join(", ")
),
scope: "in this directory's listing".to_string(),
read_as_text: None,
passed_over: Some(left_out.iter().map(|(_, n)| n).sum()),
});
}
notes
}
pub fn map_caches(count: usize) -> Option<Note> {
let files = if count == 1 { "file" } else { "files" };
(count > 0).then(|| Note {
summary: format!("{count} cache {files} written by map() not read"),
scope: "in this directory's listing".to_string(),
read_as_text: None,
passed_over: None,
})
}
pub fn merged(
open: &[Note],
dataset: &[Note],
view: &[Note],
schema: Option<&DatasetSchema>,
) -> Vec<Note> {
let rebuilt: Option<(Note, Option<Note>)> = match (
open.iter().find_map(|n| n.passed_over),
schema.map(|s| s.skipped),
) {
(Some(covered), Some(skipped)) => note_about_skipped(skipped).map(|full| {
let remaining = SkippedFiles {
not_parquet: skipped.not_parquet.saturating_sub(covered),
..skipped
};
(full, note_about_skipped(remaining))
}),
_ => None,
};
let mut out: Vec<Note> = open.to_vec();
for note in dataset {
match &rebuilt {
Some((full, reduced)) if note == full => out.extend(reduced.clone()),
_ => out.push(note.clone()),
}
}
out.extend(view.iter().cloned());
out
}
fn absence_note(
column: &ColumnDrift,
readable: usize,
denominator: &str,
range: Option<&ColumnRange>,
scope: &str,
) -> Option<Note> {
if column.present_in == 0 || column.present_in >= readable {
return None;
}
let where_it_is = match range {
Some(ColumnRange::Only(partition)) => format!(", only {partition}"),
Some(ColumnRange::NoneBefore(partition)) => format!(", none before {partition}"),
None => String::new(),
};
Some(Note {
summary: format!(
"{} is in {} of {}{}; absent from the rest, not null",
column.name,
group_chrome(column.present_in),
denominator,
where_it_is
),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
fn conflict_note(
column: &ColumnDrift,
dataset: &DatasetSchema,
chosen: &str,
others: &[String],
scope: &str,
) -> Option<Note> {
if column.conflicting_files == 0 {
return None;
}
Some(Note {
summary: format!(
"{} is {} in {}; read as {} and not read there",
column.name,
others.join(" or "),
how_many(dataset, column.conflicting_files),
chosen
),
scope: scope.to_string(),
read_as_text: column.can_read_as_text().then(|| column.name.clone()),
passed_over: None,
})
}
fn widening_note(column: &ColumnDrift, chosen: &str, scope: &str) -> Option<Note> {
if !column.widened {
return None;
}
Some(Note {
summary: format!(
"{} is stored as more than one type; read as {chosen}",
column.name
),
scope: scope.to_string(),
read_as_text: None,
passed_over: None,
})
}
#[cfg(test)]
mod tests;