use super::*;
use polars::prelude::*;
#[test]
fn how_a_row_is_read() {
use crate::ReadMode::*;
let how = |path: &str| how_read(&Entry::for_test(Path::new(path), path));
let at = |path: &str, mode, download| {
assert_eq!(how(path), Some(HowRead { mode, download }), "{path}");
};
at("/d/a.parquet", Lazy, false);
at("/d/a.csv", Lazy, false);
at("/d/a.csv.gz", Decompressed, false);
at("/d/a.json", InMemory, false);
at("/d/a.gpx", Converted, false);
at("/d/a.arrow", Lazy, false);
at("s3://b/a.parquet", Lazy, false);
at("s3://b/a.csv", Lazy, true);
at("gs://b/a.json", InMemory, true);
at("https://example.com/a.parquet", Lazy, true);
at("s3://b/m.safetensors", InMemory, false);
at("https://example.com/m.gguf", InMemory, false);
assert_eq!(how("/d/a.parquet.gz"), None, "does not open");
assert_eq!(how("/d/README"), None);
let mut stream = Entry::for_test(Path::new("/d/x.arrow"), "x.arrow");
stream.cost.ipc_stream = true;
assert_eq!(how_read(&stream).map(|h| h.mode), Some(Converted));
let mut spec = Entry::for_test(Path::new("/d/day.l2.zst"), "day.l2.zst");
spec.format_spec = Some("acme.l2feed".into());
assert_eq!(how_read(&spec).map(|h| h.mode), Some(Decompressed));
at("/d/shop.db", Lazy, false);
at("s3://b/shop.sqlite", Lazy, true);
let mut table = Entry::for_test(Path::new("/d/shop.db/orders"), "orders");
table.table = Some(TableOf {
format: Some(crate::FileFormat::Sqlite),
kind: "table".into(),
internal: false,
});
assert_eq!(how_read(&table).map(|h| h.mode), Some(Lazy));
assert_eq!(how_read(&Entry::directory(Path::new("/d/x"))), None);
}
#[test]
fn measuring_an_arrow_file_tells_a_stream() {
let dir = tempfile::tempdir().unwrap();
let file = dir.path().join("file.arrow");
std::fs::write(&file, b"ARROW1\0\0rest").unwrap();
let stream = dir.path().join("stream.arrow");
std::fs::write(&stream, b"\xff\xff\xff\xff\x10\x01\0\0").unwrap();
for (path, is_stream) in [(file, false), (stream, true)] {
let mut entry = Entry::for_test(&path, "x.arrow");
enrich(&mut entry);
assert_eq!(entry.cost.ipc_stream, is_stream, "{}", path.display());
}
}
#[test]
fn what_is_offered_and_what_opens_are_one_list() {
for ext in [
"parquet", "csv", "tsv", "psv", "json", "jsonl", "ndjson", "arrow", "arrows", "ipc",
"feather", "avro", "orc", "xls", "xlsx", "xlsm", "xlsb",
] {
let named = PathBuf::from(format!("sales.{ext}"));
assert!(
data_format(&named).is_some(),
".{ext} opens, so the home screen must offer it"
);
}
assert_eq!(
data_format(Path::new("README.txt")),
Some(crate::FileFormat::Text)
);
assert!(data_format(Path::new("notes")).is_none());
}
#[test]
fn a_format_name_round_trips_only_through_from_name() {
use crate::FileFormat;
for format in FileFormat::ALL {
assert_eq!(
FileFormat::from_name(format.name()),
Some(format),
"{} is a name",
format.name()
);
}
assert_eq!(FileFormat::from_extension("excel"), None);
assert_eq!(FileFormat::from_name("xlsx"), None);
}
#[test]
fn a_compressed_name_reads_as_the_format_under_it() {
assert_eq!(
data_format(Path::new("sales.csv.gz")),
Some(crate::FileFormat::Csv)
);
assert_eq!(
data_format(Path::new("events.json.zst")),
Some(crate::FileFormat::Json)
);
}
#[test]
fn one_format_under_several_names_is_not_a_mixture() {
let dir = tempfile::tempdir().unwrap();
std::fs::write(dir.path().join("a.arrow"), b"x").unwrap();
std::fs::write(dir.path().join("b.ipc"), b"x").unwrap();
assert_eq!(classify_directory(dir.path()), EntryKind::MultiFile);
}
#[test]
fn a_file_datui_does_not_read_does_not_disqualify_a_directory() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "a.parquet", &["id"]);
write(dir.path(), "b.parquet", &["id"]);
std::fs::write(dir.path().join("README.txt"), b"notes").unwrap();
#[cfg(feature = "cloud")]
let objects: Vec<(String, u64)> = [
("out/a.parquet", 100u64),
("out/b.parquet", 100),
("out/README.txt", 12),
]
.iter()
.map(|(k, s)| ((*k).to_string(), *s))
.collect();
#[cfg(feature = "cloud")]
assert_eq!(
classify_directory(dir.path()),
crate::cloud::cloud_browse::look_at_listing("out/", &[], &objects).0,
"the two routes answer the same directory alike"
);
assert_eq!(classify_directory(dir.path()), EntryKind::MultiFile);
}
#[test]
fn a_label_counts_what_is_there_rather_than_naming_a_decision() {
let dir = tempfile::tempdir().unwrap();
for name in ["a.parquet", "b.parquet", "c.parquet"] {
write(dir.path(), name, &["id"]);
}
std::fs::create_dir_all(dir.path().join("archive")).unwrap();
std::fs::write(dir.path().join("notes.csv"), b"x").unwrap();
std::fs::write(dir.path().join("_SUCCESS"), b"").unwrap();
std::fs::write(dir.path().join(".part.crc"), b"").unwrap();
let entry = measured(dir.path());
assert_eq!(entry.label(), "mixed", "two formats is two formats");
assert_eq!(
entry.holds.line(true).as_deref(),
Some("3 parquet · 1 csv · 1 directory"),
"and the pane says what the label boiled down"
);
assert_eq!(entry.holds.data_files(), 4);
assert_eq!(entry.holds.directories, 1);
}
#[test]
fn a_directory_of_one_format_is_labelled_by_it() {
let dir = tempfile::tempdir().unwrap();
for i in 0..12 {
write(dir.path(), &format!("part-{i:05}.parquet"), &["id", "ts"]);
}
let entry = measured(dir.path());
assert_eq!(entry.label(), "12 parquet");
assert_eq!(entry.holds.line(true).as_deref(), Some("12 parquet"));
let plain = tempfile::tempdir().unwrap();
for i in 0..20 {
std::fs::write(plain.path().join(format!("note{i}.md")), b"x").unwrap();
}
let plain = measured(plain.path());
assert_eq!(plain.label(), "dir");
assert_eq!(plain.holds.not_read, 20);
assert_eq!(plain.holds.line(true), None);
}
#[test]
fn a_kind_that_names_itself_keeps_its_name() {
let dir = tempfile::tempdir().unwrap();
std::fs::create_dir_all(dir.path().join("year=2024")).unwrap();
std::fs::create_dir_all(dir.path().join("year=2025")).unwrap();
assert_eq!(measured(dir.path()).label(), "hive");
let lake = tempfile::tempdir().unwrap();
std::fs::create_dir_all(lake.path().join("_delta_log")).unwrap();
assert_eq!(measured(lake.path()).label(), "delta");
let unlooked = Entry::new(PathBuf::from("/nowhere"), EntryKind::Unknown);
assert_eq!(unlooked.label(), "");
}
#[test]
fn a_directory_read_whole_does_not_claim_there_is_more() {
let dir = tempfile::tempdir().unwrap();
for i in 0..MAX_ENTRIES_PER_DIR {
std::fs::write(dir.path().join(format!("f{i:05}.csv")), b"x").unwrap();
}
let holds = look_at_directory(dir.path()).1;
assert!(!holds.truncated, "every entry was read");
assert_eq!(holds.label(), format!("{MAX_ENTRIES_PER_DIR} csv"));
std::fs::write(dir.path().join("one-more.csv"), b"x").unwrap();
let holds = look_at_directory(dir.path()).1;
assert!(holds.truncated, "and now there is more than was read");
assert!(holds.label().contains('+'));
}
#[test]
fn a_lake_table_is_not_counted() {
let dir = tempfile::tempdir().unwrap();
std::fs::create_dir_all(dir.path().join("_delta_log")).unwrap();
write(dir.path(), "part-00000.parquet", &["id"]);
write(dir.path(), "part-00001.parquet", &["id"]);
let (kind, holds) = look_at_directory(dir.path());
assert_eq!(kind, EntryKind::Delta);
assert!(holds.is_empty(), "and its label is the format's own name");
let entry = measured(dir.path());
assert_eq!(entry.label(), "delta");
}
#[test]
fn a_cut_short_listing_does_not_claim_a_directory_is_empty() {
let seen = Holds {
skipped: 5000,
truncated: true,
..Default::default()
};
assert_eq!(seen.label(), "dir+");
let whole = Holds {
skipped: 3,
..Default::default()
};
assert_eq!(whole.label(), "dir");
let nothing_yet = Holds {
truncated: true,
..Default::default()
};
assert!(!nothing_yet.is_empty());
assert_eq!(nothing_yet.label(), "dir+");
assert!(Holds::default().is_empty());
let mixed = Holds {
formats: vec![("parquet".to_string(), 3), ("csv".to_string(), 2)],
truncated: true,
..Default::default()
};
assert_eq!(mixed.label(), "mixed", "more files cannot unmake it");
}
#[test]
fn a_long_name_keeps_both_ends() {
let name = ".part-00000-8f3a91c2-7b4d-4e19-a6f0-c1d2e3f4a5b6-c000.snappy.parquet.crc";
let line = crate::glyphs::fit_middle(name, 24);
assert!(line.starts_with(".part-00000"), "the head: {line}");
assert!(line.ends_with(".crc"), "and the tail: {line}");
assert!(!line.contains("8f3a91c2"), "the middle goes: {line}");
assert!(line.chars().count() <= 24, "{line}");
}
#[test]
fn what_a_directory_holds_reads_the_same_twice() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "a.parquet", &["id"]);
for marker in [
"_SUCCESS",
"_committed_9",
"_committed_1",
".crc",
"_started_4",
] {
std::fs::write(dir.path().join(marker), b"").unwrap();
}
let first = look_at_directory(dir.path()).1;
for _ in 0..8 {
assert_eq!(look_at_directory(dir.path()).1, first);
}
assert_eq!(
first.skipped_names,
vec![".crc", "_SUCCESS", "_committed_1", "_committed_9"],
"the first four by name, of five"
);
assert_eq!(first.skipped, 5);
}
#[test]
fn a_directories_numbers_are_what_opening_it_gives() {
let dir = tempfile::tempdir().unwrap();
for name in ["a.parquet", "b.parquet", "c.parquet"] {
write(dir.path(), name, &["id", "legacy"]);
}
let archive = dir.path().join("archive");
std::fs::create_dir_all(&archive).unwrap();
for i in 0..20 {
write(&archive, &format!("old-{i}.parquet"), &["id", "legacy"]);
}
let entry = measured(dir.path());
assert_eq!(
entry.label(),
"3 parquet",
"three files are directly inside"
);
assert_eq!(
entry.holds.line(true).as_deref(),
Some("3 parquet · 1 directory")
);
assert_eq!(entry.rows, Some(23), "and opening it reads all of them");
}
#[test]
fn a_width_is_a_floor_only_when_a_footer_went_unread() {
let dir = tempfile::tempdir().unwrap();
for name in ["a.parquet", "b.parquet", "c.parquet"] {
write(dir.path(), name, &["id", "ts"]);
}
let archive = dir.path().join("archive");
std::fs::create_dir_all(&archive).unwrap();
for i in 0..MAX_FOOTERS_PER_DATASET + 6 {
write(
&archive,
&format!("old-{i:03}.parquet"),
&["wholly", "different"],
);
}
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::Directory, "not one table");
assert_eq!(entry.label(), "3 parquet");
assert_eq!(entry.cols, Some(2), "id and ts");
assert!(
!entry.cols_sampled,
"all three of its own footers were read"
);
}
#[test]
fn a_big_directory_is_still_found_by_a_column_one_level_down() {
let dir = tempfile::tempdir().unwrap();
for name in ["a.parquet", "b.parquet", "c.parquet"] {
write(dir.path(), name, &["id", "ts"]);
}
let archive = dir.path().join("archive");
std::fs::create_dir_all(&archive).unwrap();
for i in 0..MAX_FOOTERS_PER_DATASET + 6 {
write(
&archive,
&format!("old-{i:03}.parquet"),
&["wholly", "different"],
);
}
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::Directory);
assert_eq!(entry.label(), "3 parquet");
assert_eq!(entry.cols, Some(2), "id and ts");
assert!(
entry.columns.contains(&"wholly".to_string()),
"{:?}",
entry.columns
);
assert!(entry.columns.contains(&"id".to_string()));
}
#[test]
fn a_width_over_a_directories_own_files_is_a_floor_when_there_are_too_many() {
let dir = tempfile::tempdir().unwrap();
for i in 0..MAX_FOOTERS_PER_DATASET + 6 {
write(
dir.path(),
&format!("f-{i:03}.parquet"),
&[&format!("c{i}")],
);
}
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::Directory, "not one table");
assert_eq!(entry.label(), "70 parquet");
assert!(
entry.cols_sampled,
"three of seventy footers were read, so the width is a floor"
);
}
#[test]
fn a_directory_read_as_one_table_is_sized_by_everything_under_it() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "a.parquet", &["id", "ts"]);
write(dir.path(), "b.parquet", &["id", "ts"]);
let more = dir.path().join("more");
std::fs::create_dir_all(&more).unwrap();
write(&more, "c.parquet", &["id", "ts"]);
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::MultiFile, "one table");
let all: u64 = [
dir.path().join("a.parquet"),
dir.path().join("b.parquet"),
more.join("c.parquet"),
]
.iter()
.map(|p| std::fs::metadata(p).unwrap().len())
.sum();
assert_eq!(entry.size, Some(all));
assert_eq!(entry.rows, Some(3));
}
#[test]
fn nothing_counted_is_the_only_thing_holds_calls_empty() {
assert!(Holds::default().is_empty());
let one = |f: fn(&mut Holds)| {
let mut h = Holds::default();
f(&mut h);
h
};
for (what, holds) in [
("a data file", one(|h| h.formats.push(("csv".into(), 1)))),
("a directory", one(|h| h.directories = 1)),
("a partition", one(|h| h.partitions = 1)),
("a file it cannot read", one(|h| h.not_read = 1)),
("a writer's own file", one(|h| h.skipped = 1)),
(
"the name of one",
one(|h| h.skipped_names.push("_SUCCESS".into())),
),
("a listing cut short", one(|h| h.truncated = true)),
] {
assert!(!holds.is_empty(), "{what} is something to say");
}
}
#[test]
fn formats_that_tie_are_ordered_by_name_whatever_order_they_arrived_in() {
use crate::FileFormat;
let mut counts = vec![
(FileFormat::Json, 2),
(FileFormat::Csv, 2),
(FileFormat::Parquet, 5),
];
order_formats(&mut counts);
assert_eq!(
counts,
vec![
(FileFormat::Parquet, 5),
(FileFormat::Csv, 2),
(FileFormat::Json, 2)
]
);
}
#[test]
fn the_label_and_the_read_pick_the_same_format_on_a_tie() {
use crate::FileFormat;
let tmp = tempfile::TempDir::new().unwrap();
for name in ["a.csv", "b.csv", "c.parquet", "d.parquet"] {
std::fs::write(tmp.path().join(name), b"x").unwrap();
}
let (_, holds) = look_at_directory(tmp.path());
assert_eq!(
holds.formats.first().map(|(f, n)| (f.as_str(), *n)),
Some(("parquet", 2)),
"the label names Parquet first: {:?}",
holds.formats
);
match directory_format(tmp.path()) {
DirectoryFormat::Mixed { format, .. } => assert_eq!(
format,
FileFormat::Parquet,
"and so does the reader the open picks"
),
other => panic!("a directory of two formats is mixed, got {other:?}"),
}
}
#[test]
fn a_hugging_face_dataset_is_its_shards() {
use crate::FileFormat;
let tmp = tempfile::TempDir::new().unwrap();
for name in [
"data-00000-of-00002.arrow",
"data-00001-of-00002.arrow",
"dataset_info.json",
"state.json",
] {
std::fs::write(tmp.path().join(name), b"x").unwrap();
}
let (kind, holds) = look_at_directory(tmp.path());
assert_eq!(kind, EntryKind::MultiFile);
assert_eq!(holds.formats, [("arrow".to_string(), 2)]);
assert_eq!(holds.skipped, 2);
assert_eq!(holds.skipped_names, ["dataset_info.json", "state.json"]);
match directory_format(tmp.path()) {
DirectoryFormat::One(FileFormat::Arrow, files) => assert_eq!(files.len(), 2),
other => panic!("the shards are the dataset, got {other:?}"),
}
std::fs::remove_file(tmp.path().join("data-00001-of-00002.arrow")).unwrap();
assert!(matches!(
directory_format(tmp.path()),
DirectoryFormat::One(FileFormat::Arrow, _)
));
let json = tempfile::TempDir::new().unwrap();
for name in ["state.json", "other.json"] {
std::fs::write(json.path().join(name), b"{}").unwrap();
}
assert_eq!(
look_at_directory(json.path()).1.formats,
[("json".to_string(), 2)]
);
}
#[test]
fn partitions_carry_a_directory_only_while_they_are_the_most_of_it() {
let laid_out = |strays: usize| {
let dir = tempfile::tempdir().unwrap();
for year in ["year=2024", "year=2025"] {
let part = dir.path().join(year);
std::fs::create_dir_all(&part).unwrap();
write(&part, "data.parquet", &["id"]);
}
for i in 0..strays {
write(dir.path(), &format!("stray-{i}.parquet"), &["id"]);
}
classify_directory(dir.path())
};
assert_eq!(
laid_out(2),
EntryKind::Hive,
"two partitions against two files beside them"
);
assert_ne!(
laid_out(3),
EntryKind::Hive,
"one more file than partitions is a directory that holds a key=value"
);
}
#[test]
fn a_folder_marker_is_bookkeeping_even_beside_a_partition() {
assert!(is_bookkeeping("year=2024_$folder$"));
assert!(is_bookkeeping("alpha_$folder$"));
assert!(!is_bookkeeping("year=2024"), "the partition itself is data");
assert!(
!is_bookkeeping("_date=2024-01-01"),
"Spark partitions on internal columns"
);
}
#[test]
fn a_table_hidden_under_a_directory_still_downgrades_it() {
let dir = tempfile::tempdir().unwrap();
for name in ["a.parquet", "b.parquet"] {
write(dir.path(), name, &["id", "ts"]);
}
let archive = dir.path().join("archive");
std::fs::create_dir_all(&archive).unwrap();
write(
&archive,
"other.parquet",
&["wholly", "different", "columns"],
);
let entry = measured(dir.path());
assert_eq!(
entry.kind,
EntryKind::Directory,
"a union over these is not one table"
);
assert_eq!(entry.rows, None);
assert_eq!(entry.label(), "2 parquet");
assert_eq!(entry.cols, Some(2), "id and ts");
assert!(entry.columns.contains(&"wholly".to_string()));
assert!(entry.columns.contains(&"id".to_string()));
let own: u64 = ["a.parquet", "b.parquet"]
.iter()
.map(|n| std::fs::metadata(dir.path().join(n)).unwrap().len())
.sum();
assert_eq!(entry.size, Some(own));
}
#[test]
fn a_hive_tree_of_another_format_is_still_laid_out() {
let dir = tempfile::tempdir().unwrap();
for year in ["year=2024", "year=2025"] {
let part = dir.path().join(year);
std::fs::create_dir_all(&part).unwrap();
std::fs::write(part.join("data.csv"), b"id\n1\n").unwrap();
}
std::fs::write(dir.path().join("summary.csv"), b"id\n1\n").unwrap();
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::Hive);
assert!(entry.cost.partitions.is_some(), "the layout is named");
assert_eq!(entry.rows, None, "and nothing is invented about its rows");
}
#[test]
fn a_hive_dataset_is_described_despite_a_stray_file_at_its_root() {
let dir = tempfile::tempdir().unwrap();
for year in ["year=2024", "year=2025"] {
let part = dir.path().join(year);
std::fs::create_dir_all(&part).unwrap();
write(&part, "data.parquet", &["id"]);
}
std::fs::write(dir.path().join("schema.json"), b"{}").unwrap();
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::Hive);
assert_eq!(entry.holds.one_format(), Some("json"), "its own only file");
assert_eq!(entry.rows, Some(2), "and the dataset is still counted");
assert_eq!(entry.cols, Some(2), "`id` and the partition column `year`");
}
#[test]
fn a_hive_dataset_is_described_despite_a_stray_file() {
let dir = tempfile::tempdir().unwrap();
for year in ["year=2024", "year=2025"] {
let part = dir.path().join(year);
std::fs::create_dir_all(&part).unwrap();
write(&part, "data.parquet", &["id"]);
}
std::fs::write(dir.path().join("year=2024/notes.csv"), b"x").unwrap();
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::Hive);
assert_eq!(entry.rows, Some(2), "the dataset is still counted");
assert!(
entry.cost.partitions.is_some(),
"and its layout still named"
);
}
#[test]
fn a_directory_is_not_described_by_files_it_does_not_name() {
let dir = tempfile::tempdir().unwrap();
for name in ["a.json", "b.json", "c.json"] {
std::fs::write(dir.path().join(name), b"{}").unwrap();
}
let under = dir.path().join("derived");
std::fs::create_dir_all(&under).unwrap();
write(&under, "one.parquet", &["id", "ts", "amount"]);
write(&under, "two.parquet", &["id", "ts", "amount"]);
let entry = measured(dir.path());
assert_eq!(entry.label(), "3 json");
assert_eq!(
entry.cols, None,
"the Parquet below it is not this directory's shape"
);
assert_eq!(entry.rows, None);
assert!(entry.columns.is_empty());
}
#[test]
fn a_dataset_row_that_counted_nothing_is_still_described() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "a.parquet", &["id", "ts"]);
write(dir.path(), "b.parquet", &["id", "ts"]);
let mut entry = Entry {
kind: EntryKind::MultiFile,
..Entry::for_test(dir.path(), "data")
};
assert!(entry.holds.one_format().is_none(), "nothing counted");
enrich(&mut entry);
assert!(entry.size.is_some(), "the footers were read");
assert_eq!(entry.rows, Some(2));
assert_eq!(entry.cols, Some(2), "id and ts");
}
#[test]
fn a_parquet_file_named_like_a_writers_file_is_still_measured() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "_2024_sales.parquet", &["id", "amount"]);
let mut entry = Entry::new(dir.path().join("_2024_sales.parquet"), EntryKind::File);
enrich(&mut entry);
assert_eq!(entry.rows, Some(1), "its footer was read");
assert_eq!(entry.cols, Some(2));
assert!(schema_preview(&entry).is_some(), "and the pane shows it");
assert!(is_bookkeeping("_2024_sales.parquet"));
assert_eq!(classify_directory(dir.path()), EntryKind::Directory);
}
#[test]
fn a_hive_preview_types_its_keys_as_the_scan_does() {
use polars::prelude::DataType;
let dir = tempfile::tempdir().unwrap();
let leaf = dir.path().join("day=2024-01-02/flag=true/n=3/x=1.5");
std::fs::create_dir_all(&leaf).unwrap();
write(&leaf, "part.parquet", &["id"]);
let mut entry = Entry::directory(dir.path());
entry.kind = EntryKind::Hive;
let preview = schema_preview(&entry).expect("a footer to read");
let types: Vec<(&str, &DataType)> = preview.iter().map(|(n, t)| (n.as_str(), t)).collect();
assert_eq!(
types[..4],
[
("day", &DataType::Date),
("flag", &DataType::Boolean),
("n", &DataType::Int64),
("x", &DataType::Float64),
]
);
assert_eq!(types[4].0, "id");
}
#[cfg(feature = "cloud")]
#[test]
fn extensionless_part_files_are_data_on_both_routes() {
let dir = tempfile::tempdir().unwrap();
let table = dir.path().join("occurrence.parquet");
std::fs::create_dir_all(&table).unwrap();
std::fs::write(table.join("000001"), b"PAR1").unwrap();
std::fs::write(table.join("000002"), b"PAR1").unwrap();
let objects: Vec<(String, u64)> = [
"gbif/occurrence.parquet/000001",
"gbif/occurrence.parquet/000002",
]
.iter()
.map(|k| ((*k).to_string(), 10u64))
.collect();
assert_eq!(
classify_directory(&table),
crate::cloud::cloud_browse::look_at_listing("gbif/occurrence.parquet/", &[], &objects).0,
"the two routes answer the same directory alike"
);
assert_eq!(classify_directory(&table), EntryKind::MultiFile);
}
#[test]
fn extensionless_part_files_are_measured_not_just_offered() {
let dir = tempfile::tempdir().unwrap();
let table = dir.path().join("occurrence.parquet");
std::fs::create_dir_all(&table).unwrap();
write(&table, "000001", &["id", "species"]);
write(&table, "000002", &["id", "species"]);
let entry = measured(&table);
assert_eq!(entry.kind, EntryKind::MultiFile);
assert_eq!(entry.rows, Some(2), "both footers were read");
assert_eq!(entry.cols, Some(2));
assert!(
schema_preview(&entry).is_some(),
"and the schema pane shows what those footers said, rather than asking \
for a full read of files already read"
);
let mut listed = scan_dir_progressive(&table, |_| {}).entries;
assert_eq!(
listed.iter().map(|e| e.name.as_str()).collect::<Vec<_>>(),
vec!["000001", "000002"],
"the directory the label promises is not an empty listing"
);
let part = listed.first_mut().expect("a part file is listed");
enrich(part);
assert_eq!(part.rows, Some(1), "a part file counts its own rows");
assert_eq!(part.cols, Some(2));
}
#[cfg(feature = "cloud")]
#[test]
fn one_partition_beside_files_datui_cannot_read_answers_alike() {
let dir = tempfile::tempdir().unwrap();
std::fs::create_dir_all(dir.path().join("notes=old")).unwrap();
for note in ["README.md", "LICENSE", "logo.png"] {
std::fs::write(dir.path().join(note), b"x").unwrap();
}
let objects: Vec<(String, u64)> = ["out/README.md", "out/LICENSE", "out/logo.png"]
.iter()
.map(|k| ((*k).to_string(), 12u64))
.collect();
assert_eq!(
classify_directory(dir.path()),
crate::cloud::cloud_browse::look_at_listing(
"out/",
&["out/notes=old/".to_string()],
&objects
)
.0,
"the two routes answer the same directory alike"
);
}
#[test]
fn a_partition_named_like_a_writers_file_is_still_a_partition() {
let dir = tempfile::tempdir().unwrap();
let mut directories = Vec::new();
for day in ["2024-01-01", "2024-01-02", "2024-01-03"] {
std::fs::create_dir_all(dir.path().join(format!("_date={day}"))).unwrap();
directories.push(format!("events/_date={day}/"));
}
#[cfg(feature = "cloud")]
assert_eq!(
classify_directory(dir.path()),
crate::cloud::cloud_browse::look_at_listing("events/", &directories, &[]).0,
"the two routes answer the same directory alike"
);
assert_eq!(classify_directory(dir.path()), EntryKind::Hive);
assert!(!is_bookkeeping("_date=2024-01-01"));
assert!(is_bookkeeping("_temporary"));
}
#[cfg(feature = "cloud")]
#[test]
fn a_writers_own_directory_is_skipped_on_both_routes() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "part-00000.parquet", &["id"]);
write(dir.path(), "part-00001.parquet", &["id"]);
std::fs::create_dir_all(dir.path().join("_temporary")).unwrap();
std::fs::create_dir_all(dir.path().join("notes")).unwrap();
std::fs::create_dir_all(dir.path().join("archive")).unwrap();
let local = classify_directory(dir.path());
let directories: Vec<String> = ["out/_temporary/", "out/notes/", "out/archive/"]
.iter()
.map(|f| (*f).to_string())
.collect();
let objects: Vec<(String, u64)> = [
("out/part-00000.parquet", 100u64),
("out/part-00001.parquet", 100),
]
.iter()
.map(|(k, s)| ((*k).to_string(), *s))
.collect();
let cloud = crate::cloud::cloud_browse::look_at_listing("out/", &directories, &objects).0;
assert_eq!(
local, cloud,
"the two routes answer the same directory alike"
);
assert_eq!(local, EntryKind::MultiFile);
}
#[cfg(unix)]
#[test]
fn every_entry_is_in_one_count() {
let dir = tempfile::tempdir().unwrap();
std::fs::write(dir.path().join("real.csv"), b"id\n1\n").unwrap();
std::os::unix::fs::symlink(dir.path().join("gone"), dir.path().join("broken.csv")).unwrap();
std::fs::write(dir.path().join("notes.md"), b"x").unwrap();
std::fs::write(dir.path().join("_SUCCESS"), b"").unwrap();
let holds = look_at_directory(dir.path()).1;
assert_eq!(holds.data_files(), 1);
assert_eq!(holds.not_read, 2, "the note and the broken link");
assert_eq!(holds.skipped, 1);
assert_eq!(holds.line(true).as_deref(), Some("1 csv"));
}
#[cfg(unix)]
#[test]
fn a_name_with_nothing_behind_it_is_not_a_data_file() {
let dir = tempfile::tempdir().unwrap();
std::os::unix::fs::symlink(dir.path().join("gone.csv"), dir.path().join("a.csv")).unwrap();
std::os::unix::fs::symlink(dir.path().join("gone.csv"), dir.path().join("b.csv")).unwrap();
assert_eq!(
classify_directory(dir.path()),
EntryKind::Directory,
"two broken symlinks are not a dataset"
);
}
#[test]
fn a_model_directory_is_its_weights() {
let dir = tempfile::tempdir().unwrap();
for name in [
"model-00001-of-00002.safetensors",
"model-00002-of-00002.safetensors",
"config.json",
"generation_config.json",
"tokenizer.json",
"tokenizer_config.json",
"model.safetensors.index.json",
] {
std::fs::write(dir.path().join(name), b"x").unwrap();
}
let DirectoryFormat::Mixed {
format,
files,
passed_over,
} = directory_format(dir.path())
else {
panic!("weights and JSON are two formats");
};
assert_eq!(format, crate::FileFormat::Safetensors);
assert_eq!(files.len(), 3, "the shards, and the index for its metadata");
assert_eq!(passed_over, [(crate::FileFormat::Json, 4)]);
let (kind, holds) = look_at_directory(dir.path());
assert_eq!(kind, EntryKind::MultiFile, "opened as one");
assert_eq!(holds.label(), "2 safetensors", "the shards, not the index");
assert_eq!(
data_format(Path::new("model.safetensors.index.json")),
Some(crate::FileFormat::Safetensors)
);
std::fs::write(dir.path().join("data.parquet"), b"x").unwrap();
let (kind, holds) = look_at_directory(dir.path());
assert_eq!(
(kind, holds.label().as_str()),
(EntryKind::Directory, "mixed")
);
}
#[test]
fn signed_files_are_sniffed_by_their_first_bytes() {
let dir = tempfile::tempdir().unwrap();
let gguf = dir.path().join("weights");
std::fs::write(&gguf, b"GGUF\x03\x00\x00\x00").unwrap();
let st = dir.path().join("checkpoint.bin");
let mut bytes = 2u64.to_le_bytes().to_vec();
bytes.extend_from_slice(b"{}");
std::fs::write(&st, &bytes).unwrap();
let text = dir.path().join("notes");
std::fs::write(&text, b"just some text").unwrap();
assert_eq!(sniff_format(&gguf), Some(crate::FileFormat::Gguf));
let opened = |path: &Path| crate::formats::readers::sniff_open(path, None);
assert_eq!(opened(&st), Some(crate::FileFormat::Safetensors));
assert_eq!(opened(&text), None);
let midi = dir.path().join("song.bin");
std::fs::write(&midi, b"MThd\0\0\0\x06\0\0\0\x01\0\x60").unwrap();
assert_eq!(opened(&midi), Some(crate::FileFormat::Midi));
}
#[test]
fn a_format_that_cannot_be_read_as_many_is_not_offered_as_one() {
for ext in ["tsv", "psv", "xlsx", "xlsb"] {
let dir = tempfile::tempdir().unwrap();
std::fs::write(dir.path().join(format!("a.{ext}")), b"x").unwrap();
std::fs::write(dir.path().join(format!("b.{ext}")), b"x").unwrap();
assert_eq!(
classify_directory(dir.path()),
EntryKind::Directory,
"a directory of .{ext} has no reader that takes a list"
);
}
for ext in [
"parquet", "csv", "json", "jsonl", "ndjson", "arrow", "arrows", "ipc", "feather", "avro",
"orc",
] {
let dir = tempfile::tempdir().unwrap();
std::fs::write(dir.path().join(format!("a.{ext}")), b"x").unwrap();
std::fs::write(dir.path().join(format!("b.{ext}")), b"x").unwrap();
assert_eq!(
classify_directory(dir.path()),
EntryKind::MultiFile,
".{ext} reads as many files"
);
}
}
#[cfg(feature = "cloud")]
#[test]
fn a_writers_own_file_is_skipped_whatever_order_it_is_listed_in() {
let dir = tempfile::tempdir().unwrap();
for part in 0..8 {
write(dir.path(), &format!("{part}.parquet"), &["season"]);
}
std::fs::write(dir.path().join("_metadata.json"), b"{}").unwrap();
let local = classify_directory(dir.path());
let mut keys: Vec<(String, u64)> = vec![("jolpica/2000/_metadata.json".into(), 2)];
for part in 0..8 {
keys.push((format!("jolpica/2000/{part}.parquet"), 100));
}
keys.sort();
let cloud = crate::cloud::cloud_browse::look_at_listing("jolpica/2000/", &[], &keys).0;
assert_eq!(
local, cloud,
"the two routes answer the same directory alike"
);
assert_eq!(local, EntryKind::MultiFile);
}
#[test]
fn job_files_are_skipped_on_every_route() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "part-00000.parquet", &["id"]);
write(dir.path(), "part-00001.parquet", &["id"]);
for marker in [
"_SUCCESS",
"_committed_1727",
"_committed_1728",
"_started_1727",
".part.crc",
] {
std::fs::write(dir.path().join(marker), b"").unwrap();
}
assert_eq!(
classify_directory(dir.path()),
EntryKind::MultiFile,
"five markers beside two data files do not outvote them"
);
#[cfg(feature = "cloud")]
let keys: Vec<(String, u64)> = [
("out/_SUCCESS", 0u64),
("out/_committed_1727", 12),
("out/_committed_1728", 12),
("out/_started_1727", 12),
("out/.part.crc", 8),
("out/part-00000.parquet", 100),
("out/part-00001.parquet", 100),
]
.iter()
.map(|(k, s)| ((*k).to_string(), *s))
.collect();
#[cfg(feature = "cloud")]
assert_eq!(
crate::cloud::cloud_browse::look_at_listing("out/", &[], &keys).0,
EntryKind::MultiFile,
"and the same in a bucket"
);
}
fn write(dir: &Path, name: &str, columns: &[&str]) {
let mut frame = DataFrame::new(
1,
columns
.iter()
.map(|c| Column::new((*c).into(), &[1i32]))
.collect::<Vec<_>>(),
)
.unwrap();
let file = std::fs::File::create(dir.join(name)).unwrap();
ParquetWriter::new(file).finish(&mut frame).unwrap();
}
fn write_nested(dir: &Path, name: &str, struct_name: &str, fields: &[&str]) {
let inner = DataFrame::new(
1,
fields
.iter()
.map(|f| Column::new((*f).into(), &[1i32]))
.collect::<Vec<_>>(),
)
.unwrap();
let nested = inner
.into_struct(struct_name.into())
.into_series()
.into_column();
let mut frame = DataFrame::new(1, vec![Column::new("id".into(), &[1i32]), nested]).unwrap();
let file = std::fs::File::create(dir.join(name)).unwrap();
ParquetWriter::new(file).finish(&mut frame).unwrap();
}
fn measured(dir: &Path) -> Entry {
let (kind, holds) = look_at_directory(dir);
let mut entry = Entry::new(dir.to_path_buf(), kind);
entry.holds = holds;
enrich(&mut entry);
entry
}
#[test]
fn holds_written_as_folders_still_reads() {
let old: Holds = serde_json::from_str(r#"{"folders":3,"partitions":2}"#).unwrap();
assert_eq!((old.directories, old.partitions), (3, 2));
let new = serde_json::to_string(&old).unwrap();
assert!(new.contains(r#""directories":3"#), "{new}");
}
#[test]
fn a_directory_of_separate_tables_is_not_a_dataset() {
let dir = tempfile::tempdir().unwrap();
write(
dir.path(),
"circuits.parquet",
&["circuit_id", "lat", "lng"],
);
write(
dir.path(),
"drivers.parquet",
&["driver_id", "code", "nationality"],
);
write(
dir.path(),
"laps.parquet",
&["lap", "position", "time_millis"],
);
assert_eq!(
classify_directory(dir.path()),
EntryKind::MultiFile,
"the filenames alone still say multi"
);
let entry = measured(dir.path());
assert_eq!(
entry.kind,
EntryKind::Directory,
"reading the footers says otherwise"
);
assert_eq!(
entry.rows, None,
"a sum across separate tables is not a row count"
);
assert_eq!(
entry.cols,
Some(9),
"the union of what the directory holds is still a true answer to what is in it"
);
assert_eq!(entry.label(), "3 parquet", "and the label counts the files");
}
#[test]
fn a_directory_whose_files_each_bring_a_column_is_a_place_to_look_inside() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "old.parquet", &["id", "ts", "amount"]);
write(dir.path(), "new.parquet", &["id", "ts", "amt"]);
assert_eq!(
classify_directory(dir.path()),
EntryKind::MultiFile,
"the names alone still say two Parquet files"
);
let entry = measured(dir.path());
assert_eq!(
entry.kind,
EntryKind::Directory,
"and the footers say neither file's columns are in the other's"
);
assert_eq!(entry.label(), "2 parquet", "which the label still reports");
assert_eq!(entry.rows, None, "a sum over two tables is not a number");
}
#[test]
fn a_directory_of_one_table_stays_a_dataset() {
let dir = tempfile::tempdir().unwrap();
for part in 0..3 {
write(
dir.path(),
&format!("part-0000{part}.parquet"),
&["id", "ts", "amount"],
);
}
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::MultiFile);
assert_eq!(entry.rows, Some(3));
assert_eq!(entry.cols, Some(3));
}
#[test]
fn a_dataset_that_gained_columns_stays_a_dataset() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "2009.parquet", &["id", "ts"]);
write(dir.path(), "2015.parquet", &["id", "ts", "fee"]);
write(
dir.path(),
"2025.parquet",
&["id", "ts", "fee", "witness", "address", "value"],
);
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::MultiFile);
assert_eq!(entry.rows, Some(3));
assert_eq!(
entry.columns,
vec!["id", "ts", "fee", "witness", "address", "value"],
"every column any file has, in the order they first appear — not the \
2009 shape"
);
assert_eq!(entry.cols, Some(6), "and the count is of those");
}
#[test]
fn a_hive_dataset_that_gained_columns_reports_all_of_them() {
let dir = tempfile::tempdir().unwrap();
for (part, columns) in [
("year=2009", &["id", "ts"][..]),
("year=2025", &["id", "ts", "address"][..]),
] {
let sub = dir.path().join(part);
std::fs::create_dir_all(&sub).unwrap();
write(&sub, "part-0.parquet", columns);
}
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::Hive);
assert_eq!(entry.columns, vec!["id", "ts", "address"]);
assert_eq!(entry.cols, Some(4), "and the partition column `year`");
}
#[test]
fn a_directory_too_large_to_count_still_reports_the_columns_it_gained() {
let dir = tempfile::tempdir().unwrap();
for part in 0..MAX_FOOTERS_PER_DATASET + 1 {
let mut columns = vec!["id".to_string(), "ts".to_string()];
if part > MAX_FOOTERS_PER_DATASET / 2 {
columns.push("address".to_string());
}
let refs: Vec<&str> = columns.iter().map(String::as_str).collect();
write(dir.path(), &format!("part-{part:03}.parquet"), &refs);
}
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::MultiFile, "still one table");
assert_eq!(entry.rows, None, "too many files to count");
assert!(
entry.columns.contains(&"address".to_string()),
"the column the dataset gained is in the row: {:?}",
entry.columns
);
}
#[test]
fn a_lake_table_is_not_a_directory_of_parquet_files() {
for (marker, expected) in [
("_delta_log", EntryKind::Delta),
(".hoodie", EntryKind::Hudi),
] {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "part-0.parquet", &["id", "amount"]);
write(dir.path(), "part-1.parquet", &["id", "amount"]);
write(dir.path(), "part-2.parquet", &["id", "amount"]);
let log = dir.path().join(marker);
std::fs::create_dir_all(&log).unwrap();
std::fs::write(log.join("00000000000000000000.json"), b"{}").unwrap();
assert_eq!(
classify_directory(dir.path()),
expected,
"{marker} says what this directory is"
);
let entry = measured(dir.path());
assert_eq!(entry.kind, expected);
assert_eq!(
entry.rows, None,
"and no row count is claimed for it: summing the footers would count \
the rows the log says are gone"
);
assert!(!entry.kind.is_dataset(), "it does not open as one table");
}
}
#[test]
fn an_iceberg_root_is_metadata_beside_data() {
let iceberg = tempfile::tempdir().unwrap();
let data = iceberg.path().join("data");
let metadata = iceberg.path().join("metadata");
std::fs::create_dir_all(&data).unwrap();
std::fs::create_dir_all(&metadata).unwrap();
write(&data, "00000-0-abc.parquet", &["id", "amount"]);
write(&data, "00001-0-def.parquet", &["id", "amount"]);
std::fs::write(metadata.join("v2.metadata.json"), b"{}").unwrap();
std::fs::write(metadata.join("snap-1.avro"), b"x").unwrap();
assert_eq!(classify_directory(iceberg.path()), EntryKind::Iceberg);
let plain = tempfile::tempdir().unwrap();
std::fs::create_dir_all(plain.path().join("data")).unwrap();
std::fs::create_dir_all(plain.path().join("metadata")).unwrap();
std::fs::write(plain.path().join("metadata/notes.txt"), b"x").unwrap();
assert_eq!(
classify_directory(plain.path()),
EntryKind::Directory,
"no *.metadata.json, so no Iceberg table"
);
let no_data = tempfile::tempdir().unwrap();
let metadata = no_data.path().join("metadata");
std::fs::create_dir_all(&metadata).unwrap();
std::fs::write(metadata.join("v1.metadata.json"), b"{}").unwrap();
write(no_data.path(), "part-0.parquet", &["id"]);
write(no_data.path(), "part-1.parquet", &["id"]);
assert_eq!(
classify_directory(no_data.path()),
EntryKind::MultiFile,
"metadata with no data/ beside it is somebody's directory, not a table root"
);
}
#[test]
fn a_file_and_a_directory_of_it_count_the_same_columns() {
let dir = tempfile::tempdir().unwrap();
write_nested(dir.path(), "one.parquet", "inputs", &["address", "value"]);
let mut file = Entry::new(dir.path().join("one.parquet"), EntryKind::File);
enrich(&mut file);
assert_eq!(
file.cols,
Some(2),
"`id` and `inputs`, which is what opening it shows: {:?}",
file.columns
);
assert!(
file.columns.iter().any(|c| c == "inputs.address"),
"the leaves are still searchable: {:?}",
file.columns
);
write_nested(dir.path(), "two.parquet", "inputs", &["address", "value"]);
let directory = measured(dir.path());
assert_eq!(directory.kind, EntryKind::MultiFile);
assert_eq!(
directory.cols, file.cols,
"and a directory of them says the same number"
);
}
#[test]
fn a_dotted_column_name_is_its_own_column() {
let dir = tempfile::tempdir().unwrap();
write(dir.path(), "flat.parquet", &["id", "user.id", "user.name"]);
let mut file = Entry::new(dir.path().join("flat.parquet"), EntryKind::File);
enrich(&mut file);
assert_eq!(file.cols, Some(3), "three columns: {:?}", file.columns);
}
#[test]
fn a_writer_change_does_not_double_the_column_count() {
let dir = tempfile::tempdir().unwrap();
write_nested(dir.path(), "old.parquet", "inputs", &["address"]);
write_nested(dir.path(), "new.parquet", "inputs", &["address", "value"]);
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::MultiFile, "still one table");
assert_eq!(
entry.cols,
Some(2),
"one `inputs`, not one per shape of it: {:?}",
entry.columns
);
assert!(
entry.columns.len() > 2,
"while every leaf stays searchable: {:?}",
entry.columns
);
}
#[test]
fn a_sampled_column_count_says_it_is_a_floor() {
let dir = tempfile::tempdir().unwrap();
for part in 0..MAX_FOOTERS_PER_DATASET * 2 {
write(
dir.path(),
&format!("part-{part:04}.parquet"),
&["id", "ts"],
);
}
let entry = measured(dir.path());
assert_eq!(entry.rows, None, "too many files to count");
assert!(entry.cols.is_some(), "but the width is still worth having");
assert!(
entry.cols_sampled,
"and it is marked as the floor it is, not presented as a total"
);
let small = tempfile::tempdir().unwrap();
write(small.path(), "a.parquet", &["id", "ts"]);
write(small.path(), "b.parquet", &["id", "ts"]);
assert!(!measured(small.path()).cols_sampled);
}
#[test]
fn the_files_a_directory_offers_come_back_in_order() {
let dir = tempfile::tempdir().unwrap();
for name in ["c.parquet", "a.parquet", "d.parquet", "b.parquet"] {
write(dir.path(), name, &["id"]);
}
let files = parquet_files_under(dir.path());
let names: Vec<String> = files
.iter()
.map(|p| p.file_name().unwrap().to_string_lossy().into_owned())
.collect();
assert_eq!(
names,
vec!["a.parquet", "b.parquet", "c.parquet", "d.parquet"],
"sorted, not in the order the directory was written"
);
}
#[test]
fn a_directory_past_the_budget_keeps_the_directories_first_files() {
let dir = tempfile::tempdir().unwrap();
for part in 0..MAX_FOOTERS_PER_DATASET * 3 {
write(dir.path(), &format!("part-{part:04}.parquet"), &["id"]);
}
let files = parquet_files_under(dir.path());
assert_eq!(
files.len(),
MAX_FOOTERS_PER_DATASET + 1,
"one past the budget, which is what says there are too many to count"
);
let names: Vec<String> = files
.iter()
.map(|p| p.file_name().unwrap().to_string_lossy().into_owned())
.collect();
let expected: Vec<String> = (0..=MAX_FOOTERS_PER_DATASET)
.map(|part| format!("part-{part:04}.parquet"))
.collect();
assert_eq!(
names, expected,
"the directory's first files, not the listing's"
);
}
#[test]
fn a_lake_table_is_recognized_among_its_data_files() {
let dir = tempfile::tempdir().unwrap();
for part in 0..32 {
write(dir.path(), &format!("part-{part:03}.parquet"), &["id"]);
}
std::fs::create_dir_all(dir.path().join("_delta_log")).unwrap();
assert_eq!(classify_directory(dir.path()), EntryKind::Delta);
}
#[test]
fn a_directory_too_large_to_count_is_still_checked() {
let dir = tempfile::tempdir().unwrap();
for table in 0..MAX_FOOTERS_PER_DATASET + 1 {
write(
dir.path(),
&format!("table_{table:03}.parquet"),
&[&format!("{table}_id"), &format!("{table}_value")],
);
}
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::Directory);
assert_eq!(entry.rows, None, "too many files to count either way");
}
#[test]
fn a_large_directory_of_one_table_stays_a_dataset() {
let dir = tempfile::tempdir().unwrap();
for part in 0..MAX_FOOTERS_PER_DATASET + 1 {
write(
dir.path(),
&format!("part-{part:05}.parquet"),
&["id", "ts"],
);
}
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::MultiFile);
}
#[test]
fn a_downgraded_directory_keeps_every_column_its_files_have() {
let dir = tempfile::tempdir().unwrap();
write(
dir.path(),
"circuits.parquet",
&["circuit_id", "lat", "lng"],
);
write(
dir.path(),
"drivers.parquet",
&["driver_id", "code", "nationality"],
);
let seasons = dir.path().join("seasons");
std::fs::create_dir_all(&seasons).unwrap();
write(&seasons, "2024.parquet", &["season_year", "round"]);
let entry = measured(dir.path());
assert_eq!(entry.kind, EntryKind::Directory);
for column in [
"circuit_id",
"lat",
"lng",
"driver_id",
"code",
"nationality",
"season_year",
"round",
] {
assert!(
entry.columns.iter().any(|c| c == column),
"{column} in {:?}",
entry.columns
);
}
}
#[test]
fn a_name_no_reader_takes_is_refused_before_opening() {
let refused = |name: &str| unreadable_by_name(std::path::Path::new(name));
assert!(refused("gs://b/ml/onnx/pipeline_rf.onnx"));
assert!(refused("model.onnx.gz"));
assert!(refused("README.md"));
for readable in [
"a.csv",
"a.CSV",
"a.csv.gz",
"a.parquet",
"a.xlsx",
"data.gz",
"part-0000",
] {
assert!(!refused(readable), "{readable}");
}
}