use ::polars::prelude::*;
use super::csv::{self, StringTypes};
use super::*;
use crate::ParseStringsTarget;
fn piped(head: &[u8]) -> Option<FileFormat> {
sniff(head, None, Asked::Pipe, |_| true)
}
#[test]
fn text_formats_by_their_first_bytes() {
let said = [
(
&b"8=FIX.4.4\x019=5\x0135=0\x0110=000\x01\n"[..],
FileFormat::Fix,
),
(
b"$timescale 1ns $end\n$scope module top $end\n",
FileFormat::Vcd,
),
(
b"aspirin\n RDKit\n\n 0 0 0 0 0 0 0 0 0 0999 V2000\nM END\n$$$$\n",
FileFormat::Sdf,
),
(b"$GPGGA,1,2", FileFormat::Nmea),
(b"<?xml version=\"1.0\"?>\n<gpx>", FileFormat::Gpx),
];
for (head, format) in said {
assert_eq!(piped(head), Some(format), "{format:?}");
}
assert_eq!(piped(b"a,b\n1,2\n"), None);
}
#[test]
fn a_compressed_file_by_its_name_or_what_it_holds() {
use std::io::Write;
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("session.log.gz");
let mut gz =
flate2::write::GzEncoder::new(std::fs::File::create(&path).unwrap(), Default::default());
gz.write_all(b"20260101-00:00:00 : 8=FIX.4.2|9=5|35=0|10=000|\n")
.unwrap();
gz.finish().unwrap();
assert_eq!(sniff_open(&path, None), Some(FileFormat::Fix));
for (name, format) in [
("lib.sdf.gz", Some(FileFormat::Sdf)),
("a.nmea.gz", Some(FileFormat::Nmea)),
("a.csv.gz", None),
] {
assert_eq!(sniff_open(&dir.path().join(name), None), format, "{name}");
}
}
#[test]
fn a_signature_is_believed_where_it_says() {
let elf = b"\x7fELF\x02\x01\x01\0\0\0\0\0\0\0\0\0";
assert_eq!(piped(elf), Some(FileFormat::Elf));
assert_eq!(sniff(elf, None, Asked::Listing, |_| true), None);
assert_eq!(
sniff(elf, None, Asked::Tables, |_| true),
Some(FileFormat::Elf)
);
assert_eq!(piped(b"ORC\x00"), None);
assert_eq!(
sniff(b"ORC\x00", None, Asked::Listing, |_| true),
Some(FileFormat::Orc)
);
let npy = b"\x93NUMPY\x01\x00";
assert_eq!(piped(npy), Some(FileFormat::Numpy));
assert_eq!(sniff(npy, None, Asked::Tables, |_| true), None);
}
#[test]
fn the_copy_docs_name_each_format_s_reader() {
let page =
std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("../../docs/user-guide/copying.md");
let text = std::fs::read_to_string(&page).expect("the copying page");
let start = text.find("| Format | Reader |").expect("the reader table");
let rows: Vec<(String, String)> = text[start..]
.lines()
.skip(2)
.take_while(|l| l.starts_with('|'))
.map(|l| {
let (format, reader) = l.trim_matches('|').split_once(" | ").expect("two cells");
(format.trim().to_string(), reader.trim().to_string())
})
.collect();
let said: Vec<&str> = rows.iter().map(|(f, _)| f.as_str()).collect();
let titles: Vec<&str> = FileFormat::ALL.iter().map(|f| f.title()).collect();
assert_eq!(said, titles, "one row per format, in --format's order");
for ((title, reader), format) in rows.iter().zip(FileFormat::ALL) {
let expected = of(format)
.python
.as_ref()
.map_or("`df = ...`".to_string(), |p| format!("`{}`", p.call));
assert_eq!(*reader, expected, "{title}");
}
}
#[test]
fn readers_agree_with_their_descriptors() {
for format in FileFormat::ALL {
let reader = of(format);
assert_eq!(
reader.tables.is_some(),
format.holds_tables(),
"{}",
format.name()
);
if format.reads_into() {
assert!(reader.convert.is_some(), "{}", format.name());
}
if reader.facts.is_some() {
assert!(format.summary_tab().is_some(), "{}", format.name());
}
#[cfg(feature = "cloud")]
if reader.bucket_scan.is_some() {
assert!(format.reads_bucket_prefix(), "{}", format.name());
}
}
}
fn names(df: &DataFrame) -> Vec<String> {
df.get_column_names()
.iter()
.map(|name| name.to_string())
.collect()
}
fn mixed_strings() -> LazyFrame {
df!(
"id" => &[1i64, 2, 3, 4],
"amount" => &[" 10 ", "20", "", " 40"],
"day" => &["2024-01-01", "2024-01-02", " ", "2024-01-04"],
"at" => &["2024-01-01T10:00:00Z", "2024-01-01T11:00:00Z", "", "2024-01-01T12:00:00Z"],
"score" => &[0.5f64, 1.5, 2.5, 3.5],
"word" => &["a", " b", "c ", ""],
"empty" => &[None::<&str>, None, None, None],
)
.unwrap()
.lazy()
}
#[test]
fn the_inference_sample_holds_only_its_targets() {
let targets: Vec<String> = ["amount", "day", "at", "word", "empty"]
.map(String::from)
.to_vec();
let sample = csv::string_inference_sample(mixed_strings(), &targets, 3).unwrap();
assert_eq!(names(&sample), ["amount", "day", "at", "word", "empty"]);
assert_eq!(sample.height(), 3);
let blank = lit(PlSmallStr::from_static(""));
let wide = mixed_strings()
.limit(3)
.with_columns(
targets
.iter()
.map(|c| {
col(c.as_str())
.str()
.strip_chars(lit(PlSmallStr::from_static(" \t\n\r")))
})
.collect::<Vec<_>>(),
)
.with_columns(
targets
.iter()
.map(|c| {
when(col(c.as_str()).eq(blank.clone()))
.then(Null {}.lit())
.otherwise(col(c.as_str()))
.alias(c.as_str())
})
.collect::<Vec<_>>(),
)
.collect()
.unwrap();
assert_eq!(wide.width(), 7);
assert!(sample.equals_missing(&wide.select(targets.iter().map(String::as_str)).unwrap()));
let one = csv::string_inference_sample(mixed_strings(), &["day".to_string()], 1_000).unwrap();
assert_eq!(names(&one), ["day"]);
assert_eq!(one.height(), 4);
}
#[test]
fn string_inference_keeps_the_whole_frame() {
let utc = DataType::Datetime(TimeUnit::Microseconds, Some(TimeZone::UTC));
let typed = |target: &ParseStringsTarget, types: StringTypes| {
csv::type_string_columns(
mixed_strings(),
target,
1_000,
types,
&mut Vec::new(),
&[],
&mut Vec::new(),
)
.unwrap()
.collect()
.unwrap()
};
let all = StringTypes {
dates: true,
numbers: true,
};
let df = typed(&ParseStringsTarget::All, all);
let schema = df.schema();
let order = names(&df);
assert_eq!(
order,
["id", "amount", "day", "at", "score", "word", "empty"]
);
assert_eq!(schema.get("id"), Some(&DataType::Int64));
assert_eq!(schema.get("amount"), Some(&DataType::Int64));
assert_eq!(schema.get("day"), Some(&DataType::Date));
assert_eq!(schema.get("at"), Some(&utc));
assert_eq!(schema.get("score"), Some(&DataType::Float64));
assert_eq!(schema.get("word"), Some(&DataType::String));
assert_eq!(schema.get("empty"), Some(&DataType::String));
let amount: Vec<Option<i64>> = df.column("amount").unwrap().i64().unwrap().iter().collect();
assert_eq!(amount, [Some(10), Some(20), None, Some(40)]);
assert_eq!(df.column("day").unwrap().null_count(), 1);
assert_eq!(df.column("at").unwrap().null_count(), 1);
let word: Vec<Option<&str>> = df.column("word").unwrap().str().unwrap().iter().collect();
assert_eq!(word, [Some("a"), Some("b"), Some("c"), Some("")]);
let score: Vec<Option<f64>> = df.column("score").unwrap().f64().unwrap().iter().collect();
assert_eq!(score, [Some(0.5), Some(1.5), Some(2.5), Some(3.5)]);
let some = typed(
&ParseStringsTarget::Columns(vec!["day".into(), "id".into(), "missing".into()]),
all,
);
assert_eq!(names(&some), order);
assert_eq!(some.schema().get("day"), Some(&DataType::Date));
assert_eq!(some.schema().get("amount"), Some(&DataType::String));
assert_eq!(
some.column("amount").unwrap().str().unwrap().get(0),
Some(" 10 ")
);
let none = typed(&ParseStringsTarget::Columns(vec!["id".into()]), all);
assert!(none.equals_missing(&mixed_strings().collect().unwrap()));
let json = super::polars::apply_parse_dates_to_json_lazyframe(
mixed_strings(),
&crate::OpenOptions::default(),
&mut Vec::new(),
)
.unwrap()
.collect()
.unwrap();
assert_eq!(names(&json), order);
assert_eq!(json.schema().get("day"), Some(&DataType::Date));
assert_eq!(json.schema().get("at"), Some(&utc));
assert_eq!(json.schema().get("amount"), Some(&DataType::String));
assert_eq!(
json.column("word").unwrap().str().unwrap().get(1),
Some(" b")
);
}
#[test]
fn test_from_parquet() {
let path = crate::tests::sample_data_dir().join("people.parquet");
let schema = super::polars::parquet(&path)
.unwrap()
.collect_schema()
.unwrap();
assert!(!schema.is_empty());
}
#[test]
fn test_from_ipc() {
use ::polars::prelude::IpcWriter;
use std::io::BufWriter;
let mut df = df!(
"x" => &[1_i32, 2, 3],
"y" => &["a", "b", "c"]
)
.unwrap();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("datui_test_ipc.arrow");
let file = std::fs::File::create(&path).unwrap();
let mut writer = BufWriter::new(file);
IpcWriter::new(&mut writer).finish(&mut df).unwrap();
drop(writer);
let schema = super::polars::ipc(&path).unwrap().collect_schema().unwrap();
assert_eq!(schema.len(), 2);
assert!(schema.contains("x"));
assert!(schema.contains("y"));
}
#[test]
fn test_from_avro() {
use ::polars::io::avro::AvroWriter;
use std::io::BufWriter;
let mut df = df!(
"id" => &[1_i32, 2, 3],
"name" => &["alice", "bob", "carol"]
)
.unwrap();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("datui_test_avro.avro");
let file = std::fs::File::create(&path).unwrap();
let mut writer = BufWriter::new(file);
AvroWriter::new(&mut writer).finish(&mut df).unwrap();
drop(writer);
let schema = super::polars::avro(&path)
.unwrap()
.collect_schema()
.unwrap();
assert_eq!(schema.len(), 2);
assert!(schema.contains("id"));
assert!(schema.contains("name"));
}
#[test]
fn test_from_orc() {
use arrow::array::{Int64Array, StringArray};
use arrow::datatypes::{DataType, Field, Schema};
use arrow::record_batch::RecordBatch;
use orc_rust::ArrowWriterBuilder;
use std::io::BufWriter;
use std::sync::Arc;
let schema = Arc::new(Schema::new(vec![
Field::new("id", DataType::Int64, false),
Field::new("name", DataType::Utf8, false),
]));
let id_array = Arc::new(Int64Array::from(vec![1_i64, 2, 3]));
let name_array = Arc::new(StringArray::from(vec!["a", "b", "c"]));
let batch = RecordBatch::try_new(schema.clone(), vec![id_array, name_array]).unwrap();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("datui_test_orc.orc");
let file = std::fs::File::create(&path).unwrap();
let writer = BufWriter::new(file);
let mut orc_writer = ArrowWriterBuilder::new(writer, schema).try_build().unwrap();
orc_writer.write(&batch).unwrap();
orc_writer.close().unwrap();
let schema = super::polars::orc(&path).unwrap().collect_schema().unwrap();
assert_eq!(schema.len(), 2);
assert!(schema.contains("id"));
assert!(schema.contains("name"));
}