datui_lib/loading/open_options.rs
1//! How a dataset is opened: the options a command line, the Python binding or the home
2//! screen hand the load, and what a read reports back on them.
3
4use std::path::PathBuf;
5use std::sync::Arc;
6
7use crate::config::{InferTypes, ParquetSchema};
8use crate::{AppConfig, CompressionFormat, FileFormat, cli};
9
10/// Which CSV string columns to trim and parse (date/datetime/time/duration/int/float). Default: all. None = disabled (e.g. --infer-types=off).
11#[derive(Clone, Debug)]
12pub enum ParseStringsTarget {
13 /// Apply to all string columns.
14 All,
15 /// Apply only to these columns (must exist and be string type).
16 Columns(Vec<String>),
17}
18
19/// Which CSV dialect options were typed on the command line. A delimited spec's
20/// options replace config values but not these (#651).
21#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
22pub struct TypedDialect {
23 pub delimiter: bool,
24 pub comment_char: bool,
25 pub skip_initial_space: bool,
26 pub header_rows: bool,
27 pub skip_lines: bool,
28}
29
30impl TypedDialect {
31 pub fn from_args(args: &cli::Args) -> Self {
32 Self {
33 delimiter: args.delimiter.is_some(),
34 comment_char: args.comment.is_some(),
35 skip_initial_space: args.skip_initial_space.is_some(),
36 header_rows: !args.header_rows.is_empty(),
37 skip_lines: args.skip_lines.is_some(),
38 }
39 }
40}
41
42#[derive(Clone)]
43pub struct OpenOptions {
44 pub delimiter: Option<u8>,
45 pub has_header: Option<bool>,
46 pub skip_lines: Option<usize>,
47 pub skip_rows: Option<usize>,
48 /// Skip this many rows at the end of the file (e.g. vendor footer or trailing garbage). Applied after load for CSV.
49 pub skip_tail_rows: Option<usize>,
50 pub compression: Option<CompressionFormat>,
51 /// When set, bypass extension-based format detection and use this format (e.g. for URLs or temp files without extension).
52 pub format: Option<FileFormat>,
53 pub pages_lookahead: Option<usize>,
54 pub pages_lookback: Option<usize>,
55 pub max_buffered_rows: Option<usize>,
56 pub max_buffered_mb: Option<usize>,
57 pub row_numbers: bool,
58 /// Neither the flag nor the config said: `#` is on for text and logs
59 /// ([`crate::config::RowNumbers::Auto`]), decided once the format is known.
60 pub row_numbers_auto: bool,
61 pub row_start_index: usize,
62 /// When true, use hive load path for directory/glob; single file uses normal load.
63 pub hive: bool,
64 /// Data files in the opened directory this read passes over, by format and count: a
65 /// mixed directory reads as its commonest format, and the dataset says what it left
66 /// out. Empty for every other open.
67 pub left_out: Vec<(FileFormat, usize)>,
68 /// Set (`"Delta"`, `"Iceberg"` or `"Hudi"`) when a lake table's plain files are read.
69 /// They are not the table (deleted rows, old versions and compaction leftovers count),
70 /// so a note and a footer chip always say so.
71 pub read_as_plain_files_of: Option<&'static str>,
72 /// How the directory's files differed, for footerless formats only (Parquet footers
73 /// give the exact per-column version).
74 pub files_disagree: crate::formats::schema_union::Disagreement,
75 /// When true (default), infer Hive/partitioned Parquet schema from one file for faster "Reading schema". When false, use Polars collect_schema().
76 pub single_spine_schema: bool,
77 /// `--view NAME`: the view to apply to the dataset named on the command
78 /// line, once it is on screen. Applied to that open only; what later opens get
79 /// is `[views] auto_apply`'s business.
80 pub view: Option<String>,
81 /// When true, CSV and JSON string columns that look like dates or ISO 8601 timestamps become Date or Datetime.
82 pub parse_dates: bool,
83 /// When set, trim and parse CSV string columns: None = off, Some(true) = all columns, Some(cols) = those columns only.
84 pub parse_strings: Option<ParseStringsTarget>,
85 /// Sample size (rows) for inferring types when parse_strings is enabled; single file or multiple/partitioned.
86 pub parse_strings_sample_rows: usize,
87 /// When true, decompress a compressed CSV, TSV or PSV into memory (eager read). When false (default), decompress to a temp file and use lazy scan.
88 pub decompress_in_memory: bool,
89 /// Directory for decompression temp files. None = system default (e.g. TMPDIR).
90 pub temp_dir: Option<std::path::PathBuf>,
91 /// `--table`: which table of a file that holds several: an NMEA log's sentence
92 /// types, an Excel sheet (0-based index or name), a spec's variant. `None` is the
93 /// file's main table.
94 pub table: Option<String>,
95 /// When true, use Polars streaming engine for LazyFrame collect when the streaming feature is enabled.
96 pub polars_streaming: bool,
97 /// Null value specs for CSV: global strings and/or "COL=VAL" for per-column. Empty = use Polars default.
98 pub null_values: Option<Vec<String>>,
99 /// Number of rows to use when inferring CSV schema. None = Polars default (100). Larger values reduce risk of inferring wrong type (e.g. int then N/A).
100 pub infer_schema_length: Option<usize>,
101 /// When true, CSV reader ignores parse errors and continues with the next batch.
102 pub ignore_errors: bool,
103 /// CSV lines starting with this are comments, wherever they are (`commentChar`).
104 pub comment_char: Option<String>,
105 /// The 1-based lines, counted from the top of the file, that hold a CSV's header
106 /// (`headerRows`). Empty = Polars' own header. A layout flag, like `skip_lines`.
107 pub header_rows: Vec<usize>,
108 /// What joins a column's pieces when `header_rows` names several lines.
109 pub header_join: String,
110 /// Ignore the spaces after a CSV delimiter (`skipInitialSpace`).
111 pub skip_initial_space: bool,
112 /// Which dialect options were typed on the command line, so a spec leaves them.
113 pub typed_dialect: TypedDialect,
114 /// The debug overlay (session info, performance, query): `DATUI_DEBUG=1`.
115 pub debug: bool,
116 /// The split of a Hugging Face cache directory this read chose, the others and the
117 /// `map()` files it left out. Found by the read, or by a bucket listing, and carried
118 /// to the dataset as `left_out` is. `None` for every other open.
119 pub splits: Option<Arc<crate::formats::hf_splits::Splits>>,
120 /// Where each Arrow input's rows are once its streams are converted: the IPC files
121 /// read in place and the streams' rows in the converted file the scan names. Set
122 /// by the load, after a conversion or a bucket's listing, never by a request.
123 pub arrow_parts: Option<Arc<Vec<crate::formats::ipc_stream::Part>>>,
124 /// `--format FILE`: read the path through this format spec, whatever else
125 /// matches it. A URL is fetched when the open starts, into `spec_fetched`.
126 pub spec_file: Option<PathBuf>,
127 /// The spec a remote `spec_file` names, fetched by the open (`Phase::ReadingSpec`).
128 pub spec_fetched: Option<Arc<crate::formats::Spec>>,
129 /// `--dict FILE`: FIX dictionaries and DBC files over the search path's, each
130 /// taken by the reader of its kind.
131 pub dicts: Vec<PathBuf>,
132 /// The format spec named by `--format NAME`, or picked with `b`.
133 pub spec_name: Option<String>,
134 /// What a read through a format spec found, carried from the scan to the dataset.
135 pub format_read: Option<Arc<crate::formats::Read>>,
136 /// What the open did to the rows its reader gave — CSV column names trimmed, text
137 /// columns typed — as Python method calls for Copy as Python. Found by the scan,
138 /// carried to the dataset as `left_out` is. Empty for every other open.
139 pub read_python: Vec<String>,
140 /// `[read] audio_float`: integer audio samples as float in [-1, 1].
141 pub normalize: bool,
142 /// A SQLite table opened in place, carried from the scan to the dataset.
143 pub sqlite: Option<Arc<SqliteOpen>>,
144 /// What the reader found besides the frame: a window read straight from the file,
145 /// its row count, its Info panel tab, other tables and notes. Found by the scan and
146 /// carried to the dataset as `left_out` is. `None` for a reader with nothing to add.
147 pub opened: Option<Arc<crate::formats::members::Opened>>,
148 /// The delimited spec the file is read through, once chosen: its dialect is in
149 /// these options, and the read's units and metadata ride with it to the dataset.
150 pub delimited: Option<Arc<crate::formats::delimited_spec::DelimitedRead>>,
151 /// How the scan found the data is read: lazily, through a copy, or into memory.
152 /// Found by the scan and carried to the dataset for the Info panel's `Read:` line.
153 pub read_mode: Option<crate::ReadMode>,
154 /// The format was guessed from the first bytes of text no format's signature
155 /// claims, rather than named, so a note can say how to read it otherwise.
156 pub format_guessed: bool,
157 /// What the read of several files found to say of them, for the dataset's notes.
158 /// Found by the scan and carried to the dataset as `left_out` is.
159 pub read_notes: Vec<crate::notes::Note>,
160 /// The columns the read gave a type and the frame before, carried to the dataset
161 /// as `left_out` is, for the count of the values that did not fit.
162 pub typing: crate::formats::readers::Typing,
163 /// `--hex`: show the file's bytes in the hex view, whatever it holds.
164 pub hex: bool,
165 /// `--hex-width N`: the bytes a row of the hex view holds.
166 pub record_size: Option<usize>,
167 /// `--follow`: show rows as they are appended to the file, or arrive on standard
168 /// input, until stopped.
169 pub follow: bool,
170 /// Where the followed file's complete records end, counted by the scan and carried
171 /// to the dataset as `left_out` is, for its watcher to read on from.
172 pub tail: Option<Arc<crate::loading::follow::Tail>>,
173 /// Standard input still being copied to the file a follow reads.
174 pub spool: Option<Arc<crate::loading::follow::SpoolHandle>>,
175 /// Standard input shown as it arrives without `--follow`: read by the follow's
176 /// watcher until it ends, the view staying where it is rather than at the end.
177 pub pipe: bool,
178 /// `--tee FILE`: standard input is recorded to FILE, which is what is read.
179 pub tee: Option<PathBuf>,
180 /// `--tee-raw`: FILE is the bytes exactly as they came, a WAV header included.
181 pub tee_raw: bool,
182 /// `--force`: FILE may replace a file that is there.
183 pub force: bool,
184 /// The dataset the home screen's preview built and read the first page of, for
185 /// this open to install rather than read again. Taken once.
186 pub prepared: Option<crate::home::home_preview::Handoff>,
187 /// A file of the built-in catalog: downloaded without asking when it is small.
188 pub download_unasked: Option<UnaskedDownload>,
189}
190
191/// A remote file downloaded without asking: one the built-in catalog lists, at most
192/// `limit` bytes by what the server says or, when it says nothing, by the catalog.
193#[derive(Debug, Clone, Copy, PartialEq, Eq)]
194pub struct UnaskedDownload {
195 pub limit: u64,
196 pub listed: Option<u64>,
197}
198
199impl UnaskedDownload {
200 /// The largest built-in catalog file downloaded without asking.
201 pub const LIMIT: u64 = 50 * 1024 * 1024;
202
203 /// Whether a file of `size` bytes, if the server said, is downloaded unasked.
204 pub fn covers(&self, size: Option<u64>) -> bool {
205 size.or(self.listed)
206 .is_some_and(|bytes| bytes <= self.limit)
207 }
208}
209
210impl OpenOptions {
211 pub fn new() -> Self {
212 Self {
213 delimiter: None,
214 has_header: None,
215 skip_lines: None,
216 skip_rows: None,
217 skip_tail_rows: None,
218 left_out: Vec::new(),
219 read_python: Vec::new(),
220 read_as_plain_files_of: None,
221 files_disagree: Default::default(),
222 compression: None,
223 format: None,
224 pages_lookahead: None,
225 pages_lookback: None,
226 max_buffered_rows: None,
227 max_buffered_mb: None,
228 row_numbers: false,
229 row_numbers_auto: true,
230 row_start_index: 1,
231 hive: false,
232 single_spine_schema: true,
233 view: None,
234 parse_dates: true,
235 parse_strings: None,
236 parse_strings_sample_rows: 1000,
237 decompress_in_memory: false,
238 temp_dir: None,
239 table: None,
240 polars_streaming: true,
241 null_values: None,
242 infer_schema_length: None,
243 ignore_errors: false,
244 comment_char: None,
245 header_rows: Vec::new(),
246 header_join: crate::formats::csv_dialect::DEFAULT_HEADER_JOIN.to_string(),
247 skip_initial_space: false,
248 typed_dialect: TypedDialect::default(),
249 debug: false,
250 spec_file: None,
251 spec_fetched: None,
252 dicts: Vec::new(),
253 spec_name: None,
254 format_read: None,
255 normalize: false,
256 sqlite: None,
257 opened: None,
258 splits: None,
259 arrow_parts: None,
260 delimited: None,
261 read_mode: None,
262 format_guessed: false,
263 read_notes: Vec::new(),
264 typing: Default::default(),
265 hex: false,
266 record_size: None,
267 follow: false,
268 tail: None,
269 spool: None,
270 pipe: false,
271 tee: None,
272 tee_raw: false,
273 force: false,
274 prepared: None,
275 download_unasked: None,
276 }
277 }
278}
279
280impl Default for OpenOptions {
281 fn default() -> Self {
282 Self::new()
283 }
284}
285
286impl OpenOptions {
287 #[cfg(test)]
288 pub fn with_skip_lines(mut self, skip_lines: usize) -> Self {
289 self.skip_lines = Some(skip_lines);
290 self
291 }
292
293 /// The separator a delimited file is read with: `--delimiter` when given, else
294 /// the one its format implies (`FileFormat::separator`).
295 pub fn separator_or(&self, format_default: u8) -> u8 {
296 self.delimiter.unwrap_or(format_default)
297 }
298
299 /// The lines `--header-rows` named, unless the file is being read without a
300 /// header (`--no-header`, or `H`), which reads them as data.
301 pub fn header_rows(&self) -> Option<&[usize]> {
302 (!self.header_rows.is_empty() && self.has_header != Some(false))
303 .then_some(self.header_rows.as_slice())
304 }
305
306 /// When loading CSV: use Polars try_parse_dates only if parse_strings is not set.
307 /// When parse_strings is set we do our own date parsing (with strict: false), so we disable
308 /// Polars' try_parse_dates to avoid "could not find an appropriate format" errors.
309 pub fn csv_try_parse_dates(&self) -> bool {
310 self.parse_strings.is_none() && self.parse_dates
311 }
312
313 /// The S3 settings every cloud path uses: environment over `[cloud]` config. `run()`
314 /// folds this into the App's config so opening, sizing, downloading, discovery and
315 /// listing agree. The environment is read here, so `OpenOptions::default()` callers
316 /// (Python) honor it too.
317 pub fn effective_cloud(
318 &self,
319 cloud: &crate::config::CloudConfig,
320 ) -> crate::config::CloudConfig {
321 let mut merged = cloud.clone();
322 merged.overlay(crate::config::CloudConfig::from_env(
323 &crate::cloud::cloud_env::var,
324 ));
325 merged
326 }
327}
328
329impl OpenOptions {
330 /// The options the command line and the config give an open. The config has had
331 /// `-c` laid over it; a flag here beats both.
332 pub fn from_args_and_config(args: &cli::Args, config: &AppConfig) -> Self {
333 let mut opts = OpenOptions::new();
334
335 // A file's layout: command line only. Set in config, these applied to every
336 // file opened and silently cut rows from the ones they did not describe (#289).
337 opts.delimiter = args.delimiter;
338 opts.skip_lines = args.skip_lines;
339 opts.skip_rows = args.skip_rows;
340 opts.skip_tail_rows = args.footer_rows;
341 opts.has_header = args.no_header.then_some(false);
342 opts.header_rows = args.header_rows.iter().map(|&n| n as usize).collect();
343 opts.view = args.view.clone();
344 opts.compression = args.compression;
345
346 // A spec's name is looked up on the search path when the file is opened.
347 opts.format = args.format.as_ref().and_then(cli::FormatChoice::builtin);
348 opts.spec_name = args
349 .format
350 .as_ref()
351 .and_then(|f| f.spec().map(str::to_string));
352 opts.spec_file = args
353 .format
354 .as_ref()
355 .and_then(|f| f.spec_file().map(std::path::Path::to_path_buf));
356 opts.dicts = args.dict.clone();
357 opts.table = args.table.clone();
358 opts.hex = args.hex;
359 opts.record_size = args.hex_width.map(usize::from);
360 opts.hive = args.hive;
361 opts.follow = args.follow;
362 opts.tee = args.tee.clone();
363 opts.tee_raw = args.tee_raw;
364 opts.force = args.force;
365
366 opts.pages_lookahead = Some(config.performance.pages_ahead);
367 opts.pages_lookback = Some(config.performance.pages_behind);
368 opts.max_buffered_rows = Some(config.performance.max_buffered_rows);
369 opts.max_buffered_mb = Some(config.performance.max_buffered_mb());
370 let row_numbers = args
371 .row_numbers
372 .map(crate::config::RowNumbers::from)
373 .unwrap_or(config.display.row_numbers);
374 opts.row_numbers = row_numbers == crate::config::RowNumbers::On;
375 opts.row_numbers_auto = row_numbers == crate::config::RowNumbers::Auto;
376 opts.row_start_index = config.display.row_numbers_start;
377 opts.single_spine_schema = config.read.parquet_schema == ParquetSchema::Union;
378 opts.decompress_in_memory = config.read.decompress_in_memory;
379 opts.normalize = config.read.audio_float;
380 opts.polars_streaming = config.performance.streaming;
381
382 // Typing string columns, dates among them: the flag, else `read.infer_types`.
383 let infer = match &args.infer_types {
384 Some(cli::InferTypes::All) => InferTypes::Switch(true),
385 Some(cli::InferTypes::Off) => InferTypes::Switch(false),
386 Some(cli::InferTypes::Columns(cols)) => InferTypes::Columns(cols.clone()),
387 None => config.read.infer_types.clone(),
388 };
389 (opts.parse_strings, opts.parse_dates) = match infer {
390 InferTypes::Switch(false) => (None, false),
391 InferTypes::Switch(true) => (Some(ParseStringsTarget::All), true),
392 InferTypes::Columns(cols) => (Some(ParseStringsTarget::Columns(cols)), true),
393 };
394
395 // CSV dialect: a flag beats the config.
396 let csv = &config.csv;
397 opts.comment_char = args.comment.clone().or_else(|| csv.comment.clone());
398 opts.header_join = csv.header_join.clone();
399 opts.skip_initial_space = args.skip_initial_space.unwrap_or(csv.skip_initial_space);
400 opts.typed_dialect = TypedDialect::from_args(args);
401 opts.ignore_errors = args.ignore_errors.unwrap_or(csv.ignore_errors);
402 // `--null` replaces the config's list, as every flag replaces its key.
403 let nulls = if args.null.is_empty() {
404 csv.null_values.clone()
405 } else {
406 args.null.clone()
407 };
408 opts.null_values = (!nulls.is_empty()).then_some(nulls);
409 // One row count for one guess, Polars' and datui's alike.
410 let infer_rows = args.infer_rows.unwrap_or(csv.infer_rows);
411 opts.infer_schema_length = Some(infer_rows);
412 opts.parse_strings_sample_rows = infer_rows;
413
414 opts.temp_dir = args.temp_dir.clone().or_else(|| {
415 config
416 .read
417 .temp_dir
418 .as_deref()
419 .map(crate::config::expand_config_path)
420 });
421
422 opts
423 }
424}
425
426impl From<&cli::Args> for OpenOptions {
427 fn from(args: &cli::Args) -> Self {
428 let config = AppConfig::default();
429 Self::from_args_and_config(args, &config)
430 }
431}
432
433/// What a directory read found about itself, filled by the pass that picks files and
434/// reader and carried back for the Notes: what datui did, not what the data is (the
435/// footer notes, written later, are that).
436#[derive(Debug, Clone, Default)]
437pub struct ReadReport {
438 /// Data files in the directory this read passed over, by format and count. A
439 /// directory of more than one format is read as the commonest of them; this is the
440 /// rest.
441 pub left_out: Vec<(FileFormat, usize)>,
442 /// How the files read differed. See [`OpenOptions::files_disagree`].
443 pub files_disagree: crate::formats::schema_union::Disagreement,
444 /// The reader the files were read with, where the read chose it: a directory's
445 /// commonest format, or a file's extension. Carried back as `OpenOptions::format`,
446 /// so what is on screen knows whether it has a header row to turn off.
447 pub format: Option<FileFormat>,
448 /// What a read through a format spec found. See `OpenOptions::format_read`.
449 pub format_read: Option<Arc<crate::formats::Read>>,
450 /// See [`OpenOptions::read_python`].
451 pub read_python: Vec<String>,
452 /// A SQLite table opened in place. See `OpenOptions::sqlite`.
453 pub sqlite: Option<Arc<SqliteOpen>>,
454 /// See [`OpenOptions::opened`].
455 pub opened: Option<Arc<crate::formats::members::Opened>>,
456 /// The split a Hugging Face cache directory was read as. See `OpenOptions::splits`.
457 pub splits: Option<Arc<crate::formats::hf_splits::Splits>>,
458 /// What a read through a delimited spec found. See `OpenOptions::delimited`.
459 pub delimited: Option<Arc<crate::formats::delimited_spec::DelimitedRead>>,
460 /// The table the read opened where the open named none: a database's only table.
461 /// Carried back as `OpenOptions::table`, so the dataset says which table it is.
462 pub table: Option<String>,
463 /// The format was guessed from the text's first bytes. See
464 /// [`OpenOptions::format_guessed`].
465 pub guessed: bool,
466 /// See [`OpenOptions::read_notes`].
467 pub read_notes: Vec<crate::notes::Note>,
468 /// See [`OpenOptions::typing`].
469 pub typing: crate::formats::readers::Typing,
470}
471
472/// A SQLite table opened in place, carried from the scan to the dataset.
473pub struct SqliteOpen {
474 pub pushdown: Arc<dyn crate::formats::pushdown::Pushdown>,
475 /// Taken by the dataset, which stops the table's statements when it goes.
476 pub hold: std::sync::Mutex<Option<crate::formats::sqlite::Hold>>,
477 pub other_tables: Vec<String>,
478}
479
480impl std::fmt::Debug for SqliteOpen {
481 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
482 f.debug_struct("SqliteOpen")
483 .field("other_tables", &self.other_tables)
484 .finish_non_exhaustive()
485 }
486}