datui_lib/open_options.rs
1//! How a dataset is opened: the options a command line, the Python binding or the home
2//! screen hand the load, and what a read reports back on them.
3
4use std::path::PathBuf;
5use std::sync::Arc;
6
7use crate::config::{InferTypes, ParquetSchema};
8use crate::{AppConfig, CompressionFormat, FileFormat, cli};
9
10/// Which CSV string columns to trim and parse (date/datetime/time/duration/int/float). Default: all. None = disabled (e.g. --infer-types=off).
11#[derive(Clone, Debug)]
12pub enum ParseStringsTarget {
13 /// Apply to all string columns.
14 All,
15 /// Apply only to these columns (must exist and be string type).
16 Columns(Vec<String>),
17}
18
19/// Which CSV dialect options were typed on the command line. A delimited spec's
20/// options replace config values but not these (#651).
21#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
22pub struct TypedDialect {
23 pub delimiter: bool,
24 pub comment_char: bool,
25 pub skip_initial_space: bool,
26 pub header_rows: bool,
27 pub skip_lines: bool,
28}
29
30impl TypedDialect {
31 pub fn from_args(args: &cli::Args) -> Self {
32 Self {
33 delimiter: args.delimiter.is_some(),
34 comment_char: args.comment.is_some(),
35 skip_initial_space: args.skip_initial_space.is_some(),
36 header_rows: !args.header_rows.is_empty(),
37 skip_lines: args.skip_lines.is_some(),
38 }
39 }
40}
41
42#[derive(Clone)]
43pub struct OpenOptions {
44 pub delimiter: Option<u8>,
45 pub has_header: Option<bool>,
46 pub skip_lines: Option<usize>,
47 pub skip_rows: Option<usize>,
48 /// Skip this many rows at the end of the file (e.g. vendor footer or trailing garbage). Applied after load for CSV.
49 pub skip_tail_rows: Option<usize>,
50 pub compression: Option<CompressionFormat>,
51 /// When set, bypass extension-based format detection and use this format (e.g. for URLs or temp files without extension).
52 pub format: Option<FileFormat>,
53 pub pages_lookahead: Option<usize>,
54 pub pages_lookback: Option<usize>,
55 pub max_buffered_rows: Option<usize>,
56 pub max_buffered_mb: Option<usize>,
57 pub row_numbers: bool,
58 /// Neither the flag nor the config said: `#` is on for text and logs
59 /// ([`crate::config::RowNumbers::Auto`]), decided once the format is known.
60 pub row_numbers_auto: bool,
61 pub row_start_index: usize,
62 /// When true, use hive load path for directory/glob; single file uses normal load.
63 pub hive: bool,
64 /// Data files in the directory being opened that this read passes over, by format and
65 /// count.
66 ///
67 /// A directory of more than one format is read as the commonest of them — a thousand
68 /// CSVs and one stray JSON is a directory of CSVs — and this is what the stray was,
69 /// so the dataset can say what it left out rather than the directory being refused
70 /// over it. Empty for every other open, which is all of them but one.
71 pub left_out: Vec<(FileFormat, usize)>,
72 /// Set when the directory being opened is a lake table and this read is of its plain
73 /// files: `"Delta"`, `"Iceberg"` or `"Hudi"`.
74 ///
75 /// The files are not the table. A delete leaves its rows on disk, an update leaves
76 /// the version it replaced, and compaction leaves both sides — so this read counts
77 /// rows no query of the table would return. datui does it anyway, because the
78 /// alternative was a directory the user could see and could not read at all, and
79 /// every other engine at least lets you look. What makes it honest rather than wrong
80 /// is that it is never silent: a note and a chip in the control bar say so, and both
81 /// are load-bearing.
82 pub read_as_plain_files_of: Option<&'static str>,
83 /// How the directory's own files differed, when they did.
84 ///
85 /// Only for the formats with no footer. A Parquet dataset's footers are read
86 /// anyway, and say this per column and per file in far more detail — which columns,
87 /// in how many files, and where — so saying it twice would be one vague note above
88 /// several exact ones.
89 pub files_disagree: crate::schema_union::Disagreement,
90 /// When true (default), infer Hive/partitioned Parquet schema from one file for faster "Reading schema". When false, use Polars collect_schema().
91 pub single_spine_schema: bool,
92 /// `--view NAME`: the view to apply to the dataset named on the command
93 /// line, once it is on screen. Applied to that open only; what later opens get
94 /// is `[views] auto_apply`'s business.
95 pub view: Option<String>,
96 /// When true, CSV and JSON string columns that look like dates or ISO 8601 timestamps become Date or Datetime.
97 pub parse_dates: bool,
98 /// When set, trim and parse CSV string columns: None = off, Some(true) = all columns, Some(cols) = those columns only.
99 pub parse_strings: Option<ParseStringsTarget>,
100 /// Sample size (rows) for inferring types when parse_strings is enabled; single file or multiple/partitioned.
101 pub parse_strings_sample_rows: usize,
102 /// When true, decompress a compressed CSV, TSV or PSV into memory (eager read). When false (default), decompress to a temp file and use lazy scan.
103 pub decompress_in_memory: bool,
104 /// Directory for decompression temp files. None = system default (e.g. TMPDIR).
105 pub temp_dir: Option<std::path::PathBuf>,
106 /// `--table`: which table of a file that holds several: an NMEA log's sentence
107 /// types, an Excel sheet (0-based index or name), a spec's variant. `None` is the
108 /// file's main table.
109 pub table: Option<String>,
110 /// When true, use Polars streaming engine for LazyFrame collect when the streaming feature is enabled.
111 pub polars_streaming: bool,
112 /// Null value specs for CSV: global strings and/or "COL=VAL" for per-column. Empty = use Polars default.
113 pub null_values: Option<Vec<String>>,
114 /// Number of rows to use when inferring CSV schema. None = Polars default (100). Larger values reduce risk of inferring wrong type (e.g. int then N/A).
115 pub infer_schema_length: Option<usize>,
116 /// When true, CSV reader ignores parse errors and continues with the next batch.
117 pub ignore_errors: bool,
118 /// CSV lines starting with this are comments, wherever they are (`commentChar`).
119 pub comment_char: Option<String>,
120 /// The 1-based lines, counted from the top of the file, that hold a CSV's header
121 /// (`headerRows`). Empty = Polars' own header. A layout flag, like `skip_lines`.
122 pub header_rows: Vec<usize>,
123 /// What joins a column's pieces when `header_rows` names several lines.
124 pub header_join: String,
125 /// Ignore the spaces after a CSV delimiter (`skipInitialSpace`).
126 pub skip_initial_space: bool,
127 /// Which dialect options were typed on the command line, so a spec leaves them.
128 pub typed_dialect: TypedDialect,
129 /// The debug overlay (session info, performance, query): `DATUI_DEBUG=1`.
130 pub debug: bool,
131 /// The split of a Hugging Face cache directory this read chose, the others and the
132 /// `map()` files it left out. Found by the read, or by a bucket listing, and carried
133 /// to the dataset as `left_out` is. `None` for every other open.
134 pub splits: Option<Arc<crate::hf_splits::Splits>>,
135 /// Where each Arrow input's rows are once its streams are converted: the IPC files
136 /// read in place and the streams' rows in the converted file the scan names. Set
137 /// by the load, after a conversion or a bucket's listing, never by a request.
138 pub arrow_parts: Option<Arc<Vec<crate::ipc_stream::Part>>>,
139 /// `--format FILE`: read the path through this format spec, whatever else
140 /// matches it. A URL is fetched when the open starts, into `spec_fetched`.
141 pub spec_file: Option<PathBuf>,
142 /// The spec a remote `spec_file` names, fetched by the open (`Phase::ReadingSpec`).
143 pub spec_fetched: Option<Arc<crate::formats::Spec>>,
144 /// `--dict FILE`: FIX dictionaries and DBC files over the search path's, each
145 /// taken by the reader of its kind.
146 pub dicts: Vec<PathBuf>,
147 /// The format spec named by `--format NAME`, or picked with `b`.
148 pub spec_name: Option<String>,
149 /// What a read through a format spec found, carried from the scan to the dataset.
150 pub format_read: Option<Arc<crate::formats::Read>>,
151 /// What the open did to the rows its reader gave — CSV column names trimmed, text
152 /// columns typed — as Python method calls for Copy as Python. Found by the scan,
153 /// carried to the dataset as `left_out` is. Empty for every other open.
154 pub read_python: Vec<String>,
155 /// `[read] audio_float`: integer audio samples as float in [-1, 1].
156 pub normalize: bool,
157 /// A SQLite table opened in place, carried from the scan to the dataset.
158 pub sqlite: Option<Arc<SqliteOpen>>,
159 /// What the reader found besides the frame: a window read straight from the file,
160 /// its row count, its Info panel tab, other tables and notes. Found by the scan and
161 /// carried to the dataset as `left_out` is. `None` for a reader with nothing to add.
162 pub opened: Option<Arc<crate::members::Opened>>,
163 /// The delimited spec the file is read through, once chosen: its dialect is in
164 /// these options, and the read's units and metadata ride with it to the dataset.
165 pub delimited: Option<Arc<crate::delimited_spec::DelimitedRead>>,
166 /// How the scan found the data is read: lazily, through a copy, or into memory.
167 /// Found by the scan and carried to the dataset for the Info panel's `Read:` line.
168 pub read_mode: Option<crate::ReadMode>,
169 /// The format was guessed from the first bytes of text no format's signature
170 /// claims, rather than named, so a note can say how to read it otherwise.
171 pub format_guessed: bool,
172 /// What the read of several files found to say of them, for the dataset's notes.
173 /// Found by the scan and carried to the dataset as `left_out` is.
174 pub read_notes: Vec<crate::notes::Note>,
175 /// The columns the read gave a type and the frame before, carried to the dataset
176 /// as `left_out` is, for the count of the values that did not fit.
177 pub typing: crate::widgets::datatable::Typing,
178 /// `--hex`: show the file's bytes in the hex view, whatever it holds.
179 pub hex: bool,
180 /// `--hex-width N`: the bytes a row of the hex view holds.
181 pub record_size: Option<usize>,
182 /// `--follow`: show rows as they are appended to the file, or arrive on standard
183 /// input, until stopped.
184 pub follow: bool,
185 /// Where the followed file's complete records end, counted by the scan and carried
186 /// to the dataset as `left_out` is, for its watcher to read on from.
187 pub tail: Option<Arc<crate::follow::Tail>>,
188 /// Standard input still being copied to the file a follow reads.
189 pub spool: Option<Arc<crate::follow::SpoolHandle>>,
190 /// Standard input shown as it arrives without `--follow`: read by the follow's
191 /// watcher until it ends, the view staying where it is rather than at the end.
192 pub pipe: bool,
193 /// `--tee FILE`: standard input is recorded to FILE, which is what is read.
194 pub tee: Option<PathBuf>,
195 /// `--tee-raw`: FILE is the bytes exactly as they came, a WAV header included.
196 pub tee_raw: bool,
197 /// `--force`: FILE may replace a file that is there.
198 pub force: bool,
199 /// The dataset the home screen's preview built and read the first page of, for
200 /// this open to install rather than read again. Taken once.
201 pub prepared: Option<crate::home_preview::Handoff>,
202 /// A file of the built-in catalog: downloaded without asking when it is small.
203 pub download_unasked: Option<UnaskedDownload>,
204}
205
206/// A remote file downloaded without asking: one the built-in catalog lists, at most
207/// `limit` bytes by what the server says or, when it says nothing, by the catalog.
208#[derive(Debug, Clone, Copy, PartialEq, Eq)]
209pub struct UnaskedDownload {
210 pub limit: u64,
211 pub listed: Option<u64>,
212}
213
214impl UnaskedDownload {
215 /// The largest built-in catalog file downloaded without asking.
216 pub const LIMIT: u64 = 50 * 1024 * 1024;
217
218 /// Whether a file of `size` bytes, if the server said, is downloaded unasked.
219 pub fn covers(&self, size: Option<u64>) -> bool {
220 size.or(self.listed)
221 .is_some_and(|bytes| bytes <= self.limit)
222 }
223}
224
225impl OpenOptions {
226 pub fn new() -> Self {
227 Self {
228 delimiter: None,
229 has_header: None,
230 skip_lines: None,
231 skip_rows: None,
232 skip_tail_rows: None,
233 left_out: Vec::new(),
234 read_python: Vec::new(),
235 read_as_plain_files_of: None,
236 files_disagree: Default::default(),
237 compression: None,
238 format: None,
239 pages_lookahead: None,
240 pages_lookback: None,
241 max_buffered_rows: None,
242 max_buffered_mb: None,
243 row_numbers: false,
244 row_numbers_auto: true,
245 row_start_index: 1,
246 hive: false,
247 single_spine_schema: true,
248 view: None,
249 parse_dates: true,
250 parse_strings: None,
251 parse_strings_sample_rows: 1000,
252 decompress_in_memory: false,
253 temp_dir: None,
254 table: None,
255 polars_streaming: true,
256 null_values: None,
257 infer_schema_length: None,
258 ignore_errors: false,
259 comment_char: None,
260 header_rows: Vec::new(),
261 header_join: crate::csv_dialect::DEFAULT_HEADER_JOIN.to_string(),
262 skip_initial_space: false,
263 typed_dialect: TypedDialect::default(),
264 debug: false,
265 spec_file: None,
266 spec_fetched: None,
267 dicts: Vec::new(),
268 spec_name: None,
269 format_read: None,
270 normalize: false,
271 sqlite: None,
272 opened: None,
273 splits: None,
274 arrow_parts: None,
275 delimited: None,
276 read_mode: None,
277 format_guessed: false,
278 read_notes: Vec::new(),
279 typing: Default::default(),
280 hex: false,
281 record_size: None,
282 follow: false,
283 tail: None,
284 spool: None,
285 pipe: false,
286 tee: None,
287 tee_raw: false,
288 force: false,
289 prepared: None,
290 download_unasked: None,
291 }
292 }
293}
294
295impl Default for OpenOptions {
296 fn default() -> Self {
297 Self::new()
298 }
299}
300
301impl OpenOptions {
302 pub fn with_skip_lines(mut self, skip_lines: usize) -> Self {
303 self.skip_lines = Some(skip_lines);
304 self
305 }
306
307 pub fn with_skip_rows(mut self, skip_rows: usize) -> Self {
308 self.skip_rows = Some(skip_rows);
309 self
310 }
311
312 pub fn with_delimiter(mut self, delimiter: u8) -> Self {
313 self.delimiter = Some(delimiter);
314 self
315 }
316
317 pub fn with_has_header(mut self, has_header: bool) -> Self {
318 self.has_header = Some(has_header);
319 self
320 }
321
322 /// The separator a delimited file is read with: `--delimiter` when given, else
323 /// the one its format implies (`FileFormat::separator`).
324 pub fn separator_or(&self, format_default: u8) -> u8 {
325 self.delimiter.unwrap_or(format_default)
326 }
327
328 pub fn with_compression(mut self, compression: CompressionFormat) -> Self {
329 self.compression = Some(compression);
330 self
331 }
332
333 /// The lines `--header-rows` named, unless the file is being read without a
334 /// header (`--no-header`, or `H`), which reads them as data.
335 pub fn header_rows(&self) -> Option<&[usize]> {
336 (!self.header_rows.is_empty() && self.has_header != Some(false))
337 .then_some(self.header_rows.as_slice())
338 }
339
340 /// When loading CSV: use Polars try_parse_dates only if parse_strings is not set.
341 /// When parse_strings is set we do our own date parsing (with strict: false), so we disable
342 /// Polars' try_parse_dates to avoid "could not find an appropriate format" errors.
343 pub fn csv_try_parse_dates(&self) -> bool {
344 self.parse_strings.is_none() && self.parse_dates
345 }
346
347 /// The S3 settings every cloud path uses: the environment over the `[cloud]`
348 /// config. `run()` folds this into the config the `App` keeps,
349 /// so opening, sizing, downloading, discovery and listing all see one answer and a
350 /// bucket that is listed is reached the way it will be opened. The environment is
351 /// read here, not when the options are built, so a caller that starts from
352 /// `OpenOptions::default()` — the Python bindings do — still honours it.
353 pub fn effective_cloud(
354 &self,
355 cloud: &crate::config::CloudConfig,
356 ) -> crate::config::CloudConfig {
357 let mut merged = cloud.clone();
358 merged.overlay(crate::config::CloudConfig::from_env(&crate::cloud_env::var));
359 merged
360 }
361}
362
363impl OpenOptions {
364 /// The options the command line and the config give an open. The config has had
365 /// `-c` laid over it; a flag here beats both.
366 pub fn from_args_and_config(args: &cli::Args, config: &AppConfig) -> Self {
367 let mut opts = OpenOptions::new();
368
369 // A file's layout: command line only. Set in config, these applied to every
370 // file opened and silently cut rows from the ones they did not describe (#289).
371 opts.delimiter = args.delimiter;
372 opts.skip_lines = args.skip_lines;
373 opts.skip_rows = args.skip_rows;
374 opts.skip_tail_rows = args.footer_rows;
375 opts.has_header = args.no_header.then_some(false);
376 opts.header_rows = args.header_rows.iter().map(|&n| n as usize).collect();
377 opts.view = args.view.clone();
378 opts.compression = args.compression;
379
380 // A spec's name is looked up on the search path when the file is opened.
381 opts.format = args.format.as_ref().and_then(cli::FormatChoice::builtin);
382 opts.spec_name = args
383 .format
384 .as_ref()
385 .and_then(|f| f.spec().map(str::to_string));
386 opts.spec_file = args
387 .format
388 .as_ref()
389 .and_then(|f| f.spec_file().map(std::path::Path::to_path_buf));
390 opts.dicts = args.dict.clone();
391 opts.table = args.table.clone();
392 opts.hex = args.hex;
393 opts.record_size = args.hex_width.map(usize::from);
394 opts.hive = args.hive;
395 opts.follow = args.follow;
396 opts.tee = args.tee.clone();
397 opts.tee_raw = args.tee_raw;
398 opts.force = args.force;
399
400 opts.pages_lookahead = Some(config.performance.pages_ahead);
401 opts.pages_lookback = Some(config.performance.pages_behind);
402 opts.max_buffered_rows = Some(config.performance.max_buffered_rows);
403 opts.max_buffered_mb = Some(config.performance.max_buffered_mb());
404 let row_numbers = args
405 .row_numbers
406 .map(crate::config::RowNumbers::from)
407 .unwrap_or(config.display.row_numbers);
408 opts.row_numbers = row_numbers == crate::config::RowNumbers::On;
409 opts.row_numbers_auto = row_numbers == crate::config::RowNumbers::Auto;
410 opts.row_start_index = config.display.row_numbers_start;
411 opts.single_spine_schema = config.read.parquet_schema == ParquetSchema::Union;
412 opts.decompress_in_memory = config.read.decompress_in_memory;
413 opts.normalize = config.read.audio_float;
414 opts.polars_streaming = config.performance.streaming;
415
416 // Typing string columns, dates among them: the flag, else `read.infer_types`.
417 let infer = match &args.infer_types {
418 Some(cli::InferTypes::All) => InferTypes::Switch(true),
419 Some(cli::InferTypes::Off) => InferTypes::Switch(false),
420 Some(cli::InferTypes::Columns(cols)) => InferTypes::Columns(cols.clone()),
421 None => config.read.infer_types.clone(),
422 };
423 (opts.parse_strings, opts.parse_dates) = match infer {
424 InferTypes::Switch(false) => (None, false),
425 InferTypes::Switch(true) => (Some(ParseStringsTarget::All), true),
426 InferTypes::Columns(cols) => (Some(ParseStringsTarget::Columns(cols)), true),
427 };
428
429 // CSV dialect: a flag beats the config.
430 let csv = &config.csv;
431 opts.comment_char = args.comment.clone().or_else(|| csv.comment.clone());
432 opts.header_join = csv.header_join.clone();
433 opts.skip_initial_space = args.skip_initial_space.unwrap_or(csv.skip_initial_space);
434 opts.typed_dialect = TypedDialect::from_args(args);
435 opts.ignore_errors = args.ignore_errors.unwrap_or(csv.ignore_errors);
436 // `--null` replaces the config's list, as every flag replaces its key.
437 let nulls = if args.null.is_empty() {
438 csv.null_values.clone()
439 } else {
440 args.null.clone()
441 };
442 opts.null_values = (!nulls.is_empty()).then_some(nulls);
443 // One row count for one guess, Polars' and datui's alike.
444 let infer_rows = args.infer_rows.unwrap_or(csv.infer_rows);
445 opts.infer_schema_length = Some(infer_rows);
446 opts.parse_strings_sample_rows = infer_rows;
447
448 opts.temp_dir = args.temp_dir.clone().or_else(|| {
449 config
450 .read
451 .temp_dir
452 .as_deref()
453 .map(crate::config::expand_config_path)
454 });
455
456 opts
457 }
458}
459
460impl From<&cli::Args> for OpenOptions {
461 fn from(args: &cli::Args) -> Self {
462 // Use default config if creating from args alone
463 let config = AppConfig::default();
464 Self::from_args_and_config(args, &config)
465 }
466}
467
468/// What a read of a directory found out about itself on the way through.
469///
470/// Filled by the pass that actually picks the files and the reader, and carried back on
471/// the options so the dataset can say it in the Notes. Everything here is about what
472/// datui *did*, not about what the data is — the footer notes are the other half, and
473/// they are written later, by whatever read the footers.
474#[derive(Debug, Clone, Default)]
475pub struct ReadReport {
476 /// Data files in the directory this read passed over, by format and count. A
477 /// directory of more than one format is read as the commonest of them; this is the
478 /// rest.
479 pub left_out: Vec<(FileFormat, usize)>,
480 /// How the files read differed. See [`OpenOptions::files_disagree`].
481 pub files_disagree: crate::schema_union::Disagreement,
482 /// The reader the files were read with, where the read chose it: a directory's
483 /// commonest format, or a file's extension. Carried back as `OpenOptions::format`,
484 /// so what is on screen knows whether it has a header row to turn off.
485 pub format: Option<FileFormat>,
486 /// What a read through a format spec found. See `OpenOptions::format_read`.
487 pub format_read: Option<Arc<crate::formats::Read>>,
488 /// See [`OpenOptions::read_python`].
489 pub read_python: Vec<String>,
490 /// A SQLite table opened in place. See `OpenOptions::sqlite`.
491 pub sqlite: Option<Arc<SqliteOpen>>,
492 /// See [`OpenOptions::opened`].
493 pub opened: Option<Arc<crate::members::Opened>>,
494 /// The split a Hugging Face cache directory was read as. See `OpenOptions::splits`.
495 pub splits: Option<Arc<crate::hf_splits::Splits>>,
496 /// What a read through a delimited spec found. See `OpenOptions::delimited`.
497 pub delimited: Option<Arc<crate::delimited_spec::DelimitedRead>>,
498 /// The table the read opened where the open named none: a database's only table.
499 /// Carried back as `OpenOptions::table`, so the dataset says which table it is.
500 pub table: Option<String>,
501 /// The format was guessed from the text's first bytes. See
502 /// [`OpenOptions::format_guessed`].
503 pub guessed: bool,
504 /// See [`OpenOptions::read_notes`].
505 pub read_notes: Vec<crate::notes::Note>,
506 /// See [`OpenOptions::typing`].
507 pub typing: crate::widgets::datatable::Typing,
508}
509
510/// A SQLite table opened in place, carried from the scan to the dataset.
511pub struct SqliteOpen {
512 pub pushdown: Arc<dyn crate::pushdown::Pushdown>,
513 /// Taken by the dataset, which stops the table's statements when it goes.
514 pub hold: std::sync::Mutex<Option<crate::sqlite::Hold>>,
515 pub other_tables: Vec<String>,
516}
517
518impl std::fmt::Debug for SqliteOpen {
519 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
520 f.debug_struct("SqliteOpen")
521 .field("other_tables", &self.other_tables)
522 .finish_non_exhaustive()
523 }
524}