Skip to main content

datui_lib/
open_options.rs

1//! How a dataset is opened: the options a command line, the Python binding or the home
2//! screen hand the load, and what a read reports back on them.
3
4use std::path::PathBuf;
5use std::sync::Arc;
6
7use crate::config::{InferTypes, ParquetSchema};
8use crate::{AppConfig, CompressionFormat, FileFormat, cli};
9
10/// Which CSV string columns to trim and parse (date/datetime/time/duration/int/float). Default: all. None = disabled (e.g. --infer-types=off).
11#[derive(Clone, Debug)]
12pub enum ParseStringsTarget {
13    /// Apply to all string columns.
14    All,
15    /// Apply only to these columns (must exist and be string type).
16    Columns(Vec<String>),
17}
18
19/// Which CSV dialect options were typed on the command line. A delimited spec's
20/// options replace config values but not these (#651).
21#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
22pub struct TypedDialect {
23    pub delimiter: bool,
24    pub comment_char: bool,
25    pub skip_initial_space: bool,
26    pub header_rows: bool,
27    pub skip_lines: bool,
28}
29
30impl TypedDialect {
31    pub fn from_args(args: &cli::Args) -> Self {
32        Self {
33            delimiter: args.delimiter.is_some(),
34            comment_char: args.comment.is_some(),
35            skip_initial_space: args.skip_initial_space.is_some(),
36            header_rows: !args.header_rows.is_empty(),
37            skip_lines: args.skip_lines.is_some(),
38        }
39    }
40}
41
42#[derive(Clone)]
43pub struct OpenOptions {
44    pub delimiter: Option<u8>,
45    pub has_header: Option<bool>,
46    pub skip_lines: Option<usize>,
47    pub skip_rows: Option<usize>,
48    /// Skip this many rows at the end of the file (e.g. vendor footer or trailing garbage). Applied after load for CSV.
49    pub skip_tail_rows: Option<usize>,
50    pub compression: Option<CompressionFormat>,
51    /// When set, bypass extension-based format detection and use this format (e.g. for URLs or temp files without extension).
52    pub format: Option<FileFormat>,
53    pub pages_lookahead: Option<usize>,
54    pub pages_lookback: Option<usize>,
55    pub max_buffered_rows: Option<usize>,
56    pub max_buffered_mb: Option<usize>,
57    pub row_numbers: bool,
58    /// Neither the flag nor the config said: `#` is on for text and logs
59    /// ([`crate::config::RowNumbers::Auto`]), decided once the format is known.
60    pub row_numbers_auto: bool,
61    pub row_start_index: usize,
62    /// When true, use hive load path for directory/glob; single file uses normal load.
63    pub hive: bool,
64    /// Data files in the directory being opened that this read passes over, by format and
65    /// count.
66    ///
67    /// A directory of more than one format is read as the commonest of them — a thousand
68    /// CSVs and one stray JSON is a directory of CSVs — and this is what the stray was,
69    /// so the dataset can say what it left out rather than the directory being refused
70    /// over it. Empty for every other open, which is all of them but one.
71    pub left_out: Vec<(FileFormat, usize)>,
72    /// Set when the directory being opened is a lake table and this read is of its plain
73    /// files: `"Delta"`, `"Iceberg"` or `"Hudi"`.
74    ///
75    /// The files are not the table. A delete leaves its rows on disk, an update leaves
76    /// the version it replaced, and compaction leaves both sides — so this read counts
77    /// rows no query of the table would return. datui does it anyway, because the
78    /// alternative was a directory the user could see and could not read at all, and
79    /// every other engine at least lets you look. What makes it honest rather than wrong
80    /// is that it is never silent: a note and a chip in the control bar say so, and both
81    /// are load-bearing.
82    pub read_as_plain_files_of: Option<&'static str>,
83    /// How the directory's own files differed, when they did.
84    ///
85    /// Only for the formats with no footer. A Parquet dataset's footers are read
86    /// anyway, and say this per column and per file in far more detail — which columns,
87    /// in how many files, and where — so saying it twice would be one vague note above
88    /// several exact ones.
89    pub files_disagree: crate::schema_union::Disagreement,
90    /// When true (default), infer Hive/partitioned Parquet schema from one file for faster "Reading schema". When false, use Polars collect_schema().
91    pub single_spine_schema: bool,
92    /// `--view NAME`: the view to apply to the dataset named on the command
93    /// line, once it is on screen. Applied to that open only; what later opens get
94    /// is `[views] auto_apply`'s business.
95    pub view: Option<String>,
96    /// When true, CSV and JSON string columns that look like dates or ISO 8601 timestamps become Date or Datetime.
97    pub parse_dates: bool,
98    /// When set, trim and parse CSV string columns: None = off, Some(true) = all columns, Some(cols) = those columns only.
99    pub parse_strings: Option<ParseStringsTarget>,
100    /// Sample size (rows) for inferring types when parse_strings is enabled; single file or multiple/partitioned.
101    pub parse_strings_sample_rows: usize,
102    /// When true, decompress a compressed CSV, TSV or PSV into memory (eager read). When false (default), decompress to a temp file and use lazy scan.
103    pub decompress_in_memory: bool,
104    /// Directory for decompression temp files. None = system default (e.g. TMPDIR).
105    pub temp_dir: Option<std::path::PathBuf>,
106    /// `--table`: which table of a file that holds several: an NMEA log's sentence
107    /// types, an Excel sheet (0-based index or name), a spec's variant. `None` is the
108    /// file's main table.
109    pub table: Option<String>,
110    /// When true, use Polars streaming engine for LazyFrame collect when the streaming feature is enabled.
111    pub polars_streaming: bool,
112    /// Null value specs for CSV: global strings and/or "COL=VAL" for per-column. Empty = use Polars default.
113    pub null_values: Option<Vec<String>>,
114    /// Number of rows to use when inferring CSV schema. None = Polars default (100). Larger values reduce risk of inferring wrong type (e.g. int then N/A).
115    pub infer_schema_length: Option<usize>,
116    /// When true, CSV reader ignores parse errors and continues with the next batch.
117    pub ignore_errors: bool,
118    /// CSV lines starting with this are comments, wherever they are (`commentChar`).
119    pub comment_char: Option<String>,
120    /// The 1-based lines, counted from the top of the file, that hold a CSV's header
121    /// (`headerRows`). Empty = Polars' own header. A layout flag, like `skip_lines`.
122    pub header_rows: Vec<usize>,
123    /// What joins a column's pieces when `header_rows` names several lines.
124    pub header_join: String,
125    /// Ignore the spaces after a CSV delimiter (`skipInitialSpace`).
126    pub skip_initial_space: bool,
127    /// Which dialect options were typed on the command line, so a spec leaves them.
128    pub typed_dialect: TypedDialect,
129    /// The debug overlay (session info, performance, query): `DATUI_DEBUG=1`.
130    pub debug: bool,
131    /// The split of a Hugging Face cache directory this read chose, the others and the
132    /// `map()` files it left out. Found by the read, or by a bucket listing, and carried
133    /// to the dataset as `left_out` is. `None` for every other open.
134    pub splits: Option<Arc<crate::hf_splits::Splits>>,
135    /// Where each Arrow input's rows are once its streams are converted: the IPC files
136    /// read in place and the streams' rows in the converted file the scan names. Set
137    /// by the load, after a conversion or a bucket's listing, never by a request.
138    pub arrow_parts: Option<Arc<Vec<crate::ipc_stream::Part>>>,
139    /// `--format FILE`: read the path through this format spec, whatever else
140    /// matches it. A URL is fetched when the open starts, into `spec_fetched`.
141    pub spec_file: Option<PathBuf>,
142    /// The spec a remote `spec_file` names, fetched by the open (`Phase::ReadingSpec`).
143    pub spec_fetched: Option<Arc<crate::formats::Spec>>,
144    /// `--dict FILE`: FIX dictionaries and DBC files over the search path's, each
145    /// taken by the reader of its kind.
146    pub dicts: Vec<PathBuf>,
147    /// The format spec named by `--format NAME`, or picked with `b`.
148    pub spec_name: Option<String>,
149    /// What a read through a format spec found, carried from the scan to the dataset.
150    pub format_read: Option<Arc<crate::formats::Read>>,
151    /// What the open did to the rows its reader gave — CSV column names trimmed, text
152    /// columns typed — as Python method calls for Copy as Python. Found by the scan,
153    /// carried to the dataset as `left_out` is. Empty for every other open.
154    pub read_python: Vec<String>,
155    /// `[read] audio_float`: integer audio samples as float in [-1, 1].
156    pub normalize: bool,
157    /// A SQLite table opened in place, carried from the scan to the dataset.
158    pub sqlite: Option<Arc<SqliteOpen>>,
159    /// What the reader found besides the frame: a window read straight from the file,
160    /// its row count, its Info panel tab, other tables and notes. Found by the scan and
161    /// carried to the dataset as `left_out` is. `None` for a reader with nothing to add.
162    pub opened: Option<Arc<crate::members::Opened>>,
163    /// The delimited spec the file is read through, once chosen: its dialect is in
164    /// these options, and the read's units and metadata ride with it to the dataset.
165    pub delimited: Option<Arc<crate::delimited_spec::DelimitedRead>>,
166    /// How the scan found the data is read: lazily, through a copy, or into memory.
167    /// Found by the scan and carried to the dataset for the Info panel's `Read:` line.
168    pub read_mode: Option<crate::ReadMode>,
169    /// The format was guessed from the first bytes of text no format's signature
170    /// claims, rather than named, so a note can say how to read it otherwise.
171    pub format_guessed: bool,
172    /// What the read of several files found to say of them, for the dataset's notes.
173    /// Found by the scan and carried to the dataset as `left_out` is.
174    pub read_notes: Vec<crate::notes::Note>,
175    /// The columns the read gave a type and the frame before, carried to the dataset
176    /// as `left_out` is, for the count of the values that did not fit.
177    pub typing: crate::widgets::datatable::Typing,
178    /// `--hex`: show the file's bytes in the hex view, whatever it holds.
179    pub hex: bool,
180    /// `--hex-width N`: the bytes a row of the hex view holds.
181    pub record_size: Option<usize>,
182    /// `--follow`: show rows as they are appended to the file, or arrive on standard
183    /// input, until stopped.
184    pub follow: bool,
185    /// Where the followed file's complete records end, counted by the scan and carried
186    /// to the dataset as `left_out` is, for its watcher to read on from.
187    pub tail: Option<Arc<crate::follow::Tail>>,
188    /// Standard input still being copied to the file a follow reads.
189    pub spool: Option<Arc<crate::follow::SpoolHandle>>,
190    /// Standard input shown as it arrives without `--follow`: read by the follow's
191    /// watcher until it ends, the view staying where it is rather than at the end.
192    pub pipe: bool,
193    /// `--tee FILE`: standard input is recorded to FILE, which is what is read.
194    pub tee: Option<PathBuf>,
195    /// `--tee-raw`: FILE is the bytes exactly as they came, a WAV header included.
196    pub tee_raw: bool,
197    /// `--force`: FILE may replace a file that is there.
198    pub force: bool,
199    /// The dataset the home screen's preview built and read the first page of, for
200    /// this open to install rather than read again. Taken once.
201    pub prepared: Option<crate::home_preview::Handoff>,
202    /// A file of the built-in catalog: downloaded without asking when it is small.
203    pub download_unasked: Option<UnaskedDownload>,
204}
205
206/// A remote file downloaded without asking: one the built-in catalog lists, at most
207/// `limit` bytes by what the server says or, when it says nothing, by the catalog.
208#[derive(Debug, Clone, Copy, PartialEq, Eq)]
209pub struct UnaskedDownload {
210    pub limit: u64,
211    pub listed: Option<u64>,
212}
213
214impl UnaskedDownload {
215    /// The largest built-in catalog file downloaded without asking.
216    pub const LIMIT: u64 = 50 * 1024 * 1024;
217
218    /// Whether a file of `size` bytes, if the server said, is downloaded unasked.
219    pub fn covers(&self, size: Option<u64>) -> bool {
220        size.or(self.listed)
221            .is_some_and(|bytes| bytes <= self.limit)
222    }
223}
224
225impl OpenOptions {
226    pub fn new() -> Self {
227        Self {
228            delimiter: None,
229            has_header: None,
230            skip_lines: None,
231            skip_rows: None,
232            skip_tail_rows: None,
233            left_out: Vec::new(),
234            read_python: Vec::new(),
235            read_as_plain_files_of: None,
236            files_disagree: Default::default(),
237            compression: None,
238            format: None,
239            pages_lookahead: None,
240            pages_lookback: None,
241            max_buffered_rows: None,
242            max_buffered_mb: None,
243            row_numbers: false,
244            row_numbers_auto: true,
245            row_start_index: 1,
246            hive: false,
247            single_spine_schema: true,
248            view: None,
249            parse_dates: true,
250            parse_strings: None,
251            parse_strings_sample_rows: 1000,
252            decompress_in_memory: false,
253            temp_dir: None,
254            table: None,
255            polars_streaming: true,
256            null_values: None,
257            infer_schema_length: None,
258            ignore_errors: false,
259            comment_char: None,
260            header_rows: Vec::new(),
261            header_join: crate::csv_dialect::DEFAULT_HEADER_JOIN.to_string(),
262            skip_initial_space: false,
263            typed_dialect: TypedDialect::default(),
264            debug: false,
265            spec_file: None,
266            spec_fetched: None,
267            dicts: Vec::new(),
268            spec_name: None,
269            format_read: None,
270            normalize: false,
271            sqlite: None,
272            opened: None,
273            splits: None,
274            arrow_parts: None,
275            delimited: None,
276            read_mode: None,
277            format_guessed: false,
278            read_notes: Vec::new(),
279            typing: Default::default(),
280            hex: false,
281            record_size: None,
282            follow: false,
283            tail: None,
284            spool: None,
285            pipe: false,
286            tee: None,
287            tee_raw: false,
288            force: false,
289            prepared: None,
290            download_unasked: None,
291        }
292    }
293}
294
295impl Default for OpenOptions {
296    fn default() -> Self {
297        Self::new()
298    }
299}
300
301impl OpenOptions {
302    pub fn with_skip_lines(mut self, skip_lines: usize) -> Self {
303        self.skip_lines = Some(skip_lines);
304        self
305    }
306
307    pub fn with_skip_rows(mut self, skip_rows: usize) -> Self {
308        self.skip_rows = Some(skip_rows);
309        self
310    }
311
312    pub fn with_delimiter(mut self, delimiter: u8) -> Self {
313        self.delimiter = Some(delimiter);
314        self
315    }
316
317    pub fn with_has_header(mut self, has_header: bool) -> Self {
318        self.has_header = Some(has_header);
319        self
320    }
321
322    /// The separator a delimited file is read with: `--delimiter` when given, else
323    /// the one its format implies (`FileFormat::separator`).
324    pub fn separator_or(&self, format_default: u8) -> u8 {
325        self.delimiter.unwrap_or(format_default)
326    }
327
328    pub fn with_compression(mut self, compression: CompressionFormat) -> Self {
329        self.compression = Some(compression);
330        self
331    }
332
333    /// The lines `--header-rows` named, unless the file is being read without a
334    /// header (`--no-header`, or `H`), which reads them as data.
335    pub fn header_rows(&self) -> Option<&[usize]> {
336        (!self.header_rows.is_empty() && self.has_header != Some(false))
337            .then_some(self.header_rows.as_slice())
338    }
339
340    /// When loading CSV: use Polars try_parse_dates only if parse_strings is not set.
341    /// When parse_strings is set we do our own date parsing (with strict: false), so we disable
342    /// Polars' try_parse_dates to avoid "could not find an appropriate format" errors.
343    pub fn csv_try_parse_dates(&self) -> bool {
344        self.parse_strings.is_none() && self.parse_dates
345    }
346
347    /// The S3 settings every cloud path uses: the environment over the `[cloud]`
348    /// config. `run()` folds this into the config the `App` keeps,
349    /// so opening, sizing, downloading, discovery and listing all see one answer and a
350    /// bucket that is listed is reached the way it will be opened. The environment is
351    /// read here, not when the options are built, so a caller that starts from
352    /// `OpenOptions::default()` — the Python bindings do — still honours it.
353    pub fn effective_cloud(
354        &self,
355        cloud: &crate::config::CloudConfig,
356    ) -> crate::config::CloudConfig {
357        let mut merged = cloud.clone();
358        merged.overlay(crate::config::CloudConfig::from_env(&crate::cloud_env::var));
359        merged
360    }
361}
362
363impl OpenOptions {
364    /// The options the command line and the config give an open. The config has had
365    /// `-c` laid over it; a flag here beats both.
366    pub fn from_args_and_config(args: &cli::Args, config: &AppConfig) -> Self {
367        let mut opts = OpenOptions::new();
368
369        // A file's layout: command line only. Set in config, these applied to every
370        // file opened and silently cut rows from the ones they did not describe (#289).
371        opts.delimiter = args.delimiter;
372        opts.skip_lines = args.skip_lines;
373        opts.skip_rows = args.skip_rows;
374        opts.skip_tail_rows = args.footer_rows;
375        opts.has_header = args.no_header.then_some(false);
376        opts.header_rows = args.header_rows.iter().map(|&n| n as usize).collect();
377        opts.view = args.view.clone();
378        opts.compression = args.compression;
379
380        // A spec's name is looked up on the search path when the file is opened.
381        opts.format = args.format.as_ref().and_then(cli::FormatChoice::builtin);
382        opts.spec_name = args
383            .format
384            .as_ref()
385            .and_then(|f| f.spec().map(str::to_string));
386        opts.spec_file = args
387            .format
388            .as_ref()
389            .and_then(|f| f.spec_file().map(std::path::Path::to_path_buf));
390        opts.dicts = args.dict.clone();
391        opts.table = args.table.clone();
392        opts.hex = args.hex;
393        opts.record_size = args.hex_width.map(usize::from);
394        opts.hive = args.hive;
395        opts.follow = args.follow;
396        opts.tee = args.tee.clone();
397        opts.tee_raw = args.tee_raw;
398        opts.force = args.force;
399
400        opts.pages_lookahead = Some(config.performance.pages_ahead);
401        opts.pages_lookback = Some(config.performance.pages_behind);
402        opts.max_buffered_rows = Some(config.performance.max_buffered_rows);
403        opts.max_buffered_mb = Some(config.performance.max_buffered_mb());
404        let row_numbers = args
405            .row_numbers
406            .map(crate::config::RowNumbers::from)
407            .unwrap_or(config.display.row_numbers);
408        opts.row_numbers = row_numbers == crate::config::RowNumbers::On;
409        opts.row_numbers_auto = row_numbers == crate::config::RowNumbers::Auto;
410        opts.row_start_index = config.display.row_numbers_start;
411        opts.single_spine_schema = config.read.parquet_schema == ParquetSchema::Union;
412        opts.decompress_in_memory = config.read.decompress_in_memory;
413        opts.normalize = config.read.audio_float;
414        opts.polars_streaming = config.performance.streaming;
415
416        // Typing string columns, dates among them: the flag, else `read.infer_types`.
417        let infer = match &args.infer_types {
418            Some(cli::InferTypes::All) => InferTypes::Switch(true),
419            Some(cli::InferTypes::Off) => InferTypes::Switch(false),
420            Some(cli::InferTypes::Columns(cols)) => InferTypes::Columns(cols.clone()),
421            None => config.read.infer_types.clone(),
422        };
423        (opts.parse_strings, opts.parse_dates) = match infer {
424            InferTypes::Switch(false) => (None, false),
425            InferTypes::Switch(true) => (Some(ParseStringsTarget::All), true),
426            InferTypes::Columns(cols) => (Some(ParseStringsTarget::Columns(cols)), true),
427        };
428
429        // CSV dialect: a flag beats the config.
430        let csv = &config.csv;
431        opts.comment_char = args.comment.clone().or_else(|| csv.comment.clone());
432        opts.header_join = csv.header_join.clone();
433        opts.skip_initial_space = args.skip_initial_space.unwrap_or(csv.skip_initial_space);
434        opts.typed_dialect = TypedDialect::from_args(args);
435        opts.ignore_errors = args.ignore_errors.unwrap_or(csv.ignore_errors);
436        // `--null` replaces the config's list, as every flag replaces its key.
437        let nulls = if args.null.is_empty() {
438            csv.null_values.clone()
439        } else {
440            args.null.clone()
441        };
442        opts.null_values = (!nulls.is_empty()).then_some(nulls);
443        // One row count for one guess, Polars' and datui's alike.
444        let infer_rows = args.infer_rows.unwrap_or(csv.infer_rows);
445        opts.infer_schema_length = Some(infer_rows);
446        opts.parse_strings_sample_rows = infer_rows;
447
448        opts.temp_dir = args.temp_dir.clone().or_else(|| {
449            config
450                .read
451                .temp_dir
452                .as_deref()
453                .map(crate::config::expand_config_path)
454        });
455
456        opts
457    }
458}
459
460impl From<&cli::Args> for OpenOptions {
461    fn from(args: &cli::Args) -> Self {
462        // Use default config if creating from args alone
463        let config = AppConfig::default();
464        Self::from_args_and_config(args, &config)
465    }
466}
467
468/// What a read of a directory found out about itself on the way through.
469///
470/// Filled by the pass that actually picks the files and the reader, and carried back on
471/// the options so the dataset can say it in the Notes. Everything here is about what
472/// datui *did*, not about what the data is — the footer notes are the other half, and
473/// they are written later, by whatever read the footers.
474#[derive(Debug, Clone, Default)]
475pub struct ReadReport {
476    /// Data files in the directory this read passed over, by format and count. A
477    /// directory of more than one format is read as the commonest of them; this is the
478    /// rest.
479    pub left_out: Vec<(FileFormat, usize)>,
480    /// How the files read differed. See [`OpenOptions::files_disagree`].
481    pub files_disagree: crate::schema_union::Disagreement,
482    /// The reader the files were read with, where the read chose it: a directory's
483    /// commonest format, or a file's extension. Carried back as `OpenOptions::format`,
484    /// so what is on screen knows whether it has a header row to turn off.
485    pub format: Option<FileFormat>,
486    /// What a read through a format spec found. See `OpenOptions::format_read`.
487    pub format_read: Option<Arc<crate::formats::Read>>,
488    /// See [`OpenOptions::read_python`].
489    pub read_python: Vec<String>,
490    /// A SQLite table opened in place. See `OpenOptions::sqlite`.
491    pub sqlite: Option<Arc<SqliteOpen>>,
492    /// See [`OpenOptions::opened`].
493    pub opened: Option<Arc<crate::members::Opened>>,
494    /// The split a Hugging Face cache directory was read as. See `OpenOptions::splits`.
495    pub splits: Option<Arc<crate::hf_splits::Splits>>,
496    /// What a read through a delimited spec found. See `OpenOptions::delimited`.
497    pub delimited: Option<Arc<crate::delimited_spec::DelimitedRead>>,
498    /// The table the read opened where the open named none: a database's only table.
499    /// Carried back as `OpenOptions::table`, so the dataset says which table it is.
500    pub table: Option<String>,
501    /// The format was guessed from the text's first bytes. See
502    /// [`OpenOptions::format_guessed`].
503    pub guessed: bool,
504    /// See [`OpenOptions::read_notes`].
505    pub read_notes: Vec<crate::notes::Note>,
506    /// See [`OpenOptions::typing`].
507    pub typing: crate::widgets::datatable::Typing,
508}
509
510/// A SQLite table opened in place, carried from the scan to the dataset.
511pub struct SqliteOpen {
512    pub pushdown: Arc<dyn crate::pushdown::Pushdown>,
513    /// Taken by the dataset, which stops the table's statements when it goes.
514    pub hold: std::sync::Mutex<Option<crate::sqlite::Hold>>,
515    pub other_tables: Vec<String>,
516}
517
518impl std::fmt::Debug for SqliteOpen {
519    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
520        f.debug_struct("SqliteOpen")
521            .field("other_tables", &self.other_tables)
522            .finish_non_exhaustive()
523    }
524}