Skip to main content

datui_lib/loading/
open_options.rs

1//! How a dataset is opened: the options a command line, the Python binding or the home
2//! screen hand the load, and what a read reports back on them.
3
4use std::path::PathBuf;
5use std::sync::Arc;
6
7use crate::config::{InferTypes, ParquetSchema};
8use crate::{AppConfig, CompressionFormat, FileFormat, cli};
9
10/// Which CSV string columns to trim and parse (date/datetime/time/duration/int/float). Default: all. None = disabled (e.g. --infer-types=off).
11#[derive(Clone, Debug)]
12pub enum ParseStringsTarget {
13    /// Apply to all string columns.
14    All,
15    /// Apply only to these columns (must exist and be string type).
16    Columns(Vec<String>),
17}
18
19/// Which CSV dialect options were typed on the command line. A delimited spec's
20/// options replace config values but not these (#651).
21#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
22pub struct TypedDialect {
23    pub delimiter: bool,
24    pub comment_char: bool,
25    pub skip_initial_space: bool,
26    pub header_rows: bool,
27    pub skip_lines: bool,
28}
29
30impl TypedDialect {
31    pub fn from_args(args: &cli::Args) -> Self {
32        Self {
33            delimiter: args.delimiter.is_some(),
34            comment_char: args.comment.is_some(),
35            skip_initial_space: args.skip_initial_space.is_some(),
36            header_rows: !args.header_rows.is_empty(),
37            skip_lines: args.skip_lines.is_some(),
38        }
39    }
40}
41
42#[derive(Clone)]
43pub struct OpenOptions {
44    pub delimiter: Option<u8>,
45    pub has_header: Option<bool>,
46    pub skip_lines: Option<usize>,
47    pub skip_rows: Option<usize>,
48    /// Skip this many rows at the end of the file (e.g. vendor footer or trailing garbage). Applied after load for CSV.
49    pub skip_tail_rows: Option<usize>,
50    pub compression: Option<CompressionFormat>,
51    /// When set, bypass extension-based format detection and use this format (e.g. for URLs or temp files without extension).
52    pub format: Option<FileFormat>,
53    pub pages_lookahead: Option<usize>,
54    pub pages_lookback: Option<usize>,
55    pub max_buffered_rows: Option<usize>,
56    pub max_buffered_mb: Option<usize>,
57    pub row_numbers: bool,
58    /// Neither the flag nor the config said: `#` is on for text and logs
59    /// ([`crate::config::RowNumbers::Auto`]), decided once the format is known.
60    pub row_numbers_auto: bool,
61    pub row_start_index: usize,
62    /// When true, use hive load path for directory/glob; single file uses normal load.
63    pub hive: bool,
64    /// Data files in the opened directory this read passes over, by format and count: a
65    /// mixed directory reads as its commonest format, and the dataset says what it left
66    /// out. Empty for every other open.
67    pub left_out: Vec<(FileFormat, usize)>,
68    /// Set (`"Delta"`, `"Iceberg"` or `"Hudi"`) when a lake table's plain files are read.
69    /// They are not the table (deleted rows, old versions and compaction leftovers count),
70    /// so a note and a footer chip always say so.
71    pub read_as_plain_files_of: Option<&'static str>,
72    /// How the directory's files differed, for footerless formats only (Parquet footers
73    /// give the exact per-column version).
74    pub files_disagree: crate::formats::schema_union::Disagreement,
75    /// When true (default), infer Hive/partitioned Parquet schema from one file for faster "Reading schema". When false, use Polars collect_schema().
76    pub single_spine_schema: bool,
77    /// `--view NAME`: the view to apply to the dataset named on the command
78    /// line, once it is on screen. Applied to that open only; what later opens get
79    /// is `[views] auto_apply`'s business.
80    pub view: Option<String>,
81    /// When true, CSV and JSON string columns that look like dates or ISO 8601 timestamps become Date or Datetime.
82    pub parse_dates: bool,
83    /// When set, trim and parse CSV string columns: None = off, Some(true) = all columns, Some(cols) = those columns only.
84    pub parse_strings: Option<ParseStringsTarget>,
85    /// Sample size (rows) for inferring types when parse_strings is enabled; single file or multiple/partitioned.
86    pub parse_strings_sample_rows: usize,
87    /// When true, decompress a compressed CSV, TSV or PSV into memory (eager read). When false (default), decompress to a temp file and use lazy scan.
88    pub decompress_in_memory: bool,
89    /// Directory for decompression temp files. None = system default (e.g. TMPDIR).
90    pub temp_dir: Option<std::path::PathBuf>,
91    /// `--table`: which table of a file that holds several: an NMEA log's sentence
92    /// types, an Excel sheet (0-based index or name), a spec's variant. `None` is the
93    /// file's main table.
94    pub table: Option<String>,
95    /// When true, use Polars streaming engine for LazyFrame collect when the streaming feature is enabled.
96    pub polars_streaming: bool,
97    /// Null value specs for CSV: global strings and/or "COL=VAL" for per-column. Empty = use Polars default.
98    pub null_values: Option<Vec<String>>,
99    /// Number of rows to use when inferring CSV schema. None = Polars default (100). Larger values reduce risk of inferring wrong type (e.g. int then N/A).
100    pub infer_schema_length: Option<usize>,
101    /// When true, CSV reader ignores parse errors and continues with the next batch.
102    pub ignore_errors: bool,
103    /// CSV lines starting with this are comments, wherever they are (`commentChar`).
104    pub comment_char: Option<String>,
105    /// The 1-based lines, counted from the top of the file, that hold a CSV's header
106    /// (`headerRows`). Empty = Polars' own header. A layout flag, like `skip_lines`.
107    pub header_rows: Vec<usize>,
108    /// What joins a column's pieces when `header_rows` names several lines.
109    pub header_join: String,
110    /// Ignore the spaces after a CSV delimiter (`skipInitialSpace`).
111    pub skip_initial_space: bool,
112    /// Which dialect options were typed on the command line, so a spec leaves them.
113    pub typed_dialect: TypedDialect,
114    /// The debug overlay (session info, performance, query): `DATUI_DEBUG=1`.
115    pub debug: bool,
116    /// The split of a Hugging Face cache directory this read chose, the others and the
117    /// `map()` files it left out. Found by the read, or by a bucket listing, and carried
118    /// to the dataset as `left_out` is. `None` for every other open.
119    pub splits: Option<Arc<crate::formats::hf_splits::Splits>>,
120    /// Where each Arrow input's rows are once its streams are converted: the IPC files
121    /// read in place and the streams' rows in the converted file the scan names. Set
122    /// by the load, after a conversion or a bucket's listing, never by a request.
123    pub arrow_parts: Option<Arc<Vec<crate::formats::ipc_stream::Part>>>,
124    /// `--format FILE`: read the path through this format spec, whatever else
125    /// matches it. A URL is fetched when the open starts, into `spec_fetched`.
126    pub spec_file: Option<PathBuf>,
127    /// The spec a remote `spec_file` names, fetched by the open (`Phase::ReadingSpec`).
128    pub spec_fetched: Option<Arc<crate::formats::Spec>>,
129    /// `--dict FILE`: FIX dictionaries and DBC files over the search path's, each
130    /// taken by the reader of its kind.
131    pub dicts: Vec<PathBuf>,
132    /// The format spec named by `--format NAME`, or picked with `b`.
133    pub spec_name: Option<String>,
134    /// What a read through a format spec found, carried from the scan to the dataset.
135    pub format_read: Option<Arc<crate::formats::Read>>,
136    /// What the open did to the rows its reader gave — CSV column names trimmed, text
137    /// columns typed — as Python method calls for Copy as Python. Found by the scan,
138    /// carried to the dataset as `left_out` is. Empty for every other open.
139    pub read_python: Vec<String>,
140    /// `[read] audio_float`: integer audio samples as float in [-1, 1].
141    pub normalize: bool,
142    /// A SQLite table opened in place, carried from the scan to the dataset.
143    pub sqlite: Option<Arc<SqliteOpen>>,
144    /// What the reader found besides the frame: a window read straight from the file,
145    /// its row count, its Info panel tab, other tables and notes. Found by the scan and
146    /// carried to the dataset as `left_out` is. `None` for a reader with nothing to add.
147    pub opened: Option<Arc<crate::formats::members::Opened>>,
148    /// The delimited spec the file is read through, once chosen: its dialect is in
149    /// these options, and the read's units and metadata ride with it to the dataset.
150    pub delimited: Option<Arc<crate::formats::delimited_spec::DelimitedRead>>,
151    /// How the scan found the data is read: lazily, through a copy, or into memory.
152    /// Found by the scan and carried to the dataset for the Info panel's `Read:` line.
153    pub read_mode: Option<crate::ReadMode>,
154    /// The format was guessed from the first bytes of text no format's signature
155    /// claims, rather than named, so a note can say how to read it otherwise.
156    pub format_guessed: bool,
157    /// What the read of several files found to say of them, for the dataset's notes.
158    /// Found by the scan and carried to the dataset as `left_out` is.
159    pub read_notes: Vec<crate::notes::Note>,
160    /// The columns the read gave a type and the frame before, carried to the dataset
161    /// as `left_out` is, for the count of the values that did not fit.
162    pub typing: crate::formats::readers::Typing,
163    /// `--hex`: show the file's bytes in the hex view, whatever it holds.
164    pub hex: bool,
165    /// `--hex-width N`: the bytes a row of the hex view holds.
166    pub record_size: Option<usize>,
167    /// `--follow`: show rows as they are appended to the file, or arrive on standard
168    /// input, until stopped.
169    pub follow: bool,
170    /// Where the followed file's complete records end, counted by the scan and carried
171    /// to the dataset as `left_out` is, for its watcher to read on from.
172    pub tail: Option<Arc<crate::loading::follow::Tail>>,
173    /// Standard input still being copied to the file a follow reads.
174    pub spool: Option<Arc<crate::loading::follow::SpoolHandle>>,
175    /// Standard input shown as it arrives without `--follow`: read by the follow's
176    /// watcher until it ends, the view staying where it is rather than at the end.
177    pub pipe: bool,
178    /// `--tee FILE`: standard input is recorded to FILE, which is what is read.
179    pub tee: Option<PathBuf>,
180    /// `--tee-raw`: FILE is the bytes exactly as they came, a WAV header included.
181    pub tee_raw: bool,
182    /// `--force`: FILE may replace a file that is there.
183    pub force: bool,
184    /// The dataset the home screen's preview built and read the first page of, for
185    /// this open to install rather than read again. Taken once.
186    pub prepared: Option<crate::home::home_preview::Handoff>,
187    /// A file of the built-in catalog: downloaded without asking when it is small.
188    pub download_unasked: Option<UnaskedDownload>,
189}
190
191/// A remote file downloaded without asking: one the built-in catalog lists, at most
192/// `limit` bytes by what the server says or, when it says nothing, by the catalog.
193#[derive(Debug, Clone, Copy, PartialEq, Eq)]
194pub struct UnaskedDownload {
195    pub limit: u64,
196    pub listed: Option<u64>,
197}
198
199impl UnaskedDownload {
200    /// The largest built-in catalog file downloaded without asking.
201    pub const LIMIT: u64 = 50 * 1024 * 1024;
202
203    /// Whether a file of `size` bytes, if the server said, is downloaded unasked.
204    pub fn covers(&self, size: Option<u64>) -> bool {
205        size.or(self.listed)
206            .is_some_and(|bytes| bytes <= self.limit)
207    }
208}
209
210impl OpenOptions {
211    pub fn new() -> Self {
212        Self {
213            delimiter: None,
214            has_header: None,
215            skip_lines: None,
216            skip_rows: None,
217            skip_tail_rows: None,
218            left_out: Vec::new(),
219            read_python: Vec::new(),
220            read_as_plain_files_of: None,
221            files_disagree: Default::default(),
222            compression: None,
223            format: None,
224            pages_lookahead: None,
225            pages_lookback: None,
226            max_buffered_rows: None,
227            max_buffered_mb: None,
228            row_numbers: false,
229            row_numbers_auto: true,
230            row_start_index: 1,
231            hive: false,
232            single_spine_schema: true,
233            view: None,
234            parse_dates: true,
235            parse_strings: None,
236            parse_strings_sample_rows: 1000,
237            decompress_in_memory: false,
238            temp_dir: None,
239            table: None,
240            polars_streaming: true,
241            null_values: None,
242            infer_schema_length: None,
243            ignore_errors: false,
244            comment_char: None,
245            header_rows: Vec::new(),
246            header_join: crate::formats::csv_dialect::DEFAULT_HEADER_JOIN.to_string(),
247            skip_initial_space: false,
248            typed_dialect: TypedDialect::default(),
249            debug: false,
250            spec_file: None,
251            spec_fetched: None,
252            dicts: Vec::new(),
253            spec_name: None,
254            format_read: None,
255            normalize: false,
256            sqlite: None,
257            opened: None,
258            splits: None,
259            arrow_parts: None,
260            delimited: None,
261            read_mode: None,
262            format_guessed: false,
263            read_notes: Vec::new(),
264            typing: Default::default(),
265            hex: false,
266            record_size: None,
267            follow: false,
268            tail: None,
269            spool: None,
270            pipe: false,
271            tee: None,
272            tee_raw: false,
273            force: false,
274            prepared: None,
275            download_unasked: None,
276        }
277    }
278}
279
280impl Default for OpenOptions {
281    fn default() -> Self {
282        Self::new()
283    }
284}
285
286impl OpenOptions {
287    #[cfg(test)]
288    pub fn with_skip_lines(mut self, skip_lines: usize) -> Self {
289        self.skip_lines = Some(skip_lines);
290        self
291    }
292
293    /// The separator a delimited file is read with: `--delimiter` when given, else
294    /// the one its format implies (`FileFormat::separator`).
295    pub fn separator_or(&self, format_default: u8) -> u8 {
296        self.delimiter.unwrap_or(format_default)
297    }
298
299    /// The lines `--header-rows` named, unless the file is being read without a
300    /// header (`--no-header`, or `H`), which reads them as data.
301    pub fn header_rows(&self) -> Option<&[usize]> {
302        (!self.header_rows.is_empty() && self.has_header != Some(false))
303            .then_some(self.header_rows.as_slice())
304    }
305
306    /// When loading CSV: use Polars try_parse_dates only if parse_strings is not set.
307    /// When parse_strings is set we do our own date parsing (with strict: false), so we disable
308    /// Polars' try_parse_dates to avoid "could not find an appropriate format" errors.
309    pub fn csv_try_parse_dates(&self) -> bool {
310        self.parse_strings.is_none() && self.parse_dates
311    }
312
313    /// The S3 settings every cloud path uses: environment over `[cloud]` config. `run()`
314    /// folds this into the App's config so opening, sizing, downloading, discovery and
315    /// listing agree. The environment is read here, so `OpenOptions::default()` callers
316    /// (Python) honor it too.
317    pub fn effective_cloud(
318        &self,
319        cloud: &crate::config::CloudConfig,
320    ) -> crate::config::CloudConfig {
321        let mut merged = cloud.clone();
322        merged.overlay(crate::config::CloudConfig::from_env(
323            &crate::cloud::cloud_env::var,
324        ));
325        merged
326    }
327}
328
329impl OpenOptions {
330    /// The options the command line and the config give an open. The config has had
331    /// `-c` laid over it; a flag here beats both.
332    pub fn from_args_and_config(args: &cli::Args, config: &AppConfig) -> Self {
333        let mut opts = OpenOptions::new();
334
335        // A file's layout: command line only. Set in config, these applied to every
336        // file opened and silently cut rows from the ones they did not describe (#289).
337        opts.delimiter = args.delimiter;
338        opts.skip_lines = args.skip_lines;
339        opts.skip_rows = args.skip_rows;
340        opts.skip_tail_rows = args.footer_rows;
341        opts.has_header = args.no_header.then_some(false);
342        opts.header_rows = args.header_rows.iter().map(|&n| n as usize).collect();
343        opts.view = args.view.clone();
344        opts.compression = args.compression;
345
346        // A spec's name is looked up on the search path when the file is opened.
347        opts.format = args.format.as_ref().and_then(cli::FormatChoice::builtin);
348        opts.spec_name = args
349            .format
350            .as_ref()
351            .and_then(|f| f.spec().map(str::to_string));
352        opts.spec_file = args
353            .format
354            .as_ref()
355            .and_then(|f| f.spec_file().map(std::path::Path::to_path_buf));
356        opts.dicts = args.dict.clone();
357        opts.table = args.table.clone();
358        opts.hex = args.hex;
359        opts.record_size = args.hex_width.map(usize::from);
360        opts.hive = args.hive;
361        opts.follow = args.follow;
362        opts.tee = args.tee.clone();
363        opts.tee_raw = args.tee_raw;
364        opts.force = args.force;
365
366        opts.pages_lookahead = Some(config.performance.pages_ahead);
367        opts.pages_lookback = Some(config.performance.pages_behind);
368        opts.max_buffered_rows = Some(config.performance.max_buffered_rows);
369        opts.max_buffered_mb = Some(config.performance.max_buffered_mb());
370        let row_numbers = args
371            .row_numbers
372            .map(crate::config::RowNumbers::from)
373            .unwrap_or(config.display.row_numbers);
374        opts.row_numbers = row_numbers == crate::config::RowNumbers::On;
375        opts.row_numbers_auto = row_numbers == crate::config::RowNumbers::Auto;
376        opts.row_start_index = config.display.row_numbers_start;
377        opts.single_spine_schema = config.read.parquet_schema == ParquetSchema::Union;
378        opts.decompress_in_memory = config.read.decompress_in_memory;
379        opts.normalize = config.read.audio_float;
380        opts.polars_streaming = config.performance.streaming;
381
382        // Typing string columns, dates among them: the flag, else `read.infer_types`.
383        let infer = match &args.infer_types {
384            Some(cli::InferTypes::All) => InferTypes::Switch(true),
385            Some(cli::InferTypes::Off) => InferTypes::Switch(false),
386            Some(cli::InferTypes::Columns(cols)) => InferTypes::Columns(cols.clone()),
387            None => config.read.infer_types.clone(),
388        };
389        (opts.parse_strings, opts.parse_dates) = match infer {
390            InferTypes::Switch(false) => (None, false),
391            InferTypes::Switch(true) => (Some(ParseStringsTarget::All), true),
392            InferTypes::Columns(cols) => (Some(ParseStringsTarget::Columns(cols)), true),
393        };
394
395        // CSV dialect: a flag beats the config.
396        let csv = &config.csv;
397        opts.comment_char = args.comment.clone().or_else(|| csv.comment.clone());
398        opts.header_join = csv.header_join.clone();
399        opts.skip_initial_space = args.skip_initial_space.unwrap_or(csv.skip_initial_space);
400        opts.typed_dialect = TypedDialect::from_args(args);
401        opts.ignore_errors = args.ignore_errors.unwrap_or(csv.ignore_errors);
402        // `--null` replaces the config's list, as every flag replaces its key.
403        let nulls = if args.null.is_empty() {
404            csv.null_values.clone()
405        } else {
406            args.null.clone()
407        };
408        opts.null_values = (!nulls.is_empty()).then_some(nulls);
409        // One row count for one guess, Polars' and datui's alike.
410        let infer_rows = args.infer_rows.unwrap_or(csv.infer_rows);
411        opts.infer_schema_length = Some(infer_rows);
412        opts.parse_strings_sample_rows = infer_rows;
413
414        opts.temp_dir = args.temp_dir.clone().or_else(|| {
415            config
416                .read
417                .temp_dir
418                .as_deref()
419                .map(crate::config::expand_config_path)
420        });
421
422        opts
423    }
424}
425
426impl From<&cli::Args> for OpenOptions {
427    fn from(args: &cli::Args) -> Self {
428        let config = AppConfig::default();
429        Self::from_args_and_config(args, &config)
430    }
431}
432
433/// What a directory read found about itself, filled by the pass that picks files and
434/// reader and carried back for the Notes: what datui did, not what the data is (the
435/// footer notes, written later, are that).
436#[derive(Debug, Clone, Default)]
437pub struct ReadReport {
438    /// Data files in the directory this read passed over, by format and count. A
439    /// directory of more than one format is read as the commonest of them; this is the
440    /// rest.
441    pub left_out: Vec<(FileFormat, usize)>,
442    /// How the files read differed. See [`OpenOptions::files_disagree`].
443    pub files_disagree: crate::formats::schema_union::Disagreement,
444    /// The reader the files were read with, where the read chose it: a directory's
445    /// commonest format, or a file's extension. Carried back as `OpenOptions::format`,
446    /// so what is on screen knows whether it has a header row to turn off.
447    pub format: Option<FileFormat>,
448    /// What a read through a format spec found. See `OpenOptions::format_read`.
449    pub format_read: Option<Arc<crate::formats::Read>>,
450    /// See [`OpenOptions::read_python`].
451    pub read_python: Vec<String>,
452    /// A SQLite table opened in place. See `OpenOptions::sqlite`.
453    pub sqlite: Option<Arc<SqliteOpen>>,
454    /// See [`OpenOptions::opened`].
455    pub opened: Option<Arc<crate::formats::members::Opened>>,
456    /// The split a Hugging Face cache directory was read as. See `OpenOptions::splits`.
457    pub splits: Option<Arc<crate::formats::hf_splits::Splits>>,
458    /// What a read through a delimited spec found. See `OpenOptions::delimited`.
459    pub delimited: Option<Arc<crate::formats::delimited_spec::DelimitedRead>>,
460    /// The table the read opened where the open named none: a database's only table.
461    /// Carried back as `OpenOptions::table`, so the dataset says which table it is.
462    pub table: Option<String>,
463    /// The format was guessed from the text's first bytes. See
464    /// [`OpenOptions::format_guessed`].
465    pub guessed: bool,
466    /// See [`OpenOptions::read_notes`].
467    pub read_notes: Vec<crate::notes::Note>,
468    /// See [`OpenOptions::typing`].
469    pub typing: crate::formats::readers::Typing,
470}
471
472/// A SQLite table opened in place, carried from the scan to the dataset.
473pub struct SqliteOpen {
474    pub pushdown: Arc<dyn crate::formats::pushdown::Pushdown>,
475    /// Taken by the dataset, which stops the table's statements when it goes.
476    pub hold: std::sync::Mutex<Option<crate::formats::sqlite::Hold>>,
477    pub other_tables: Vec<String>,
478}
479
480impl std::fmt::Debug for SqliteOpen {
481    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
482        f.debug_struct("SqliteOpen")
483            .field("other_tables", &self.other_tables)
484            .finish_non_exhaustive()
485    }
486}