Skip to main content

datui_lib/widgets/
datatable.rs

1use color_eyre::Result;
2use std::borrow::Cow;
3use std::collections::HashSet;
4use std::sync::Arc;
5use std::{fs, fs::File, path::Path, path::PathBuf};
6
7use polars::frame::PivotColumnNaming;
8use polars::io::HiveOptions;
9use polars::prelude::*;
10use ratatui::{
11    buffer::Buffer,
12    layout::Rect,
13    style::{Color, Modifier, Style},
14    text::{Line, Span, Text},
15    widgets::{
16        Block, Borders, Cell, HighlightSpacing, Padding, Paragraph, Row, StatefulWidget, Table,
17        TableState, Widget,
18    },
19};
20
21use crate::error_display::user_message_from_polars;
22use crate::filter_modal::FilterStatement;
23use crate::local_copy::RemoteObject;
24use crate::numfmt::{self, CellFormatter, NumberFormatSettings};
25use crate::pivot_melt_modal::{MeltSpec, PivotAggregation, PivotSpec, ReshapeSource};
26use crate::python_script::{SidebarFilter, Step, py_str};
27use crate::query::{ParsedQuery, parse_query_over};
28use crate::statistics::collect_lazy;
29use crate::unfinished::{Claim, Writer};
30use crate::widgets::column_paging::{ColumnMove, CursorMove, OnScreen, Room};
31use crate::widgets::column_widths::{ColumnWidths, PageMeasure, WidthChoice};
32use crate::{CompressionFormat, OpenOptions, ParseStringsTarget};
33use polars::io::csv::read::NullValues;
34use polars::prelude::StrptimeOptions;
35use std::io::{BufReader, Read};
36
37use calamine::{Data, Reader, open_workbook_auto};
38use chrono::{NaiveDate, NaiveDateTime, NaiveTime};
39use orc_rust::ArrowReaderBuilder;
40use tempfile::NamedTempFile;
41
42use arrow::array::types::{
43    Date32Type, Date64Type, Float32Type, Float64Type, Int8Type, Int16Type, Int32Type, Int64Type,
44    TimestampMillisecondType, UInt8Type, UInt16Type, UInt32Type, UInt64Type,
45};
46use arrow::array::{Array, AsArray};
47use arrow::record_batch::RecordBatch;
48
49/// `agg` over `values`, one cell of a pivot.
50fn pivot_agg_expr(agg: PivotAggregation, values: Expr) -> Expr {
51    match agg {
52        PivotAggregation::Last => values.last(),
53        PivotAggregation::First => values.first(),
54        PivotAggregation::Min => values.min(),
55        PivotAggregation::Max => values.max(),
56        PivotAggregation::Avg => values.mean(),
57        PivotAggregation::Med => values.median(),
58        PivotAggregation::Std => values.std(1),
59        PivotAggregation::Count => values.len(),
60    }
61}
62
63/// The most columns a pivot may make. Past it the table, the schema and every view
64/// of them slow to a crawl; a pivot on a column with this many values is almost
65/// always a mistake (an id or a timestamp picked for Columns).
66pub const PIVOT_COLUMN_LIMIT: usize = 10_000;
67
68/// A pivot of the view as it was when planned, to be read off the UI thread.
69pub struct PivotJob {
70    view: LazyFrame,
71    spec: PivotSpec,
72    streaming: bool,
73}
74
75impl PivotJob {
76    /// A pivot of `view`: the builder's preview runs one over a few rows in memory.
77    pub(crate) fn new(view: LazyFrame, spec: PivotSpec, streaming: bool) -> Self {
78        Self {
79            view,
80            spec,
81            streaming,
82        }
83    }
84
85    /// The pivoted frame, in one pass over the view.
86    ///
87    /// The lazy pivot has to be told its new columns before it runs, which would mean a
88    /// distinct pass over the view and then the pivot over it again. Instead each cell is
89    /// aggregated by a group-by on the index and pivot columns in one pass; the new
90    /// columns are read off that result and the pivot runs over it in memory. The new
91    /// columns come out alphabetical with a trailing `null` column, as the eager pivot
92    /// ordered them, and index rows keep first-seen order.
93    pub fn run(self) -> Result<DataFrame> {
94        let on = self.spec.pivot_column.as_str();
95        let value = self.spec.value_column.as_str();
96        let index: Vec<PlSmallStr> = if self.spec.index.is_empty() {
97            self.view
98                .clone()
99                .collect_schema()?
100                .iter_names()
101                .filter(|name| name.as_str() != on && name.as_str() != value)
102                .cloned()
103                .collect()
104        } else {
105            self.spec.index.iter().map(PlSmallStr::from).collect()
106        };
107        // `Expr::Column`, not `col`: a header may contain `*` or `^`, and names are
108        // literal.
109        let keys: Vec<Expr> = index
110            .iter()
111            .cloned()
112            .chain([PlSmallStr::from(on)])
113            .map(Expr::Column)
114            .collect();
115        let cells = collect_lazy(
116            self.view.group_by_stable(keys).agg([pivot_agg_expr(
117                self.spec.aggregation,
118                Expr::Column(PlSmallStr::from(value)),
119            )
120            .alias(value)]),
121            self.streaming,
122        )?;
123        let on_columns = cells
124            .clone()
125            .lazy()
126            .select([Expr::Column(PlSmallStr::from(on))])
127            .unique(None, UniqueKeepStrategy::Any)
128            .sort([on], SortMultipleOptions::default().with_nulls_last(true))
129            .collect()?;
130        // Refused before the pivot builds them: the cells are already in memory, the
131        // columns would be the expensive part.
132        if on_columns.height() > PIVOT_COLUMN_LIMIT {
133            return Err(color_eyre::eyre::eyre!(
134                "Pivot would make {} columns from {on}; the limit is {}. Filter first, or pivot a column with fewer values",
135                numfmt::group_chrome(on_columns.height()),
136                numfmt::group_chrome(PIVOT_COLUMN_LIMIT),
137            ));
138        }
139        let (cells, on_columns) = Self::pivot_dates_as_text(cells, on_columns, on)?;
140        // One row per index and pivot value now, so `first` is that cell. A count sums
141        // instead, so a pair with no rows counts 0 rather than null, as it always did.
142        let cell = match self.spec.aggregation {
143            PivotAggregation::Count => element().sum(),
144            _ => element().first(),
145        };
146        let pivoted = cells
147            .lazy()
148            .pivot(
149                by_name([on], true, false),
150                Arc::new(on_columns),
151                by_name(index, true, false),
152                by_name([value], true, false),
153                cell,
154                true,
155                PlSmallStr::from_static("_"),
156                PivotColumnNaming::Auto,
157            )
158            .collect()?;
159        Ok(pivoted)
160    }
161
162    /// Polars names the new columns by casting the pivot values to text, which
163    /// panics on a date past the calendar. When one is there, the values become
164    /// their text first, such a date its stored number, after they are ordered as
165    /// dates. Both frames are already in memory.
166    fn pivot_dates_as_text(
167        cells: DataFrame,
168        on_columns: DataFrame,
169        on: &str,
170    ) -> Result<(DataFrame, DataFrame)> {
171        let values = on_columns.column(on)?.as_materialized_series();
172        if crate::exact::calendar_without_out_of_range(values)?.is_none() {
173            return Ok((cells, on_columns));
174        }
175        let text = |mut df: DataFrame| -> Result<DataFrame> {
176            let values = df.column(on)?.as_materialized_series();
177            let values = crate::past_calendar::cast_text(
178                values,
179                polars::chunked_array::cast::CastOptions::NonStrict,
180            )?;
181            df.with_column(values.into_column())?;
182            Ok(df)
183        };
184        Ok((text(cells)?, text(on_columns)?))
185    }
186}
187
188pub struct DataTableState {
189    lf: LazyFrame,
190    /// `lf` before its sort, when it has one. See [`DataTableState::analysis_lf`].
191    unsorted_lf: Option<LazyFrame>,
192    original_lf: LazyFrame,
193    original_schema: Arc<Schema>,
194    /// What the sidebar filters and sort are applied to: the active query's result (DSL,
195    /// SQL or fuzzy), the last pivot/melt, or `original_lf` when there is none. The
196    /// pipeline is original → query/reshape (`base_lf`) → filters → sort (`lf`) → column
197    /// order (at collect). Filters therefore never discard the query.
198    base_lf: LazyFrame,
199    df: Option<DataFrame>,        // Scrollable columns dataframe
200    locked_df: Option<DataFrame>, // Locked columns dataframe
201    pub table_state: TableState,
202    start_row: usize,
203    pub visible_rows: usize,
204    pub termcol_index: usize,
205    /// The column cursor's column, by name, so it follows hide, reorder and freeze.
206    /// `None` is the first column. See [`Self::current_column`].
207    cursor_column: Option<String>,
208    /// Where the cursor stood in `column_order` when placed: a column hidden from
209    /// under it hands the cursor to the one now in its place.
210    cursor_at: usize,
211    /// The cursor may be off screen (the order, the frozen count or the room
212    /// changed): the next draw brings it back, scrolling as little as it takes.
213    reveal_cursor: bool,
214    pub visible_termcols: usize,
215    /// The scrolling side as last drawn, which a sideways page is planned in. `None`
216    /// before the first draw.
217    scroll_room: Option<Room>,
218    /// Sideways moves waiting on the next draw to measure columns not drawn yet, in
219    /// the order asked. See [`Self::scroll_columns`].
220    column_moves: Vec<WaitingMove>,
221    /// The pages `]` went, from and to, so `[` straight after goes back exactly.
222    page_trail: Vec<(usize, usize)>,
223    /// Which columns the last draw showed, while some are off screen.
224    on_screen: Option<OnScreen>,
225    /// Where the last frame drew the rows and columns, for a click.
226    drawn: Option<DrawnTable>,
227    error: Option<PolarsError>,
228    pub suppress_error_display: bool, // When true, don't show errors in main view (e.g., when query input is active)
229    schema: Arc<Schema>,
230    num_rows: usize,
231    /// When true, collect() skips the len() query.
232    num_rows_valid: bool,
233    /// The dataset's own row count, remembered from the last moment the frame was
234    /// pristine. Lets the control bar say "417 of 1,000" under a filter or query
235    /// without a second count; `None` until a pristine count has resolved.
236    pristine_rows: Option<usize>,
237    /// Bumped whenever `lf` changes (via `invalidate_num_rows`). A background `len()`
238    /// count carries the generation it was spawned under; a result whose generation no
239    /// longer matches is stale (the data changed) and is dropped. Decoupled from
240    /// `task_generation` so a mere scroll doesn't invalidate / restart an in-flight count.
241    ///
242    /// Seeded from a process-wide counter rather than zero, so the value is unique
243    /// across datasets as well as across mutations of one. Starting every state at
244    /// zero meant a count still running for the dataset you just closed matched the
245    /// one you just opened, and set its row count to the wrong number.
246    len_generation: u64,
247    /// Taken afresh whenever `original_lf` is replaced. A checkpoint records it, so one
248    /// taken over other data is never put back over this data.
249    root_generation: u64,
250    /// The local Parquet hive directory the data was loaded from, whose per-file footer
251    /// counts sum to the exact row count while the frame is the scan as loaded
252    /// (`is_pristine`) — far cheaper than a `len()` data scan over a huge/partitioned set.
253    parquet_count_dir: Option<PathBuf>,
254    /// What finding and reading this dataset cost.
255    ///
256    /// On the dataset rather than on the app, for the reason `dataset_generation` is
257    /// bumped per dataset that reaches the screen rather than per open started: an open
258    /// that fails leaves the last dataset up, and its figures have to stay with it. A
259    /// meter the app held would by then be the failed load's.
260    measurements: Arc<crate::measurements::Meter>,
261    filters: Vec<FilterStatement>,
262    sort_columns: Vec<String>,
263    /// Per entry of `sort_columns`, whether that column runs descending. Always the
264    /// same length as `sort_columns`.
265    sort_descending: Vec<bool>,
266    sort_ascending: bool,
267    /// Last executed DSL query. At most one of the three `active_*` queries is set: running
268    /// one clears the other two.
269    active_query: String,
270    /// Last executed SQL (Sql tab).
271    active_sql_query: String,
272    /// The leading columns the SQL in effect orders by, as named in its result, and
273    /// whether each runs descending: the header's sort marks while the sidebar sorts
274    /// nothing. Empty for an ORDER BY of an expression.
275    query_order: Vec<(String, bool)>,
276    /// Last executed fuzzy search (Fuzzy tab).
277    active_fuzzy_query: String,
278    column_order: Vec<String>,   // Order of columns for display
279    locked_columns_count: usize, // Number of locked columns (from left)
280    /// What the last layout made of the frozen columns: the count asked for, and how
281    /// many of them fit frozen beside a usable scrolling column. The rest scroll until
282    /// a wider window has room again; a different count asked for starts over.
283    frozen_fit: (usize, usize),
284    /// The width each column is drawn at, by column identity, so paging, reordering,
285    /// hiding and opening a sidebar move nothing. Learned by the renderer from rows it
286    /// formats anyway; not part of a rollback, since it describes columns, not a view.
287    widths: ColumnWidths,
288    /// The grouped view a drill-down left, restored exactly by `drill_up`.
289    grouped: Option<GroupedView>,
290    /// The rows behind a grouped query result, so Enter can drill from an aggregate.
291    group_source: Option<GroupSource>,
292    /// The last pivot/melt result, while one is in effect. SQL runs against it rather
293    /// than the data as loaded (see `query_root`).
294    reshaped_lf: Option<LazyFrame>,
295    drilled_down_group_index: Option<usize>, // Index of the group we're viewing
296    drilled_down_group_key: Option<Vec<String>>, // Key values of the drilled down group
297    drilled_down_group_key_columns: Option<Vec<String>>, // Key column names of the drilled down group
298    pages_lookahead: usize,
299    pages_lookback: usize,
300    max_buffered_rows: usize, // 0 = no limit
301    max_buffered_mb: usize,   // 0 = no limit
302    /// True for a scan of an object store, where a buffer fill is a ranged read of
303    /// whole row groups. See `is_remote_source`.
304    remote_source: bool,
305    /// Where each row group of a remote Parquet object starts, with the total as the
306    /// last entry, from its footer. See `record_row_groups`.
307    row_group_offsets: Option<Vec<usize>>,
308    /// The files of a remote dataset, when it is many. See `RemoteFiles`.
309    remote_files: Option<RemoteFiles>,
310    /// Each remote object the dataset reads, by URL, with its size and tag from the
311    /// listing or the footer read that opened it. What a Data Quality local copy
312    /// would fetch.
313    remote_objects: Option<Arc<std::collections::HashMap<String, RemoteObject>>>,
314    /// What the footers said about a many-file dataset's columns: where the schema came
315    /// from, and which columns are not in every file. `None` for a single file.
316    dataset_schema: Option<crate::schema_union::DatasetSchema>,
317    /// Whether the frame still carries the scan's hidden drift column. True from the
318    /// open of a dataset whose files differ; false once a query or reshape has built a
319    /// new frame, which has no file behind each row any more.
320    drift_column_present: bool,
321    /// What each drift group is missing, shared with the renderer so a frame costs no
322    /// allocation. Indexed by the drift column's values.
323    drift_groups: Arc<Vec<crate::schema_union::DriftGroup>>,
324    /// The two above as the dataset was opened, so a reset returns to them.
325    drift_at_open: bool,
326    groups_at_open: Arc<Vec<crate::schema_union::DriftGroup>>,
327    /// The data as loaded carries each row's place in the source in the hidden row
328    /// index (lines), which `#` shows while the frame is the scan's.
329    source_rows_at_open: bool,
330    /// The sorted or filtered view numbers its rows itself, `#` being on and the data
331    /// as loaded carrying no place of its own: a row index over the base, under the
332    /// filters and sort. Taken only while `#` is on, because a row index between a
333    /// scan and a filter keeps the filter from being pushed into the scan.
334    view_numbered: bool,
335    /// Lines still being indexed behind the first rows: the frames grow as they are.
336    indexing: Option<Arc<crate::lines::Lines>>,
337    /// The lines of several files, which `#` numbers by their line in their own file.
338    numbering: Option<Arc<crate::lines::Lines>>,
339    /// The dataset's row count from a sample of its footers, until it is counted.
340    row_estimate: Option<crate::schema_union::RowEstimate>,
341    /// The notes the lines gave when they opened, replaced once they are all indexed.
342    indexing_notes: Vec<crate::notes::Note>,
343    /// Whether the open guessed the lines were text, which their notes say.
344    indexing_guessed: bool,
345    /// Where each file's rows begin in the dataset, and the drift group of each file.
346    /// Together they turn a row's place in the dataset into what its file was missing.
347    drift_file_starts: Vec<usize>,
348    drift_file_group: Vec<u32>,
349    /// Rows in the dataset as the footers counted them, so the last file's length is
350    /// known without asking what the view currently holds.
351    drift_dataset_rows: usize,
352    /// The dataset as its footers found it, kept beside the view because reading a
353    /// column as text needs the types the files actually hold — which is the very
354    /// thing the view no longer says.
355    dataset_at_open: Option<crate::schema_union::DatasetSchema>,
356    /// Columns being read as text from every file rather than as the type most rows
357    /// have. Empty for a dataset as opened.
358    read_as_text: Vec<PlSmallStr>,
359    /// Each file's path or URL, in scan order, so a row can be traced to the file it
360    /// came from and an export can name it.
361    drift_files: Vec<String>,
362    /// Set while the dataset is on screen from a footer or two and the rest are still
363    /// to be read. Cleared when their answer joins. See [`FootersJoin`].
364    footers_pending: Option<FootersJoin>,
365    /// What datui noticed about the dataset, from the footers it had to read anyway.
366    notes: Vec<crate::notes::Note>,
367    /// Whether the Info panel has been opened since the notes were gathered. Belongs to
368    /// the dataset, so opening another one offers its notes afresh.
369    notes_seen: bool,
370    /// The notes as the dataset was opened, so a reset and a drill up restore them.
371    notes_at_open: Vec<crate::notes::Note>,
372    /// Notes about the view rather than the dataset: what the filter and sort on
373    /// screen are leaving out. Recomputed whenever either changes, so clearing them
374    /// takes the note away with them.
375    view_notes: Vec<crate::notes::Note>,
376    /// Notes about the read itself rather than about what it found: which files this
377    /// open passed over, and whether it is reading a lake table's plain files.
378    ///
379    /// Their own list because they are settled before a footer is read, and
380    /// [`Self::notes`] is written from the footers when those land — so a note put
381    /// there at open time would be overwritten by the dataset's own. They also outlive
382    /// a reshape, which the footer notes do not: a query changes what is on screen, not
383    /// which files were read to get it.
384    open_notes: Vec<crate::notes::Note>,
385    /// The lake format whose plain files this dataset is, if it is one. See
386    /// [`crate::OpenOptions::read_as_plain_files_of`].
387    not_the_table: Option<&'static str>,
388    /// What a read through a format spec found: the spec, why, and its notes.
389    format_read: Option<Arc<crate::formats::Read>>,
390    /// What a read through a delimited spec found: units and metadata.
391    delimited: Option<Arc<crate::delimited_spec::DelimitedRead>>,
392    /// The fixed records the data as loaded is, while it still is: a window of a
393    /// pristine view starts its columns at the window rather than decoding from row 0.
394    fixed_window: Option<Arc<dyn crate::pushdown::Windowed>>,
395    /// A source that runs the sidebar's filters and sort itself (a SQLite table), while
396    /// the data as loaded is the root: see [`Self::pushed_view`].
397    pushdown: Option<Arc<dyn crate::pushdown::Pushdown>>,
398    /// Stops what the source runs when this state goes.
399    source_hold: Option<crate::sqlite::Hold>,
400    /// How the open reads the data. See [`crate::OpenOptions::read_mode`].
401    read_mode: Option<crate::ReadMode>,
402    /// The format the open read. See [`OpenFacts::read_as`].
403    read_as: Option<crate::FileFormat>,
404    /// The data was downloaded from a remote source before it was read.
405    fetched: bool,
406    /// What the file said besides its rows. See [`OpenFacts::detail`].
407    detail: Option<Arc<crate::text_formats::Detail>>,
408    /// Each loaded column's unit, from the file. See [`OpenFacts::units`].
409    file_units: Arc<Vec<(String, String)>>,
410    /// Uncompressed bytes per row of each column, from the Parquet footer, for
411    /// `bytes_per_row` before anything has been collected.
412    column_bytes: Vec<(String, usize)>,
413    /// Bytes per row of the last buffer collected, which outranks the estimate from
414    /// the schema.
415    observed_bytes_per_row: Option<usize>,
416    buffered_start_row: usize,
417    buffered_end_row: usize,
418    /// Full buffered DataFrame (all columns in column_order) for the current buffer range.
419    /// When set, column scroll (scroll_left/scroll_right) only re-slices columns without re-collecting from LazyFrame.
420    buffered_df: Option<DataFrame>,
421    proximity_threshold: usize,
422    /// The first row of the last page drawn whole. See [`Self::start_to_draw`].
423    drawn_start: usize,
424    row_numbers: bool,
425    row_start_index: usize,
426    /// Last applied pivot spec, if current lf is result of a pivot. Used for views.
427    last_pivot_spec: Option<PivotSpec>,
428    /// Last applied melt spec, if current lf is result of a melt. Used for views.
429    last_melt_spec: Option<MeltSpec>,
430    /// The query, filters and sort the pivot or melt in effect ran over, for a view to
431    /// replay before it. `None` while none is in effect, or when it ran over the data as
432    /// loaded.
433    reshape_source: Option<ReshapeSource>,
434    /// How `base_lf` was built from the data as loaded, step by step, for Copy as
435    /// Python. Set with every new base; empty for the data as loaded.
436    base_steps: Vec<Step>,
437    /// What the open did to the rows its reader gave, as Python method calls: names
438    /// trimmed, text columns typed.
439    read_python: Vec<String>,
440    /// What the read of several files has to say of them: files passed over, columns
441    /// not every file has. Carried to the dataset's notes.
442    read_notes: Vec<crate::notes::Note>,
443    /// Each column's unit, from the first of several files read through a spec that
444    /// has the column; `None` when the first file's header said them all.
445    read_units: Option<Vec<(String, String)>>,
446    /// The columns the read gave a type, and the frame before it did.
447    typing: Typing,
448    /// The notes on the values the types made null, once counted.
449    unfit_notes: Option<Vec<crate::notes::Note>>,
450    /// The view's own column types and columns made from others, in the order asked:
451    /// a step of `lf`, before the filters, as a spec's `[columns]` would say them.
452    column_changes: Vec<crate::column_types::ColumnChange>,
453    /// Bumped with every change to `column_changes`, so a count of what they made null
454    /// answers for the changes it was asked about.
455    changes_version: u64,
456    /// The notes on the values the view's types made null: for the version counted.
457    changes_unfit: Option<(u64, Vec<crate::notes::Note>)>,
458    /// Steps of a saved view whose columns this data does not have.
459    changes_dropped: Vec<crate::notes::Note>,
460    /// How `reshaped_lf` was built, while there is one: what SQL runs over.
461    reshape_steps: Option<Vec<Step>>,
462    /// Which loaded column each column of the base is (see [`Lineage`]).
463    lineage: Lineage,
464    /// The same for the pivot or melt in effect, which SQL runs against.
465    reshape_lineage: Lineage,
466    /// When set, dataset was loaded with hive partitioning; partition column names for Info panel and predicate pushdown.
467    partition_columns: Option<Vec<String>>,
468    /// When set, decompressed CSV was written to this temp file; kept alive so the file exists for lazy scan.
469    /// Shared with any view that scans it, and removed with the last.
470    decompress_temp_file: Option<Arc<Decompressed>>,
471    /// The downloaded remote file this dataset was opened from, held while it is scanned.
472    download: Option<crate::download::TempDownload>,
473    /// The files a GPS log was read into, which the frame scans; held as `download` is.
474    converted: Vec<crate::download::TempDownload>,
475    /// The file's other tables, as `--table` names them; see [`OpenFacts::other_tables`].
476    other_tables: Vec<String>,
477    /// When true, use Polars streaming engine for LazyFrame collect when the streaming feature is enabled.
478    polars_streaming: bool,
479    /// When true, `collect()` / `apply_transformations()` skip the blocking collect.
480    /// The caller is responsible for triggering an async collect afterwards.
481    defer_collect: bool,
482    /// Set by the render code when `visible_rows` changes. The App event loop checks this
483    /// after each render and triggers an async collect if needed.
484    pub needs_recollect: bool,
485    /// The watcher of the file this dataset follows (`--follow`), while it does.
486    follow: Option<crate::follow::Follow>,
487    /// For a followed view that filters or sorts the file's rows: points where the
488    /// view's rows before a file row are known (view rows, file row), ascending, for
489    /// the count generation they hold for. The next count reads on from the last; a
490    /// filtered window from the one before it.
491    follow_known: Option<(u64, Vec<(usize, usize)>)>,
492    /// The sample this view's rows are, and the view it was drawn from, while the
493    /// view has one: the step between the source and the query.
494    sampled: Option<Box<Sampled>>,
495}
496
497/// A view's sample: the step between the source and the query. The view's frames
498/// scan [`Self::frame`], the chunks kept so far, which grows as the draw goes on.
499pub struct Sampled {
500    /// The view the sample was drawn from, as it stood: what clearing the sample
501    /// returns to.
502    source: Box<DataTableState>,
503    sample: crate::sampling::Sample,
504    rows: Arc<crate::table_sample::SampleRows>,
505    /// The frame the view's plans scan: the chunks taken so far, on their buffers.
506    frame: Arc<DataFrame>,
507    /// Drawn from the view's query or filters, which the sample then stands for,
508    /// rather than from the source under them.
509    through: bool,
510    /// What the draw read, once it ended; `None` while it runs.
511    drawn: Option<crate::table_sample::Drawn>,
512    /// How a random sample of a stream is drawn, which a view keeps.
513    path: Option<crate::table_sample::DrawPath>,
514}
515
516impl Sampled {
517    pub fn sample(&self) -> &crate::sampling::Sample {
518        &self.sample
519    }
520
521    /// The view the sample was drawn from.
522    pub fn source(&self) -> &DataTableState {
523        &self.source
524    }
525
526    /// Whether the sample was drawn from the view's query or filters.
527    pub fn through(&self) -> bool {
528        self.through
529    }
530
531    /// Whether these are the rows `rows` holds: the sample a draw fills.
532    pub(crate) fn holds(&self, rows: &Arc<crate::table_sample::SampleRows>) -> bool {
533        Arc::ptr_eq(&self.rows, rows)
534    }
535
536    /// Whether rows are still arriving.
537    pub fn drawing(&self) -> bool {
538        self.drawn.is_none()
539    }
540
541    pub fn drawn(&self) -> Option<&crate::table_sample::Drawn> {
542        self.drawn.as_ref()
543    }
544
545    /// How a random sample of a stream is drawn: what draws the same rows again.
546    pub fn path(&self) -> Option<crate::table_sample::DrawPath> {
547        self.path
548    }
549
550    /// The frame the view's plans scan.
551    #[cfg(test)]
552    pub(crate) fn frame(&self) -> &DataFrame {
553        &self.frame
554    }
555
556    /// Rows the view has taken of the sample.
557    pub fn rows(&self) -> usize {
558        self.frame.height()
559    }
560
561    /// Bytes the sample's rows take.
562    pub fn bytes(&self) -> usize {
563        self.rows.bytes()
564    }
565
566    /// Why memory stopped the draw, if it did.
567    pub fn stopped(&self) -> Option<String> {
568        self.rows.stopped()
569    }
570
571    /// The footer's segment: `sample 100,000 of 36.8M`, `sample 1,234+` while it is
572    /// drawn, `sample about 100,000 of 36.8M` when kept row by row by chance.
573    pub fn label(&self) -> String {
574        let rows = crate::numfmt::group_chrome(self.rows());
575        let Some(drawn) = &self.drawn else {
576            return format!("sample {rows}+");
577        };
578        let about = if drawn.about { "about " } else { "" };
579        let cut = if drawn.cut { ", stopped" } else { "" };
580        match drawn.total {
581            Some(total) if total > self.rows() => {
582                format!(
583                    "sample {about}{rows} of {}{cut}",
584                    crate::discover::format_rows(total)
585                )
586            }
587            _ => format!("sample {rows}{cut}"),
588        }
589    }
590}
591
592/// What string-column inference may turn a column into, besides Time.
593#[derive(Clone, Copy)]
594struct StringTypes {
595    /// Date and Datetime.
596    dates: bool,
597    /// Duration, Int64 and Float64, and trimming the columns that stay text.
598    numbers: bool,
599}
600
601/// Inferred type for an Excel column (preserves numbers, bools, dates; avoids stringifying).
602#[derive(Clone, Copy)]
603enum ExcelColType {
604    Int64,
605    Float64,
606    Boolean,
607    Utf8,
608    Date,
609    Datetime,
610}
611
612/// Which loaded column each shown column is, as (shown name, loaded name), so a
613/// delimited spec's unit stays on a column that holds the loaded values, renamed or
614/// not, and never lands on a computed column that reuses a name. `None` while every
615/// column is the loaded column of its name.
616type Lineage = Option<Arc<Vec<(String, String)>>>;
617
618/// `pairs`, each a shown name and the name of a column of a frame whose lineage is
619/// `root`, traced back to the loaded columns. A name `root` does not know is dropped.
620fn traced(root: &Lineage, pairs: Vec<(String, String)>) -> Lineage {
621    let pairs = match root {
622        None => pairs,
623        Some(root) => pairs
624            .into_iter()
625            .filter_map(|(shown, from)| {
626                root.iter()
627                    .find(|(name, _)| *name == from)
628                    .map(|(_, loaded)| (shown, loaded.clone()))
629            })
630            .collect(),
631    };
632    Some(Arc::new(pairs))
633}
634
635/// Each of `exprs` that is a column unchanged, renamed or not: its output name and
636/// the column's.
637fn passed_through(exprs: &[Expr]) -> Vec<(String, String)> {
638    exprs
639        .iter()
640        .filter_map(|e| {
641            let Expr::Column(from) = e.clone().meta().undo_aliases() else {
642                return None;
643            };
644            let shown = e.clone().meta().output_name().ok()?;
645            Some((shown.to_string(), from.to_string()))
646        })
647        .collect()
648}
649
650/// The grouped view and the pipeline state that produced it, saved by a drill-down so
651/// filters and sort inside the group work on the group and `drill_up` restores the
652/// grouped view as it was.
653#[derive(Clone)]
654struct GroupedView {
655    lf: LazyFrame,
656    base_lf: LazyFrame,
657    filters: Vec<FilterStatement>,
658    sort_columns: Vec<String>,
659    sort_descending: Vec<bool>,
660    sort_ascending: bool,
661    /// Whether `lf` carries the hidden drift column, and what its groups mean. Saved
662    /// with the frame so drilling back up restores the cells it explains, along with
663    /// the notes that explain them.
664    drift: bool,
665    drift_groups: Arc<Vec<crate::schema_union::DriftGroup>>,
666    /// Whether `lf` numbers its rows itself (`#`).
667    view_numbered: bool,
668    notes: Vec<crate::notes::Note>,
669    group_source: Option<GroupSource>,
670    /// Where the user was, so coming back puts the cursor on the group drilled into
671    /// with the key columns still frozen.
672    column_order: Vec<String>,
673    locked_columns_count: usize,
674    start_row: usize,
675    termcol_index: usize,
676    cursor_column: Option<String>,
677    selected: Option<usize>,
678    /// Drilled into from Value Counts rather than from a grouped row.
679    by_value: bool,
680    /// How `base_lf` was built, for Copy as Python.
681    base_steps: Vec<Step>,
682    lineage: Lineage,
683}
684
685/// The rows a grouped result was computed from and how its keys were computed, so a
686/// drill-down can find a group's rows even when the result holds only aggregates.
687/// Recorded by the query that grouped rather than inferred from the result's columns,
688/// which may be renamed or computed.
689#[derive(Clone)]
690struct GroupSource {
691    /// The rows before grouping, after any filter the query applied first.
692    rows: LazyFrame,
693    /// Each key's column in the result, with the expression that computes it from `rows`.
694    keys: Vec<(PlSmallStr, Expr)>,
695    /// Columns `rows` carries only to compute keys, left out of a drill.
696    scratch: Vec<PlSmallStr>,
697    /// Whether the result's list columns are each group's rows, as a `by` query's are.
698    /// A SQL result's lists are values it computed, such as `ARRAY_AGG`.
699    rows_in_lists: bool,
700    /// The same as Copy as Python steps: how `rows` was built, and each key as
701    /// Python code, aliases undone. None where the script cannot say.
702    python_rows: Option<Vec<Step>>,
703    python_keys: Vec<Option<String>>,
704    /// Which loaded column each column of `rows` is.
705    lineage: Lineage,
706}
707
708/// One field the row inspector lists: a column of the frame on screen.
709#[derive(Debug, Clone, PartialEq)]
710pub struct InspectField {
711    pub name: String,
712    pub dtype: DataType,
713    /// Hidden from the table, so not among the rows read for it.
714    pub hidden: bool,
715}
716
717impl InspectField {
718    /// Whether the table's rows hold this field's value: a hidden column is not
719    /// read for them, and a binary one is read as a stub.
720    pub fn buffered(&self) -> bool {
721        !self.hidden && !matches!(self.dtype, DataType::Binary)
722    }
723}
724
725/// The selected row as the buffer holds it; see [`DataTableState::inspect_row`].
726#[derive(Clone)]
727pub struct InspectRow {
728    /// The row's index in the view.
729    pub row: usize,
730    /// The frame it is a row of (`len_generation`), which a sort, a filter or a
731    /// query replaces.
732    pub frame: u64,
733    /// The row number the table shows.
734    pub display_row: usize,
735    /// The row, one column per table column, raw.
736    pub values: DataFrame,
737    /// Which drift group its file is in, when the files differ.
738    pub drift_group: Option<u32>,
739}
740
741/// What a null means where the files of a dataset differ.
742#[derive(Debug, Clone, Copy, PartialEq, Eq)]
743pub enum NullKind {
744    Null,
745    Absent,
746    Conflict,
747}
748
749/// The row a drill into a group reads, from [`DataTableState::drill_row`].
750pub enum DrillRow {
751    /// Taken from the rows on screen.
752    Buffered(DataFrame),
753    /// Not on hand: the one-row frame to collect, off the UI thread.
754    Read(Box<LazyFrame>),
755}
756
757/// A group's rows, drilled from its row of a grouped view.
758struct GroupRows {
759    lf: LazyFrame,
760    /// The key columns of the grouped view and the row's values in them, as text.
761    key_columns: Vec<String>,
762    key_values: Vec<String>,
763    /// Columns of `lf` that hold the keys as they stand, to lead the view.
764    lead: Vec<String>,
765    /// How `lf` is built, as Copy as Python steps.
766    steps: Vec<Step>,
767    /// Which loaded column each column of `lf` is.
768    lineage: Lineage,
769}
770
771/// The view as it stood before a query or view replaced it: a checkpoint. A query plans
772/// without reading anything and can still fail once it runs — a value that will not
773/// cast — and then the table goes back to this, rows and all, rather than keep a
774/// frame that fails on every scroll. Frames and buffers are shared, not copied.
775///
776/// Taken by [`DataTableState::rollback_point`] or [`DataTableState::try_transition`],
777/// put back by [`DataTableState::roll_back`].
778pub struct ViewRollback {
779    /// The data as loaded when this was taken; see [`DataTableState::roll_back`].
780    root_generation: u64,
781    /// A count of this frame that came back after it was replaced, to return with it.
782    counted: Option<CountedRows>,
783    drawn_start: usize,
784    lf: LazyFrame,
785    unsorted_lf: Option<LazyFrame>,
786    base_lf: LazyFrame,
787    df: Option<DataFrame>,
788    locked_df: Option<DataFrame>,
789    table_state: TableState,
790    start_row: usize,
791    termcol_index: usize,
792    cursor_column: Option<String>,
793    cursor_at: usize,
794    schema: Arc<Schema>,
795    num_rows: usize,
796    num_rows_valid: bool,
797    len_generation: u64,
798    filters: Vec<FilterStatement>,
799    sort_columns: Vec<String>,
800    sort_descending: Vec<bool>,
801    sort_ascending: bool,
802    active_query: String,
803    active_sql_query: String,
804    query_order: Vec<(String, bool)>,
805    active_fuzzy_query: String,
806    column_order: Vec<String>,
807    locked_columns_count: usize,
808    /// The frozen fit `df` was sliced for; a later layout may have changed it.
809    frozen_fit: (usize, usize),
810    grouped: Option<GroupedView>,
811    reshaped_lf: Option<LazyFrame>,
812    last_pivot_spec: Option<PivotSpec>,
813    last_melt_spec: Option<MeltSpec>,
814    reshape_source: Option<ReshapeSource>,
815    base_steps: Vec<Step>,
816    reshape_steps: Option<Vec<Step>>,
817    lineage: Lineage,
818    reshape_lineage: Lineage,
819    group_source: Option<GroupSource>,
820    drilled_down_group_index: Option<usize>,
821    drilled_down_group_key: Option<Vec<String>>,
822    drilled_down_group_key_columns: Option<Vec<String>>,
823    drift_column_present: bool,
824    view_numbered: bool,
825    drift_groups: Arc<Vec<crate::schema_union::DriftGroup>>,
826    notes: Vec<crate::notes::Note>,
827    notes_seen: bool,
828    view_notes: Vec<crate::notes::Note>,
829    column_changes: Vec<crate::column_types::ColumnChange>,
830    changes_version: u64,
831    changes_dropped: Vec<crate::notes::Note>,
832    observed_bytes_per_row: Option<usize>,
833    buffered_start_row: usize,
834    buffered_end_row: usize,
835    buffered_df: Option<DataFrame>,
836}
837
838impl ViewRollback {
839    /// A background count of frame `len_generation` came back while this checkpoint
840    /// was waiting. Kept when the frame is the one this restores, so the rows and the
841    /// count return together; returns whether it was.
842    pub fn count_landed(
843        &mut self,
844        len_generation: u64,
845        rows: usize,
846        file_row_groups: Option<&[Vec<usize>]>,
847    ) -> bool {
848        let ours = len_generation == self.len_generation;
849        if ours {
850            self.counted = Some(CountedRows {
851                rows,
852                file_row_groups: file_row_groups.map(<[_]>::to_vec),
853            });
854        }
855        ours
856    }
857}
858
859/// A row count read in the background: the total, and for a remote dataset of many
860/// files, the rows in each row group of each file.
861struct CountedRows {
862    rows: usize,
863    file_row_groups: Option<Vec<Vec<usize>>>,
864}
865
866/// The query bar a result came from, with its text. At most one is active at a time.
867enum ActiveQuery {
868    Dsl(String),
869    #[cfg(feature = "sql")]
870    Sql(String),
871    Fuzzy(String),
872}
873
874/// Parameters for a background buffer load. Produced by `prepare_async_collect()`.
875pub struct CollectRequest {
876    /// LazyFrame to collect (sliced to the buffer range, with column selection applied).
877    pub lf: LazyFrame,
878    /// Whether to use Polars streaming engine.
879    pub polars_streaming: bool,
880    /// Buffer start row in the full dataset.
881    pub buffer_start: usize,
882    /// Buffer end row in the full dataset.
883    pub buffer_end: usize,
884    /// Row count for the full (unsliced) dataset. Only meaningful when `count_known`.
885    pub num_rows: usize,
886    /// Whether `num_rows` is the true total. False for a first buffer rendered before
887    /// the background `len()` count has resolved; in that case `num_rows` is provisional.
888    pub count_known: bool,
889    /// How the worker fits the rows it reads to the buffer: [`FillPlan::fit`].
890    pub plan: FillPlan,
891}
892
893/// What the worker that reads a fill needs to make it the buffer: the rows on hand it
894/// runs on from or up to, the view, and the caps, as they were when it was planned.
895///
896/// A trim copies the rows it keeps when a slice would keep the fill allocated behind
897/// them (see [`trim_rows`]): up to the byte budget, too long for the UI thread (#483).
898/// The worker does it, and `apply_async_collect` installs what it hands back as it is.
899pub struct FillPlan {
900    buffer_start: usize,
901    buffer_end: usize,
902    num_rows: usize,
903    count_known: bool,
904    /// Lines were still being indexed when the read was planned: a short read ends
905    /// where the indexing had got to, not the file.
906    indexing: bool,
907    /// The rows on hand and their first row, when the fill is planned to be stitched
908    /// on to them. Shared, not copied.
909    held: Option<(DataFrame, usize)>,
910    view_start: usize,
911    view_len: usize,
912    max_rows: usize,
913    max_mb: usize,
914}
915
916impl FillPlan {
917    /// Make the buffer of `df`, the rows read for the planned range: stitched on to
918    /// the rows on hand when it runs on from them or up to them, then cut to the caps
919    /// around the view.
920    pub fn fit(mut self, df: DataFrame) -> CollectResult {
921        let returned = df.height();
922        let bytes_per_row = (returned > 0).then(|| (df.estimated_size() / returned).max(1));
923        // A shape mismatch (the columns changed underneath) keeps the fetched rows alone.
924        let (df, start, seam) = match self.held.take() {
925            Some((mut held, held_start)) if held_start + held.height() == self.buffer_start => {
926                let seam = held.height();
927                match held.vstack_mut(&df) {
928                    Ok(_) => (held, held_start, Some(seam)),
929                    Err(_) => (df, self.buffer_start, None),
930                }
931            }
932            Some((held, held_start))
933                if returned > 0 && self.buffer_start + returned == held_start =>
934            {
935                match df.vstack(&held) {
936                    Ok(joined) => (joined, self.buffer_start, Some(returned)),
937                    Err(_) => (df, self.buffer_start, None),
938                }
939            }
940            _ => (df, self.buffer_start, None),
941        };
942        let (df, start) = self.cut_to_caps(df, start, seam);
943        CollectResult {
944            df,
945            start,
946            returned,
947            bytes_per_row,
948            buffer_start: self.buffer_start,
949            buffer_end: self.buffer_end,
950            num_rows: self.num_rows,
951            count_known: self.count_known,
952            indexing: self.indexing,
953        }
954    }
955
956    /// Cut `df`, spanning `[start, start + df.height())`, to the row cap and the byte
957    /// budget. The rows kept are centered on the view rather than taken from the head:
958    /// a jump near the end of the dataset would otherwise drop exactly the rows the
959    /// view needs. Returns the rows kept and their first row.
960    ///
961    /// The budget bounds the rows held between collects, not the collect itself: the
962    /// fill, the operators upstream of it and an eager source frame all take memory of
963    /// their own.
964    fn cut_to_caps(&self, df: DataFrame, start: usize, seam: Option<usize>) -> (DataFrame, usize) {
965        let total = df.height();
966        if total == 0 {
967            return (df, start);
968        }
969        // The row cap as well: a row group stitched on to the rows on hand can run over it.
970        let mut max_rows = total;
971        if self.max_rows > 0 {
972            max_rows = max_rows.min(self.max_rows);
973        }
974        if self.max_mb > 0 {
975            let bytes_per_row = (df.estimated_size() / total).max(1);
976            max_rows = max_rows.min(self.max_mb * 1024 * 1024 / bytes_per_row);
977        }
978        let max_rows = max_rows.max(1);
979        if max_rows >= total {
980            return (df, start);
981        }
982        let view_off = self.view_start.saturating_sub(start).min(total);
983        let view_len = self.view_len.max(1).min(total);
984        let view_center = view_off + view_len / 2;
985        let mut keep_start = view_center.saturating_sub(max_rows / 2);
986        if keep_start + max_rows > total {
987            keep_start = total - max_rows;
988        }
989        let kept = max_rows.min(total - keep_start);
990        (trim_rows(df, keep_start, kept, seam), start + keep_start)
991    }
992}
993
994/// Builds a scan of some of a dataset's files, as the full scan reads them, with the
995/// columns named in the second argument read as text from every file rather than as
996/// the type most rows have.
997pub type FileScan = Arc<dyn Fn(&[String], &[PlSmallStr]) -> PolarsResult<LazyFrame> + Send + Sync>;
998/// Counts the rows in each row group of every file of a dataset. Blocks.
999pub type FileCounter = Arc<
1000    dyn Fn(&Arc<crate::schema_union::FooterProgress>) -> Result<Vec<Vec<usize>>, String>
1001        + Send
1002        + Sync,
1003>;
1004/// Reads every footer of a dataset that opened from a couple of them, and returns what
1005/// they say. `None` when they could not be read, in which case the dataset stays as it
1006/// opened. Blocks, and counts itself off against the progress it is given.
1007pub type FootersJoin =
1008    Arc<dyn Fn(&Arc<crate::schema_union::FooterProgress>) -> Option<FootersFound> + Send + Sync>;
1009/// What reading every footer turned up, and everything built from it that the dataset
1010/// has to be given together — the schema and the scans that read at that schema.
1011pub struct FootersFound {
1012    /// Every column every file has, and which files disagree about what.
1013    pub dataset: crate::schema_union::DatasetSchema,
1014    /// The scan that reads the dataset whole.
1015    pub lf: LazyFrame,
1016    /// Each file's rows, in scan order.
1017    pub file_rows: Vec<usize>,
1018    /// Every file listed, in scan order — including any whose footer would not read.
1019    /// The dataset's per-file findings index this, so it is the whole list.
1020    pub files: Vec<String>,
1021    /// Each file's row groups, or empty if a footer would not parse.
1022    pub row_groups: Vec<Vec<usize>>,
1023    /// How to read part of a remote dataset rather than all of it. `None` for one that
1024    /// does not read by file.
1025    pub remote: Option<RemoteRead>,
1026    /// The row count the footers read say, when they were a sample.
1027    pub estimate: Option<crate::schema_union::RowEstimate>,
1028}
1029
1030/// How a remote dataset reads some of its files, as the pass behind an open found them.
1031///
1032/// The three travel together because they describe one list. The scan is built at a
1033/// schema — the one the dataset opened with has never heard of the columns this pass
1034/// found — and the counter answers one entry per file it was given, which has to be the
1035/// same list `urls` holds or the answer is dropped on a length check and the dataset
1036/// never learns its own size.
1037pub struct RemoteRead {
1038    /// The files that will open, which is not every file listed.
1039    pub urls: Vec<String>,
1040    pub scan: FileScan,
1041    pub count: FileCounter,
1042}
1043
1044impl From<RemoteRead> for RemoteFiles {
1045    fn from(read: RemoteRead) -> Self {
1046        RemoteFiles {
1047            urls: Arc::new(read.urls),
1048            scan: read.scan,
1049            count: read.count,
1050            offsets: None,
1051        }
1052    }
1053}
1054
1055/// A compressed CSV's decompressed copy, then the open's claim on it: dropped in that
1056/// order, so the claim goes only once the file has. See [`crate::unfinished`].
1057struct Decompressed {
1058    file: NamedTempFile,
1059    _claim: Claim,
1060}
1061
1062impl Decompressed {
1063    fn path(&self) -> &Path {
1064        self.file.path()
1065    }
1066}
1067
1068/// A dataset of many files, and how to read only some of them: a remote one, or a
1069/// local Hive directory once every footer is known.
1070///
1071/// Polars reads a scan of many files in order: row 900,000 is reached by reading every
1072/// file before it, and a count is a read of all of them. Once each file's rows are
1073/// known, from its footer, a buffer is a scan of just the files holding its rows.
1074#[derive(Clone)]
1075pub struct RemoteFiles {
1076    /// Every file, in scan order.
1077    pub urls: Arc<Vec<String>>,
1078    pub scan: FileScan,
1079    pub count: FileCounter,
1080    /// Where each file's rows start, with the total last. Known once counted.
1081    pub offsets: Option<Vec<usize>>,
1082}
1083
1084/// What an open learned about a dataset besides its frame and schema: given to the
1085/// state once, by [`DataTableState::with_open`], so the count, the row groups, the
1086/// files and the notes all describe the same open. Each field's default means the open
1087/// did not find it.
1088#[derive(Default)]
1089pub struct OpenFacts {
1090    /// A scan of an object store in place: a buffer is one window of whole row groups.
1091    pub remote_source: bool,
1092    /// Each file's row groups in scan order, from the footers; one entry for a single
1093    /// object. Gives the count. With `remote_files`, one entry per file it lists.
1094    pub row_groups: Vec<Vec<usize>>,
1095    /// The files of a remote dataset of many, and how to read some of them.
1096    pub remote_files: Option<RemoteFiles>,
1097    /// Each remote object the dataset reads, as the listing or footer found it.
1098    pub remote_objects: Vec<RemoteObject>,
1099    /// What the footers said about a many-file dataset's columns.
1100    pub dataset: Option<DatasetAtOpen>,
1101    /// The pass that reads the rest of the footers, for a dataset opened from a few.
1102    pub footers_pending: Option<FootersJoin>,
1103    /// Each column's uncompressed bytes per row, from the footers.
1104    pub column_bytes: Vec<(String, usize)>,
1105    /// The local Parquet hive directory whose footers sum to the count.
1106    pub parquet_count_dir: Option<PathBuf>,
1107    /// What finding and reading the dataset cost.
1108    pub measurements: Arc<crate::measurements::Meter>,
1109    /// What the open itself has to say. See [`DataTableState::open_notes`].
1110    pub open_notes: Vec<crate::notes::Note>,
1111    /// The lake format whose plain files this dataset is. See
1112    /// [`DataTableState::not_the_table`].
1113    pub not_the_table: Option<&'static str>,
1114    /// What a read through a format spec found.
1115    pub format_read: Option<Arc<crate::formats::Read>>,
1116    /// What a read through a delimited spec found.
1117    pub delimited: Option<Arc<crate::delimited_spec::DelimitedRead>>,
1118    /// The downloaded file the frame scans, held for as long as the state lives.
1119    pub download: Option<crate::download::TempDownload>,
1120    /// The files a GPS log was read into, which the frame scans.
1121    pub converted: Vec<crate::download::TempDownload>,
1122    /// The file's other tables, each as `--table` names it with how many rows it holds
1123    /// where that is known, for the Info panel's Schema tab. Empty for a file of one.
1124    pub other_tables: Vec<String>,
1125    /// A source that runs the sidebar's filters and sort itself: a SQLite table.
1126    pub pushdown: Option<Arc<dyn crate::pushdown::Pushdown>>,
1127    /// What stops that source's statements when the dataset goes.
1128    pub hold: Option<crate::sqlite::Hold>,
1129    /// How the open reads the data. See [`crate::OpenOptions::read_mode`].
1130    pub read_mode: Option<crate::ReadMode>,
1131    /// The format the open read the data as, after sniffing and spec matching: what
1132    /// the scan chose, which a file's name may not say. Copy as Python and the export
1133    /// default follow it.
1134    pub read_as: Option<crate::FileFormat>,
1135    /// The data was downloaded from a remote source before it was read: not a local
1136    /// stream's conversion or standard input's spool, which are held as downloads are.
1137    pub fetched: bool,
1138    /// What the file said besides its rows, for the Info panel.
1139    pub detail: Option<Arc<crate::text_formats::Detail>>,
1140    /// Rows read straight from a reader that decodes them from the file (a NumPy
1141    /// array, an audio file's frames), and how many it holds: a page deep in the table,
1142    /// and the count, need no row index.
1143    pub records: Option<(Arc<dyn crate::pushdown::Windowed>, usize)>,
1144    /// Each column's unit, where the file says one.
1145    pub units: Vec<(String, String)>,
1146    /// Lines still being indexed behind the first rows: the frames grow as they are.
1147    pub indexing: Option<Arc<crate::lines::Lines>>,
1148    /// The lines of several files, which `#` numbers by their line in their own file.
1149    pub numbering: Option<Arc<crate::lines::Lines>>,
1150    /// The columns the read gave a type, for the count of what did not fit.
1151    pub typing: Typing,
1152}
1153
1154/// The footers' account of a dataset of many files.
1155pub struct DatasetAtOpen {
1156    pub schema: crate::schema_union::DatasetSchema,
1157    /// Each file's row count in scan order; empty unless every one is known, which is
1158    /// when the scan numbers its rows.
1159    pub file_rows: Vec<usize>,
1160    /// Every file's path or URL, in scan order.
1161    pub files: Vec<String>,
1162}
1163
1164/// The rows an export writes. See [`DataTableState::export_frame`].
1165pub struct ExportFrame {
1166    lf: LazyFrame,
1167    files: Option<SourceFiles>,
1168}
1169
1170impl ExportFrame {
1171    /// Rows that are not the view's, such as a column's value counts.
1172    pub fn of(lf: LazyFrame) -> Self {
1173        Self { lf, files: None }
1174    }
1175}
1176
1177/// The dataset's files in scan order, and the row each starts at.
1178struct SourceFiles {
1179    names: Arc<Vec<String>>,
1180    starts: Arc<Vec<usize>>,
1181}
1182
1183impl ExportFrame {
1184    /// The name of the column an export adds when asked to say where each row is from.
1185    pub const SOURCE_FILE_COLUMN: &'static str = "source_file";
1186
1187    /// The plan. Naming the files reads the schema, which may resolve the scan, so
1188    /// this belongs off the UI thread.
1189    ///
1190    /// The names are mapped from the row index batch by batch, so a streamed export
1191    /// still never holds every row.
1192    pub fn into_lazy(self) -> PolarsResult<LazyFrame> {
1193        let Some(SourceFiles { names, starts }) = self.files else {
1194            return Ok(self.lf);
1195        };
1196        let mut lf = self.lf;
1197        let schema = lf.collect_schema()?;
1198        let name = Self::free_name(schema.iter_names().map(|n| n.as_str()));
1199        let index = crate::schema_union::DRIFT_COLUMN;
1200        let file_of = move |rows: Column| -> PolarsResult<Column> {
1201            let rows = rows.strict_cast(&DataType::UInt64)?;
1202            let named: StringChunked = rows
1203                .u64()?
1204                .iter()
1205                .map(|row| {
1206                    let row = row? as usize;
1207                    let file = starts
1208                        .partition_point(|&start| start <= row)
1209                        .saturating_sub(1);
1210                    names.get(file).map(String::as_str)
1211                })
1212                .collect();
1213            Ok(named.with_name(rows.name().clone()).into_column())
1214        };
1215        // Added last, after the dataset's own columns, as the index comes off.
1216        Ok(lf
1217            .with_column(
1218                col(index)
1219                    .map(file_of, |_, field| {
1220                        Ok(Field::new(field.name().clone(), DataType::String))
1221                    })
1222                    .alias(name),
1223            )
1224            .drop(by_name([index], true, false)))
1225    }
1226
1227    /// A name for the source-file column that no column already has.
1228    ///
1229    /// `source_file` is a name a dataset may well use itself — a directory of per-file
1230    /// extracts is exactly this feature's audience — and adding a column by a name
1231    /// already present replaces it, silently, in the file the user takes away.
1232    fn free_name<'a>(taken: impl Iterator<Item = &'a str>) -> String {
1233        let taken: HashSet<&str> = taken.collect();
1234        std::iter::once(Self::SOURCE_FILE_COLUMN.to_string())
1235            .chain((1..).map(|n| format!("{}_{n}", Self::SOURCE_FILE_COLUMN)))
1236            .find(|candidate| !taken.contains(candidate.as_str()))
1237            .expect("some suffix is free")
1238    }
1239}
1240
1241/// Result of a background buffer load, made by [`FillPlan::fit`] on the worker and
1242/// installed as it is by `apply_async_collect()`.
1243pub struct CollectResult {
1244    /// The buffer: the rows read, stitched and cut to the caps.
1245    df: DataFrame,
1246    /// The first row of `df`.
1247    start: usize,
1248    /// Rows the read returned, before the stitch and the cut.
1249    returned: usize,
1250    /// Bytes per row of the rows read, to plan the next fill by.
1251    bytes_per_row: Option<usize>,
1252    /// The range the read was planned for.
1253    buffer_start: usize,
1254    buffer_end: usize,
1255    num_rows: usize,
1256    /// See `CollectRequest::count_known`.
1257    count_known: bool,
1258    /// See `FillPlan::indexing`.
1259    indexing: bool,
1260}
1261
1262impl CollectResult {
1263    /// The rows read, as the buffer will hold them.
1264    pub(crate) fn rows(&self) -> &DataFrame {
1265        &self.df
1266    }
1267}
1268
1269/// Rows the display buffer may hold when `performance.max_buffered_rows` is not set. Also
1270/// the window a remote scan buffers when the cap is switched off.
1271pub const DEFAULT_MAX_BUFFERED_ROWS: usize = 100_000;
1272
1273/// Seeds `DataTableState::len_generation`. Unique per state, so a row count spawned
1274/// for one dataset can never be mistaken for a valid result for another.
1275static NEXT_LEN_GENERATION: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1);
1276
1277fn next_len_generation() -> u64 {
1278    NEXT_LEN_GENERATION.fetch_add(1, std::sync::atomic::Ordering::Relaxed)
1279}
1280
1281/// The `[start, end)` row ranges of the files flagged in `conflicts`, merged where
1282/// they touch.
1283///
1284/// `starts[i]` is where file `i`'s rows begin in the dataset and `total` is how many
1285/// rows the dataset has, so the last file's end is known without a start after it.
1286///
1287/// Runs, not files: a vendor who wrote a column as text for a month wrote a
1288/// contiguous stretch of files, and the predicate built from this is one term per run
1289/// however many files the stretch holds. Merging changes no row's fate — three
1290/// touching ranges keep out exactly what one joined range does — which is why it is
1291/// pinned here, where the runs themselves can be counted, rather than by a test of
1292/// what ends up on screen.
1293///
1294/// Post-conditions, for any `starts` ascending and `conflicts` of the same length:
1295/// - a row is in some run exactly when the file it belongs to is flagged;
1296/// - the runs are ascending and no two of them touch or overlap;
1297/// - a file of no rows produces no run of its own, and never splits one.
1298fn conflicting_row_runs(starts: &[usize], total: usize, conflicts: &[bool]) -> Vec<(usize, usize)> {
1299    let mut runs: Vec<(usize, usize)> = Vec::new();
1300    for (file, start) in starts.iter().copied().enumerate() {
1301        if !conflicts.get(file).copied().unwrap_or(false) {
1302            continue;
1303        }
1304        let end = starts.get(file + 1).copied().unwrap_or(total);
1305        match runs.last_mut() {
1306            Some(last) if last.1 == start => last.1 = end,
1307            _ => runs.push((start, end)),
1308        }
1309    }
1310    // A file of no rows leaves an empty range, which keeps no row out and would make
1311    // the "no two touch" post-condition depend on which files happen to be empty.
1312    runs.retain(|(start, end)| start < end);
1313    runs
1314}
1315
1316/// Options for a sort, one direction per column. Nulls go last in both directions, as
1317/// in pandas, DuckDB and spreadsheets; Polars would otherwise put them first either way.
1318/// Ties keep their order: each page is its own sort-then-slice, and an unstable sort
1319/// orders ties differently for a slice at the top (a top-k) than for one further
1320/// down, so pages would repeat and skip rows, and the inspector's one-row read would
1321/// find another row.
1322fn sort_options(descending: Vec<bool>) -> SortMultipleOptions {
1323    let n = descending.len();
1324    SortMultipleOptions::default()
1325        .with_order_descending_multi(descending)
1326        .with_nulls_last_multi(vec![true; n])
1327        .with_maintain_order(true)
1328}
1329
1330/// The columns a SQL statement's plan orders its result by, leading ones first, and
1331/// whether each runs descending: down from the top through what keeps the order (a
1332/// LIMIT, a projection of plain columns) to the sort. Stops at the first key that
1333/// is an expression rather than a column; empty when no sort is on top.
1334#[cfg(feature = "sql")]
1335fn ordered_by(plan: &polars::lazy::dsl::DslPlan) -> Vec<(String, bool)> {
1336    use polars::lazy::dsl::DslPlan;
1337    let mut node = plan;
1338    loop {
1339        node = match node {
1340            DslPlan::Slice { input, .. }
1341            | DslPlan::Filter { input, .. }
1342            | DslPlan::Cache { input, .. } => input,
1343            DslPlan::IR { dsl, .. } => dsl,
1344            DslPlan::Select { expr, input, .. }
1345                if expr.iter().all(|e| matches!(e, Expr::Column(_))) =>
1346            {
1347                input
1348            }
1349            DslPlan::Sort {
1350                by_column,
1351                sort_options,
1352                ..
1353            } => {
1354                let descending = &sort_options.descending;
1355                return by_column
1356                    .iter()
1357                    .map_while(|e| match e {
1358                        Expr::Column(name) => Some(name.to_string()),
1359                        _ => None,
1360                    })
1361                    .enumerate()
1362                    .map(|(i, name)| {
1363                        let down = descending
1364                            .get(i)
1365                            .or(descending.first())
1366                            .copied()
1367                            .unwrap_or(false);
1368                        (name, down)
1369                    })
1370                    .collect();
1371            }
1372            _ => return Vec::new(),
1373        };
1374    }
1375}
1376
1377/// `plan` giving its rows in one order on every read. Each page is its own read of
1378/// the view, so a node free to return rows in any order lets pages repeat some rows
1379/// and skip others, a `LIMIT` keep different groups on each read, and a sort's ties
1380/// arrive in a different order each time. Every sort keeps tied rows in the order
1381/// they come, as [`sort_options`] does (Polars SQL sorts unstably and offers no
1382/// option), and every grouping, distinct, union and join keeps its input's order,
1383/// except a grouping sorted by all its keys (see [`sorts_by_group_keys`]). Only the
1384/// parts of the plan holding such a node are rewritten.
1385#[cfg(feature = "sql")]
1386fn stable_order(plan: &mut polars::lazy::dsl::DslPlan) {
1387    order_stably(plan, false);
1388}
1389
1390/// [`stable_order`], where `groups_sorted` says a sort above `plan` orders the rows
1391/// of the grouping it reads through `plan`.
1392#[cfg(feature = "sql")]
1393fn order_stably(plan: &mut polars::lazy::dsl::DslPlan, groups_sorted: bool) {
1394    use polars::lazy::dsl::DslPlan;
1395    let unordered = |node: &DslPlan| match node {
1396        DslPlan::Sort { sort_options, .. } => !sort_options.maintain_order,
1397        DslPlan::GroupBy { maintain_order, .. } => !maintain_order,
1398        DslPlan::Distinct { options, .. } => !options.maintain_order,
1399        DslPlan::Union { args, .. } => !args.maintain_order,
1400        DslPlan::Join { options, .. } => options.args.maintain_order == MaintainOrderJoin::None,
1401        _ => false,
1402    };
1403    if !plan.into_iter().any(unordered) {
1404        return;
1405    }
1406    // Passed down the path sorts_by_group_keys walked to the grouping, and no other.
1407    let inputs_sorted = match plan {
1408        DslPlan::Sort { .. } => sorts_by_group_keys(plan),
1409        DslPlan::Select { .. } | DslPlan::IR { .. } => groups_sorted,
1410        _ => false,
1411    };
1412    match plan {
1413        DslPlan::Sort { sort_options, .. } => sort_options.maintain_order = true,
1414        DslPlan::GroupBy { maintain_order, .. } if !groups_sorted => *maintain_order = true,
1415        DslPlan::Distinct { options, .. } => options.maintain_order = true,
1416        DslPlan::Union { args, .. } => args.maintain_order = true,
1417        DslPlan::Join { options, .. } => {
1418            Arc::make_mut(options).args.maintain_order = MaintainOrderJoin::LeftRight;
1419        }
1420        _ => {}
1421    }
1422    if let DslPlan::IR { dsl, .. } = plan {
1423        // A plan asked for its schema is wrapped as IR, which would run as converted:
1424        // rewrite the plan it came from, and leave the IR behind.
1425        let mut inner = Arc::unwrap_or_clone(dsl.clone());
1426        order_stably(&mut inner, inputs_sorted);
1427        *plan = inner;
1428        return;
1429    }
1430    for_each_input(plan, &mut |input| order_stably(input, inputs_sorted));
1431}
1432
1433/// Whether `sort` sorts the rows of a grouping by every one of its keys, so the
1434/// order the groups arrive in never shows and keeping it is wasted time (#523).
1435/// Keys are unique per group, so such a sort has no ties, wherever it puts NULLs: a
1436/// NULL key is one group, and NaN and -0.0 group as the sort compares them. The
1437/// groups must reach the sort through projections that only pass or rename
1438/// columns: a filter or a computed column could depend on the order they arrive
1439/// in, as `ROW_NUMBER() OVER ()` does. A key counts only as a plain column of the
1440/// sort, which is what polars-sql makes of an alias, an ordinal or a key's own
1441/// name. It evaluates any other expression against the grouped columns, which
1442/// already hold the keys: `ORDER BY x % 4 * 2` over `GROUP BY x % 4 * 2` sorts by
1443/// the key's `% 4 * 2`, and ties keys that differ.
1444#[cfg(feature = "sql")]
1445fn sorts_by_group_keys(sort: &polars::lazy::dsl::DslPlan) -> bool {
1446    use polars::lazy::dsl::DslPlan;
1447    let DslPlan::Sort {
1448        input, by_column, ..
1449    } = sort
1450    else {
1451        return false;
1452    };
1453    // The sort's columns, under the names they have at each node on the way down.
1454    let mut names: Vec<PlSmallStr> = by_column
1455        .iter()
1456        .filter_map(|e| match e {
1457            Expr::Column(name) => Some(name.clone()),
1458            _ => None,
1459        })
1460        .collect();
1461    let mut node: &DslPlan = input;
1462    loop {
1463        match node {
1464            DslPlan::Select { input, expr, .. } => {
1465                // (output name, input name) of each column.
1466                let Some(renames) = expr
1467                    .iter()
1468                    .map(|e| match e {
1469                        Expr::Column(c) => Some((c, c)),
1470                        Expr::Alias(inner, alias) => match &**inner {
1471                            Expr::Column(c) => Some((alias, c)),
1472                            _ => None,
1473                        },
1474                        _ => None,
1475                    })
1476                    .collect::<Option<Vec<_>>>()
1477                else {
1478                    return false;
1479                };
1480                names = names
1481                    .iter()
1482                    .filter_map(|name| {
1483                        renames
1484                            .iter()
1485                            .find(|(out, _)| *out == name)
1486                            .map(|(_, source)| (*source).clone())
1487                    })
1488                    .collect();
1489                node = input;
1490            }
1491            DslPlan::IR { dsl, .. } => node = dsl,
1492            DslPlan::GroupBy {
1493                keys,
1494                options,
1495                apply: None,
1496                ..
1497            } if **options == GroupbyOptions::default() => {
1498                return keys.iter().all(|key| {
1499                    let meta = key.clone().meta();
1500                    !meta.has_multiple_outputs()
1501                        && meta.output_name().is_ok_and(|key| names.contains(&key))
1502                });
1503            }
1504            _ => return false,
1505        }
1506    }
1507}
1508
1509/// `plan` with an `IN (SELECT …)` subquery's values counted once instead of once per
1510/// row. polars-sql adds the values as a one-row list column and filters on
1511/// `col.first().list.len()` and `col.first().list.contains(NULL)`. The streaming
1512/// engine repeats that `first()` for every row of a batch and the list kernels copy
1513/// the list into each: rows times values, 26 GB for a page of a 100k-row table where
1514/// a third of the rows match (#509). Asked of the values exploded, the questions read
1515/// the one list; an empty list explodes to no rows, so both answers are unchanged.
1516/// Only those questions are rewritten: a user's own `ARRAY_LENGTH(FIRST(l))` differs
1517/// once exploded when the first list is NULL.
1518#[cfg(feature = "sql")]
1519fn count_subquery_values_once(plan: &mut polars::lazy::dsl::DslPlan) {
1520    use polars::lazy::dsl::{DslPlan, FunctionExpr, ListFunction};
1521    fn ask_once(e: Expr, names: &[PlSmallStr]) -> Expr {
1522        if !asks_of_subquery_values(&e, names) {
1523            return e;
1524        }
1525        let Expr::Function {
1526            mut input,
1527            function: FunctionExpr::ListExpr(function),
1528        } = e
1529        else {
1530            return e;
1531        };
1532        let values = input.swap_remove(0).explode(ExplodeOptions {
1533            empty_as_null: false,
1534            keep_nulls: true,
1535        });
1536        match function {
1537            ListFunction::Length => values.len(),
1538            _ => values.null_count().gt(lit(0)),
1539        }
1540    }
1541    if !plan.into_iter().any(asks_per_row) {
1542        return;
1543    }
1544    match plan {
1545        // A plan asked for its schema is wrapped as IR, which would run as converted:
1546        // rewrite the plan it came from, and leave the IR behind.
1547        DslPlan::IR { dsl, .. } => {
1548            let mut inner = Arc::unwrap_or_clone(dsl.clone());
1549            count_subquery_values_once(&mut inner);
1550            *plan = inner;
1551            return;
1552        }
1553        DslPlan::Filter { input, predicate } => {
1554            let names = subquery_value_columns(input);
1555            *predicate = predicate.clone().map_expr(|e| ask_once(e, &names));
1556        }
1557        _ => {}
1558    }
1559    for_each_input(plan, &mut count_subquery_values_once);
1560}
1561
1562/// The columns of `schema`, `plan`'s columns, that carry what polars-sql added to
1563/// hold `IN` subqueries' values (see [`subquery_value_columns`]). A WHERE's projection
1564/// drops them, but a QUALIFY keeps them in its result, a list of every value on every
1565/// row, and a statement reading its result as a table carries them on (#519), under
1566/// a join's suffix when both sides hold one. Matched by the name polars-sql gave
1567/// them, which is unique to the process, so no column of the user's is taken for one.
1568#[cfg(feature = "sql")]
1569fn leftover_subquery_value_columns(
1570    plan: &mut polars::lazy::dsl::DslPlan,
1571    schema: &Schema,
1572) -> Vec<PlSmallStr> {
1573    use polars::lazy::dsl::DslPlan;
1574    fn find(plan: &mut DslPlan, values: &mut Vec<PlSmallStr>, suffixes: &mut Vec<PlSmallStr>) {
1575        match plan {
1576            DslPlan::IR { dsl, .. } => {
1577                let mut inner = Arc::unwrap_or_clone(dsl.clone());
1578                find(&mut inner, values, suffixes);
1579                return;
1580            }
1581            DslPlan::Join { options, .. } => suffixes.push(options.args.suffix().clone()),
1582            _ => {}
1583        }
1584        values.extend(subquery_value_columns(plan));
1585        for_each_input(plan, &mut |input| find(input, values, suffixes));
1586    }
1587    fn carries(name: &str, values: &[PlSmallStr], suffixes: &[PlSmallStr]) -> bool {
1588        values.iter().any(|v| v == name)
1589            || suffixes.iter().any(|s| {
1590                name.strip_suffix(s.as_str())
1591                    .is_some_and(|rest| carries(rest, values, suffixes))
1592            })
1593    }
1594    let (mut values, mut suffixes) = (Vec::new(), Vec::new());
1595    find(plan, &mut values, &mut suffixes);
1596    if values.is_empty() {
1597        return Vec::new();
1598    }
1599    suffixes.retain(|s| !s.is_empty());
1600    schema
1601        .iter_names()
1602        .filter(|name| carries(name, &values, &suffixes))
1603        .cloned()
1604        .collect()
1605}
1606
1607/// Whether `node` filters on a question of an `IN` subquery's values that the
1608/// streaming engine answers once per row.
1609#[cfg(feature = "sql")]
1610fn asks_per_row(node: &polars::lazy::dsl::DslPlan) -> bool {
1611    let polars::lazy::dsl::DslPlan::Filter { input, predicate } = node else {
1612        return false;
1613    };
1614    let names = subquery_value_columns(input);
1615    !names.is_empty()
1616        && predicate
1617            .into_iter()
1618            .any(|e| asks_of_subquery_values(e, &names))
1619}
1620
1621/// The columns polars-sql adds beside `plan` to hold `IN` subqueries' values: each
1622/// subquery is selected as one aliased list and concatenated horizontally, broadcast
1623/// to the frame's rows (`SQLContext::process_subqueries`).
1624#[cfg(feature = "sql")]
1625fn subquery_value_columns(plan: &polars::lazy::dsl::DslPlan) -> Vec<PlSmallStr> {
1626    use polars::lazy::dsl::DslPlan;
1627    match plan {
1628        DslPlan::HConcat { inputs, options } if options.broadcast_unit_length => inputs
1629            .iter()
1630            .skip(1)
1631            .filter_map(|input| match input {
1632                DslPlan::Select { expr, .. } => match expr.as_slice() {
1633                    [Expr::Alias(_, name)] => Some(name.clone()),
1634                    _ => None,
1635                },
1636                _ => None,
1637            })
1638            .collect(),
1639        DslPlan::IR { dsl, .. } => subquery_value_columns(dsl),
1640        _ => Vec::new(),
1641    }
1642}
1643
1644/// Whether `e` is polars-sql asking how many values an `IN` subquery returned, or
1645/// whether one is NULL: `list.len` or `list.contains(NULL)` of `col(name).first()`,
1646/// where `name` is one of `names`, the columns holding the values.
1647#[cfg(feature = "sql")]
1648fn asks_of_subquery_values(e: &Expr, names: &[PlSmallStr]) -> bool {
1649    use polars::lazy::dsl::{FunctionExpr, ListFunction};
1650    let values = |e: &Expr| {
1651        matches!(e, Expr::Agg(AggExpr::First(c))
1652            if matches!(&**c, Expr::Column(name) if names.contains(name)))
1653    };
1654    match e {
1655        Expr::Function {
1656            input,
1657            function: FunctionExpr::ListExpr(ListFunction::Length),
1658        } => matches!(input.as_slice(), [set] if values(set)),
1659        Expr::Function {
1660            input,
1661            function: FunctionExpr::ListExpr(ListFunction::Contains { nulls_equal: true }),
1662        } => {
1663            matches!(input.as_slice(), [set, Expr::Literal(item)] if values(set) && item.is_null())
1664        }
1665        _ => false,
1666    }
1667}
1668
1669/// The in-memory frame `lf` scans, when it is a scan of one.
1670fn scanned_frame(lf: &LazyFrame) -> Option<Arc<DataFrame>> {
1671    match &lf.logical_plan {
1672        polars::lazy::dsl::DslPlan::DataFrameScan { df, .. } => Some(df.clone()),
1673        _ => None,
1674    }
1675}
1676
1677/// Calls `f` on each plan `plan` reads from.
1678pub(crate) fn for_each_input(
1679    plan: &mut polars::lazy::dsl::DslPlan,
1680    f: &mut dyn FnMut(&mut polars::lazy::dsl::DslPlan),
1681) {
1682    use polars::lazy::dsl::DslPlan;
1683    match plan {
1684        DslPlan::Sort { input, .. }
1685        | DslPlan::Select { input, .. }
1686        | DslPlan::GroupBy { input, .. }
1687        | DslPlan::Filter { input, .. }
1688        | DslPlan::Distinct { input, .. }
1689        | DslPlan::Slice { input, .. }
1690        | DslPlan::HStack { input, .. }
1691        | DslPlan::MatchToSchema { input, .. }
1692        | DslPlan::MapFunction { input, .. }
1693        | DslPlan::Sink { input, .. }
1694        | DslPlan::Cache { input, .. }
1695        | DslPlan::Pivot { input, .. } => f(Arc::make_mut(input)),
1696        DslPlan::Union { inputs, .. }
1697        | DslPlan::HConcat { inputs, .. }
1698        | DslPlan::SinkMultiple { inputs } => inputs.iter_mut().for_each(f),
1699        DslPlan::PipeWithSchema { input, .. } => {
1700            let mut inputs = input.to_vec();
1701            inputs.iter_mut().for_each(&mut *f);
1702            *input = inputs.into();
1703        }
1704        DslPlan::Join {
1705            input_left,
1706            input_right,
1707            ..
1708        } => {
1709            f(Arc::make_mut(input_left));
1710            f(Arc::make_mut(input_right));
1711        }
1712        DslPlan::Gather { input, idxs, .. } => {
1713            f(Arc::make_mut(input));
1714            f(Arc::make_mut(idxs));
1715        }
1716        DslPlan::ExtContext { input, contexts } => {
1717            f(Arc::make_mut(input));
1718            contexts.iter_mut().for_each(f);
1719        }
1720        _ => {}
1721    }
1722}
1723
1724/// A string's in-memory width when nothing says otherwise: the view plus a short value.
1725const STRING_BYTES_GUESS: usize = 40;
1726
1727/// Bytes a row of `columns` takes in memory, estimated from the schema: the width of
1728/// each fixed-size type; for a string the footer's average in `column_bytes` (or a
1729/// guess) plus its view; for a nested column the footer's average, else a guess.
1730/// Binary columns are buffered as a stub (see `binary_stub_exprs`).
1731fn estimate_bytes_per_row(
1732    schema: &Schema,
1733    columns: &[String],
1734    column_bytes: &[(String, usize)],
1735) -> usize {
1736    let footer_width = |name: &String| {
1737        column_bytes
1738            .iter()
1739            .find(|(n, _)| n == name)
1740            .map(|(_, w)| *w)
1741    };
1742    columns
1743        .iter()
1744        .map(|name| match schema.get(name.as_str()) {
1745            Some(DataType::String) => 16 + footer_width(name).unwrap_or(STRING_BYTES_GUESS - 16),
1746            Some(DataType::Binary) => 16 + binary_stub().len(),
1747            Some(DataType::Boolean) => 1,
1748            Some(DataType::Null) => 0,
1749            Some(dtype) if dtype.is_primitive_numeric() || dtype.is_temporal() => {
1750                match dtype.to_physical() {
1751                    DataType::Int8 | DataType::UInt8 => 1,
1752                    DataType::Int16 | DataType::UInt16 => 2,
1753                    DataType::Int32 | DataType::UInt32 | DataType::Float32 => 4,
1754                    DataType::Int128 => 16,
1755                    _ => 8,
1756                }
1757            }
1758            Some(DataType::Decimal(..)) => 16,
1759            _ => footer_width(name).unwrap_or(64),
1760        })
1761        .sum::<usize>()
1762        .max(1)
1763}
1764
1765/// The rows `[offset, offset + len)` of `df`, copied when a slice of them would keep
1766/// much more allocated than they are. A `seam` inside them, where a stitch joined two
1767/// fills, stays a chunk boundary (see [`compact_rows`]).
1768///
1769/// A slice keeps every chunk it touches. A fill read in many chunks (a Parquet or CSV
1770/// scan) lets the rest go with a slice alone; one read in a single chunk, a stitched
1771/// union or a string column sharing its parent's data would keep the whole fill. A
1772/// chunk that is itself a slice of more is not seen through.
1773fn trim_rows(df: DataFrame, offset: usize, len: usize, seam: Option<usize>) -> DataFrame {
1774    if backing_rows(&df, offset, len) > len + len / 4 {
1775        compact_rows(df, offset, len, seam)
1776    } else {
1777        df.slice(offset as i64, len)
1778    }
1779}
1780
1781/// The most rows any column of `df` keeps allocated behind the slice `[offset, offset
1782/// + len)`: every chunk the slice touches, whole.
1783fn backing_rows(df: &DataFrame, offset: usize, len: usize) -> usize {
1784    let end = offset + len;
1785    df.columns()
1786        .iter()
1787        .filter_map(Column::as_series)
1788        .map(|s| {
1789            let mut start = 0;
1790            let mut touched = 0;
1791            for chunk in s.chunks() {
1792                let chunk_end = start + chunk.len();
1793                if start < end && offset < chunk_end {
1794                    touched += chunk.len();
1795                }
1796                start = chunk_end;
1797            }
1798            touched
1799        })
1800        .max()
1801        .unwrap_or(len)
1802}
1803
1804/// The rows `[offset, offset + len)` of `df` in storage of their own: one chunk a
1805/// column, or two when `seam` falls inside them, so a later cut down to one side of a
1806/// stitch (`holds_buffer`) is a slice that lets the other side go.
1807///
1808/// A slice keeps the whole of its parent allocated, and neither `rechunk` (a lone chunk
1809/// is left as it is) nor `take` (a string column keeps its parent's data buffers) is
1810/// sure to let go of it. Polars' builders with `ShareStrategy::Never` copy every
1811/// physical type, nested children and string bytes included. A constant column stays
1812/// one value: built out, it would be a copy of the value per row.
1813///
1814/// Each column of `df` is let go of once it is copied, so the copy costs about one
1815/// column's kept rows over `df` rather than all of them. On the collect worker the
1816/// rows on screen are still held meanwhile (#483).
1817fn compact_rows(df: DataFrame, offset: usize, len: usize, seam: Option<usize>) -> DataFrame {
1818    use polars::series::builder::SeriesBuilder;
1819    use polars_arrow::array::builder::ShareStrategy;
1820    #[cfg(test)]
1821    tests::COMPACTIONS.with(|count| count.set(count.get() + 1));
1822    let len = len.min(df.height().saturating_sub(offset));
1823    let pieces = match seam.filter(|&seam| offset < seam && seam < offset + len) {
1824        Some(seam) => vec![(offset, seam - offset), (seam, offset + len - seam)],
1825        None => vec![(offset, len)],
1826    };
1827    let copy = |series: &Series, (offset, len): (usize, usize)| {
1828        let mut builder = SeriesBuilder::new(series.dtype().clone());
1829        builder.reserve(len);
1830        builder.subslice_extend(series, offset, len, ShareStrategy::Never);
1831        builder.freeze(series.name().clone())
1832    };
1833    let columns = df
1834        .into_columns()
1835        .into_iter()
1836        .map(|column| match column {
1837            Column::Scalar(constant) => {
1838                Column::new_scalar(constant.name().clone(), constant.scalar().clone(), len)
1839            }
1840            Column::Series(series) => {
1841                let mut kept = copy(&series, pieces[0]);
1842                for &piece in &pieces[1..] {
1843                    if kept.append_owned(copy(&series, piece)).is_err() {
1844                        kept = copy(&series, (offset, len));
1845                        break;
1846                    }
1847                }
1848                kept.into_column()
1849            }
1850        })
1851        .collect();
1852    // Cannot fail: the names are one frame's and every column was built to `len` rows.
1853    DataFrame::new(len, columns).unwrap_or_else(|_| DataFrame::empty_with_height(len))
1854}
1855
1856/// Shrink `[buffer_start, buffer_end)` to at most `max_len` rows, kept around the view
1857/// `[view_start, view_end)` and inside `[floor, ceil)`.
1858fn shrink_around_view(
1859    view_start: usize,
1860    view_end: usize,
1861    max_len: usize,
1862    floor: usize,
1863    ceil: usize,
1864    buffer_start: &mut usize,
1865    buffer_end: &mut usize,
1866) {
1867    if buffer_end.saturating_sub(*buffer_start) <= max_len {
1868        return;
1869    }
1870    let view_len = view_end.saturating_sub(view_start);
1871    if view_len >= max_len {
1872        *buffer_start = view_start;
1873        *buffer_end = (view_start + max_len).min(ceil);
1874        return;
1875    }
1876    let half = (max_len - view_len) / 2;
1877    *buffer_end = (view_end + half).min(ceil);
1878    *buffer_start = buffer_end.saturating_sub(max_len).max(floor);
1879    if *buffer_start > view_start {
1880        *buffer_start = view_start;
1881    }
1882    *buffer_end = (*buffer_start + max_len).min(ceil);
1883}
1884
1885/// The most files one buffer read opens, beyond those the view itself spans.
1886const MAX_FILES_PER_BUFFER: usize = 16;
1887
1888/// Narrow `[start, end)` to at most `max_files` files, keeping every file the view
1889/// `[view_start, view_end)` lies in and adding the ones after it first.
1890fn limit_files(
1891    offsets: &[usize],
1892    view_start: usize,
1893    view_end: usize,
1894    start: usize,
1895    end: usize,
1896    max_files: usize,
1897) -> (usize, usize) {
1898    let (Some((first, last)), Some((view_first, view_last))) = (
1899        files_holding(offsets, start, end.saturating_sub(start)),
1900        files_holding(
1901            offsets,
1902            view_start,
1903            view_end.saturating_sub(view_start).max(1),
1904        ),
1905    ) else {
1906        return (start, end);
1907    };
1908    // An empty file is not opened (see `window_of`), so it costs nothing to reach past.
1909    let opened = |from: usize, to: usize| (from..=to).filter(|&i| holds_rows(offsets, i)).count();
1910    if opened(first, last) <= max_files {
1911        return (start, end);
1912    }
1913    let (mut lo, mut hi) = (view_first.max(first), view_last.min(last));
1914    let mut files = opened(lo, hi);
1915    while files < max_files && (hi < last || lo > first) {
1916        if hi < last {
1917            hi += 1;
1918            files += usize::from(holds_rows(offsets, hi));
1919        }
1920        if files < max_files && lo > first {
1921            lo -= 1;
1922            files += usize::from(holds_rows(offsets, lo));
1923        }
1924    }
1925    (start.max(offsets[lo]), end.min(offsets[hi + 1]))
1926}
1927
1928/// Whether file `i` has any rows, given where each file's rows start.
1929fn holds_rows(offsets: &[usize], i: usize) -> bool {
1930    offsets[i + 1] > offsets[i]
1931}
1932
1933/// The files from `first` to `last` that hold rows: a window reads these and passes
1934/// over the empty ones, which a dataset written a file a day can be mostly made of.
1935fn files_with_rows(offsets: &[usize], first: usize, last: usize) -> Vec<usize> {
1936    (first..=last).filter(|&i| holds_rows(offsets, i)).collect()
1937}
1938
1939/// The first and last files holding rows `[start, start + len)`, given where each file's
1940/// rows start (`offsets`, with the total last). `None` when the rows lie past the end.
1941fn files_holding(offsets: &[usize], start: usize, len: usize) -> Option<(usize, usize)> {
1942    let files = offsets.len().checked_sub(1)?;
1943    let total = *offsets.last()?;
1944    if files == 0 || len == 0 || start >= total {
1945        return None;
1946    }
1947    let end = (start + len).min(total);
1948    // The file a row is in: the last one starting at or before it. Empty files start
1949    // where the next one does and are skipped over.
1950    let file_of = |row: usize| offsets.partition_point(|&o| o <= row).saturating_sub(1);
1951    Some((file_of(start), file_of(end - 1).min(files - 1)))
1952}
1953
1954/// Rows `[start, start + len)` of `lf` as `all_columns`. With `files` counted, a scan
1955/// of only the files holding them, so a window deep in a remote dataset does not read
1956/// every file before it. With `records`, the rows read straight from the source.
1957fn window_of(
1958    lf: &LazyFrame,
1959    files: Option<&RemoteFiles>,
1960    records: Option<&dyn crate::pushdown::Windowed>,
1961    read_as_text: &[PlSmallStr],
1962    start: usize,
1963    len: usize,
1964    all_columns: Vec<Expr>,
1965) -> PolarsResult<LazyFrame> {
1966    // Polars gives an anonymous scan no row offset, so a slice deep in the view would
1967    // read every row before it; the source starts the window there instead.
1968    if let Some(records) = records {
1969        return Ok(records.window(start, len)?.select(all_columns));
1970    }
1971    if let Some((files, offsets)) = files.and_then(|f| f.offsets.as_ref().map(|o| (f, o)))
1972        && let Some((first, last)) = files_holding(offsets, start, len)
1973    {
1974        // The window's first file holds its first row, so leaving out the empty files
1975        // after it does not move the slice.
1976        let urls: Vec<String> = files_with_rows(offsets, first, last)
1977            .into_iter()
1978            .map(|i| files.urls[i].clone())
1979            .collect();
1980        let lf = (files.scan)(&urls, read_as_text)?;
1981        return Ok(lf
1982            .select(all_columns)
1983            .slice((start - offsets[first]) as i64, len as u32));
1984    }
1985    Ok(lf
1986        .clone()
1987        .select(all_columns)
1988        .slice(start as i64, len as u32))
1989}
1990
1991/// The rows of a view, for a reader off the UI thread: read a window at a time as a
1992/// page is, or from the buffer the table already holds.
1993#[derive(Clone)]
1994pub(crate) struct ViewRows {
1995    lf: LazyFrame,
1996    files: Option<RemoteFiles>,
1997    /// See [`DataTableState::window_now`].
1998    records: Option<Arc<dyn crate::pushdown::Windowed>>,
1999    read_as_text: Vec<PlSmallStr>,
2000    /// The buffer on hand and the view row it starts at.
2001    pub(crate) buffer: Option<(DataFrame, usize)>,
2002    /// The view's row count, when it is known.
2003    pub(crate) num_rows: Option<usize>,
2004    pub(crate) streaming: bool,
2005    /// Any window of the view reads all of it: see [`sees_every_row_first`].
2006    pub(crate) whole: bool,
2007    /// A window of the view reads every row before it: see [`reads_up_to_a_window`].
2008    pub(crate) reads_up_to: bool,
2009}
2010
2011/// Whether `lf` has to see every row before it gives its first: a sort, a group by or a
2012/// pivot under it. Then a window of it costs as much as all of it.
2013pub(crate) fn sees_every_row_first(lf: &LazyFrame) -> bool {
2014    use polars::lazy::dsl::DslPlan;
2015    lf.logical_plan.into_iter().any(|node| {
2016        matches!(
2017            node,
2018            DslPlan::Sort { .. } | DslPlan::GroupBy { .. } | DslPlan::Pivot { .. }
2019        )
2020    })
2021}
2022
2023/// Whether a window of `lf` reads every row before it: a filter, which has to test
2024/// them to know which row is the window's first, or a scan with no row index to skip
2025/// by, such as a CSV. Parquet and IPC skip to a window.
2026pub(crate) fn reads_up_to_a_window(lf: &LazyFrame) -> bool {
2027    use polars::lazy::dsl::{DslPlan, FileScanDsl};
2028    lf.logical_plan.into_iter().any(|node| match node {
2029        DslPlan::Filter { .. } => true,
2030        DslPlan::Scan { scan_type, .. } => !matches!(
2031            **scan_type,
2032            FileScanDsl::Parquet { .. } | FileScanDsl::Ipc { .. }
2033        ),
2034        _ => false,
2035    })
2036}
2037
2038impl ViewRows {
2039    /// Rows `[start, start + len)` of the view as `exprs`.
2040    pub(crate) fn window(
2041        &self,
2042        start: usize,
2043        len: usize,
2044        exprs: Vec<Expr>,
2045    ) -> PolarsResult<LazyFrame> {
2046        window_of(
2047            &self.lf,
2048            self.files.as_ref(),
2049            self.records.as_deref(),
2050            &self.read_as_text,
2051            start,
2052            len,
2053            exprs,
2054        )
2055    }
2056
2057    /// The view `lf`, with `buffer` on hand from row `buffer_start`.
2058    #[cfg(test)]
2059    pub(crate) fn of(lf: LazyFrame, buffer: Option<(DataFrame, usize)>) -> Self {
2060        Self {
2061            whole: sees_every_row_first(&lf),
2062            reads_up_to: reads_up_to_a_window(&lf),
2063            lf,
2064            files: None,
2065            records: None,
2066            read_as_text: Vec::new(),
2067            buffer,
2068            num_rows: None,
2069            streaming: false,
2070        }
2071    }
2072}
2073
2074/// Snap `[start, end)` outward to the row groups it touches, given where each group
2075/// starts (`offsets`, with the total last).
2076///
2077/// Polars fetches a row group whole for any slice that touches it, so the groups the
2078/// view `[view_start, view_end)` lies in are always taken whole: paging inside them then
2079/// costs nothing. The other groups the window reaches into are added while the result
2080/// stays within `cap` rows (0 for no cap), the ones ahead of the view first.
2081fn align_to_row_groups(
2082    offsets: &[usize],
2083    view_start: usize,
2084    view_end: usize,
2085    start: usize,
2086    end: usize,
2087    cap: usize,
2088) -> (usize, usize) {
2089    let Some(groups) = offsets.len().checked_sub(1).filter(|n| *n > 0) else {
2090        return (start, end);
2091    };
2092    let group_of = |row: usize| {
2093        offsets
2094            .partition_point(|&o| o <= row)
2095            .saturating_sub(1)
2096            .min(groups - 1)
2097    };
2098    let last_row = |s: usize, e: usize| e.saturating_sub(1).max(s);
2099    let (mut lo, mut hi) = (
2100        group_of(view_start),
2101        group_of(last_row(view_start, view_end)),
2102    );
2103    let (want_lo, want_hi) = (group_of(start), group_of(last_row(start, end)));
2104    let fits = |lo: usize, hi: usize| cap == 0 || offsets[hi + 1] - offsets[lo] <= cap;
2105    loop {
2106        if hi < want_hi && fits(lo, hi + 1) {
2107            hi += 1;
2108        } else if lo > want_lo && fits(lo - 1, hi) {
2109            lo -= 1;
2110        } else {
2111            break;
2112        }
2113    }
2114    (offsets[lo], offsets[hi + 1])
2115}
2116
2117impl DataTableState {
2118    pub fn new(
2119        lf: LazyFrame,
2120        pages_lookahead: Option<usize>,
2121        pages_lookback: Option<usize>,
2122        max_buffered_rows: Option<usize>,
2123        max_buffered_mb: Option<usize>,
2124        polars_streaming: bool,
2125    ) -> Result<Self> {
2126        let (schema, source_rows_at_open) = Self::without_source_rows(lf.clone().collect_schema()?);
2127        let column_order: Vec<String> = schema.iter_names().map(|s| s.to_string()).collect();
2128        Ok(Self {
2129            unsorted_lf: None,
2130            original_lf: lf.clone(),
2131            original_schema: schema.clone(),
2132            base_lf: lf.clone(),
2133            lf,
2134            df: None,
2135            locked_df: None,
2136            table_state: TableState::default(),
2137            start_row: 0,
2138            visible_rows: 0,
2139            termcol_index: 0,
2140            visible_termcols: 0,
2141            scroll_room: None,
2142            column_moves: Vec::new(),
2143            page_trail: Vec::new(),
2144            on_screen: None,
2145            drawn: None,
2146            error: None,
2147            suppress_error_display: false,
2148            schema,
2149            num_rows: 0,
2150            num_rows_valid: false,
2151            pristine_rows: None,
2152            len_generation: next_len_generation(),
2153            root_generation: next_len_generation(),
2154            parquet_count_dir: None,
2155            measurements: Arc::new(crate::measurements::Meter::default()),
2156            filters: Vec::new(),
2157            sort_columns: Vec::new(),
2158            sort_descending: Vec::new(),
2159            sort_ascending: true,
2160            cursor_column: None,
2161            cursor_at: 0,
2162            reveal_cursor: false,
2163            active_query: String::new(),
2164            active_sql_query: String::new(),
2165            query_order: Vec::new(),
2166            active_fuzzy_query: String::new(),
2167            column_order,
2168            locked_columns_count: 0,
2169            frozen_fit: (0, 0),
2170            widths: ColumnWidths::default(),
2171            grouped: None,
2172            group_source: None,
2173            reshaped_lf: None,
2174            drilled_down_group_index: None,
2175            drilled_down_group_key: None,
2176            drilled_down_group_key_columns: None,
2177            pages_lookahead: pages_lookahead.unwrap_or(3),
2178            pages_lookback: pages_lookback.unwrap_or(3),
2179            max_buffered_rows: max_buffered_rows.unwrap_or(DEFAULT_MAX_BUFFERED_ROWS),
2180            max_buffered_mb: max_buffered_mb.unwrap_or(512),
2181            remote_source: false,
2182            row_group_offsets: None,
2183            remote_files: None,
2184            remote_objects: None,
2185            dataset_schema: None,
2186            drift_column_present: false,
2187            drift_groups: Arc::new(Vec::new()),
2188            drift_at_open: false,
2189            groups_at_open: Arc::new(Vec::new()),
2190            source_rows_at_open,
2191            view_numbered: false,
2192            indexing: None,
2193            numbering: None,
2194            row_estimate: None,
2195            indexing_notes: Vec::new(),
2196            indexing_guessed: false,
2197            drift_file_starts: Vec::new(),
2198            drift_file_group: Vec::new(),
2199            drift_files: Vec::new(),
2200            footers_pending: None,
2201            notes: Vec::new(),
2202            open_notes: Vec::new(),
2203            not_the_table: None,
2204            format_read: None,
2205            delimited: None,
2206            fixed_window: None,
2207            pushdown: None,
2208            source_hold: None,
2209            read_mode: None,
2210            read_as: None,
2211            fetched: false,
2212            detail: None,
2213            file_units: Arc::new(Vec::new()),
2214            notes_seen: false,
2215            notes_at_open: Vec::new(),
2216            view_notes: Vec::new(),
2217            drift_dataset_rows: 0,
2218            dataset_at_open: None,
2219            read_as_text: Vec::new(),
2220            column_bytes: Vec::new(),
2221            observed_bytes_per_row: None,
2222            buffered_start_row: 0,
2223            buffered_end_row: 0,
2224            buffered_df: None,
2225            proximity_threshold: 0, // Will be set when visible_rows is known
2226            drawn_start: 0,
2227            row_numbers: false, // Will be set from options
2228            row_start_index: 1, // Will be set from options
2229            last_pivot_spec: None,
2230            last_melt_spec: None,
2231            reshape_source: None,
2232            base_steps: Vec::new(),
2233            read_python: Vec::new(),
2234            read_notes: Vec::new(),
2235            read_units: None,
2236            typing: Typing::default(),
2237            unfit_notes: None,
2238            column_changes: Vec::new(),
2239            changes_version: 0,
2240            changes_unfit: None,
2241            changes_dropped: Vec::new(),
2242            reshape_steps: None,
2243            lineage: None,
2244            reshape_lineage: None,
2245            partition_columns: None,
2246            decompress_temp_file: None,
2247            download: None,
2248            converted: Vec::new(),
2249            other_tables: Vec::new(),
2250            polars_streaming,
2251            defer_collect: false,
2252            needs_recollect: false,
2253            follow: None,
2254            follow_known: None,
2255            sampled: None,
2256        })
2257    }
2258
2259    /// `schema` without the hidden row index, and whether it had one: the rows' place
2260    /// in the source, which `#` shows, never a column of theirs.
2261    fn without_source_rows(schema: Arc<Schema>) -> (Arc<Schema>, bool) {
2262        if !schema.contains(crate::schema_union::DRIFT_COLUMN) {
2263            return (schema, false);
2264        }
2265        let mut schema = (*schema).clone();
2266        schema.shift_remove(crate::schema_union::DRIFT_COLUMN);
2267        (Arc::new(schema), true)
2268    }
2269
2270    /// Create state from an existing LazyFrame (e.g. from Python or in-memory). Uses OpenOptions for display/buffer settings.
2271    pub fn from_lazyframe(lf: LazyFrame, options: &crate::OpenOptions) -> Result<Self> {
2272        let mut state = Self::new(
2273            lf,
2274            options.pages_lookahead,
2275            options.pages_lookback,
2276            options.max_buffered_rows,
2277            options.max_buffered_mb,
2278            options.polars_streaming,
2279        )?;
2280        state.row_numbers = options.row_numbers;
2281        state.row_start_index = options.row_start_index;
2282        Ok(state)
2283    }
2284
2285    /// Create state from a pre-collected schema and LazyFrame (for phased loading). Does not call collect_schema();
2286    /// df is None so the UI can render headers while the first collect() runs.
2287    /// When `partition_columns` is Some (e.g. hive), column order is partition cols first.
2288    pub fn from_schema_and_lazyframe(
2289        schema: Arc<Schema>,
2290        lf: LazyFrame,
2291        options: &crate::OpenOptions,
2292        partition_columns: Option<Vec<String>>,
2293    ) -> Result<Self> {
2294        let (schema, source_rows_at_open) = Self::without_source_rows(schema);
2295        let column_order: Vec<String> = if let Some(ref part) = partition_columns {
2296            let part_set: HashSet<&str> = part.iter().map(String::as_str).collect();
2297            let rest: Vec<String> = schema
2298                .iter_names()
2299                .map(|s| s.to_string())
2300                .filter(|c| !part_set.contains(c.as_str()))
2301                .collect();
2302            part.iter().cloned().chain(rest).collect()
2303        } else {
2304            schema.iter_names().map(|s| s.to_string()).collect()
2305        };
2306        Ok(Self {
2307            unsorted_lf: None,
2308            original_lf: lf.clone(),
2309            original_schema: schema.clone(),
2310            base_lf: lf.clone(),
2311            lf,
2312            df: None,
2313            locked_df: None,
2314            table_state: TableState::default(),
2315            start_row: 0,
2316            visible_rows: 0,
2317            termcol_index: 0,
2318            visible_termcols: 0,
2319            scroll_room: None,
2320            column_moves: Vec::new(),
2321            page_trail: Vec::new(),
2322            on_screen: None,
2323            drawn: None,
2324            error: None,
2325            suppress_error_display: false,
2326            schema,
2327            num_rows: 0,
2328            num_rows_valid: false,
2329            pristine_rows: None,
2330            len_generation: next_len_generation(),
2331            root_generation: next_len_generation(),
2332            parquet_count_dir: None,
2333            measurements: Arc::new(crate::measurements::Meter::default()),
2334            filters: Vec::new(),
2335            sort_columns: Vec::new(),
2336            sort_descending: Vec::new(),
2337            sort_ascending: true,
2338            cursor_column: None,
2339            cursor_at: 0,
2340            reveal_cursor: false,
2341            active_query: String::new(),
2342            active_sql_query: String::new(),
2343            query_order: Vec::new(),
2344            active_fuzzy_query: String::new(),
2345            column_order,
2346            locked_columns_count: 0,
2347            frozen_fit: (0, 0),
2348            widths: ColumnWidths::default(),
2349            grouped: None,
2350            group_source: None,
2351            reshaped_lf: None,
2352            drilled_down_group_index: None,
2353            drilled_down_group_key: None,
2354            drilled_down_group_key_columns: None,
2355            pages_lookahead: options.pages_lookahead.unwrap_or(3),
2356            pages_lookback: options.pages_lookback.unwrap_or(3),
2357            max_buffered_rows: options
2358                .max_buffered_rows
2359                .unwrap_or(DEFAULT_MAX_BUFFERED_ROWS),
2360            max_buffered_mb: options.max_buffered_mb.unwrap_or(512),
2361            remote_source: false,
2362            row_group_offsets: None,
2363            remote_files: None,
2364            remote_objects: None,
2365            dataset_schema: None,
2366            drift_column_present: false,
2367            drift_groups: Arc::new(Vec::new()),
2368            drift_at_open: false,
2369            groups_at_open: Arc::new(Vec::new()),
2370            source_rows_at_open,
2371            view_numbered: false,
2372            indexing: None,
2373            numbering: None,
2374            row_estimate: None,
2375            indexing_notes: Vec::new(),
2376            indexing_guessed: false,
2377            drift_file_starts: Vec::new(),
2378            drift_file_group: Vec::new(),
2379            drift_files: Vec::new(),
2380            footers_pending: None,
2381            notes: Vec::new(),
2382            open_notes: Vec::new(),
2383            not_the_table: None,
2384            format_read: None,
2385            delimited: None,
2386            fixed_window: None,
2387            pushdown: None,
2388            source_hold: None,
2389            read_mode: None,
2390            read_as: None,
2391            fetched: false,
2392            detail: None,
2393            file_units: Arc::new(Vec::new()),
2394            notes_seen: false,
2395            notes_at_open: Vec::new(),
2396            view_notes: Vec::new(),
2397            drift_dataset_rows: 0,
2398            dataset_at_open: None,
2399            read_as_text: Vec::new(),
2400            column_bytes: Vec::new(),
2401            observed_bytes_per_row: None,
2402            buffered_start_row: 0,
2403            buffered_end_row: 0,
2404            buffered_df: None,
2405            proximity_threshold: 0,
2406            drawn_start: 0,
2407            row_numbers: options.row_numbers,
2408            row_start_index: options.row_start_index,
2409            last_pivot_spec: None,
2410            last_melt_spec: None,
2411            reshape_source: None,
2412            base_steps: Vec::new(),
2413            read_python: Vec::new(),
2414            read_notes: Vec::new(),
2415            read_units: None,
2416            typing: Typing::default(),
2417            unfit_notes: None,
2418            column_changes: Vec::new(),
2419            changes_version: 0,
2420            changes_unfit: None,
2421            changes_dropped: Vec::new(),
2422            reshape_steps: None,
2423            lineage: None,
2424            reshape_lineage: None,
2425            partition_columns,
2426            decompress_temp_file: None,
2427            download: None,
2428            converted: Vec::new(),
2429            other_tables: Vec::new(),
2430            polars_streaming: options.polars_streaming,
2431            defer_collect: false,
2432            needs_recollect: false,
2433            follow: None,
2434            follow_known: None,
2435            sampled: None,
2436        })
2437    }
2438
2439    /// The state as its open found the dataset: everything in `facts`, given at once.
2440    ///
2441    /// The one way an open's findings reach a state, taken while it is still the data as
2442    /// loaded. Applied in the order they depend on each other: the files before their row
2443    /// groups, which set the count. Once on screen, a dataset learns more only through
2444    /// [`Self::join_dataset_schema`] and [`Self::count_landed`].
2445    pub fn with_open(mut self, facts: OpenFacts) -> Self {
2446        let OpenFacts {
2447            remote_source,
2448            row_groups,
2449            remote_files,
2450            remote_objects,
2451            dataset,
2452            footers_pending,
2453            column_bytes,
2454            parquet_count_dir,
2455            measurements,
2456            open_notes,
2457            not_the_table,
2458            format_read,
2459            delimited,
2460            download,
2461            converted,
2462            other_tables,
2463            pushdown,
2464            hold,
2465            read_mode,
2466            read_as,
2467            fetched,
2468            detail,
2469            records,
2470            units,
2471            indexing,
2472            numbering,
2473            typing,
2474        } = facts;
2475        self.numbering = numbering;
2476        self.typing = typing;
2477        debug_assert!(
2478            self.is_pristine(),
2479            "an open's facts are for the data as loaded"
2480        );
2481        self.remote_source = remote_source;
2482        self.remote_files = remote_files;
2483        self.remote_objects = (!remote_objects.is_empty()).then(|| {
2484            Arc::new(
2485                remote_objects
2486                    .into_iter()
2487                    .map(|object| (object.url.clone(), object))
2488                    .collect(),
2489            )
2490        });
2491        if !row_groups.is_empty() {
2492            if self.remote_files.is_some() {
2493                self.record_file_row_groups(&row_groups);
2494            } else {
2495                let flat: Vec<usize> = row_groups.into_iter().flatten().collect();
2496                self.record_row_groups(&flat);
2497            }
2498        }
2499        if let Some(DatasetAtOpen {
2500            schema,
2501            file_rows,
2502            files,
2503        }) = dataset
2504        {
2505            self.record_dataset_schema(schema, &file_rows, &files);
2506        }
2507        self.footers_pending = footers_pending;
2508        self.column_bytes = column_bytes;
2509        self.parquet_count_dir = parquet_count_dir;
2510        self.measurements = measurements;
2511        self.open_notes = open_notes;
2512        self.not_the_table = not_the_table;
2513        self.fixed_window = format_read
2514            .as_ref()
2515            .map(|read| read.records.clone() as Arc<dyn crate::pushdown::Windowed>);
2516        if let Some(read) = &format_read {
2517            // As for audio: the reader counted the records from the file's size, and a
2518            // count through the frame would build its row index whole.
2519            self.set_num_rows(read.records.rows());
2520        }
2521        self.pushdown = pushdown;
2522        self.source_hold = hold;
2523        self.format_read = format_read;
2524        self.delimited = delimited;
2525        self.download = download;
2526        self.converted = converted;
2527        self.other_tables = other_tables;
2528        self.read_mode = read_mode;
2529        self.read_as = read_as;
2530        self.fetched = fetched;
2531        self.detail = detail;
2532        if let Some((window, rows)) = records {
2533            // The reader knows its rows; a count through the frame would build its row
2534            // index whole. Lines still being indexed know only some of theirs.
2535            if indexing.is_none() {
2536                self.set_num_rows(rows);
2537            }
2538            self.fixed_window = Some(window);
2539        }
2540        if let Some(lines) = &indexing {
2541            // The lines' own notes, as the open wrote them: replaced once every line
2542            // is in, when they can say what the whole file holds.
2543            self.indexing_guessed = self
2544                .open_notes
2545                .iter()
2546                .any(|n| n.summary.starts_with(crate::lines::GUESSED));
2547            self.indexing_notes = crate::lines::notes(lines, self.indexing_guessed);
2548        }
2549        self.indexing = indexing;
2550        self.file_units = Arc::new(units);
2551        self
2552    }
2553
2554    /// Make `lf` the data as loaded, with `schema`: the root, the base and the frame
2555    /// shown, until the caller lays the filters and sort back on. The rows and count
2556    /// read through the old root are dropped, and checkpoints taken over it no longer
2557    /// apply.
2558    fn replace_root(&mut self, lf: LazyFrame, schema: Arc<Schema>) {
2559        // The records, or the table, no longer stand for the root.
2560        self.fixed_window = None;
2561        self.pushdown = None;
2562        self.root_generation = next_len_generation();
2563        self.invalidate_num_rows();
2564        self.original_schema = schema.clone();
2565        self.schema = schema;
2566        self.original_lf = lf.clone();
2567        self.base_lf = lf.clone();
2568        self.lf = lf;
2569        self.unsorted_lf = None;
2570        self.base_steps = Vec::new();
2571        self.reshape_steps = None;
2572        self.drop_buffer();
2573    }
2574
2575    /// Make `lf` the frame shown and the base the sidebar filters and sort go on top of,
2576    /// with `schema` as its schema and every column in view. Row counts are invalidated.
2577    fn install_base(&mut self, lf: LazyFrame, schema: Arc<Schema>) {
2578        self.invalidate_num_rows();
2579        // A new frame is the user's own projection of the data; its rows no longer
2580        // stand for rows of a file, so nulls in it are just nulls, no column is marked
2581        // as missing from one, and notes about the files behind it no longer describe
2582        // what is on screen.
2583        self.drift_column_present = false;
2584        self.view_numbered = false;
2585        self.drift_groups = Arc::new(Vec::new());
2586        self.notes = Vec::new();
2587        self.view_notes = Vec::new();
2588        // Rows of the new shape are measured afresh; the old width would plan the
2589        // window of a wide frame from a narrow one, or the reverse.
2590        self.observed_bytes_per_row = None;
2591        // A new frame is in no order a query named; `sql_query` names it after.
2592        self.query_order = Vec::new();
2593        // A column may keep its name and type and hold other values now.
2594        self.widths.relearn();
2595        self.base_lf = lf.clone();
2596        self.lf = lf;
2597        self.unsorted_lf = None;
2598        // Every caller says how the base was built; one that does not leaves a script
2599        // that says so rather than one that computes something else.
2600        self.base_steps = vec![Step::Unreproducible(
2601            "datui built the view from here in a way it cannot write as Python".to_string(),
2602        )];
2603        self.schema = schema;
2604        self.column_order = self.schema.iter_names().map(|s| s.to_string()).collect();
2605        // No column is a loaded one until the caller says which are.
2606        self.lineage = Some(Arc::default());
2607        self.settle_cursor();
2608        // A query that groups records its source after installing its result.
2609        self.group_source = None;
2610        self.drop_buffer();
2611    }
2612
2613    /// Forget the rows read through the frame being replaced, so the next collect reads
2614    /// the new one. Without this a view that fits in the old buffer keeps drawing it.
2615    fn drop_buffer(&mut self) {
2616        self.buffered_start_row = 0;
2617        self.buffered_end_row = 0;
2618        self.buffered_df = None;
2619    }
2620
2621    /// The view state for a new pipeline root: no query bar text, no sidebar filters or
2622    /// sort, not drilled, the first `locked_columns_count` columns frozen, the buffer
2623    /// dropped and the cursor at the top left.
2624    fn reset_view_state(&mut self, locked_columns_count: usize) {
2625        self.forget_column_changes();
2626        self.active_query.clear();
2627        self.active_sql_query.clear();
2628        self.active_fuzzy_query.clear();
2629        self.locked_columns_count = locked_columns_count;
2630        self.filters.clear();
2631        self.sort_columns.clear();
2632        self.sort_descending.clear();
2633        self.sort_ascending = true;
2634        self.start_row = 0;
2635        self.termcol_index = 0;
2636        self.clear_column_moves();
2637        self.place_cursor_at(0);
2638        self.drilled_down_group_index = None;
2639        self.drilled_down_group_key = None;
2640        self.drilled_down_group_key_columns = None;
2641        self.grouped = None;
2642        self.drop_buffer();
2643        self.table_state.select(Some(0));
2644    }
2645
2646    /// Install a query's result as the pipeline root with `query` as the one active
2647    /// query bar. Whether a pivot or melt in effect survives is the caller's call: SQL
2648    /// runs against it, the others run over the data as loaded. The caller collects.
2649    fn install_query_result(
2650        &mut self,
2651        lf: LazyFrame,
2652        schema: Arc<Schema>,
2653        query: ActiveQuery,
2654        locked_columns_count: usize,
2655        steps: Vec<Step>,
2656    ) {
2657        self.install_base(lf, schema);
2658        self.base_steps = steps;
2659        self.reset_view_state(locked_columns_count);
2660        match query {
2661            ActiveQuery::Dsl(q) => self.active_query = q,
2662            #[cfg(feature = "sql")]
2663            ActiveQuery::Sql(q) => self.active_sql_query = q,
2664            ActiveQuery::Fuzzy(q) => self.active_fuzzy_query = q,
2665        }
2666    }
2667
2668    /// The view no longer shows the pivot or melt, so nothing may run against it.
2669    fn forget_reshape(&mut self) {
2670        self.reshaped_lf = None;
2671        self.reshape_lineage = None;
2672        self.reshape_steps = None;
2673        self.last_pivot_spec = None;
2674        self.last_melt_spec = None;
2675        self.reshape_source = None;
2676    }
2677
2678    /// Reset LazyFrame and view state to original_lf. Schema is re-fetched so it matches
2679    /// after a previous query/SQL that may have changed columns. Caller should call
2680    /// collect() afterward if display update is needed (reset/query/fuzzy do; sql_query
2681    /// relies on event loop Collect).
2682    fn reset_lf_to_original(&mut self) {
2683        let schema = self
2684            .query_source()
2685            .collect_schema()
2686            .unwrap_or_else(|_| Arc::new(Schema::with_capacity(0)));
2687        self.install_base(self.original_lf.clone(), schema);
2688        self.base_steps = Vec::new();
2689        self.reshape_steps = None;
2690        self.lineage = None;
2691        self.reshape_lineage = None;
2692        // A reset is a return to the data as opened, so the rows stand for files again
2693        // and what datui noticed about them applies once more.
2694        self.drift_column_present = self.drift_at_open;
2695        self.drift_groups = self.groups_at_open.clone();
2696        self.notes = self.notes_at_open.clone();
2697        self.reshaped_lf = None;
2698        self.reshape_source = None;
2699        self.reset_view_state(0);
2700        self.restore_footer_count();
2701    }
2702
2703    /// Back to the data as loaded, with nothing applied and no error showing.
2704    fn return_to_root(&mut self) {
2705        self.reset_lf_to_original();
2706        self.error = None;
2707        self.suppress_error_display = false;
2708        self.last_pivot_spec = None;
2709        self.last_melt_spec = None;
2710    }
2711
2712    /// Back to the data as loaded with nothing applied, for a view's steps to be laid
2713    /// on again. Reads nothing.
2714    pub(crate) fn reset_view_for_replay(&mut self) {
2715        self.return_to_root();
2716    }
2717
2718    /// Back to the table as opened: the data as loaded, nothing applied, and every
2719    /// column's width learned afresh from the first page.
2720    pub fn reset(&mut self) {
2721        self.widths = ColumnWidths::default();
2722        self.return_to_root();
2723        self.collect();
2724        if self.num_rows > 0 {
2725            self.start_row = 0;
2726        }
2727    }
2728
2729    /// A file reader's state, with the open's paging and row numbers.
2730    ///
2731    /// Streaming is always on here, whatever `options.polars_streaming` says: the
2732    /// per-format readers have always built that way.
2733    fn read_with(lf: LazyFrame, options: &OpenOptions) -> Result<Self> {
2734        let mut state = Self::new(
2735            lf,
2736            options.pages_lookahead,
2737            options.pages_lookback,
2738            options.max_buffered_rows,
2739            options.max_buffered_mb,
2740            true,
2741        )?;
2742        state.row_numbers = options.row_numbers;
2743        state.row_start_index = options.row_start_index;
2744        Ok(state)
2745    }
2746
2747    pub fn from_parquet(path: &Path, options: &OpenOptions) -> Result<Self> {
2748        let is_glob = crate::source::expands_as_glob(path);
2749        let pl_path = PlRefPath::try_from_path(path)?;
2750        let args = ScanArgsParquet {
2751            glob: is_glob,
2752            ..Default::default()
2753        };
2754        let lf = LazyFrame::scan_parquet(pl_path, args)?;
2755        Self::read_with(lf, options)
2756    }
2757
2758    /// Load multiple Parquet files and concatenate them into one LazyFrame (same schema assumed).
2759    /// How the files of one dataset are stacked into one table.
2760    ///
2761    /// `diagonal`, so a file written before a column existed brings the rest of its
2762    /// rows instead of refusing the whole directory; the column reads null for it, and
2763    /// the Notes say which files have it. `to_supertypes`, because a CSV column is
2764    /// typed by inference per file — one `N/A` makes `amount` a String in one file and
2765    /// an Int64 in the next — and without widening, name agreement is not enough to
2766    /// stack them.
2767    ///
2768    /// Both are opt-ins everywhere else: DuckDB's `union_by_name`, pyarrow's
2769    /// `unify_schemas`, Spark's `mergeSchema`. They are the default here because a
2770    /// library that unions silently becomes wrong analysis downstream, while datui
2771    /// says what it did in the Notes and keeps `Enter` on the row conservative — a
2772    /// directory whose files are not one table is gone inside, not unioned, and this is
2773    /// what the `(all files)` row behind it reads with.
2774    ///
2775    /// **Only for the formats that rule can judge**, which is CSV and NDJSON here, and
2776    /// Parquet through `lenient_scan` elsewhere. Arrow, Avro, ORC and `.json` keep
2777    /// their columns nowhere cheap to reach, so nothing looks at them before the open
2778    /// and nothing could say what a union of them had done — a silent union with no
2779    /// gate in front of it and no note behind it is the pairing this whole change
2780    /// exists to remove, not something to spread further.
2781    ///
2782    /// Identical schemas stack exactly as before: diagonal over one schema is vertical,
2783    /// and nothing is widened where nothing differs. Arrow streams converted beside IPC
2784    /// files read in place stack with it too: the streams and the files are one
2785    /// directory's table, read two ways (`App::scan_arrow_parts`).
2786    pub(crate) fn union_of_files() -> polars::prelude::UnionArgs {
2787        polars::prelude::UnionArgs {
2788            diagonal: true,
2789            to_supertypes: true,
2790            ..Default::default()
2791        }
2792    }
2793
2794    pub fn from_parquet_paths(paths: &[impl AsRef<Path>], options: &OpenOptions) -> Result<Self> {
2795        if paths.is_empty() {
2796            return Err(color_eyre::eyre::eyre!("No paths provided"));
2797        }
2798        if paths.len() == 1 {
2799            return Self::from_parquet(paths[0].as_ref(), options);
2800        }
2801        let mut lazy_frames = Vec::with_capacity(paths.len());
2802        for p in paths {
2803            let pl_path = PlRefPath::try_from_path(p.as_ref())?;
2804            let args = ScanArgsParquet {
2805                glob: crate::source::expands_as_glob(p.as_ref()),
2806                ..Default::default()
2807            };
2808            let lf = LazyFrame::scan_parquet(pl_path, args)?;
2809            lazy_frames.push(lf);
2810        }
2811        let lf = polars::prelude::concat(lazy_frames.as_slice(), Default::default())?;
2812        Self::read_with(lf, options)
2813    }
2814
2815    /// Load a single Arrow IPC / Feather v2 file (lazy).
2816    pub fn from_ipc(path: &Path, options: &OpenOptions) -> Result<Self> {
2817        let pl_path = PlRefPath::try_from_path(path)?;
2818        let args = UnifiedScanArgs {
2819            glob: crate::source::expands_as_glob(path),
2820            ..Default::default()
2821        };
2822        let lf = LazyFrame::scan_ipc(pl_path, Default::default(), args)?;
2823        Self::read_with(lf, options)
2824    }
2825
2826    /// Load multiple Arrow IPC / Feather files and concatenate into one LazyFrame.
2827    pub fn from_ipc_paths(paths: &[impl AsRef<Path>], options: &OpenOptions) -> Result<Self> {
2828        if paths.is_empty() {
2829            return Err(color_eyre::eyre::eyre!("No paths provided"));
2830        }
2831        if paths.len() == 1 {
2832            return Self::from_ipc(paths[0].as_ref(), options);
2833        }
2834        let mut lazy_frames = Vec::with_capacity(paths.len());
2835        for p in paths {
2836            let pl_path = PlRefPath::try_from_path(p.as_ref())?;
2837            let args = UnifiedScanArgs {
2838                glob: crate::source::expands_as_glob(p.as_ref()),
2839                ..Default::default()
2840            };
2841            let lf = LazyFrame::scan_ipc(pl_path, Default::default(), args)?;
2842            lazy_frames.push(lf);
2843        }
2844        let lf = polars::prelude::concat(lazy_frames.as_slice(), Default::default())?;
2845        Self::read_with(lf, options)
2846    }
2847
2848    /// Load a single Avro file (eager read, then lazy).
2849    pub fn from_avro(path: &Path, options: &OpenOptions) -> Result<Self> {
2850        let file = File::open(path)?;
2851        let df = polars::io::avro::AvroReader::new(file).finish()?;
2852        let lf = df.lazy();
2853        Self::read_with(lf, options)
2854    }
2855
2856    /// Load multiple Avro files and concatenate into one LazyFrame.
2857    pub fn from_avro_paths(paths: &[impl AsRef<Path>], options: &OpenOptions) -> Result<Self> {
2858        if paths.is_empty() {
2859            return Err(color_eyre::eyre::eyre!("No paths provided"));
2860        }
2861        if paths.len() == 1 {
2862            return Self::from_avro(paths[0].as_ref(), options);
2863        }
2864        let mut lazy_frames = Vec::with_capacity(paths.len());
2865        for p in paths {
2866            let file = File::open(p.as_ref())?;
2867            let df = polars::io::avro::AvroReader::new(file).finish()?;
2868            lazy_frames.push(df.lazy());
2869        }
2870        let lf = polars::prelude::concat(lazy_frames.as_slice(), Default::default())?;
2871        Self::read_with(lf, options)
2872    }
2873
2874    /// Load a single Excel file (xls, xlsx, xlsm, xlsb) using calamine (eager read, then lazy).
2875    /// Sheet is selected by name, or by 0-based index when no sheet is so named, via
2876    /// `options.table` (`--table`).
2877    pub fn from_excel(path: &Path, options: &OpenOptions) -> Result<Self> {
2878        Self::from_excel_with_detail(path, options).map(|(state, _)| state)
2879    }
2880
2881    /// [`Self::from_excel`], with the workbook's Excel tab for the Info panel.
2882    pub(crate) fn from_excel_with_detail(
2883        path: &Path,
2884        options: &OpenOptions,
2885    ) -> Result<(Self, crate::text_formats::Detail)> {
2886        let mut workbook =
2887            open_workbook_auto(path).map_err(|e| color_eyre::eyre::eyre!("Excel: {}", e))?;
2888        let sheet_names = workbook.sheet_names().to_vec();
2889        if sheet_names.is_empty() {
2890            return Err(color_eyre::eyre::eyre!("Excel file has no worksheets"));
2891        }
2892        // Named so a bad --table says what to ask for instead: "0 'Sales', 1 'Summary'".
2893        let sheets_on_offer = || {
2894            sheet_names
2895                .iter()
2896                .enumerate()
2897                .map(|(i, name)| format!("{} '{}'", i, name))
2898                .collect::<Vec<_>>()
2899                .join(", ")
2900        };
2901        // A sheet's name before an index: the home screen names a sheet called `2023`.
2902        let opened = match options.table.as_deref() {
2903            None => sheet_names[0].clone(),
2904            Some(name) if sheet_names.iter().any(|n| n == name) => name.to_string(),
2905            Some(sheet_sel) => match sheet_sel.parse::<usize>() {
2906                Ok(idx) => sheet_names.get(idx).cloned().ok_or_else(|| {
2907                    color_eyre::eyre::eyre!(
2908                        "Excel: no worksheet at index {}; this file has: {}",
2909                        idx,
2910                        sheets_on_offer()
2911                    )
2912                })?,
2913                Err(_) => {
2914                    return Err(color_eyre::eyre::eyre!(
2915                        "Excel: no worksheet named '{}'; this file has: {}",
2916                        sheet_sel,
2917                        sheets_on_offer()
2918                    ));
2919                }
2920            },
2921        };
2922        let range = workbook
2923            .worksheet_range(&opened)
2924            .map_err(|e| color_eyre::eyre::eyre!("Excel: {}", e))?;
2925        let detail = crate::excel::detail(&mut workbook, &opened, &range);
2926        drop(workbook);
2927        let rows: Vec<Vec<Data>> = range.rows().map(|r| r.to_vec()).collect();
2928        if rows.is_empty() {
2929            let empty_df = DataFrame::empty();
2930            return Ok((Self::read_with(empty_df.lazy(), options)?, detail));
2931        }
2932        let headers: Vec<String> = rows[0]
2933            .iter()
2934            .map(|c| calamine::DataType::as_string(c).unwrap_or_else(|| c.to_string()))
2935            .collect();
2936        let n_cols = headers.len();
2937        let mut series_vec = Vec::with_capacity(n_cols);
2938        for (col_idx, header) in headers.iter().enumerate() {
2939            let col_cells: Vec<Option<&Data>> =
2940                rows[1..].iter().map(|row| row.get(col_idx)).collect();
2941            let inferred = Self::excel_infer_column_type(&col_cells);
2942            let name = if header.is_empty() {
2943                format!("column_{}", col_idx + 1)
2944            } else {
2945                header.clone()
2946            };
2947            let series = Self::excel_column_to_series(name.as_str(), &col_cells, inferred)?;
2948            series_vec.push(series.into());
2949        }
2950        let df = DataFrame::new_infer_height(series_vec)?;
2951        Ok((Self::read_with(df.lazy(), options)?, detail))
2952    }
2953
2954    /// Infers column type: prefers Int64 for whole-number floats; infers Date/Datetime for
2955    /// calamine DateTime/DateTimeIso or for string columns that parse as ISO date/datetime.
2956    fn excel_infer_column_type(cells: &[Option<&Data>]) -> ExcelColType {
2957        use calamine::DataType as CalamineTrait;
2958        let mut has_string = false;
2959        let mut has_float = false;
2960        let mut has_int = false;
2961        let mut has_bool = false;
2962        let mut has_datetime = false;
2963        for cell in cells.iter().flatten() {
2964            if CalamineTrait::is_string(*cell) {
2965                has_string = true;
2966                break;
2967            }
2968            if CalamineTrait::is_float(*cell)
2969                || CalamineTrait::is_datetime(*cell)
2970                || CalamineTrait::is_datetime_iso(*cell)
2971            {
2972                has_float = true;
2973            }
2974            if CalamineTrait::is_int(*cell) {
2975                has_int = true;
2976            }
2977            if CalamineTrait::is_bool(*cell) {
2978                has_bool = true;
2979            }
2980            if CalamineTrait::is_datetime(*cell) || CalamineTrait::is_datetime_iso(*cell) {
2981                has_datetime = true;
2982            }
2983        }
2984        if has_string {
2985            let any_parsed = cells
2986                .iter()
2987                .flatten()
2988                .any(|c| Self::excel_cell_to_naive_datetime(c).is_some());
2989            let all_non_empty_parse = cells.iter().flatten().all(|c| {
2990                CalamineTrait::is_empty(*c) || Self::excel_cell_to_naive_datetime(c).is_some()
2991            });
2992            if any_parsed && all_non_empty_parse {
2993                if Self::excel_parsed_cells_all_midnight(cells) {
2994                    ExcelColType::Date
2995                } else {
2996                    ExcelColType::Datetime
2997                }
2998            } else {
2999                ExcelColType::Utf8
3000            }
3001        } else if has_int {
3002            ExcelColType::Int64
3003        } else if has_datetime {
3004            if Self::excel_parsed_cells_all_midnight(cells) {
3005                ExcelColType::Date
3006            } else {
3007                ExcelColType::Datetime
3008            }
3009        } else if has_float {
3010            let all_whole = cells.iter().flatten().all(|cell| {
3011                cell.as_f64()
3012                    .is_none_or(|f| f.is_finite() && (f - f.trunc()).abs() < 1e-10)
3013            });
3014            if all_whole {
3015                ExcelColType::Int64
3016            } else {
3017                ExcelColType::Float64
3018            }
3019        } else if has_bool {
3020            ExcelColType::Boolean
3021        } else {
3022            ExcelColType::Utf8
3023        }
3024    }
3025
3026    /// True if every cell that parses as datetime has time 00:00:00.
3027    fn excel_parsed_cells_all_midnight(cells: &[Option<&Data>]) -> bool {
3028        let midnight = NaiveTime::from_hms_opt(0, 0, 0).expect("valid time");
3029        cells
3030            .iter()
3031            .flatten()
3032            .filter_map(|c| Self::excel_cell_to_naive_datetime(c))
3033            .all(|dt| dt.time() == midnight)
3034    }
3035
3036    /// Converts a calamine cell to NaiveDateTime (Excel serial, DateTimeIso, or parseable string).
3037    fn excel_cell_to_naive_datetime(cell: &Data) -> Option<NaiveDateTime> {
3038        use calamine::DataType;
3039        if let Some(dt) = cell.as_datetime() {
3040            return Some(dt);
3041        }
3042        let s = cell.get_datetime_iso().or_else(|| cell.get_string())?;
3043        Self::parse_naive_datetime_str(s)
3044    }
3045
3046    /// Parses an ISO-style date/datetime string; tries FORMATS in order.
3047    fn parse_naive_datetime_str(s: &str) -> Option<NaiveDateTime> {
3048        let s = s.trim();
3049        if s.is_empty() {
3050            return None;
3051        }
3052        const FORMATS: &[&str] = &[
3053            "%Y-%m-%dT%H:%M:%S%.f",
3054            "%Y-%m-%dT%H:%M:%S",
3055            "%Y-%m-%d %H:%M:%S%.f",
3056            "%Y-%m-%d %H:%M:%S",
3057            "%Y-%m-%d",
3058        ];
3059        for fmt in FORMATS {
3060            if let Ok(dt) = NaiveDateTime::parse_from_str(s, fmt) {
3061                return Some(dt);
3062            }
3063        }
3064        if let Ok(d) = NaiveDate::parse_from_str(s, "%Y-%m-%d") {
3065            return Some(d.and_hms_opt(0, 0, 0).expect("midnight"));
3066        }
3067        None
3068    }
3069
3070    /// Build a Polars Series from a column of calamine cells using the inferred type.
3071    fn excel_column_to_series(
3072        name: &str,
3073        cells: &[Option<&Data>],
3074        col_type: ExcelColType,
3075    ) -> Result<Series> {
3076        use calamine::DataType as CalamineTrait;
3077        use polars::datatypes::TimeUnit;
3078        let series = match col_type {
3079            ExcelColType::Int64 => {
3080                let v: Vec<Option<i64>> = cells
3081                    .iter()
3082                    .map(|c| c.and_then(|cell| cell.as_i64()))
3083                    .collect();
3084                Series::new(name.into(), v)
3085            }
3086            ExcelColType::Float64 => {
3087                let v: Vec<Option<f64>> = cells
3088                    .iter()
3089                    .map(|c| c.and_then(|cell| cell.as_f64()))
3090                    .collect();
3091                Series::new(name.into(), v)
3092            }
3093            ExcelColType::Boolean => {
3094                let v: Vec<Option<bool>> = cells
3095                    .iter()
3096                    .map(|c| c.and_then(|cell| cell.get_bool()))
3097                    .collect();
3098                Series::new(name.into(), v)
3099            }
3100            ExcelColType::Utf8 => {
3101                let v: Vec<Option<String>> = cells
3102                    .iter()
3103                    .map(|c| c.and_then(|cell| cell.as_string()))
3104                    .collect();
3105                Series::new(name.into(), v)
3106            }
3107            ExcelColType::Date => {
3108                let epoch = NaiveDate::from_ymd_opt(1970, 1, 1).expect("valid date");
3109                let v: Vec<Option<i32>> = cells
3110                    .iter()
3111                    .map(|c| {
3112                        c.and_then(Self::excel_cell_to_naive_datetime)
3113                            .map(|dt| (dt.date() - epoch).num_days() as i32)
3114                    })
3115                    .collect();
3116                Series::new(name.into(), v).cast(&DataType::Date)?
3117            }
3118            ExcelColType::Datetime => {
3119                let v: Vec<Option<i64>> = cells
3120                    .iter()
3121                    .map(|c| {
3122                        c.and_then(Self::excel_cell_to_naive_datetime)
3123                            .map(|dt| dt.and_utc().timestamp_micros())
3124                    })
3125                    .collect();
3126                Series::new(name.into(), v)
3127                    .cast(&DataType::Datetime(TimeUnit::Microseconds, None))?
3128            }
3129        };
3130        Ok(series)
3131    }
3132
3133    /// Load a single ORC file (eager read via orc-rust → Arrow, then convert to Polars, then lazy).
3134    /// ORC is read fully into memory; see `docs/formats/columnar-and-json.md`.
3135    pub fn from_orc(path: &Path, options: &OpenOptions) -> Result<Self> {
3136        let file = File::open(path)?;
3137        let reader = ArrowReaderBuilder::try_new(file)
3138            .map_err(|e| color_eyre::eyre::eyre!("ORC: {}", e))?
3139            .build();
3140        let batches: Vec<RecordBatch> = reader
3141            .collect::<std::result::Result<Vec<_>, _>>()
3142            .map_err(|e| color_eyre::eyre::eyre!("ORC: {}", e))?;
3143        let df = Self::arrow_record_batches_to_dataframe(&batches)?;
3144        let lf = df.lazy();
3145        Self::read_with(lf, options)
3146    }
3147
3148    /// Load multiple ORC files and concatenate into one LazyFrame.
3149    pub fn from_orc_paths(paths: &[impl AsRef<Path>], options: &OpenOptions) -> Result<Self> {
3150        if paths.is_empty() {
3151            return Err(color_eyre::eyre::eyre!("No paths provided"));
3152        }
3153        if paths.len() == 1 {
3154            return Self::from_orc(paths[0].as_ref(), options);
3155        }
3156        let mut lazy_frames = Vec::with_capacity(paths.len());
3157        for p in paths {
3158            let file = File::open(p.as_ref())?;
3159            let reader = ArrowReaderBuilder::try_new(file)
3160                .map_err(|e| color_eyre::eyre::eyre!("ORC: {}", e))?
3161                .build();
3162            let batches: Vec<RecordBatch> = reader
3163                .collect::<std::result::Result<Vec<_>, _>>()
3164                .map_err(|e| color_eyre::eyre::eyre!("ORC: {}", e))?;
3165            let df = Self::arrow_record_batches_to_dataframe(&batches)?;
3166            lazy_frames.push(df.lazy());
3167        }
3168        let lf = polars::prelude::concat(lazy_frames.as_slice(), Default::default())?;
3169        Self::read_with(lf, options)
3170    }
3171
3172    /// Convert Arrow (arrow crate 57) RecordBatches to Polars DataFrame by value (ORC uses
3173    /// arrow 57; Polars uses polars-arrow, so we cannot use Series::from_arrow).
3174    fn arrow_record_batches_to_dataframe(batches: &[RecordBatch]) -> Result<DataFrame> {
3175        if batches.is_empty() {
3176            return Ok(DataFrame::empty());
3177        }
3178        let mut all_dfs = Vec::with_capacity(batches.len());
3179        for batch in batches {
3180            let n_cols = batch.num_columns();
3181            let schema = batch.schema();
3182            let mut series_vec = Vec::with_capacity(n_cols);
3183            for (i, col) in batch.columns().iter().enumerate() {
3184                let name = schema.field(i).name().as_str();
3185                let s = Self::arrow_array_to_polars_series(name, col)?;
3186                series_vec.push(s.into());
3187            }
3188            let df = DataFrame::new_infer_height(series_vec)?;
3189            all_dfs.push(df);
3190        }
3191        let mut out = all_dfs.remove(0);
3192        for df in all_dfs {
3193            out = out.vstack(&df)?;
3194        }
3195        Ok(out)
3196    }
3197
3198    fn arrow_array_to_polars_series(name: &str, array: &dyn Array) -> Result<Series> {
3199        use arrow::datatypes::DataType as ArrowDataType;
3200        let len = array.len();
3201        match array.data_type() {
3202            ArrowDataType::Int8 => {
3203                let a = array
3204                    .as_primitive_opt::<Int8Type>()
3205                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Int8 array"))?;
3206                let v: Vec<Option<i8>> = (0..len)
3207                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3208                    .collect();
3209                Ok(Series::new(name.into(), v))
3210            }
3211            ArrowDataType::Int16 => {
3212                let a = array
3213                    .as_primitive_opt::<Int16Type>()
3214                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Int16 array"))?;
3215                let v: Vec<Option<i16>> = (0..len)
3216                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3217                    .collect();
3218                Ok(Series::new(name.into(), v))
3219            }
3220            ArrowDataType::Int32 => {
3221                let a = array
3222                    .as_primitive_opt::<Int32Type>()
3223                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Int32 array"))?;
3224                let v: Vec<Option<i32>> = (0..len)
3225                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3226                    .collect();
3227                Ok(Series::new(name.into(), v))
3228            }
3229            ArrowDataType::Int64 => {
3230                let a = array
3231                    .as_primitive_opt::<Int64Type>()
3232                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Int64 array"))?;
3233                let v: Vec<Option<i64>> = (0..len)
3234                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3235                    .collect();
3236                Ok(Series::new(name.into(), v))
3237            }
3238            ArrowDataType::UInt8 => {
3239                let a = array
3240                    .as_primitive_opt::<UInt8Type>()
3241                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected UInt8 array"))?;
3242                let v: Vec<Option<i64>> = (0..len)
3243                    .map(|i| {
3244                        if a.is_null(i) {
3245                            None
3246                        } else {
3247                            Some(a.value(i) as i64)
3248                        }
3249                    })
3250                    .collect();
3251                Ok(Series::new(name.into(), v).cast(&DataType::UInt8)?)
3252            }
3253            ArrowDataType::UInt16 => {
3254                let a = array
3255                    .as_primitive_opt::<UInt16Type>()
3256                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected UInt16 array"))?;
3257                let v: Vec<Option<i64>> = (0..len)
3258                    .map(|i| {
3259                        if a.is_null(i) {
3260                            None
3261                        } else {
3262                            Some(a.value(i) as i64)
3263                        }
3264                    })
3265                    .collect();
3266                Ok(Series::new(name.into(), v).cast(&DataType::UInt16)?)
3267            }
3268            ArrowDataType::UInt32 => {
3269                let a = array
3270                    .as_primitive_opt::<UInt32Type>()
3271                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected UInt32 array"))?;
3272                let v: Vec<Option<u32>> = (0..len)
3273                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3274                    .collect();
3275                Ok(Series::new(name.into(), v))
3276            }
3277            ArrowDataType::UInt64 => {
3278                let a = array
3279                    .as_primitive_opt::<UInt64Type>()
3280                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected UInt64 array"))?;
3281                let v: Vec<Option<u64>> = (0..len)
3282                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3283                    .collect();
3284                Ok(Series::new(name.into(), v))
3285            }
3286            ArrowDataType::Float32 => {
3287                let a = array
3288                    .as_primitive_opt::<Float32Type>()
3289                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Float32 array"))?;
3290                let v: Vec<Option<f32>> = (0..len)
3291                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3292                    .collect();
3293                Ok(Series::new(name.into(), v))
3294            }
3295            ArrowDataType::Float64 => {
3296                let a = array
3297                    .as_primitive_opt::<Float64Type>()
3298                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Float64 array"))?;
3299                let v: Vec<Option<f64>> = (0..len)
3300                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3301                    .collect();
3302                Ok(Series::new(name.into(), v))
3303            }
3304            ArrowDataType::Boolean => {
3305                let a = array
3306                    .as_boolean_opt()
3307                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Boolean array"))?;
3308                let v: Vec<Option<bool>> = (0..len)
3309                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3310                    .collect();
3311                Ok(Series::new(name.into(), v))
3312            }
3313            ArrowDataType::Utf8 => {
3314                let a = array
3315                    .as_string_opt::<i32>()
3316                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Utf8 array"))?;
3317                let v: Vec<Option<String>> = (0..len)
3318                    .map(|i| {
3319                        if a.is_null(i) {
3320                            None
3321                        } else {
3322                            Some(a.value(i).to_string())
3323                        }
3324                    })
3325                    .collect();
3326                Ok(Series::new(name.into(), v))
3327            }
3328            ArrowDataType::LargeUtf8 => {
3329                let a = array
3330                    .as_string_opt::<i64>()
3331                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected LargeUtf8 array"))?;
3332                let v: Vec<Option<String>> = (0..len)
3333                    .map(|i| {
3334                        if a.is_null(i) {
3335                            None
3336                        } else {
3337                            Some(a.value(i).to_string())
3338                        }
3339                    })
3340                    .collect();
3341                Ok(Series::new(name.into(), v))
3342            }
3343            ArrowDataType::Date32 => {
3344                let a = array
3345                    .as_primitive_opt::<Date32Type>()
3346                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Date32 array"))?;
3347                let v: Vec<Option<i32>> = (0..len)
3348                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3349                    .collect();
3350                Ok(Series::new(name.into(), v))
3351            }
3352            ArrowDataType::Date64 => {
3353                let a = array
3354                    .as_primitive_opt::<Date64Type>()
3355                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Date64 array"))?;
3356                let v: Vec<Option<i64>> = (0..len)
3357                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3358                    .collect();
3359                Ok(Series::new(name.into(), v))
3360            }
3361            ArrowDataType::Timestamp(_, _) => {
3362                let a = array
3363                    .as_primitive_opt::<TimestampMillisecondType>()
3364                    .ok_or_else(|| color_eyre::eyre::eyre!("ORC: expected Timestamp array"))?;
3365                let v: Vec<Option<i64>> = (0..len)
3366                    .map(|i| if a.is_null(i) { None } else { Some(a.value(i)) })
3367                    .collect();
3368                Ok(Series::new(name.into(), v))
3369            }
3370            other => Err(color_eyre::eyre::eyre!(
3371                "ORC: unsupported column type {:?} for column '{}'",
3372                other,
3373                name
3374            )),
3375        }
3376    }
3377
3378    /// Build a LazyFrame for hive-partitioned Parquet only (no schema collection, no partition discovery).
3379    /// Use this for phased loading so "Scanning input" is instant; schema and partition handling are the schema phase's.
3380    pub fn scan_parquet_hive(path: &Path) -> Result<LazyFrame> {
3381        let is_glob = crate::source::expands_as_glob(path);
3382        let pl_path = PlRefPath::try_from_path(path)?;
3383        let args = ScanArgsParquet {
3384            hive_options: HiveOptions::new_enabled(),
3385            glob: is_glob,
3386            ..Default::default()
3387        };
3388        LazyFrame::scan_parquet(pl_path, args).map_err(Into::into)
3389    }
3390
3391    /// Build a LazyFrame for hive-partitioned Parquet with a pre-computed schema (avoids slow collect_schema across all files).
3392    pub fn scan_parquet_hive_with_schema(path: &Path, schema: Arc<Schema>) -> Result<LazyFrame> {
3393        let is_glob = crate::source::expands_as_glob(path);
3394        let pl_path = PlRefPath::try_from_path(path)?;
3395        let args = ScanArgsParquet {
3396            schema: Some(schema),
3397            hive_options: HiveOptions::new_enabled(),
3398            glob: is_glob,
3399            ..Default::default()
3400        };
3401        LazyFrame::scan_parquet(pl_path, args).map_err(Into::into)
3402    }
3403
3404    /// Find the first parquet file along a single spine of a hive-partitioned directory (same walk as partition discovery).
3405    /// Returns `None` if the directory is empty or has no parquet files along that spine.
3406    fn first_parquet_file_in_hive_dir(path: &Path) -> Option<std::path::PathBuf> {
3407        const MAX_DEPTH: usize = 64;
3408        Self::first_parquet_file_spine(path, 0, MAX_DEPTH)
3409    }
3410
3411    fn first_parquet_file_spine(
3412        path: &Path,
3413        depth: usize,
3414        max_depth: usize,
3415    ) -> Option<std::path::PathBuf> {
3416        if depth >= max_depth {
3417            return None;
3418        }
3419        let entries = fs::read_dir(path).ok()?;
3420        let mut first_partition_child: Option<std::path::PathBuf> = None;
3421        for entry in entries.flatten() {
3422            let child = entry.path();
3423            if child.is_file() {
3424                if crate::discover::is_parquet_path(&child) {
3425                    return Some(child);
3426                }
3427            } else if child.is_dir()
3428                && let Some(name) = child.file_name().and_then(|n| n.to_str())
3429                && name.contains('=')
3430                && first_partition_child.is_none()
3431            {
3432                first_partition_child = Some(child);
3433            }
3434        }
3435        first_partition_child.and_then(|p| Self::first_parquet_file_spine(&p, depth + 1, max_depth))
3436    }
3437
3438    /// Read schema from a single parquet file (metadata only, no data scan). Used to avoid collect_schema() over many files.
3439    fn read_schema_from_single_parquet(path: &Path) -> Result<Arc<Schema>> {
3440        let file = File::open(path)?;
3441        let mut reader = ParquetReader::new(file);
3442        let arrow_schema = reader.schema()?;
3443        let schema = Schema::from_arrow_schema(arrow_schema.as_ref());
3444        Ok(Arc::new(schema))
3445    }
3446
3447    /// Infer schema from one parquet file in a hive directory and merge with partition columns.
3448    /// Returns (merged_schema, partition_columns). Use with scan_parquet_hive_with_schema to avoid slow collect_schema().
3449    /// Only supported when path is a directory (not a glob). Returns Err if no parquet file found or read fails.
3450    pub fn schema_from_one_hive_parquet(path: &Path) -> Result<(Arc<Schema>, Vec<String>)> {
3451        let partition_columns = Self::discover_hive_partition_columns(path);
3452        let one_file = Self::first_parquet_file_in_hive_dir(path)
3453            .ok_or_else(|| color_eyre::eyre::eyre!("No parquet file found in hive directory"))?;
3454        let file_schema = Self::read_schema_from_single_parquet(&one_file)?;
3455        let values = Self::hive_partition_values(path, &one_file);
3456        let part_set: HashSet<&str> = partition_columns.iter().map(String::as_str).collect();
3457        let mut merged = Schema::with_capacity(partition_columns.len() + file_schema.len());
3458        for name in &partition_columns {
3459            merged.with_column(
3460                name.clone().into(),
3461                partition_dtype(name, &file_schema, &values),
3462            );
3463        }
3464        for (name, dtype) in file_schema.iter() {
3465            if !part_set.contains(name.as_str()) {
3466                merged.with_column(name.clone(), dtype.clone());
3467            }
3468        }
3469        Ok((Arc::new(merged), partition_columns))
3470    }
3471
3472    /// `key=value` names of the partition directories beside each one on the path to `file`.
3473    pub(crate) fn hive_partition_values(root: &Path, file: &Path) -> Vec<(String, String)> {
3474        let mut out = Vec::new();
3475        let Some(rel) = file.strip_prefix(root).ok().and_then(Path::parent) else {
3476            return out;
3477        };
3478        let mut dir = root.to_path_buf();
3479        for component in rel.components() {
3480            let Some(segment) = component.as_os_str().to_str() else {
3481                break;
3482            };
3483            if let Some((key, _)) = segment.split_once('=') {
3484                for entry in fs::read_dir(&dir).into_iter().flatten().flatten() {
3485                    let name = entry.file_name();
3486                    let Some((k, v)) = name.to_str().and_then(|n| n.split_once('=')) else {
3487                        continue;
3488                    };
3489                    if k == key && entry.path().is_dir() {
3490                        out.push((k.to_string(), v.to_string()));
3491                    }
3492                }
3493            }
3494            dir.push(segment);
3495        }
3496        out
3497    }
3498
3499    /// Discover hive partition column names (public for phased loading). Directory: single-spine walk; glob: parse pattern.
3500    pub fn discover_hive_partition_columns(path: &Path) -> Vec<String> {
3501        if path.is_dir() {
3502            Self::discover_partition_columns_from_path(path)
3503        } else {
3504            Self::discover_partition_columns_from_glob_pattern(path)
3505        }
3506    }
3507
3508    /// Discover hive partition column names from a directory path by walking a single
3509    /// "spine" (one branch) of key=value directories. Partition keys are uniform across
3510    /// the tree, so we only need one path to infer [year, month, day] etc. Returns columns
3511    /// in path order. Stops after max_depth levels to avoid runaway on malformed trees.
3512    fn discover_partition_columns_from_path(path: &Path) -> Vec<String> {
3513        const MAX_PARTITION_DEPTH: usize = 64;
3514        let mut columns = Vec::<String>::new();
3515        let mut seen = HashSet::<String>::new();
3516        Self::discover_partition_columns_spine(
3517            path,
3518            &mut columns,
3519            &mut seen,
3520            0,
3521            MAX_PARTITION_DEPTH,
3522        );
3523        columns
3524    }
3525
3526    /// Walk one branch: at this directory, find the first child that is a key=value dir,
3527    /// record the key (if not already seen), then recurse into that one child only.
3528    /// This does O(depth) read_dir calls instead of walking the entire tree.
3529    fn discover_partition_columns_spine(
3530        path: &Path,
3531        columns: &mut Vec<String>,
3532        seen: &mut HashSet<String>,
3533        depth: usize,
3534        max_depth: usize,
3535    ) {
3536        if depth >= max_depth {
3537            return;
3538        }
3539        let Ok(entries) = fs::read_dir(path) else {
3540            return;
3541        };
3542        let mut first_partition_child: Option<std::path::PathBuf> = None;
3543        for entry in entries.flatten() {
3544            let child = entry.path();
3545            if child.is_dir()
3546                && let Some(name) = child.file_name().and_then(|n| n.to_str())
3547                && let Some((key, _)) = name.split_once('=')
3548            {
3549                if !key.is_empty() && seen.insert(key.to_string()) {
3550                    columns.push(key.to_string());
3551                }
3552                if first_partition_child.is_none() {
3553                    first_partition_child = Some(child);
3554                }
3555                break;
3556            }
3557        }
3558        if let Some(one) = first_partition_child {
3559            Self::discover_partition_columns_spine(&one, columns, seen, depth + 1, max_depth);
3560        }
3561    }
3562
3563    /// Infer partition column names from a glob pattern path (e.g. "data/year=*/month=*/*.parquet").
3564    fn discover_partition_columns_from_glob_pattern(path: &Path) -> Vec<String> {
3565        let path_str = path.as_os_str().to_string_lossy();
3566        let mut columns = Vec::<String>::new();
3567        let mut seen = HashSet::<String>::new();
3568        for segment in path_str.split('/') {
3569            if let Some((key, rest)) = segment.split_once('=')
3570                && !key.is_empty()
3571                && (rest == "*" || !rest.contains('*'))
3572                && seen.insert(key.to_string())
3573            {
3574                columns.push(key.to_string());
3575            }
3576        }
3577        columns
3578    }
3579
3580    /// Load Parquet with Hive partitioning from a directory or glob path.
3581    /// When path is a directory, partition columns are discovered from path structure.
3582    /// When path contains glob (e.g. `**/*.parquet`), partition columns are inferred from the pattern (e.g. `year=*/month=*`).
3583    /// Partition columns are moved to the left in the initial LazyFrame before state is created.
3584    ///
3585    /// **Performance**: The slow part is Polars, not our code. `scan_parquet` + `collect_schema()` trigger
3586    /// path expansion (full directory tree or glob) and parquet metadata reads; we only do a single-spine
3587    /// walk for partition key discovery and cheap schema/select work.
3588    pub fn from_parquet_hive(
3589        path: &Path,
3590        pages_lookahead: Option<usize>,
3591        pages_lookback: Option<usize>,
3592        max_buffered_rows: Option<usize>,
3593        max_buffered_mb: Option<usize>,
3594        row_numbers: bool,
3595        row_start_index: usize,
3596    ) -> Result<Self> {
3597        let is_glob = crate::source::expands_as_glob(path);
3598        let pl_path = PlRefPath::try_from_path(path)?;
3599        let args = ScanArgsParquet {
3600            hive_options: HiveOptions::new_enabled(),
3601            glob: is_glob,
3602            ..Default::default()
3603        };
3604        let mut lf = LazyFrame::scan_parquet(pl_path, args)?;
3605        let schema = lf.collect_schema()?;
3606
3607        let mut discovered = if path.is_dir() {
3608            Self::discover_partition_columns_from_path(path)
3609        } else {
3610            Self::discover_partition_columns_from_glob_pattern(path)
3611        };
3612
3613        // Fallback: glob like "**/*.parquet" has no key= in the pattern, so discovery is empty.
3614        // Try discovering from a directory prefix (e.g. path.parent() or walk up until we find a dir).
3615        if discovered.is_empty() {
3616            let mut dir = path;
3617            while !dir.is_dir() {
3618                match dir.parent() {
3619                    Some(p) => dir = p,
3620                    None => break,
3621                }
3622            }
3623            if dir.is_dir() {
3624                discovered = Self::discover_partition_columns_from_path(dir);
3625            }
3626        }
3627
3628        let partition_columns: Vec<String> = discovered
3629            .into_iter()
3630            .filter(|c| schema.contains(c.as_str()))
3631            .collect();
3632
3633        let new_order: Vec<String> = if partition_columns.is_empty() {
3634            schema.iter_names().map(|s| s.to_string()).collect()
3635        } else {
3636            let part_set: HashSet<&str> = partition_columns.iter().map(String::as_str).collect();
3637            let all_names: Vec<String> = schema.iter_names().map(|s| s.to_string()).collect();
3638            let rest: Vec<String> = all_names
3639                .into_iter()
3640                .filter(|c| !part_set.contains(c.as_str()))
3641                .collect();
3642            partition_columns.iter().cloned().chain(rest).collect()
3643        };
3644
3645        if !partition_columns.is_empty() {
3646            let exprs: Vec<Expr> = new_order.iter().map(|s| col(s.as_str())).collect();
3647            lf = lf.select(exprs);
3648        }
3649
3650        let mut state = Self::new(
3651            lf,
3652            pages_lookahead,
3653            pages_lookback,
3654            max_buffered_rows,
3655            max_buffered_mb,
3656            true,
3657        )?;
3658        state.row_numbers = row_numbers;
3659        state.row_start_index = row_start_index;
3660        state.partition_columns = if partition_columns.is_empty() {
3661            None
3662        } else {
3663            Some(partition_columns)
3664        };
3665        // Ensure display order is partition-first (Self::new uses schema order; be explicit).
3666        state.set_column_order(new_order);
3667        Ok(state)
3668    }
3669
3670    pub fn set_row_numbers(&mut self, enabled: bool) {
3671        self.row_numbers = enabled;
3672    }
3673
3674    /// `#` on or off. Returns whether the view's frame changed and its rows need
3675    /// reading again: a sorted or filtered view of data with no place of its own
3676    /// numbers its rows once `#` is on.
3677    pub fn toggle_row_numbers(&mut self) -> bool {
3678        self.row_numbers = !self.row_numbers;
3679        if self.row_numbers && self.wants_view_numbers() && !self.view_numbered {
3680            self.drop_buffer();
3681            self.apply_transformations();
3682            return true;
3683        }
3684        false
3685    }
3686
3687    /// Whether the view would number its rows itself with `#` on: it is sorted or
3688    /// filtered over the scan, and the scan's rows do not carry their place.
3689    fn wants_view_numbers(&self) -> bool {
3690        // A followed file's view is read from a mark, where a row index would count
3691        // from the mark rather than the file's start.
3692        // Nor one in a store or of many files, where a row index between the scan and
3693        // the filter would read every file; nor past what a row index counts to.
3694        let too_many = self
3695            .pristine_rows
3696            .or(self.num_rows_if_valid())
3697            .is_some_and(|rows| rows > crate::row_index::MAX_ROWS);
3698        self.scan_is_the_root()
3699            && self.follow.is_none()
3700            && !self.remote_source
3701            && self.remote_files.is_none()
3702            && self.parquet_count_dir.is_none()
3703            && !too_many
3704            && !self.drift_column_present
3705            && !self.source_rows_at_open
3706            && self.pushed_view().is_none()
3707            && (!self.filters.is_empty() || !self.sort_columns.is_empty() || !self.sort_ascending)
3708    }
3709
3710    /// Whether the row-number column is shown.
3711    pub fn row_numbers(&self) -> bool {
3712        self.row_numbers
3713    }
3714
3715    /// Row number display start (0 or 1); used by go-to-line to interpret user input.
3716    pub fn row_start_index(&self) -> usize {
3717        self.row_start_index
3718    }
3719
3720    /// Decompress `path` into a new file in `temp_dir`, claimed through `writer` (see
3721    /// [`crate::unfinished`]) and given up, removed, once its open is stopped.
3722    fn decompress_compressed_csv_to_temp(
3723        path: &Path,
3724        compression: CompressionFormat,
3725        temp_dir: &Path,
3726        writer: &Writer,
3727    ) -> Result<Decompressed> {
3728        let stopped = || color_eyre::eyre::eyre!("Decompressing was stopped.");
3729        let Some((file, claim)) = writer.create(|| NamedTempFile::new_in(temp_dir))? else {
3730            return Err(stopped());
3731        };
3732        // Held from here, so a failure drops the file before the claim.
3733        let mut temp = Decompressed {
3734            file,
3735            _claim: claim,
3736        };
3737        let out = temp.file.as_file_mut();
3738        let mut reader: Box<dyn Read> = match compression {
3739            CompressionFormat::Gzip => {
3740                let f = File::open(path)?;
3741                Box::new(flate2::read::GzDecoder::new(BufReader::new(f)))
3742            }
3743            CompressionFormat::Zstd => {
3744                let f = File::open(path)?;
3745                Box::new(zstd::Decoder::new(BufReader::new(f))?)
3746            }
3747            CompressionFormat::Bzip2 => {
3748                let f = File::open(path)?;
3749                Box::new(bzip2::read::BzDecoder::new(BufReader::new(f)))
3750            }
3751            CompressionFormat::Xz => {
3752                let f = File::open(path)?;
3753                Box::new(xz2::read::XzDecoder::new(BufReader::new(f)))
3754            }
3755        };
3756        // A chunk at a time, so a stopped open stops writing rather than finishing a
3757        // file nobody will read.
3758        let mut chunk = vec![0u8; 1 << 20];
3759        loop {
3760            if writer.stopped() {
3761                return Err(stopped());
3762            }
3763            let read = match reader.read(&mut chunk) {
3764                Ok(0) => break,
3765                Ok(read) => read,
3766                Err(e) if e.kind() == std::io::ErrorKind::Interrupted => continue,
3767                Err(e) => return Err(e.into()),
3768            };
3769            std::io::Write::write_all(out, &chunk[..read])?;
3770        }
3771        out.sync_all()?;
3772        Ok(temp)
3773    }
3774
3775    /// Parse null value specs: "VAL" -> global, "COL=VAL" -> per-column (first '=' separates).
3776    fn parse_null_value_specs(specs: &[String]) -> (Vec<String>, Vec<(String, String)>) {
3777        let mut global = Vec::new();
3778        let mut per_column = Vec::new();
3779        for s in specs {
3780            if let Some(i) = s.find('=') {
3781                let (col, val) = (s[..i].to_string(), s[i + 1..].to_string());
3782                per_column.push((col, val));
3783            } else {
3784                global.push(s.clone());
3785            }
3786        }
3787        (global, per_column)
3788    }
3789
3790    /// Build Polars NullValues from parsed specs. When both global and per_column are set, schema is required (caller does schema scan).
3791    fn build_polars_null_values(
3792        global: &[String],
3793        per_column: &[(String, String)],
3794        schema: Option<&Schema>,
3795    ) -> Option<NullValues> {
3796        if global.is_empty() && per_column.is_empty() {
3797            return None;
3798        }
3799        if per_column.is_empty() {
3800            let vals: Vec<PlSmallStr> = global
3801                .iter()
3802                .map(|s| PlSmallStr::from(s.as_str()))
3803                .collect();
3804            return Some(if vals.len() == 1 {
3805                NullValues::AllColumnsSingle(vals[0].clone())
3806            } else {
3807                NullValues::AllColumns(vals)
3808            });
3809        }
3810        if global.is_empty() {
3811            let pairs: Vec<(PlSmallStr, PlSmallStr)> = per_column
3812                .iter()
3813                .map(|(c, v)| (PlSmallStr::from(c.as_str()), PlSmallStr::from(v.as_str())))
3814                .collect();
3815            return Some(NullValues::Named(pairs));
3816        }
3817        let schema = schema?;
3818        let mut pairs: Vec<(PlSmallStr, PlSmallStr)> = Vec::new();
3819        let first_global = PlSmallStr::from(global[0].as_str());
3820        for (name, _) in schema.iter() {
3821            let col_name = name.as_str();
3822            let val = per_column
3823                .iter()
3824                .rev()
3825                .find(|(c, _)| c == col_name)
3826                .map(|(_, v)| PlSmallStr::from(v.as_str()))
3827                .unwrap_or_else(|| first_global.clone());
3828            pairs.push((PlSmallStr::from(col_name), val));
3829        }
3830        Some(NullValues::Named(pairs))
3831    }
3832
3833    /// Every reader-level CSV option, set one way on every route that scans lazily: a
3834    /// file, several files, a decompressed temp file, and a prefix in a bucket.
3835    pub(crate) fn configure_csv_reader(
3836        mut reader: LazyCsvReader,
3837        options: &OpenOptions,
3838        null_values: Option<&NullValues>,
3839    ) -> LazyCsvReader {
3840        reader = reader
3841            .with_separator(options.separator_or(b','))
3842            .with_comment_prefix(options.comment_char.as_deref().map(PlSmallStr::from));
3843        if let Some(rows) = options.header_rows() {
3844            // The header lines are read apart (`csv_header_names`); Polars starts
3845            // after the last of them, with `--skip-lines` counted from the same top
3846            // and `--skip-rows` counted after.
3847            let last = rows.iter().copied().max().unwrap_or(0);
3848            reader = reader
3849                .with_has_header(false)
3850                .with_skip_lines(last.max(options.skip_lines.unwrap_or(0)))
3851                .with_skip_rows_after_header(options.skip_rows.unwrap_or(0));
3852        } else {
3853            if let Some(skip_lines) = options.skip_lines {
3854                reader = reader.with_skip_lines(skip_lines);
3855            }
3856            if let Some(skip_rows) = options.skip_rows {
3857                reader = reader.with_skip_rows(skip_rows);
3858            }
3859            if let Some(has_header) = options.has_header {
3860                reader = reader.with_has_header(has_header);
3861            }
3862        }
3863        if let Some(n) = options.infer_schema_length {
3864            reader = reader.with_infer_schema_length(Some(n));
3865        }
3866        reader
3867            .with_ignore_errors(options.ignore_errors)
3868            // A followed file's later rows may have a field too many; they are counted
3869            // as not fitting rather than failing the read.
3870            .with_truncate_ragged_lines(options.follow)
3871            .with_try_parse_dates(options.csv_try_parse_dates())
3872            .with_null_values(null_values.cloned())
3873            // One byte that is not UTF-8 is a U+FFFD where it stands, not a file that
3874            // cannot be read past it.
3875            .with_encoding(CsvEncoding::LossyUtf8)
3876    }
3877
3878    /// [`Self::configure_csv_reader`] for the in-memory readers, which take options
3879    /// rather than a builder.
3880    fn eager_csv_read_options(
3881        options: &OpenOptions,
3882        null_values: Option<&NullValues>,
3883    ) -> CsvReadOptions {
3884        let mut read_options = CsvReadOptions::default();
3885        if let Some(rows) = options.header_rows() {
3886            let last = rows.iter().copied().max().unwrap_or(0);
3887            read_options.has_header = false;
3888            read_options.skip_lines = last.max(options.skip_lines.unwrap_or(0));
3889            read_options.skip_rows_after_header = options.skip_rows.unwrap_or(0);
3890        } else {
3891            if let Some(skip_lines) = options.skip_lines {
3892                read_options.skip_lines = skip_lines;
3893            }
3894            if let Some(skip_rows) = options.skip_rows {
3895                read_options.skip_rows = skip_rows;
3896            }
3897            if let Some(has_header) = options.has_header {
3898                read_options.has_header = has_header;
3899            }
3900        }
3901        if let Some(n) = options.infer_schema_length {
3902            read_options.infer_schema_length = Some(n);
3903        }
3904        read_options.ignore_errors = options.ignore_errors;
3905        read_options.map_parse_options(|opts| {
3906            opts.with_separator(options.separator_or(b','))
3907                .with_comment_prefix(
3908                    options
3909                        .comment_char
3910                        .as_deref()
3911                        .map(polars::io::csv::read::CommentPrefix::new_from_str),
3912                )
3913                .with_try_parse_dates(options.csv_try_parse_dates())
3914                .with_null_values(null_values.cloned())
3915                .with_encoding(CsvEncoding::LossyUtf8)
3916        })
3917    }
3918
3919    /// The columns a CSV reader will produce, from one row, for building null_values
3920    /// when both global and per-column are set.
3921    pub(crate) fn csv_schema_for_null_values(
3922        reader: LazyCsvReader,
3923        options: &OpenOptions,
3924    ) -> Result<Arc<Schema>> {
3925        let mut lf =
3926            Self::configure_csv_reader(reader.with_n_rows(Some(1)), options, None).finish()?;
3927        lf.collect_schema().map_err(color_eyre::eyre::Report::from)
3928    }
3929
3930    /// Build Polars NullValues from options, for the CSV at `path` with the header
3931    /// lines `header` read from it.
3932    fn build_null_values_for_csv(
3933        options: &OpenOptions,
3934        path: &Path,
3935        header: Option<&[String]>,
3936    ) -> Result<Option<NullValues>> {
3937        Self::build_null_values_with(options, header, || {
3938            Self::csv_schema_for_null_values(Self::csv_reader_of(path)?, options)
3939        })
3940    }
3941
3942    /// Build Polars NullValues from options. `schema` is the reader's own columns, read
3943    /// only when a spec names a column: the user names it as it is shown (trimmed, or
3944    /// from `--header-rows`), and the reader knows it by what it parsed.
3945    pub(crate) fn build_null_values_with(
3946        options: &OpenOptions,
3947        header: Option<&[String]>,
3948        schema: impl FnOnce() -> Result<Arc<Schema>>,
3949    ) -> Result<Option<NullValues>> {
3950        let specs = match &options.null_values {
3951            None => return Ok(None),
3952            Some(s) if s.is_empty() => return Ok(None),
3953            Some(s) => s.as_slice(),
3954        };
3955        let (global, mut per_column) = Self::parse_null_value_specs(specs);
3956        if per_column.is_empty() {
3957            return Ok(Self::build_polars_null_values(&global, &per_column, None));
3958        }
3959        let schema = match schema() {
3960            Ok(schema) => schema,
3961            // Nothing follows the header lines: no value to read as null.
3962            Err(e)
3963                if header.is_some()
3964                    && matches!(
3965                        e.downcast_ref::<PolarsError>(),
3966                        Some(PolarsError::NoData(_))
3967                    ) =>
3968            {
3969                return Ok(None);
3970            }
3971            Err(e) => return Err(e),
3972        };
3973        let raw: Vec<PlSmallStr> = schema.iter_names().cloned().collect();
3974        let shown = crate::csv_dialect::shown_names(&raw, header);
3975        for (column, _) in per_column.iter_mut() {
3976            if let Some(i) = shown.iter().position(|s| s == column) {
3977                *column = raw[i].to_string();
3978            }
3979        }
3980        Ok(Self::build_polars_null_values(
3981            &global,
3982            &per_column,
3983            Some(schema.as_ref()),
3984        ))
3985    }
3986
3987    /// The null values `--null` gives the column shown as `column`.
3988    pub(crate) fn csv_null_values_for(options: &OpenOptions, column: &str) -> Vec<String> {
3989        let (global, per_column) =
3990            Self::parse_null_value_specs(options.null_values.as_deref().unwrap_or_default());
3991        let mut values: Vec<String> = per_column
3992            .into_iter()
3993            .filter(|(c, _)| c == column)
3994            .map(|(_, v)| v)
3995            .collect();
3996        values.extend(global);
3997        values
3998    }
3999
4000    /// The names `--header-rows` gives the columns of the CSV `source` holds, or
4001    /// `None` when it is not in effect.
4002    fn csv_header_names<R: std::io::BufRead>(
4003        options: &OpenOptions,
4004        source: impl FnOnce() -> std::io::Result<R>,
4005    ) -> Result<Option<Vec<String>>> {
4006        let Some(rows) = options.header_rows() else {
4007            return Ok(None);
4008        };
4009        Ok(Some(crate::csv_dialect::header_names(
4010            source()?,
4011            rows,
4012            &options.header_join,
4013            options.separator_or(b','),
4014            options.comment_char.as_deref(),
4015        )?))
4016    }
4017
4018    /// A lazy CSV reader of the file at `path`, by its path; or, when the file ends in
4019    /// a run of NULs, of its text before them, mapped and read in place.
4020    pub(crate) fn csv_reader_of(path: &Path) -> Result<LazyCsvReader> {
4021        let glob = crate::source::expands_as_glob(path);
4022        if !glob
4023            && path.is_file()
4024            && let Ok(Some(text)) = crate::nul_tail::text_buffer(path)
4025        {
4026            return Ok(LazyCsvReader::new_with_sources(
4027                polars::lazy::dsl::ScanSources::Buffers(Arc::from([text])),
4028            ));
4029        }
4030        Ok(LazyCsvReader::new(PlRefPath::try_from_path(path)?).with_glob(glob))
4031    }
4032
4033    /// [`Self::csv_header_names`] for a file on disk, compressed with `compression`
4034    /// or not.
4035    pub(crate) fn csv_header_names_of(
4036        options: &OpenOptions,
4037        path: &Path,
4038        compression: Option<CompressionFormat>,
4039    ) -> Result<Option<Vec<String>>> {
4040        Self::csv_header_names(options, || Self::text_source(path, compression))
4041    }
4042
4043    /// The text of the file at `path`, through its decompressor when it has one.
4044    pub(crate) fn text_source(
4045        path: &Path,
4046        compression: Option<CompressionFormat>,
4047    ) -> std::io::Result<Box<dyn std::io::BufRead>> {
4048        let file = File::open(path)?;
4049        if compression.is_none()
4050            && let Some(len) = crate::nul_tail::text_len(&file)?
4051        {
4052            return Ok(Box::new(BufReader::new(file.take(len))));
4053        }
4054        let file = BufReader::new(file);
4055        Ok(match compression {
4056            None => Box::new(file),
4057            Some(CompressionFormat::Gzip) => {
4058                Box::new(BufReader::new(flate2::read::GzDecoder::new(file)))
4059            }
4060            Some(CompressionFormat::Zstd) => {
4061                Box::new(BufReader::new(zstd::Decoder::with_buffer(file)?))
4062            }
4063            Some(CompressionFormat::Bzip2) => {
4064                Box::new(BufReader::new(bzip2::read::BzDecoder::new(file)))
4065            }
4066            Some(CompressionFormat::Xz) => {
4067                Box::new(BufReader::new(xz2::read::XzDecoder::new(file)))
4068            }
4069        })
4070    }
4071
4072    /// What every CSV read does once Polars has parsed it: name the columns (trimmed,
4073    /// or from `--header-rows`), skip the padding after a delimiter, type the text
4074    /// columns, and drop the footer.
4075    /// `read` gets the steps that Python can repeat.
4076    fn finish_csv_frame(
4077        lf: LazyFrame,
4078        options: &OpenOptions,
4079        header: Option<&[String]>,
4080        read: &mut Vec<String>,
4081        typing: &mut Typing,
4082    ) -> Result<LazyFrame> {
4083        let lf = Self::name_csv_columns(lf, header, Some(read))?;
4084        Self::finish_csv_values(lf, options, read, typing)
4085    }
4086
4087    /// [`crate::csv_dialect::name_columns`], with the renames as Python in `read`.
4088    /// Names from `--header-rows` are not recorded: Copy as Python does not write that
4089    /// read.
4090    fn name_csv_columns(
4091        mut lf: LazyFrame,
4092        header: Option<&[String]>,
4093        read: Option<&mut Vec<String>>,
4094    ) -> Result<LazyFrame> {
4095        if let (None, Some(read)) = (header, read) {
4096            let raw: Vec<PlSmallStr> = lf.collect_schema()?.iter_names().cloned().collect();
4097            let shown = crate::csv_dialect::shown_names(&raw, None);
4098            let renames: Vec<String> = raw
4099                .iter()
4100                .zip(&shown)
4101                .filter(|(raw, shown)| raw.as_str() != shown.as_str())
4102                .map(|(raw, shown)| format!("{}: {}", py_str(raw), py_str(shown)))
4103                .collect();
4104            if !renames.is_empty() {
4105                read.push(format!(".rename({{{}}})", renames.join(", ")));
4106            }
4107        }
4108        Ok(crate::csv_dialect::name_columns(lf, header)?)
4109    }
4110
4111    /// [`Self::finish_csv_frame`] after the names, for frames already named: several
4112    /// files are named one at a time and stacked first.
4113    fn finish_csv_values(
4114        mut lf: LazyFrame,
4115        options: &OpenOptions,
4116        read: &mut Vec<String>,
4117        typing: &mut Typing,
4118    ) -> Result<LazyFrame> {
4119        if options.skip_initial_space {
4120            lf = crate::csv_dialect::skip_initial_space(lf, |column| {
4121                Self::csv_null_values_for(options, column)
4122            })?;
4123        }
4124        // Read without a header (`H`), the columns have no names to derive from, or to
4125        // type by. Derived columns read the file's text, before any column is typed.
4126        let spec = options
4127            .delimited
4128            .as_ref()
4129            .filter(|_| options.has_header != Some(false))
4130            .map(|read| read.delimited());
4131        if let Some(spec) = spec {
4132            lf = spec.derive(lf)?;
4133            lf = Self::declare_types(lf, &spec.types, typing)?;
4134        }
4135        let typed: Vec<String> = typing.typed.iter().map(|t| t.column.clone()).collect();
4136        lf = Self::apply_parse_strings_to_csv_lazyframe(lf, options, read, &typed, typing)?;
4137        Self::apply_skip_tail_rows_csv(lf, options)
4138    }
4139
4140    /// `lf` with each column `types` names read as its type, lazily; `typing` records
4141    /// them and the frame before, for the count of the values that did not fit, and a
4142    /// note names the ones the frame does not have.
4143    fn declare_types(
4144        mut lf: LazyFrame,
4145        types: &[(String, crate::column_types::ColumnType)],
4146        typing: &mut Typing,
4147    ) -> Result<LazyFrame> {
4148        if types.is_empty() {
4149            return Ok(lf);
4150        }
4151        let schema = lf.collect_schema()?;
4152        let mut exprs = Vec::with_capacity(types.len());
4153        let mut missing = Vec::new();
4154        for (name, ty) in types {
4155            match schema.get(name.as_str()) {
4156                Some(from) => {
4157                    exprs.push(ty.expr(name, from).alias(name.as_str()));
4158                    typing.typed.push(crate::column_types::Typed {
4159                        column: name.clone(),
4160                        ty: ty.clone(),
4161                        from: from.clone(),
4162                    });
4163                }
4164                None => missing.push(name.as_str()),
4165            }
4166        }
4167        if !missing.is_empty() {
4168            typing.notes.push(crate::notes::Note {
4169                summary: format!(
4170                    "typed in the spec, not in the file: {}",
4171                    crate::notes::some_names(&missing)
4172                ),
4173                scope: "the spec's [columns]".to_string(),
4174                read_as_text: None,
4175                passed_over: None,
4176            });
4177        }
4178        if exprs.is_empty() {
4179            return Ok(lf);
4180        }
4181        typing.source = Some(lf.clone());
4182        Ok(lf.with_columns(exprs))
4183    }
4184
4185    /// `reader`, the scan of the file at `path`, reading some columns as text: those a
4186    /// spec gives a type, which [`Self::declare_types`] types, a value that does not fit
4187    /// null rather than a failed read; and, while `read.infer_types` types text, those
4188    /// whose first rows hold a number with a leading zero (`02134`), which Polars would
4189    /// read as an integer and lose. `window` is those rows when the read has them;
4190    /// otherwise they are read, up to the rows a scan infers its types from.
4191    pub(crate) fn scan_some_as_text(
4192        reader: LazyCsvReader,
4193        options: &OpenOptions,
4194        header: Option<&[String]>,
4195        path: &Path,
4196        window: Option<&[Vec<String>]>,
4197        text: &mut Vec<String>,
4198    ) -> Result<LazyCsvReader> {
4199        if options.has_header == Some(false) {
4200            return Ok(reader);
4201        }
4202        let names: Vec<String> = options
4203            .delimited
4204            .as_ref()
4205            .map(|read| {
4206                read.delimited()
4207                    .types
4208                    .iter()
4209                    .map(|(name, _)| name.clone())
4210                    .collect()
4211            })
4212            .unwrap_or_default();
4213        let zeros: Vec<usize> = match &options.parse_strings {
4214            None => Vec::new(),
4215            Some(_) => {
4216                let read;
4217                let window = match window {
4218                    Some(window) => window,
4219                    None => {
4220                        read = crate::spec_union::head_window(path, options).unwrap_or_default();
4221                        &read
4222                    }
4223                };
4224                let width = window.iter().map(Vec::len).max().unwrap_or(0);
4225                (0..width)
4226                    .filter(|&at| {
4227                        window.iter().any(|row| {
4228                            row.get(at)
4229                                .is_some_and(|v| crate::column_types::has_leading_zero(v))
4230                        })
4231                    })
4232                    .collect()
4233            }
4234        };
4235        if names.is_empty() && zeros.is_empty() {
4236            return Ok(reader);
4237        }
4238        let header = header.map(<[String]>::to_vec);
4239        let target = options.parse_strings.clone();
4240        let read_as_text = Arc::new(std::sync::Mutex::new(Vec::new()));
4241        let said = read_as_text.clone();
4242        let reader = reader.with_schema_modify(move |mut schema| {
4243            let raw: Vec<PlSmallStr> = schema.iter_names().cloned().collect();
4244            let shown = crate::csv_dialect::shown_names(&raw, header.as_deref());
4245            for (at, (raw, shown)) in raw.iter().zip(&shown).enumerate() {
4246                let inferred = match &target {
4247                    Some(ParseStringsTarget::All) => true,
4248                    Some(ParseStringsTarget::Columns(columns)) => columns.contains(shown),
4249                    None => false,
4250                };
4251                if names.contains(shown) || (inferred && zeros.contains(&at)) {
4252                    schema.with_column(raw.clone(), DataType::String);
4253                    if let Ok(mut said) = said.lock() {
4254                        said.push(raw.to_string());
4255                    }
4256                }
4257            }
4258            Ok(schema)
4259        })?;
4260        if let Ok(mut read) = read_as_text.lock() {
4261            text.append(&mut read);
4262        }
4263        Ok(reader)
4264    }
4265
4266    /// If options.skip_tail_rows is set, run a count query and slice the LazyFrame to drop that many rows from the end. Used for CSV with trailing garbage/footer.
4267    pub(crate) fn apply_skip_tail_rows_csv(
4268        lf: LazyFrame,
4269        options: &OpenOptions,
4270    ) -> Result<LazyFrame> {
4271        let n = match options.skip_tail_rows {
4272            None | Some(0) => return Ok(lf),
4273            Some(n) => n,
4274        };
4275        let count_df = collect_lazy(lf.clone().select([len()]), options.polars_streaming)
4276            .map_err(color_eyre::eyre::Report::from)?;
4277        let total: u32 = match count_df.get(0) {
4278            Some(col) => match col.first() {
4279                Some(AnyValue::UInt32(v)) => *v,
4280                _ => return Ok(lf),
4281            },
4282            _ => {
4283                return Ok(lf);
4284            }
4285        };
4286        let keep = total.saturating_sub(n as u32);
4287        Ok(lf.slice(0, keep))
4288    }
4289
4290    /// The first date format `sample` reads in. `None` when none does, so Polars is
4291    /// never handed `format: None`, which can fail.
4292    fn infer_date_format_from_sample(sample: &str) -> Option<&'static str> {
4293        crate::column_types::formats_reading(&DataType::Date, sample)
4294            .first()
4295            .copied()
4296    }
4297
4298    fn infer_datetime_format_from_sample(sample: &str) -> Option<&'static str> {
4299        crate::column_types::formats_reading(
4300            &DataType::Datetime(TimeUnit::Microseconds, None),
4301            sample,
4302        )
4303        .first()
4304        .copied()
4305    }
4306
4307    /// Parse a string ChunkedArray into a Duration ChunkedArray (nanoseconds). Uses Polars duration
4308    /// format (e.g. `1d`, `2h30m`, `-1w2d`). Invalid or null inputs become null in the output.
4309    fn string_chunked_to_duration_ns(str_ca: &StringChunked) -> DurationChunked {
4310        let name = str_ca.name().clone();
4311        let vals: Vec<Option<i64>> = str_ca
4312            .iter()
4313            .map(|opt_s| {
4314                opt_s.and_then(|s| {
4315                    polars::time::Duration::try_parse(s)
4316                        .ok()
4317                        .map(|d| d.duration_ns())
4318                })
4319            })
4320            .collect();
4321        let int_ca = Int64Chunked::from_iter_options(name, vals.into_iter());
4322        int_ca.into_duration(TimeUnit::Nanoseconds)
4323    }
4324
4325    fn infer_time_format_from_sample(sample: &str) -> Option<&'static str> {
4326        crate::column_types::formats_reading(&DataType::Time, sample)
4327            .first()
4328            .copied()
4329    }
4330
4331    /// Apply trim and type inference to CSV string columns when --infer-types is enabled.
4332    /// Samples up to `options.parse_strings_sample_rows` rows to infer types, then overlays lazy exprs (trim then cast) on the LazyFrame.
4333    fn apply_parse_strings_to_csv_lazyframe(
4334        lf: LazyFrame,
4335        options: &OpenOptions,
4336        read: &mut Vec<String>,
4337        except: &[String],
4338        typing: &mut Typing,
4339    ) -> Result<LazyFrame> {
4340        let Some(target) = &options.parse_strings else {
4341            return Ok(lf);
4342        };
4343        let before = lf.clone();
4344        let mut typed = Vec::new();
4345        let lf = Self::type_string_columns(
4346            lf,
4347            target,
4348            options.parse_strings_sample_rows,
4349            StringTypes {
4350                dates: options.parse_dates,
4351                numbers: true,
4352            },
4353            read,
4354            except,
4355            &mut typed,
4356        )?;
4357        // The columns it typed are counted as the spec's are, over the frame before
4358        // either: the spec's typing leaves these columns as they were read.
4359        if !typed.is_empty() {
4360            typing.source.get_or_insert(before);
4361            typing.typed.extend(typed);
4362        }
4363        Ok(lf)
4364    }
4365
4366    /// Dates and timestamps a JSON file holds as strings, typed the way a CSV's are.
4367    /// JSON already says which values are numbers, so a string only ever becomes a
4368    /// date, datetime or time, and one that is none of those is left as it was read.
4369    pub(crate) fn apply_parse_dates_to_json_lazyframe(
4370        lf: LazyFrame,
4371        options: &OpenOptions,
4372        read: &mut Vec<String>,
4373    ) -> Result<LazyFrame> {
4374        if !options.parse_dates {
4375            return Ok(lf);
4376        }
4377        Self::type_string_columns(
4378            lf,
4379            &ParseStringsTarget::All,
4380            options.parse_strings_sample_rows,
4381            StringTypes {
4382                dates: true,
4383                numbers: false,
4384            },
4385            read,
4386            &[],
4387            &mut Vec::new(),
4388        )
4389    }
4390
4391    /// A string column read as a microsecond Datetime with `format`. Exact, so a naive
4392    /// format cannot match the front of a value that carries an offset and drop it; not
4393    /// strict, so a value past the sample that does not parse is null.
4394    fn datetime_from_str(expr: Expr, format: &str) -> Expr {
4395        expr.str().to_datetime(
4396            Some(TimeUnit::Microseconds),
4397            None,
4398            StrptimeOptions {
4399                format: Some(PlSmallStr::from(format)),
4400                strict: false,
4401                exact: true,
4402                cache: true,
4403            },
4404            lit(PlSmallStr::from_static("raise")),
4405        )
4406    }
4407
4408    /// The rows string inference reads: the first `sample_rows` of the `targets` only,
4409    /// trimmed so inference sees "1" not " 1 ", with blanks as null so "all null" and
4410    /// the accept test see normalized values. Only the targets are selected, so the
4411    /// other columns are neither decoded nor held; the frame the table shows keeps them.
4412    fn string_inference_sample(
4413        lf: LazyFrame,
4414        targets: &[String],
4415        sample_rows: usize,
4416    ) -> PolarsResult<DataFrame> {
4417        let whitespace_pat = lit(PlSmallStr::from_static(" \t\n\r"));
4418        let trimmed: Vec<Expr> = targets
4419            .iter()
4420            .map(|c| {
4421                let name = PlSmallStr::from(c.as_str());
4422                col(name.clone())
4423                    .str()
4424                    .strip_chars(whitespace_pat.clone())
4425                    .alias(name)
4426            })
4427            .collect();
4428        let blank_to_null: Vec<Expr> = targets
4429            .iter()
4430            .map(|c| {
4431                let name = PlSmallStr::from(c.as_str());
4432                when(col(name.clone()).eq(lit(PlSmallStr::from_static(""))))
4433                    .then(Null {}.lit())
4434                    .otherwise(col(name.clone()))
4435                    .alias(name)
4436            })
4437            .collect();
4438        lf.limit(sample_rows as u32)
4439            .select(trimmed)
4440            .with_columns(blank_to_null)
4441            .collect()
4442    }
4443
4444    /// Type string columns from the first `sample_rows` rows: trim, then keep the first
4445    /// of Date, Datetime, Time, Duration, Int64 and Float64 (as `types` allows) that
4446    /// parses every sampled value, as lazy expressions over `lf`.
4447    fn type_string_columns(
4448        lf: LazyFrame,
4449        target: &ParseStringsTarget,
4450        sample_rows: usize,
4451        types: StringTypes,
4452        read: &mut Vec<String>,
4453        except: &[String],
4454        typed: &mut Vec<crate::column_types::Typed>,
4455    ) -> Result<LazyFrame> {
4456        // The scan already inferred the schema; the sample below is the one read.
4457        let schema = lf.clone().collect_schema()?;
4458        let string_cols: Vec<String> = schema
4459            .iter()
4460            // A column given a type keeps it.
4461            .filter(|(name, _)| !except.iter().any(|e| e == name.as_str()))
4462            .filter(|(_name, dtype)| **dtype == DataType::String)
4463            .map(|(name, _)| name.to_string())
4464            .collect();
4465        let target_cols: Vec<String> = match target {
4466            ParseStringsTarget::All => string_cols,
4467            ParseStringsTarget::Columns(c) => c
4468                .iter()
4469                .filter(|name| string_cols.contains(name))
4470                .cloned()
4471                .collect(),
4472        };
4473        if target_cols.is_empty() {
4474            return Ok(lf);
4475        }
4476        use polars::datatypes::TimeUnit;
4477        let whitespace_pat = lit(PlSmallStr::from_static(" \t\n\r"));
4478        let sample_df = Self::string_inference_sample(lf.clone(), &target_cols, sample_rows)?;
4479        log::debug!(
4480            target: "datui",
4481            "string inference sample: {} rows x {} columns for {} targets, {} bytes",
4482            sample_df.height(),
4483            sample_df.width(),
4484            target_cols.len(),
4485            sample_df.estimated_size()
4486        );
4487        let mut exprs = Vec::with_capacity(target_cols.len());
4488        // The same typing as Python, for Copy as Python.
4489        let mut python = Vec::with_capacity(target_cols.len());
4490        for col_name in &target_cols {
4491            let name = PlSmallStr::from(col_name.as_str());
4492            let s = sample_df.column(col_name.as_str())?;
4493            let null_before = s.null_count();
4494            let len = s.len();
4495            // Accept type if we didn't introduce new nulls (null_after <= null_before).
4496            let accept_type = |null_after: usize| null_after <= null_before;
4497            // Inference order: Date → Datetime → Time → Duration → Int64 → Float64 → String.
4498            enum InferredType {
4499                Date,
4500                Datetime,
4501                Time,
4502                Duration,
4503                Int64,
4504                Float64,
4505                String,
4506            }
4507            let (inferred, date_fmt, datetime_fmt, time_fmt) = if null_before == len {
4508                // Column is all null (including blanks treated as null): leave as string.
4509                (InferredType::String, None, None, None)
4510            } else {
4511                match s.str() {
4512                    Err(_) => (InferredType::String, None, None, None),
4513                    Ok(str_ca) => {
4514                        let first_val: Option<&str> = str_ca
4515                            .iter()
4516                            .find_map(|o: Option<&str>| o.filter(|s: &&str| !s.is_empty()));
4517                        // `02134`, `007`: a ZIP code or an ID, not a number.
4518                        let zeros = str_ca
4519                            .iter()
4520                            .flatten()
4521                            .any(crate::column_types::has_leading_zero);
4522                        let (mut t, mut date_fmt, mut datetime_fmt, mut time_fmt) = match str_ca
4523                            .as_date(None, true)
4524                        {
4525                            Ok(as_date) if types.dates && accept_type(as_date.null_count()) => {
4526                                let fmt = first_val.and_then(Self::infer_date_format_from_sample);
4527                                if fmt.is_some() {
4528                                    (InferredType::Date, fmt.map(String::from), None, None)
4529                                } else {
4530                                    (InferredType::String, None, None, None)
4531                                }
4532                            }
4533                            _ => (InferredType::String, None, None, None),
4534                        };
4535                        if matches!(t, InferredType::String)
4536                            && types.dates
4537                            && let Some(fmt) =
4538                                first_val.and_then(Self::infer_datetime_format_from_sample)
4539                        {
4540                            // Judged by the expression the table will run, so a column
4541                            // whose values disagree (an offset on some, none on others)
4542                            // fails here and stays text.
4543                            let parsed = sample_df
4544                                .clone()
4545                                .lazy()
4546                                .select([Self::datetime_from_str(col(name.clone()), fmt)])
4547                                .collect()?;
4548                            if accept_type(parsed.column(col_name.as_str())?.null_count()) {
4549                                (t, date_fmt, datetime_fmt, time_fmt) =
4550                                    (InferredType::Datetime, None, Some(fmt.to_string()), None);
4551                            }
4552                        }
4553                        if matches!(t, InferredType::String) {
4554                            (t, date_fmt, datetime_fmt, time_fmt) = match str_ca.as_time(None, true)
4555                            {
4556                                Ok(as_time) if accept_type(as_time.null_count()) => {
4557                                    let fmt =
4558                                        first_val.and_then(Self::infer_time_format_from_sample);
4559                                    if fmt.is_some() {
4560                                        (InferredType::Time, None, None, fmt.map(String::from))
4561                                    } else {
4562                                        (InferredType::String, None, None, None)
4563                                    }
4564                                }
4565                                _ => (InferredType::String, None, None, None),
4566                            };
4567                        }
4568                        if matches!(t, InferredType::String) && types.numbers {
4569                            let duration_ca = Self::string_chunked_to_duration_ns(str_ca);
4570                            (t, date_fmt, datetime_fmt, time_fmt) =
4571                                if accept_type(duration_ca.null_count()) {
4572                                    (InferredType::Duration, None, None, None)
4573                                } else {
4574                                    (InferredType::String, None, None, None)
4575                                };
4576                        }
4577                        if matches!(t, InferredType::String) && types.numbers && !zeros {
4578                            (t, date_fmt, datetime_fmt, time_fmt) =
4579                                match s.strict_cast(&DataType::Int64) {
4580                                    Ok(as_int) if accept_type(as_int.null_count()) => {
4581                                        (InferredType::Int64, None, None, None)
4582                                    }
4583                                    _ => (InferredType::String, None, None, None),
4584                                };
4585                        }
4586                        if matches!(t, InferredType::String) && types.numbers && !zeros {
4587                            (t, date_fmt, datetime_fmt, time_fmt) =
4588                                match s.strict_cast(&DataType::Float64) {
4589                                    Ok(as_float) if accept_type(as_float.null_count()) => {
4590                                        (InferredType::Float64, None, None, None)
4591                                    }
4592                                    _ => (InferredType::String, None, None, None),
4593                                };
4594                        }
4595                        (t, date_fmt, datetime_fmt, time_fmt)
4596                    }
4597                }
4598            };
4599            let base = col(PlSmallStr::from(col_name.as_str()))
4600                .str()
4601                .strip_chars(whitespace_pat.clone());
4602            let trimmed = format!(
4603                "pl.col({}).str.strip_chars(\" \\t\\n\\r\")",
4604                py_str(col_name)
4605            );
4606            let blank_null = format!("{trimmed}.replace(\"\", None)");
4607            let format_arg = |f: &Option<String>| match f {
4608                Some(f) => format!("{}, ", py_str(f)),
4609                None => String::new(),
4610            };
4611            python.push(match &inferred {
4612                InferredType::Date => format!(
4613                    "{blank_null}.str.to_date({}strict=False)",
4614                    format_arg(&date_fmt)
4615                ),
4616                InferredType::Datetime => format!(
4617                    "{blank_null}.str.to_datetime({}time_unit=\"us\", strict=False)",
4618                    format_arg(&datetime_fmt)
4619                ),
4620                InferredType::Time => format!(
4621                    "{blank_null}.str.to_time({}strict=False)",
4622                    format_arg(&time_fmt)
4623                ),
4624                InferredType::Duration => crate::python_script::py_comment(&format!(
4625                    "{col_name}: datui reads these as durations (\"1d2h\"); Polars has no parser for them"
4626                )),
4627                InferredType::Int64 => {
4628                    format!("{blank_null}.cast(pl.Int64, strict=False)")
4629                }
4630                InferredType::Float64 => {
4631                    format!("{blank_null}.cast(pl.Float64, strict=False)")
4632                }
4633                InferredType::String if types.numbers => trimmed.clone(),
4634                InferredType::String => String::new(),
4635            });
4636            // The one way a column is given a type: the spec's and the table's too.
4637            let ty = |dtype: DataType, format: Option<String>| crate::column_types::ColumnType {
4638                dtype,
4639                format,
4640            };
4641            let ty = match inferred {
4642                InferredType::Date => ty(DataType::Date, date_fmt),
4643                InferredType::Datetime => ty(
4644                    DataType::Datetime(TimeUnit::Microseconds, None),
4645                    datetime_fmt,
4646                ),
4647                InferredType::Time => ty(DataType::Time, time_fmt),
4648                InferredType::Duration => ty(DataType::Duration(TimeUnit::Nanoseconds), None),
4649                InferredType::Int64 => ty(DataType::Int64, None),
4650                InferredType::Float64 => ty(DataType::Float64, None),
4651                // Trimmed where every column is text; left as read where the
4652                // writer chose a string.
4653                InferredType::String if types.numbers => {
4654                    exprs.push(base.alias(name));
4655                    continue;
4656                }
4657                InferredType::String => continue,
4658            };
4659            let expr = ty.expr(col_name, &DataType::String).alias(name);
4660            typed.push(crate::column_types::Typed {
4661                column: col_name.clone(),
4662                ty,
4663                from: DataType::String,
4664            });
4665            exprs.push(expr);
4666        }
4667        let python: Vec<String> = python.into_iter().filter(|p| !p.is_empty()).collect();
4668        if !python.is_empty() {
4669            read.push(".with_columns(".to_string());
4670            read.extend(python.into_iter().map(|p| {
4671                if p.starts_with('#') {
4672                    format!("    {p}")
4673                } else {
4674                    format!("    {p},")
4675                }
4676            }));
4677            read.push(")".to_string());
4678        }
4679        Ok(lf.with_columns(exprs))
4680    }
4681
4682    pub fn from_csv(path: &Path, options: &OpenOptions) -> Result<Self> {
4683        Self::from_delimited(path, b',', options)
4684    }
4685
4686    /// `path` decompressed to a temporary copy in `temp_dir`, written through `writer`:
4687    /// a stopped open stops the copy, and quitting removes it.
4688    pub(crate) fn decompress_to_copy(
4689        path: &Path,
4690        compression: CompressionFormat,
4691        temp_dir: &Path,
4692        writer: &Writer,
4693    ) -> Result<crate::download::TempDownload> {
4694        let Decompressed { file, _claim } =
4695            Self::decompress_compressed_csv_to_temp(path, compression, temp_dir, writer)?;
4696        Ok(crate::download::TempDownload::held(file, Some(_claim)))
4697    }
4698
4699    /// As [`Self::from_delimited`], for an open: a compressed file is decompressed
4700    /// through `writer`, so the open's stop and quitting reach the copy.
4701    pub(crate) fn from_delimited_for_open(
4702        path: &Path,
4703        delimiter: u8,
4704        options: &OpenOptions,
4705        writer: &Writer,
4706    ) -> Result<Self> {
4707        Self::read_delimited(path, delimiter, options, writer)
4708    }
4709
4710    /// A delimited text file, split on `delimiter` (its format's separator) unless
4711    /// `--delimiter` says otherwise. CSV, TSV and PSV are one reader, so every CSV
4712    /// option means the same thing for all three.
4713    pub fn from_delimited(path: &Path, delimiter: u8, options: &OpenOptions) -> Result<Self> {
4714        Self::read_delimited(path, delimiter, options, &Writer::default())
4715    }
4716
4717    fn read_delimited(
4718        path: &Path,
4719        delimiter: u8,
4720        options: &OpenOptions,
4721        writer: &Writer,
4722    ) -> Result<Self> {
4723        // Settled once here: every reader below asks `options.separator_or(b',')`.
4724        let options = &OpenOptions {
4725            delimiter: Some(options.separator_or(delimiter)),
4726            ..options.clone()
4727        };
4728
4729        // Determine compression format: explicit option, or auto-detect from extension
4730        let compression = options
4731            .compression
4732            .or_else(|| CompressionFormat::from_extension(path));
4733
4734        if let Some(compression) = compression {
4735            if options.decompress_in_memory {
4736                // Eager read: decompress into memory, then CSV read
4737                let (df, header) = match compression {
4738                    CompressionFormat::Gzip | CompressionFormat::Zstd => {
4739                        let header = Self::csv_header_names_of(options, path, Some(compression))?;
4740                        let nv = Self::build_null_values_for_csv(options, path, header.as_deref())?;
4741                        let read_options = Self::eager_csv_read_options(options, nv.as_ref());
4742                        let df = crate::csv_dialect::read_after_header(
4743                            read_options
4744                                .try_into_reader_with_file_path(Some(path.into()))?
4745                                .finish(),
4746                            header.as_deref(),
4747                        )?;
4748                        (df, header)
4749                    }
4750                    CompressionFormat::Bzip2 | CompressionFormat::Xz => {
4751                        let file = BufReader::new(File::open(path)?);
4752                        let mut decompressed = Vec::new();
4753                        if compression == CompressionFormat::Bzip2 {
4754                            bzip2::read::BzDecoder::new(file).read_to_end(&mut decompressed)?;
4755                        } else {
4756                            xz2::read::XzDecoder::new(file).read_to_end(&mut decompressed)?;
4757                        }
4758                        crate::nul_tail::trim(&mut decompressed);
4759                        let header = Self::csv_header_names(options, || {
4760                            Ok(std::io::Cursor::new(decompressed.as_slice()))
4761                        })?;
4762                        // Column names for per-column null values come from the bytes: the file on
4763                        // disk is still compressed.
4764                        let nv = Self::build_null_values_with(options, header.as_deref(), || {
4765                            let one_row =
4766                                Self::eager_csv_read_options(options, None).with_n_rows(Some(1));
4767                            let df = CsvReader::new(std::io::Cursor::new(decompressed.as_slice()))
4768                                .with_options(one_row)
4769                                .finish()?;
4770                            Ok(df.schema().clone())
4771                        })?;
4772                        let read_options = Self::eager_csv_read_options(options, nv.as_ref());
4773                        let df = crate::csv_dialect::read_after_header(
4774                            CsvReader::new(std::io::Cursor::new(decompressed))
4775                                .with_options(read_options)
4776                                .finish(),
4777                            header.as_deref(),
4778                        )?;
4779                        (df, header)
4780                    }
4781                };
4782                let mut read = Vec::new();
4783                let mut typing = Typing::default();
4784                let lf = Self::finish_csv_frame(
4785                    df.lazy(),
4786                    options,
4787                    header.as_deref(),
4788                    &mut read,
4789                    &mut typing,
4790                )?;
4791                let mut state = Self::new(
4792                    lf,
4793                    options.pages_lookahead,
4794                    options.pages_lookback,
4795                    options.max_buffered_rows,
4796                    options.max_buffered_mb,
4797                    options.polars_streaming,
4798                )?;
4799                state.row_numbers = options.row_numbers;
4800                state.row_start_index = options.row_start_index;
4801                state.read_python = read;
4802                state.take_typing(typing);
4803                Ok(state)
4804            } else {
4805                // Decompress to temp file, then lazy scan
4806                let temp_dir = options.temp_dir.clone().unwrap_or_else(std::env::temp_dir);
4807                let temp =
4808                    Self::decompress_compressed_csv_to_temp(path, compression, &temp_dir, writer)?;
4809                let mut state = Self::scan_csv_file(temp.path(), options)?;
4810                state.decompress_temp_file = Some(Arc::new(temp));
4811                Ok(state)
4812            }
4813        } else {
4814            // For uncompressed files, use lazy scanning (more efficient)
4815            Self::scan_csv_file(path, options)
4816        }
4817    }
4818
4819    /// A compressed file read as lines: decompressed once to a file in `--temp-dir`, or
4820    /// into memory with `[read] decompress_in_memory`, then indexed.
4821    pub(crate) fn from_lines_decompressed(
4822        path: &Path,
4823        options: &OpenOptions,
4824        writer: &Writer,
4825    ) -> Result<(Self, crate::members::Opened)> {
4826        let compression = options
4827            .compression
4828            .or_else(|| CompressionFormat::from_extension(path))
4829            .ok_or_else(|| color_eyre::eyre::eyre!("{} is not compressed", path.display()))?;
4830        let (lines, temp) = if options.decompress_in_memory {
4831            let mut bytes = Vec::new();
4832            let read = std::sync::atomic::AtomicU64::new(0);
4833            crate::gps::open_reader(path, options, &read)?.read_to_end(&mut bytes)?;
4834            let name = path
4835                .file_stem()
4836                .map_or_else(String::new, |n| n.to_string_lossy().into_owned());
4837            let bytes = Arc::new(crate::fixed_records::Bytes::Owned(bytes));
4838            (crate::lines::Lines::from_bytes(vec![(name, bytes)]), None)
4839        } else {
4840            let temp_dir = options.temp_dir.clone().unwrap_or_else(std::env::temp_dir);
4841            let temp =
4842                Self::decompress_compressed_csv_to_temp(path, compression, &temp_dir, writer)?;
4843            let lines = crate::lines::Lines::open(&[temp.path().to_path_buf()], false)?;
4844            (lines, Some(Arc::new(temp)))
4845        };
4846        let lines = Arc::new(lines);
4847        let opened = crate::lines::opened(&lines, options);
4848        let mut state = Self::new(
4849            lines.lazy(),
4850            options.pages_lookahead,
4851            options.pages_lookback,
4852            options.max_buffered_rows,
4853            options.max_buffered_mb,
4854            options.polars_streaming,
4855        )?;
4856        state.row_numbers = options.row_numbers;
4857        state.row_start_index = options.row_start_index;
4858        state.decompress_temp_file = temp;
4859        Ok((state, opened))
4860    }
4861
4862    /// One uncompressed delimited file, scanned lazily. The frame is finished before
4863    /// the state is made from it, so the column order is of the names shown.
4864    fn scan_csv_file(path: &Path, options: &OpenOptions) -> Result<Self> {
4865        let header = Self::csv_header_names_of(options, path, None)?;
4866        let nv = Self::build_null_values_for_csv(options, path, header.as_deref())?;
4867        let reader = Self::csv_reader_of(path)?;
4868        let reader = Self::configure_csv_reader(reader, options, nv.as_ref());
4869        let mut typing = Typing::default();
4870        let lf = Self::scan_some_as_text(
4871            reader,
4872            options,
4873            header.as_deref(),
4874            path,
4875            None,
4876            &mut typing.text,
4877        )?
4878        .finish()?;
4879        let mut read = Vec::new();
4880        let lf = Self::finish_csv_frame(lf, options, header.as_deref(), &mut read, &mut typing)?;
4881        let mut state = Self::new(
4882            lf,
4883            options.pages_lookahead,
4884            options.pages_lookback,
4885            options.max_buffered_rows,
4886            options.max_buffered_mb,
4887            true,
4888        )?;
4889        state.row_numbers = options.row_numbers;
4890        state.row_start_index = options.row_start_index;
4891        state.read_python = read;
4892        state.take_typing(typing);
4893        Ok(state)
4894    }
4895
4896    pub fn from_csv_customize<F>(
4897        path: &Path,
4898        pages_lookahead: Option<usize>,
4899        pages_lookback: Option<usize>,
4900        max_buffered_rows: Option<usize>,
4901        max_buffered_mb: Option<usize>,
4902        func: F,
4903    ) -> Result<Self>
4904    where
4905        F: FnOnce(LazyCsvReader) -> LazyCsvReader,
4906    {
4907        let pl_path = PlRefPath::try_from_path(path)?;
4908        let reader = LazyCsvReader::new(pl_path).with_glob(crate::source::expands_as_glob(path));
4909        let lf = func(reader).finish()?;
4910        Self::new(
4911            lf,
4912            pages_lookahead,
4913            pages_lookback,
4914            max_buffered_rows,
4915            max_buffered_mb,
4916            true,
4917        )
4918    }
4919
4920    /// Load multiple CSV files (uncompressed) and concatenate into one LazyFrame.
4921    pub fn from_csv_paths(paths: &[impl AsRef<Path>], options: &OpenOptions) -> Result<Self> {
4922        if paths.is_empty() {
4923            return Err(color_eyre::eyre::eyre!("No paths provided"));
4924        }
4925        if paths.len() == 1 {
4926            return Self::from_csv(paths[0].as_ref(), options);
4927        }
4928        // Each file is named from its own header, so files whose names are padded
4929        // differently, or whose header lines say the same thing, stack by name.
4930        let mut lazy_frames = Vec::with_capacity(paths.len());
4931        // Python reads the files as one scan: the first file's renames stand for all.
4932        let mut read = Vec::new();
4933        // Files with nothing in them: no header, so no columns to stack.
4934        let mut no_header: Vec<&Path> = Vec::new();
4935        // Read through a spec, each file's header pass reads its units and the lines its
4936        // types are inferred from too, for lining the files up by name.
4937        let spec = options.delimited.as_ref().map(|read| read.delimited());
4938        let mut heads = Vec::new();
4939        let mut read_text = Vec::new();
4940        for p in paths {
4941            let p = p.as_ref();
4942            let in_file = |e: color_eyre::Report| crate::error_display::in_file(p, e);
4943            let head_read = match spec {
4944                Some(spec) => crate::spec_union::read_head(p, options, spec).map(Some),
4945                None => Ok(None),
4946            };
4947            let head = match head_read {
4948                Err(e) if crate::csv_dialect::is_blank_file(&e) => {
4949                    no_header.push(p);
4950                    continue;
4951                }
4952                head => head.map_err(in_file)?,
4953            };
4954            let header = match &head {
4955                Some(head) => head.names.clone(),
4956                None => match Self::csv_header_names_of(options, p, None) {
4957                    Err(e) if crate::csv_dialect::is_blank_file(&e) => {
4958                        no_header.push(p);
4959                        continue;
4960                    }
4961                    header => header.map_err(in_file)?,
4962                },
4963            };
4964            let nv =
4965                Self::build_null_values_for_csv(options, p, header.as_deref()).map_err(in_file)?;
4966            let reader = Self::csv_reader_of(p).map_err(in_file)?;
4967            let reader = Self::configure_csv_reader(reader, options, nv.as_ref());
4968            let window = head.as_ref().map(|head| head.window.as_slice());
4969            // Python reads the files as one scan: the first file's columns stand for all.
4970            let mut text = Vec::new();
4971            let lf =
4972                Self::scan_some_as_text(reader, options, header.as_deref(), p, window, &mut text)
4973                    .map_err(in_file)?
4974                    .finish()
4975                    .map_err(|e| in_file(e.into()))?;
4976            if lazy_frames.is_empty() {
4977                read_text = text;
4978            }
4979            let record = lazy_frames.is_empty().then_some(&mut read);
4980            // Polars reads the header line itself: a file with none has no columns, or
4981            // one with a blank name.
4982            if header.is_none() {
4983                let raw = lf.clone().collect_schema();
4984                let headless = match &raw {
4985                    Err(PolarsError::NoData(_)) => true,
4986                    Ok(schema) => {
4987                        schema.is_empty()
4988                            || (schema.len() == 1
4989                                && schema.iter_names().all(|n| n.trim().is_empty()))
4990                    }
4991                    Err(_) => false,
4992                };
4993                if headless && Self::is_blank_text(p) {
4994                    no_header.push(p);
4995                    continue;
4996                }
4997            }
4998            let named = Self::name_csv_columns(lf, header.as_deref(), record).map_err(in_file)?;
4999            lazy_frames.push(named);
5000            heads.extend(head);
5001        }
5002        if lazy_frames.is_empty() {
5003            return Err(color_eyre::eyre::eyre!(
5004                "none of these {} files has a header: each is empty, or blank",
5005                paths.len()
5006            ));
5007        }
5008        let mut notes: Vec<crate::notes::Note> =
5009            crate::notes::no_header(&no_header).into_iter().collect();
5010        let mut units = None;
5011        if spec.is_some() && lazy_frames.len() > 1 {
5012            let lined = crate::spec_union::line_up(lazy_frames, &heads, options)?;
5013            lazy_frames = lined.frames;
5014            notes.extend(lined.notes);
5015            units = Some(lined.units);
5016        }
5017        let mut typing = Typing {
5018            text: read_text,
5019            ..Typing::default()
5020        };
5021        let lf = Self::finish_csv_values(
5022            polars::prelude::concat(lazy_frames.as_slice(), Self::union_of_files())?,
5023            options,
5024            &mut read,
5025            &mut typing,
5026        )?;
5027        let mut state = Self::new(
5028            lf,
5029            options.pages_lookahead,
5030            options.pages_lookback,
5031            options.max_buffered_rows,
5032            options.max_buffered_mb,
5033            options.polars_streaming,
5034        )?;
5035        state.row_numbers = options.row_numbers;
5036        state.row_start_index = options.row_start_index;
5037        state.read_python = read;
5038        state.read_notes = notes;
5039        state.read_units = units;
5040        state.take_typing(typing);
5041        Ok(state)
5042    }
5043
5044    /// Whether the text of the file at `path`, to its NUL padding, is no more than
5045    /// white space. Read up to a bound: past it, the file holds something.
5046    fn is_blank_text(path: &Path) -> bool {
5047        const MOST: u64 = 64 << 10;
5048        let mut text = Vec::new();
5049        Self::text_source(path, None)
5050            .and_then(|source| source.take(MOST + 1).read_to_end(&mut text))
5051            .is_ok_and(|n| n as u64 <= MOST && text.iter().all(u8::is_ascii_whitespace))
5052    }
5053
5054    pub fn from_json(path: &Path, options: &OpenOptions) -> Result<Self> {
5055        Self::from_json_with_format(path, options, JsonFormat::Json)
5056    }
5057
5058    pub fn from_json_lines(path: &Path, options: &OpenOptions) -> Result<Self> {
5059        Self::from_json_with_format(path, options, JsonFormat::JsonLines)
5060    }
5061
5062    fn from_json_with_format(
5063        path: &Path,
5064        options: &OpenOptions,
5065        format: JsonFormat,
5066    ) -> Result<Self> {
5067        let file = File::open(path)?;
5068        let lf = JsonReader::new(file)
5069            .with_json_format(format)
5070            .finish()?
5071            .lazy();
5072        Self::read_with(lf, options)
5073    }
5074
5075    /// Load multiple JSON (array) files and concatenate into one LazyFrame.
5076    pub fn from_json_paths(paths: &[impl AsRef<Path>], options: &OpenOptions) -> Result<Self> {
5077        Self::from_json_with_format_paths(paths, options, JsonFormat::Json)
5078    }
5079
5080    /// Load multiple JSON Lines files and concatenate into one LazyFrame.
5081    pub fn from_json_lines_paths(
5082        paths: &[impl AsRef<Path>],
5083        options: &OpenOptions,
5084    ) -> Result<Self> {
5085        Self::from_json_with_format_paths(paths, options, JsonFormat::JsonLines)
5086    }
5087
5088    fn from_json_with_format_paths(
5089        paths: &[impl AsRef<Path>],
5090        options: &OpenOptions,
5091        format: JsonFormat,
5092    ) -> Result<Self> {
5093        if paths.is_empty() {
5094            return Err(color_eyre::eyre::eyre!("No paths provided"));
5095        }
5096        if paths.len() == 1 {
5097            return Self::from_json_with_format(paths[0].as_ref(), options, format);
5098        }
5099        let mut lazy_frames = Vec::with_capacity(paths.len());
5100        for p in paths {
5101            let file = File::open(p.as_ref())?;
5102            let lf = match &format {
5103                JsonFormat::Json => JsonReader::new(file)
5104                    .with_json_format(JsonFormat::Json)
5105                    .finish()?
5106                    .lazy(),
5107                JsonFormat::JsonLines => JsonReader::new(file)
5108                    .with_json_format(JsonFormat::JsonLines)
5109                    .finish()?
5110                    .lazy(),
5111            };
5112            lazy_frames.push(lf);
5113        }
5114        let lf = polars::prelude::concat(lazy_frames.as_slice(), Default::default())?;
5115        Self::read_with(lf, options)
5116    }
5117
5118    /// Returns true if a scroll by `rows` would trigger a collect (view would leave the buffer).
5119    /// Used so the UI only shows the throbber when actual data loading will occur.
5120    pub fn scroll_would_trigger_collect(&self, rows: i64) -> bool {
5121        if rows < 0 && self.start_row == 0 {
5122            return false;
5123        }
5124        let new_start_row = if self.start_row as i64 + rows <= 0 {
5125            0
5126        } else {
5127            if let Some(df) = self.df.as_ref()
5128                && rows > 0
5129                && df.shape().0 <= self.visible_rows
5130            {
5131                return false;
5132            }
5133            let unclamped = (self.start_row as i64 + rows) as usize;
5134            if rows > 0 {
5135                unclamped.min(self.num_rows.saturating_sub(self.visible_rows))
5136            } else {
5137                unclamped
5138            }
5139        };
5140        if new_start_row == self.start_row {
5141            return false;
5142        }
5143        let view_end = new_start_row
5144            + self
5145                .visible_rows
5146                .min(self.num_rows.saturating_sub(new_start_row));
5147        let within_buffer = new_start_row >= self.buffered_start_row
5148            && view_end <= self.buffered_end_row
5149            && self.buffered_end_row > 0;
5150        !within_buffer
5151    }
5152
5153    /// Update scroll position. If the view is within the buffer, re-slices display.
5154    /// If outside the buffer, sets the position but the caller must trigger a collect
5155    /// (synchronous or async) to load the new buffer range.
5156    /// Returns true if a collect is needed (view is outside the current buffer).
5157    pub fn slide_table(&mut self, rows: i64) -> bool {
5158        if rows < 0 && self.start_row == 0 {
5159            return false;
5160        }
5161
5162        let new_start_row = if self.start_row as i64 + rows <= 0 {
5163            0
5164        } else {
5165            if let Some(df) = self.df.as_ref()
5166                && rows > 0
5167                && df.shape().0 <= self.visible_rows
5168            {
5169                return false;
5170            }
5171            let unclamped = (self.start_row as i64 + rows) as usize;
5172            if rows > 0 {
5173                // Clamp forward scroll to keep at least visible_rows of data in view.
5174                // Without this, holding PageDown at the bottom pushes start_row past
5175                // num_rows, which makes scroll_would_trigger_collect fire repeatedly
5176                // and can leave busy stuck if the resulting collect is a no-op.
5177                unclamped.min(self.num_rows.saturating_sub(self.visible_rows))
5178            } else {
5179                unclamped
5180            }
5181        };
5182
5183        if new_start_row == self.start_row {
5184            return false;
5185        }
5186
5187        let view_end = new_start_row
5188            + self
5189                .visible_rows
5190                .min(self.num_rows.saturating_sub(new_start_row));
5191        let within_buffer = new_start_row >= self.buffered_start_row
5192            && view_end <= self.buffered_end_row
5193            && self.buffered_end_row > 0;
5194
5195        self.start_row = new_start_row;
5196
5197        if within_buffer {
5198            // Re-slice display from existing buffer.
5199            self.slice_from_buffer();
5200            if self.table_state.selected().is_none() {
5201                self.table_state.select(Some(0));
5202            }
5203            false
5204        } else {
5205            true // caller must collect
5206        }
5207    }
5208
5209    pub fn collect(&mut self) {
5210        if self.defer_collect {
5211            return;
5212        }
5213        // Update proximity threshold based on visible rows
5214        if self.visible_rows > 0 {
5215            self.proximity_threshold = self.proximity();
5216        }
5217
5218        // Run len() only when lf has changed (query, filter, sort, pivot, melt, reset, drill).
5219        if !self.num_rows_valid {
5220            self.num_rows = match collect_lazy(row_count_lf(&self.lf), self.polars_streaming) {
5221                Ok(df) => {
5222                    // The frame counts, so there is nothing wrong with it: retire a
5223                    // failure left by the frame this one replaced. `load_buffer` ends
5224                    // the same way, but the zero-row path below returns before it.
5225                    self.error = None;
5226                    match df.get(0) {
5227                        Some(col) => match col.first() {
5228                            Some(AnyValue::UInt64(len)) => *len as usize,
5229                            _ => 0,
5230                        },
5231                        _ => 0,
5232                    }
5233                }
5234                // A count that fails means the frame itself is broken — a sort or a
5235                // column order naming a column the query removed, say. Zero rows is the
5236                // wrong thing to report: it blanks the table and returns below, before
5237                // `load_buffer`, the only other place that records a failure. The caller
5238                // is then told nothing, so a broken frame reads as an empty one. Say what
5239                // went wrong instead.
5240                Err(e) => {
5241                    self.error = Some(e);
5242                    0
5243                }
5244            };
5245            self.num_rows_valid = true;
5246            self.remember_pristine_count();
5247        }
5248
5249        if self.num_rows > 0 {
5250            let max_start = self.num_rows.saturating_sub(1);
5251            if self.start_row > max_start {
5252                self.start_row = max_start;
5253            }
5254        } else {
5255            self.start_row = 0;
5256            self.buffered_start_row = 0;
5257            self.buffered_end_row = 0;
5258            self.buffered_df = None;
5259            self.df = None;
5260            self.locked_df = None;
5261            return;
5262        }
5263
5264        // Proximity-based buffer logic
5265        let view_start = self.start_row;
5266        let view_end = self.start_row + self.visible_rows.min(self.num_rows - self.start_row);
5267
5268        // Check if current view is within buffered range
5269        let within_buffer = view_start >= self.buffered_start_row
5270            && view_end <= self.buffered_end_row
5271            && self.buffered_end_row > 0;
5272
5273        // Buffer grows incrementally: initial load and each expansion add only a few pages (lookahead + lookback).
5274        // fit_window caps at max_buffered_rows and slides the window when at cap.
5275
5276        if within_buffer {
5277            let dist_to_start = view_start.saturating_sub(self.buffered_start_row);
5278            let dist_to_end = self.buffered_end_row.saturating_sub(view_end);
5279
5280            let needs_expansion_back =
5281                dist_to_start <= self.proximity_threshold && self.buffered_start_row > 0;
5282            let needs_expansion_forward =
5283                dist_to_end <= self.proximity_threshold && self.buffered_end_row < self.num_rows;
5284
5285            if !needs_expansion_back && !needs_expansion_forward {
5286                // Column scroll only: reuse cached full buffer and re-slice into locked/scroll columns.
5287                let expected_len = self
5288                    .buffered_end_row
5289                    .saturating_sub(self.buffered_start_row);
5290                if self
5291                    .buffered_df
5292                    .as_ref()
5293                    .is_some_and(|b| b.height() == expected_len)
5294                {
5295                    self.slice_buffer_into_display();
5296                    if self.table_state.selected().is_none() {
5297                        self.table_state.select(Some(0));
5298                    }
5299                    return;
5300                }
5301                self.load_buffer(self.buffered_start_row, self.buffered_end_row);
5302                if self.table_state.selected().is_none() {
5303                    self.table_state.select(Some(0));
5304                }
5305                return;
5306            }
5307
5308            let mut new_buffer_start = if needs_expansion_back {
5309                view_start.saturating_sub(self.reach_rows(self.pages_lookback))
5310            } else {
5311                self.buffered_start_row
5312            };
5313
5314            let mut new_buffer_end = if needs_expansion_forward {
5315                (view_end + self.reach_rows(self.pages_lookahead)).min(self.num_rows)
5316            } else {
5317                self.buffered_end_row
5318            };
5319
5320            self.fit_window(
5321                view_start,
5322                view_end,
5323                &mut new_buffer_start,
5324                &mut new_buffer_end,
5325            );
5326            if self.holds_buffer(new_buffer_start, new_buffer_end) {
5327                // Fitting the expansion gave back the row group already held.
5328                self.slice_buffer_into_display();
5329                if self.table_state.selected().is_none() {
5330                    self.table_state.select(Some(0));
5331                }
5332                return;
5333            }
5334            self.load_buffer(new_buffer_start, new_buffer_end);
5335        } else {
5336            // Outside buffer: either extend the previous buffer (so it grows) or load a fresh small window.
5337            // Only extend when the view is "close" to the existing buffer (e.g. user paged down a bit).
5338            // A big jump (e.g. jump to end) should load just a window around the new view, not extend
5339            // the buffer across the whole dataset.
5340            let mut new_buffer_start;
5341            let mut new_buffer_end;
5342
5343            let had_buffer = self.buffered_end_row > 0;
5344            let scrolled_past_end = had_buffer && view_start >= self.buffered_end_row;
5345            let scrolled_past_start = had_buffer && view_end <= self.buffered_start_row;
5346
5347            let extend_forward_ok = scrolled_past_end
5348                && (view_start - self.buffered_end_row) <= self.reach_rows(self.pages_lookahead);
5349            let extend_backward_ok = scrolled_past_start
5350                && (self.buffered_start_row - view_end) <= self.reach_rows(self.pages_lookback);
5351
5352            if extend_forward_ok {
5353                // View is just a few pages past buffer end; extend forward.
5354                new_buffer_start = self.buffered_start_row;
5355                new_buffer_end =
5356                    (view_end + self.reach_rows(self.pages_lookahead)).min(self.num_rows);
5357            } else if extend_backward_ok {
5358                // View is just a few pages before buffer start; extend backward.
5359                new_buffer_start = view_start.saturating_sub(self.reach_rows(self.pages_lookback));
5360                new_buffer_end = self.buffered_end_row;
5361            } else if scrolled_past_end || scrolled_past_start {
5362                // Big jump (e.g. jump to end or jump to start): load a fresh window around the view.
5363                new_buffer_start = view_start.saturating_sub(self.reach_rows(self.pages_lookback));
5364                new_buffer_end =
5365                    (view_end + self.reach_rows(self.pages_lookahead)).min(self.num_rows);
5366                let min_initial_len = self.min_buffer_len();
5367                let current_len = new_buffer_end.saturating_sub(new_buffer_start);
5368                if current_len < min_initial_len {
5369                    let need = min_initial_len.saturating_sub(current_len);
5370                    let can_extend_end = self.num_rows.saturating_sub(new_buffer_end);
5371                    let can_extend_start = new_buffer_start;
5372                    if can_extend_end >= need {
5373                        new_buffer_end = (new_buffer_end + need).min(self.num_rows);
5374                    } else if can_extend_start >= need {
5375                        new_buffer_start = new_buffer_start.saturating_sub(need);
5376                    } else {
5377                        new_buffer_end = (new_buffer_end + can_extend_end).min(self.num_rows);
5378                        new_buffer_start =
5379                            new_buffer_start.saturating_sub(need.saturating_sub(can_extend_end));
5380                    }
5381                }
5382            } else {
5383                // No buffer yet or big jump: load a fresh small window (view ± a few pages).
5384                new_buffer_start = view_start.saturating_sub(self.reach_rows(self.pages_lookback));
5385                new_buffer_end =
5386                    (view_end + self.reach_rows(self.pages_lookahead)).min(self.num_rows);
5387
5388                // Ensure at least (1 + lookahead + lookback) pages so buffer size is consistent (e.g. 364 at 52 visible).
5389                let min_initial_len = self.min_buffer_len();
5390                let current_len = new_buffer_end.saturating_sub(new_buffer_start);
5391                if current_len < min_initial_len {
5392                    let need = min_initial_len.saturating_sub(current_len);
5393                    let can_extend_end = self.num_rows.saturating_sub(new_buffer_end);
5394                    let can_extend_start = new_buffer_start;
5395                    if can_extend_end >= need {
5396                        new_buffer_end = (new_buffer_end + need).min(self.num_rows);
5397                    } else if can_extend_start >= need {
5398                        new_buffer_start = new_buffer_start.saturating_sub(need);
5399                    } else {
5400                        new_buffer_end = (new_buffer_end + can_extend_end).min(self.num_rows);
5401                        new_buffer_start =
5402                            new_buffer_start.saturating_sub(need.saturating_sub(can_extend_end));
5403                    }
5404                }
5405            }
5406
5407            self.fit_window(
5408                view_start,
5409                view_end,
5410                &mut new_buffer_start,
5411                &mut new_buffer_end,
5412            );
5413            self.load_buffer(new_buffer_start, new_buffer_end);
5414        }
5415
5416        self.slice_from_buffer();
5417        if self.table_state.selected().is_none() {
5418            self.table_state.select(Some(0));
5419        }
5420    }
5421
5422    /// Prepare the LazyFrame and parameters for an async collect, without blocking.
5423    /// Updates internal state (proximity, start_row clamping) then returns
5424    /// a `CollectRequest` if a new buffer load is needed, or `None` if the current
5425    /// buffer is sufficient (in which case display slices are already updated).
5426    ///
5427    /// Caller must ensure `num_rows_valid` (via `set_num_rows`) before calling.
5428    /// `num_rows_override`, when supplied, applies that value first.
5429    /// Column expressions for every column in `column_order`, with binary columns replaced by a
5430    /// stub literal ([`binary_stub`]) so their blobs are never read. Used both for the display
5431    /// buffer (keeps scroll/jump collects fast) and for analysis (describe/distribution/
5432    /// correlation), where reading multi-GB blobs across partitions would otherwise exhaust
5433    /// memory and freeze the process. The full bytes stay available through `lf` for export.
5434    pub(crate) fn binary_stub_exprs(&self) -> Vec<Expr> {
5435        self.column_order
5436            .iter()
5437            .map(|name| {
5438                if matches!(self.schema.get(name.as_str()), Some(DataType::Binary)) {
5439                    lit(binary_stub()).alias(name.as_str())
5440                } else {
5441                    col(name.as_str())
5442                }
5443            })
5444            .collect()
5445    }
5446
5447    /// What the footers said about each file, for the checks that measure which files
5448    /// hold which columns. Taken from the dataset as it is read *now*, so a column
5449    /// already read as text is no longer a conflict.
5450    ///
5451    /// `row_index_column` is left empty: only the caller knows which index its own
5452    /// frame carries, and every caller fills it in.
5453    fn quality_source_drift(&self) -> crate::data_quality::QualitySourceContext {
5454        crate::data_quality::QualitySourceContext {
5455            file_names: self.drift_files.clone(),
5456            file_starts: self.drift_file_starts.clone(),
5457            row_index_column: String::new(),
5458            file_group: self.drift_file_group.clone(),
5459            drift_groups: self.drift_groups.clone(),
5460            file_omitted: self
5461                .dataset_schema
5462                .as_ref()
5463                .map(|dataset| dataset.omitted.clone())
5464                .unwrap_or_default(),
5465            dataset_rows: self.drift_dataset_rows,
5466            footers_read: self
5467                .dataset_schema
5468                .as_ref()
5469                .map(|dataset| dataset.files)
5470                .unwrap_or_default(),
5471            // Attached by the run that promised the extra reads, not by every frame.
5472            conflict_scan: None,
5473        }
5474    }
5475
5476    /// How many extra one-column file reads a full data-quality run would make for the
5477    /// values a type conflict hides. Zero when the dataset's files agree.
5478    pub(crate) fn quality_conflict_reads(&self) -> usize {
5479        if !self.drift_column_present
5480            || !self
5481                .dataset_at_open
5482                .as_ref()
5483                .is_some_and(crate::schema_union::DatasetSchema::drifts)
5484        {
5485            return 0;
5486        }
5487        crate::data_quality::conflict_reads(&self.drift_file_group, &self.drift_groups)
5488    }
5489
5490    /// Reads one column of named files at the type each of them wrote it in, for the
5491    /// values a type conflict hides. `None` when the dataset's files all agree, or
5492    /// when this frame is not the dataset as it opened.
5493    pub(crate) fn quality_conflict_scan(&self) -> Option<crate::data_quality::QualityConflictScan> {
5494        let dataset = self.dataset_at_open.clone()?;
5495        if !self.drift_column_present || !dataset.drifts() {
5496            return None;
5497        }
5498        if let Some(remote) = self.remote_files.as_ref() {
5499            return Some(crate::data_quality::QualityConflictScan(
5500                remote.scan.clone(),
5501            ));
5502        }
5503        // Built from the dataset as its footers found it, for the same reason
5504        // `read_column_as_text` is: only the footers know the type each file wrote.
5505        let drift =
5506            crate::schema_union::ScanDrift::new(&self.drift_files, &dataset, &self.file_rows())
5507                .map(Arc::new);
5508        let partition_columns = self.partition_columns.clone();
5509        Some(crate::data_quality::QualityConflictScan(Arc::new(
5510            move |files: &[String], as_text: &[PlSmallStr]| {
5511                let drifts = drift.is_some();
5512                let lf = crate::schema_union::lenient_scan(
5513                    files,
5514                    dataset.schema.clone(),
5515                    None,
5516                    drift.as_deref(),
5517                    as_text,
5518                )?;
5519                Ok(crate::hoist_partition_columns(
5520                    lf,
5521                    &dataset.schema,
5522                    partition_columns.as_deref().unwrap_or(&[]),
5523                    drifts,
5524                ))
5525            },
5526        )))
5527    }
5528
5529    /// Frame and optional row-to-file map used by the data-quality worker. The hidden
5530    /// scan index is projected only while it still identifies source files; the worker
5531    /// replaces it with file names before profiling and never exposes it as user data.
5532    /// Unsorted unless `ordered`, as every tool's sample reads it: a sample drawn by
5533    /// position is then the same rows whichever tool drew it. A row range is the one
5534    /// scope whose meaning is the order on screen.
5535    pub(crate) fn data_quality_scan(
5536        &self,
5537        ordered: bool,
5538    ) -> (LazyFrame, Option<crate::data_quality::QualitySourceContext>) {
5539        let known_files =
5540            !self.drift_files.is_empty() && self.drift_files.len() == self.drift_file_starts.len();
5541        let source = if self.can_name_source_files() {
5542            Some(crate::data_quality::QualitySourceContext {
5543                row_index_column: crate::schema_union::DRIFT_COLUMN.to_string(),
5544                ..self.quality_source_drift()
5545            })
5546        } else if self.is_pristine() && known_files {
5547            Some(crate::data_quality::QualitySourceContext {
5548                row_index_column: "__datui_quality_row".to_string(),
5549                ..self.quality_source_drift()
5550            })
5551        } else {
5552            None
5553        };
5554        let mut expressions = self.binary_stub_exprs();
5555        if self.can_name_source_files() {
5556            expressions.push(col(crate::schema_union::DRIFT_COLUMN));
5557        }
5558        let lf = if ordered {
5559            self.lf.clone()
5560        } else {
5561            self.analysis_lf()
5562        }
5563        .select(expressions);
5564        let lf = if source
5565            .as_ref()
5566            .is_some_and(|mapping| mapping.row_index_column == "__datui_quality_row")
5567        {
5568            lf.with_row_index("__datui_quality_row", None)
5569        } else {
5570            lf
5571        };
5572        (lf, source)
5573    }
5574
5575    /// Source-level profiling starts from the loaded scan, independent of the
5576    /// current query, filters, sort, and column projection. Schema projection is
5577    /// deferred to the background worker so opening the plan performs no I/O.
5578    pub(crate) fn data_quality_source_scan(
5579        &self,
5580    ) -> (LazyFrame, Option<crate::data_quality::QualitySourceContext>) {
5581        let known_files =
5582            !self.drift_files.is_empty() && self.drift_files.len() == self.drift_file_starts.len();
5583        let source = if known_files {
5584            Some(crate::data_quality::QualitySourceContext {
5585                row_index_column: if self.drift_at_open {
5586                    crate::schema_union::DRIFT_COLUMN.to_string()
5587                } else {
5588                    "__datui_quality_row".to_string()
5589                },
5590                ..self.quality_source_drift()
5591            })
5592        } else {
5593            None
5594        };
5595        (self.original_lf.clone(), source)
5596    }
5597
5598    pub(crate) fn quality_source_file_count(&self) -> usize {
5599        if self.drift_files.len() == self.drift_file_starts.len() {
5600            self.drift_files.len()
5601        } else {
5602            0
5603        }
5604    }
5605
5606    pub(crate) fn quality_source_file_names(&self) -> &[String] {
5607        if self.quality_source_file_count() > 0 {
5608            &self.drift_files
5609        } else {
5610            &[]
5611        }
5612    }
5613
5614    /// The columns a Data Quality scope reads: the loaded source's for a source
5615    /// scope, the view's otherwise.
5616    pub(crate) fn quality_schema(&self, scope: &crate::data_quality::QualityScope) -> &Schema {
5617        if scope.uses_source() {
5618            &self.original_schema
5619        } else {
5620            &self.schema
5621        }
5622    }
5623
5624    pub(crate) fn quality_temporal_columns(
5625        &self,
5626        scope: &crate::data_quality::QualityScope,
5627    ) -> Vec<String> {
5628        self.quality_schema(scope)
5629            .iter()
5630            .filter(|(name, dtype)| {
5631                name.as_str() != crate::schema_union::DRIFT_COLUMN && dtype.is_temporal()
5632            })
5633            .map(|(name, _)| name.to_string())
5634            .collect()
5635    }
5636
5637    /// The scope's text columns: the ones Data Quality Setup can read as time
5638    /// through a format.
5639    pub(crate) fn quality_text_columns(
5640        &self,
5641        scope: &crate::data_quality::QualityScope,
5642    ) -> Vec<String> {
5643        self.quality_schema(scope)
5644            .iter()
5645            .filter(|(name, dtype)| {
5646                name.as_str() != crate::schema_union::DRIFT_COLUMN
5647                    && matches!(dtype, DataType::String | DataType::Categorical(..))
5648            })
5649            .map(|(name, _)| name.to_string())
5650            .collect()
5651    }
5652
5653    /// Build a temporary filtered table without changing the current pipeline. The
5654    /// caller keeps this state to restore its query, filters, sort, and buffer.
5655    /// A table of rows already read: an analysis's sample, shown in the table viewer
5656    /// with everything it offers (sort, filter, query, copy, export). The rows are in
5657    /// memory, so nothing here reads the source again.
5658    pub(crate) fn sample_view(&self, df: DataFrame) -> Result<Self> {
5659        let options = crate::OpenOptions {
5660            pages_lookahead: Some(self.pages_lookahead),
5661            pages_lookback: Some(self.pages_lookback),
5662            max_buffered_rows: Some(self.max_buffered_rows),
5663            max_buffered_mb: Some(self.max_buffered_mb),
5664            row_numbers: self.row_numbers,
5665            row_start_index: self.row_start_index,
5666            polars_streaming: self.polars_streaming,
5667            ..crate::OpenOptions::default()
5668        };
5669        let schema = df.schema().clone();
5670        let mut view = Self::from_schema_and_lazyframe(schema, df.lazy(), &options, None)?;
5671        view.visible_rows = self.visible_rows;
5672        Ok(view)
5673    }
5674
5675    pub(crate) fn quality_evidence_view(
5676        &self,
5677        scope: &crate::data_quality::QualityScope,
5678        predicate: Expr,
5679    ) -> Result<Self> {
5680        let options = crate::OpenOptions {
5681            pages_lookahead: Some(self.pages_lookahead),
5682            pages_lookback: Some(self.pages_lookback),
5683            max_buffered_rows: Some(self.max_buffered_rows),
5684            max_buffered_mb: Some(self.max_buffered_mb),
5685            row_numbers: self.row_numbers,
5686            row_start_index: self.row_start_index,
5687            polars_streaming: self.polars_streaming,
5688            ..crate::OpenOptions::default()
5689        };
5690        let (lf, schema) = self.quality_scope_frame(scope)?;
5691        let mut view = Self::from_schema_and_lazyframe(
5692            schema,
5693            lf.filter(predicate),
5694            &options,
5695            self.partition_columns.clone(),
5696        )?;
5697        if !scope.uses_source() {
5698            view.column_order = self.column_order.clone();
5699            view.locked_columns_count = self.locked_columns_count;
5700        }
5701        view.visible_rows = self.visible_rows;
5702        view.remote_source = self.remote_source;
5703        // It scans the same file, and a view captured from it must be refused too.
5704        view.decompress_temp_file = self.decompress_temp_file.clone();
5705        view.download = self.download.clone();
5706        view.converted = self.converted.clone();
5707        Ok(view)
5708    }
5709
5710    /// The rows of a Data Quality scope as a lazy frame, and their schema: the
5711    /// source's for a source scope, the view's otherwise. Nothing is read here.
5712    pub(crate) fn quality_scope_frame(
5713        &self,
5714        scope: &crate::data_quality::QualityScope,
5715    ) -> Result<(LazyFrame, Arc<Schema>)> {
5716        Ok(if scope.uses_source() {
5717            let mut lf = self.query_source();
5718            let source = if matches!(scope, crate::data_quality::QualityScope::SourceFiles(_)) {
5719                lf = lf.with_row_index("__datui_quality_row", None);
5720                Some(crate::data_quality::QualitySourceContext {
5721                    row_index_column: "__datui_quality_row".to_string(),
5722                    ..self.quality_source_drift()
5723                })
5724            } else {
5725                None
5726            };
5727            let lf = crate::data_quality::apply_quality_scope(lf, scope, source.as_ref())?;
5728            let lf = if source.is_some() {
5729                lf.drop(by_name(["__datui_quality_row"], false, false))
5730            } else {
5731                lf
5732            };
5733            (lf, self.original_schema.clone())
5734        } else {
5735            (
5736                crate::data_quality::apply_quality_scope(self.visible_lf(), scope, None)?,
5737                self.schema.clone(),
5738            )
5739        })
5740    }
5741
5742    pub fn prepare_async_collect(
5743        &mut self,
5744        num_rows_override: Option<usize>,
5745    ) -> Option<CollectRequest> {
5746        if self.visible_rows > 0 {
5747            self.proximity_threshold = self.proximity();
5748        }
5749
5750        if let Some(n) = num_rows_override {
5751            self.num_rows = n;
5752            self.num_rows_valid = true;
5753        }
5754
5755        // `bound` is the exact total when known, or `usize::MAX` while the background
5756        // `len()` is still running. Using it instead of `self.num_rows` lets us plan a
5757        // top-of-data window for first paint without waiting for the count. See
5758        // `num_rows_bound`.
5759        let count_known = self.num_rows_valid;
5760        let bound = self.num_rows_bound();
5761
5762        if count_known {
5763            if self.num_rows > 0 {
5764                let max_start = self.num_rows.saturating_sub(1);
5765                if self.start_row > max_start {
5766                    self.start_row = max_start;
5767                }
5768            } else {
5769                // Confirmed-empty dataset: clear everything.
5770                self.start_row = 0;
5771                self.buffered_start_row = 0;
5772                self.buffered_end_row = 0;
5773                self.buffered_df = None;
5774                self.df = None;
5775                self.locked_df = None;
5776                return None;
5777            }
5778        }
5779
5780        let view_start = self.start_row;
5781        let view_end = self.start_row + self.visible_rows.min(bound - self.start_row);
5782        let within_buffer = view_start >= self.buffered_start_row
5783            && view_end <= self.buffered_end_row
5784            && self.buffered_end_row > 0;
5785
5786        // Compute the buffer range using the same logic as collect().
5787        let (new_buffer_start, new_buffer_end) = if within_buffer {
5788            let dist_to_start = view_start.saturating_sub(self.buffered_start_row);
5789            let dist_to_end = self.buffered_end_row.saturating_sub(view_end);
5790            let needs_expansion_back =
5791                dist_to_start <= self.proximity_threshold && self.buffered_start_row > 0;
5792            let needs_expansion_forward =
5793                dist_to_end <= self.proximity_threshold && self.buffered_end_row < bound;
5794
5795            if !needs_expansion_back && !needs_expansion_forward {
5796                // Buffer is fine, just re-slice display.
5797                (self.buffered_start_row, self.buffered_end_row)
5798            } else {
5799                let mut s = if needs_expansion_back {
5800                    view_start.saturating_sub(self.reach_rows(self.pages_lookback))
5801                } else {
5802                    self.buffered_start_row
5803                };
5804                let mut e = if needs_expansion_forward {
5805                    (view_end + self.reach_rows(self.pages_lookahead)).min(bound)
5806                } else {
5807                    self.buffered_end_row
5808                };
5809                self.fit_window(view_start, view_end, &mut s, &mut e);
5810                (s, e)
5811            }
5812        } else {
5813            let had_buffer = self.buffered_end_row > 0;
5814            let scrolled_past_end = had_buffer && view_start >= self.buffered_end_row;
5815            let scrolled_past_start = had_buffer && view_end <= self.buffered_start_row;
5816            let extend_forward_ok = scrolled_past_end
5817                && (view_start - self.buffered_end_row) <= self.reach_rows(self.pages_lookahead);
5818            let extend_backward_ok = scrolled_past_start
5819                && (self.buffered_start_row - view_end) <= self.reach_rows(self.pages_lookback);
5820
5821            let mut s;
5822            let mut e;
5823            if extend_forward_ok {
5824                s = self.buffered_start_row;
5825                e = (view_end + self.reach_rows(self.pages_lookahead)).min(bound);
5826            } else if extend_backward_ok {
5827                s = view_start.saturating_sub(self.reach_rows(self.pages_lookback));
5828                e = self.buffered_end_row;
5829            } else {
5830                s = view_start.saturating_sub(self.reach_rows(self.pages_lookback));
5831                e = (view_end + self.reach_rows(self.pages_lookahead)).min(bound);
5832                let min_initial_len = self.min_buffer_len();
5833                let current_len = e.saturating_sub(s);
5834                if current_len < min_initial_len {
5835                    let need = min_initial_len.saturating_sub(current_len);
5836                    let can_extend_end = bound.saturating_sub(e);
5837                    let can_extend_start = s;
5838                    if can_extend_end >= need {
5839                        e = (e + need).min(bound);
5840                    } else if can_extend_start >= need {
5841                        s = s.saturating_sub(need);
5842                    } else {
5843                        e = (e + can_extend_end).min(bound);
5844                        s = s.saturating_sub(need.saturating_sub(can_extend_end));
5845                    }
5846                }
5847            }
5848            self.fit_window(view_start, view_end, &mut s, &mut e);
5849            (s, e)
5850        };
5851
5852        let buffer_size = new_buffer_end.saturating_sub(new_buffer_start);
5853        if buffer_size == 0 {
5854            return None;
5855        }
5856        // Already held: the view fits, or fitting the expansion to whole row groups
5857        // gave back the group on hand.
5858        if self.holds_buffer(new_buffer_start, new_buffer_end) {
5859            self.slice_buffer_into_display();
5860            if self.table_state.selected().is_none() {
5861                self.table_state.select(Some(0));
5862            }
5863            return None;
5864        }
5865
5866        let lf = match self.buffer_lf(new_buffer_start, buffer_size) {
5867            Ok(lf) => lf,
5868            Err(e) => {
5869                self.error = Some(e);
5870                return None;
5871            }
5872        };
5873
5874        // When the count isn't known yet, `num_rows` is provisional (the planned end of
5875        // this buffer). `apply_async_collect` keeps `num_rows_valid` false so the
5876        // background `len()` corrects it, unless the short read reveals the true end.
5877        let num_rows = if count_known {
5878            self.num_rows
5879        } else {
5880            new_buffer_end
5881        };
5882        Some(CollectRequest {
5883            lf,
5884            polars_streaming: self.polars_streaming,
5885            buffer_start: new_buffer_start,
5886            buffer_end: new_buffer_end,
5887            num_rows,
5888            count_known,
5889            plan: self.fill_plan(new_buffer_start, new_buffer_end, num_rows, count_known),
5890        })
5891    }
5892
5893    /// How a fill of `[buffer_start, buffer_end)` is to be made the buffer, from what
5894    /// is held and shown now. See [`FillPlan`].
5895    fn fill_plan(
5896        &self,
5897        buffer_start: usize,
5898        buffer_end: usize,
5899        num_rows: usize,
5900        count_known: bool,
5901    ) -> FillPlan {
5902        let held = self
5903            .abuts_buffer(buffer_start, buffer_end.saturating_sub(buffer_start))
5904            .then(|| self.buffered_df.clone())
5905            .flatten()
5906            .map(|df| (df, self.buffered_start_row));
5907        FillPlan {
5908            buffer_start,
5909            buffer_end,
5910            num_rows,
5911            count_known,
5912            indexing: self.indexing().is_some(),
5913            held,
5914            view_start: self.start_row,
5915            view_len: self.visible_rows,
5916            max_rows: self.max_buffered_rows,
5917            max_mb: self.max_buffered_mb,
5918        }
5919    }
5920
5921    /// Apply the result of a background buffer load. The worker has already stitched
5922    /// and cut it ([`FillPlan::fit`]): installing it copies nothing.
5923    pub fn apply_async_collect(&mut self, result: CollectResult) {
5924        let CollectResult {
5925            df,
5926            start,
5927            returned: returned_rows,
5928            bytes_per_row,
5929            buffer_start,
5930            buffer_end,
5931            num_rows,
5932            count_known,
5933            indexing,
5934        } = result;
5935        let requested_rows = buffer_end.saturating_sub(buffer_start);
5936
5937        if count_known {
5938            self.num_rows = num_rows;
5939            self.num_rows_valid = true;
5940        } else if returned_rows < requested_rows
5941            && (buffer_start == 0 || returned_rows > 0)
5942            // Lines still being indexed end where the indexing has got to, not the file.
5943            && !indexing
5944            && self.indexing().is_none()
5945        {
5946            // Short read: the slice ran off the end, so we now know the exact total
5947            // without waiting for the background len() count. A slice deep in the
5948            // frame that found nothing may lie past the data entirely; only the count
5949            // can say where it ends.
5950            self.num_rows = buffer_start + returned_rows;
5951            self.num_rows_valid = true;
5952        } else if !self.num_rows_valid {
5953            // Full buffer with the count still unresolved: render with a provisional
5954            // total (at least this buffer's end) and leave num_rows_valid false so the
5955            // in-flight background len() corrects it via count_landed().
5956            self.num_rows = self.num_rows.max(buffer_end);
5957        }
5958        // else: the background len() already resolved the exact count between this
5959        // buffer being requested and applied — keep it; don't downgrade to provisional.
5960        self.error = None;
5961        self.remember_pristine_count();
5962
5963        if bytes_per_row.is_some() {
5964            self.observed_bytes_per_row = bytes_per_row;
5965        }
5966        // A fill that does not hold the view's first row was planned for rows since
5967        // replaced (a synchronous collect re-planned while it was out, or the view
5968        // jumped past what the cut kept): installing it would draw rows under the wrong
5969        // numbers. Keep what is held and plan again. A fill that holds the first row
5970        // but not the whole view (the terminal grew while it was out) is kept, and the
5971        // rest fetched; a downloaded row group is too costly to throw away for a resize.
5972        // A read that came back short ends the data, so a view past it is shown by the
5973        // rows kept up to that end, and only by them: a cut may have dropped the end.
5974        let end = start + df.height();
5975        let view_end = self.start_row + self.visible_rows.max(1);
5976        let reaches_end = end >= buffer_start + returned_rows;
5977        let shows_view = start <= self.start_row
5978            && (self.start_row < end || (returned_rows < requested_rows && reaches_end));
5979        if !shows_view {
5980            self.needs_recollect = true;
5981            return;
5982        }
5983        self.release_display_buffer();
5984        self.buffered_start_row = start;
5985        self.buffered_end_row = end;
5986        self.buffered_df = Some(df);
5987        // Slice the buffered DataFrame into display DataFrames (locked + scroll columns).
5988        self.slice_buffer_into_display();
5989        if self.table_state.selected().is_none() {
5990            self.table_state.select(Some(0));
5991        }
5992        if view_end > end && end < self.num_rows {
5993            self.needs_recollect = true;
5994        }
5995    }
5996
5997    /// True when `rows` rows fetched from `start` run on from the rows on hand or up to
5998    /// them, so a fill of them is planned to be stitched on (see [`FillPlan`]).
5999    fn abuts_buffer(&self, start: usize, rows: usize) -> bool {
6000        self.stitches_buffer()
6001            && (start == self.buffered_end_row || start + rows == self.buffered_start_row)
6002    }
6003
6004    /// Invalidate num_rows cache when lf is mutated. Takes a fresh `len_generation` so any
6005    /// in-flight background count for the previous `lf` is recognized as stale. Also drops
6006    /// the cheap Parquet-footer count source: once `lf` carries a filter/query/group, the
6007    /// row count no longer equals the sum of file footers.
6008    ///
6009    /// A view of `sample`, drawn from `source` into `rows`, whose rows have the
6010    /// columns of `schema`. It starts empty and takes rows with
6011    /// [`Self::sample_grew`]. `through` when the sample was drawn from the view's
6012    /// query or filters, rather than the source under them.
6013    pub(crate) fn sampled_from(
6014        source: DataTableState,
6015        sample: crate::sampling::Sample,
6016        schema: &Schema,
6017        rows: Arc<crate::table_sample::SampleRows>,
6018        through: bool,
6019        path: Option<crate::table_sample::DrawPath>,
6020    ) -> Result<Self> {
6021        let mut view = source.sample_view(DataFrame::empty_with_schema(schema))?;
6022        let frame = scanned_frame(&view.original_lf)
6023            .ok_or_else(|| color_eyre::eyre::eyre!("a sample's frame has no rows to scan"))?;
6024        view.sampled = Some(Box::new(Sampled {
6025            source: Box::new(source),
6026            sample,
6027            rows,
6028            frame,
6029            through,
6030            drawn: None,
6031            path,
6032        }));
6033        Ok(view)
6034    }
6035
6036    /// The view's sample, while it has one.
6037    pub fn sampled(&self) -> Option<&Sampled> {
6038        self.sampled.as_deref()
6039    }
6040
6041    /// The view the sample was drawn from, or this one when it has none: where a new
6042    /// sample is drawn from.
6043    pub fn unsampled(&self) -> &DataTableState {
6044        self.sampled
6045            .as_ref()
6046            .map_or(self, |sampled| sampled.source.as_ref())
6047    }
6048
6049    /// The view the sample was drawn from, putting the sample down; `self` when it
6050    /// has none.
6051    pub(crate) fn into_unsampled(mut self) -> DataTableState {
6052        match self.sampled.take() {
6053            Some(sampled) => *sampled.source,
6054            None => self,
6055        }
6056    }
6057
6058    /// Take the chunks the draw kept since the last call: every frame reads them, so
6059    /// the query, filters and sort run over them too. `None` when there were none;
6060    /// otherwise whether the rows on hand still stand. The view stays where it is,
6061    /// and the rows on hand stand while nothing reorders them, since the new rows
6062    /// come after them.
6063    pub(crate) fn sample_grew(&mut self) -> Option<bool> {
6064        let sampled = self.sampled.as_ref()?;
6065        let chunks = sampled.rows.take_new();
6066        if chunks.is_empty() {
6067            return None;
6068        }
6069        // On the same buffers: each column takes the chunks' arrays, nothing copied.
6070        let mut frame = (*sampled.frame).clone();
6071        for chunk in &chunks {
6072            frame.vstack_mut(chunk).ok()?;
6073        }
6074        Some(self.rebind_sample(Arc::new(frame), false))
6075    }
6076
6077    /// The draw ended, having read what `drawn` says: the rows go into the order the
6078    /// source holds them, once.
6079    pub(crate) fn sample_drawn(&mut self, drawn: crate::table_sample::Drawn) {
6080        let Some(sampled) = self.sampled.as_mut() else {
6081            return;
6082        };
6083        // One chunk per column from here: the many the draw left would slow every
6084        // read, and the chunks are let go so the rows are held once.
6085        let ordered = sampled.rows.take_in_source_order().ok().flatten();
6086        // A seeded read of one file needs no path; it was not one, then.
6087        sampled.path = drawn.path;
6088        sampled.drawn = Some(drawn);
6089        if let Some(frame) = ordered {
6090            self.rebind_sample(Arc::new(frame), true);
6091        }
6092    }
6093
6094    /// Every frame scans `frame` in place of the sample's last one. `reordered` when
6095    /// the rows already shown changed places. Returns whether the rows on hand stand.
6096    fn rebind_sample(&mut self, frame: Arc<DataFrame>, reordered: bool) -> bool {
6097        let Some(old) = self.sampled.as_ref().map(|sampled| sampled.frame.clone()) else {
6098            return false;
6099        };
6100        let rows_stand = !reordered
6101            && self.sort_columns.is_empty()
6102            && self.sort_ascending
6103            && self.scan_is_the_root();
6104        let rows = frame.height();
6105        self.each_frame(|lf| crate::table_sample::rebind(&mut lf.logical_plan, &old, &frame));
6106        if let Some(sampled) = self.sampled.as_mut() {
6107            sampled.frame = frame;
6108        }
6109        self.invalidate_num_rows();
6110        if self.is_pristine() {
6111            self.set_num_rows(rows);
6112        } else if self.scan_is_the_root() {
6113            self.pristine_rows = Some(rows);
6114        }
6115        if !rows_stand {
6116            self.drop_buffer();
6117        }
6118        self.needs_recollect = true;
6119        rows_stand
6120    }
6121
6122    /// Bytes a row of a sample of this view takes: of the source's columns when it
6123    /// is drawn from the source, of the view's when from the view, every column of
6124    /// either, shown or not.
6125    pub(crate) fn sample_row_bytes(&self, from_source: bool) -> usize {
6126        let schema = if from_source {
6127            &self.original_schema
6128        } else {
6129            &self.schema
6130        };
6131        let columns: Vec<String> = schema
6132            .iter_names()
6133            .filter(|name| name.as_str() != crate::schema_union::DRIFT_COLUMN)
6134            .map(|name| name.to_string())
6135            .collect();
6136        // What the table measured, when it measured these columns.
6137        if !from_source && columns.len() == self.column_order.len() {
6138            return self.bytes_per_row();
6139        }
6140        estimate_bytes_per_row(schema, &columns, &self.column_bytes)
6141    }
6142
6143    /// Draws from the shared counter rather than incrementing, so a mutation here can
6144    /// never land on the value a later dataset is about to be seeded with.
6145    pub(crate) fn invalidate_num_rows(&mut self) {
6146        self.num_rows_valid = false;
6147        self.len_generation = next_len_generation();
6148    }
6149
6150    /// True while `lf` is the data as loaded: no sidebar filter or sort, no query in
6151    /// any bar, no pivot or melt, no drill-down. Derived rather than kept, so clearing
6152    /// the filters or un-sorting makes the frame pristine again by itself.
6153    /// Whether the table shows other rows than its source holds: a filter, a query, a
6154    /// reshape or a drill. A sort alone reorders the same rows.
6155    pub(crate) fn changes_rows(&self) -> bool {
6156        !self.filters.is_empty()
6157            || !self.active_query.is_empty()
6158            || !self.active_sql_query.is_empty()
6159            || !self.active_fuzzy_query.is_empty()
6160            || self.reshaped_lf.is_some()
6161            || self.grouped.is_some()
6162            || self.drilled_down_group_index.is_some()
6163    }
6164
6165    /// Whether the view may still take its rows straight from the scan: nothing
6166    /// that picks rows (a filter, a search, a reshape, a group, a drill). A query
6167    /// may only choose columns, so it may.
6168    pub(crate) fn may_keep_scan_rows(&self) -> bool {
6169        self.filters.is_empty()
6170            && self.active_fuzzy_query.is_empty()
6171            && self.reshaped_lf.is_none()
6172            && self.grouped.is_none()
6173            && self.drilled_down_group_index.is_none()
6174    }
6175
6176    fn is_pristine(&self) -> bool {
6177        self.column_changes.is_empty()
6178            && self.filters.is_empty()
6179            && self.sort_columns.is_empty()
6180            && self.sort_ascending
6181            && self.active_query.is_empty()
6182            && self.active_sql_query.is_empty()
6183            && self.active_fuzzy_query.is_empty()
6184            && self.reshaped_lf.is_none()
6185            && self.grouped.is_none()
6186            && self.drilled_down_group_index.is_none()
6187    }
6188
6189    /// Whether the frame on screen still grows from the dataset's own scan.
6190    ///
6191    /// A filter and a sort do: `apply_transformations` rebuilds them over whatever the
6192    /// root is, so widening the root under them is exactly what should happen. A query,
6193    /// a SQL statement, a fuzzy search, a pivot, a melt and a drill-down do not — each
6194    /// makes its own result the root, with its own columns, and replacing the root
6195    /// underneath one leaves the view naming columns the frame no longer has.
6196    ///
6197    /// `grouped` and `drilled_down_group_index` are set together by a drill down and
6198    /// cleared together by a drill up, so asking both is belt and braces — kept because
6199    /// what they guard is the frame being rebuilt under a view of one group of it.
6200    /// The frame an analysis reads: the view as filtered and queried, without its
6201    /// order. No statistic depends on the order, and a sort is the one step that makes
6202    /// a sampled read of a huge table read all of it.
6203    pub fn analysis_lf(&self) -> LazyFrame {
6204        self.unsorted_lf.clone().unwrap_or_else(|| self.lf.clone())
6205    }
6206
6207    /// What the Pivot & Melt builder previews a few rows of: the view as the user
6208    /// sees its columns, without its order. A sort would make the head of a large
6209    /// table a read of all of it.
6210    pub fn preview_lf(&self) -> LazyFrame {
6211        Self::without_drift(self.analysis_lf())
6212    }
6213
6214    /// Whether the view has a sort, which [`Self::preview_lf`] leaves out.
6215    pub fn is_sorted(&self) -> bool {
6216        self.unsorted_lf.is_some()
6217    }
6218
6219    pub fn scan_is_the_root(&self) -> bool {
6220        self.active_query.is_empty()
6221            && self.active_sql_query.is_empty()
6222            && self.active_fuzzy_query.is_empty()
6223            && self.reshaped_lf.is_none()
6224            && self.grouped.is_none()
6225            && self.drilled_down_group_index.is_none()
6226    }
6227
6228    /// A pristine scan's count is its footer's: take it back, without a `len()`, when
6229    /// the frame is the scan as loaded again.
6230    fn restore_footer_count(&mut self) {
6231        if !self.is_pristine() {
6232            return;
6233        }
6234        if let Some(total) = self.row_group_offsets.as_ref().and_then(|o| o.last()) {
6235            self.set_num_rows(*total);
6236        }
6237    }
6238
6239    /// What finding and reading this dataset cost.
6240    pub fn measurements(&self) -> &Arc<crate::measurements::Meter> {
6241        &self.measurements
6242    }
6243
6244    /// The directory whose Parquet footers can be summed for an exact row count, if the
6245    /// current `lf` still allows it. See `parquet_count_dir`.
6246    pub fn parquet_count_dir(&self) -> Option<PathBuf> {
6247        self.parquet_count_dir
6248            .clone()
6249            .filter(|_| self.is_pristine())
6250    }
6251
6252    /// Current count generation. A background `len()` task captures this; its result is
6253    /// only applied if the generation still matches (i.e. the data hasn't changed since).
6254    pub fn len_generation(&self) -> u64 {
6255        self.len_generation
6256    }
6257
6258    /// Returns the cached row count when valid (same value shown in the control bar). Use this to
6259    /// avoid an extra full scan for analysis/describe when the table has already been collected.
6260    pub fn num_rows_if_valid(&self) -> Option<usize> {
6261        if self.num_rows_valid {
6262            Some(self.num_rows)
6263        } else {
6264            None
6265        }
6266    }
6267
6268    /// True when num_rows reflects the current `lf`. Used by App to decide whether
6269    /// to dispatch a background len() query before planning the buffer collect.
6270    pub fn is_num_rows_valid(&self) -> bool {
6271        self.num_rows_valid
6272    }
6273
6274    /// Effective upper bound on row indices for buffer planning. When the exact count
6275    /// is known, that's `num_rows`; when it isn't yet (first paint before the background
6276    /// `len()` resolves), treat the dataset as unbounded so we plan a top-of-data window
6277    /// (`slice(0, N)`) instead of clamping everything to a stale/zero count.
6278    fn num_rows_bound(&self) -> usize {
6279        if self.num_rows_valid {
6280            self.num_rows
6281        } else {
6282            usize::MAX
6283        }
6284    }
6285
6286    /// Apply a row count computed in the background (so prepare_async_collect doesn't
6287    /// have to fall back to a blocking len() on the UI thread).
6288    fn set_num_rows(&mut self, n: usize) {
6289        self.num_rows = n;
6290        self.num_rows_valid = true;
6291        self.remember_pristine_count();
6292        // A view past the end of a frame that turned out smaller comes back to it.
6293        if self.start_row > 0 && self.start_row >= n {
6294            self.start_row = n.saturating_sub(self.visible_rows);
6295            self.needs_recollect = true;
6296        }
6297    }
6298
6299    /// Keep the pristine frame's count for the control bar's "417 of 1,000". Only a
6300    /// count already resolved for the data as loaded — never a reason to run one.
6301    fn remember_pristine_count(&mut self) {
6302        if self.num_rows_valid && self.error.is_none() && self.is_pristine() {
6303            self.pristine_rows = Some(self.num_rows);
6304        }
6305    }
6306
6307    /// The dataset's full row count for the control bar, when the rows on screen are a
6308    /// subset of it: a sidebar filter, a query in any bar or a drill-down is active and
6309    /// the count from before it was applied is known. A pivot or melt makes rows that
6310    /// are not the dataset's, so the comparison would mislead and none is offered.
6311    /// Cheap by construction: it only reads what a pristine collect already knew.
6312    pub fn total_rows_when_subset(&self) -> Option<usize> {
6313        let subsetting = !self.filters.is_empty()
6314            || !self.active_query.is_empty()
6315            || !self.active_sql_query.is_empty()
6316            || !self.active_fuzzy_query.is_empty()
6317            || self.drilled_down_group_index.is_some();
6318        if subsetting && self.reshaped_lf.is_none() {
6319            self.pristine_rows
6320        } else {
6321            None
6322        }
6323    }
6324
6325    /// Clone of the LazyFrame for off-thread queries (e.g. background len()).
6326    pub fn lf_clone(&self) -> LazyFrame {
6327        self.lf.clone()
6328    }
6329
6330    /// Whether the current LazyFrame should use Polars streaming engine.
6331    pub fn polars_streaming_enabled(&self) -> bool {
6332        self.polars_streaming
6333    }
6334
6335    /// True when a fill that runs on from the rows on hand, or up to them, will be
6336    /// stitched on to them rather than replace them. See [`FillPlan`].
6337    pub(crate) fn stitches_buffer(&self) -> bool {
6338        self.remote_window() && self.buffer_on_hand()
6339    }
6340
6341    /// The rows on hand and the view row the first of them is, when every row of
6342    /// the buffered range is: what a find lights up as it is typed, without a read.
6343    pub(crate) fn rows_on_hand(&self) -> Option<(&DataFrame, usize)> {
6344        self.buffered_df
6345            .as_ref()
6346            .filter(|_| self.buffer_on_hand())
6347            .map(|df| (df, self.buffered_start_row))
6348    }
6349
6350    /// True when every row of the buffered range is on hand.
6351    fn buffer_on_hand(&self) -> bool {
6352        self.buffered_end_row > self.buffered_start_row
6353            && self
6354                .buffered_df
6355                .as_ref()
6356                .is_some_and(|b| b.height() == self.buffered_end_row - self.buffered_start_row)
6357    }
6358
6359    /// True when the rows on hand include `[start, end)`. The buffer is then cut down
6360    /// to that range, so a row group stitched on to cross into it is let go once the
6361    /// view has left it, rather than fetched again when the view comes back.
6362    fn holds_buffer(&mut self, start: usize, end: usize) -> bool {
6363        if !self.buffer_on_hand()
6364            || start < self.buffered_start_row
6365            || end > self.buffered_end_row
6366            || end <= start
6367        {
6368            return false;
6369        }
6370        if (start, end) != (self.buffered_start_row, self.buffered_end_row) {
6371            let offset = start - self.buffered_start_row;
6372            // Trimmed so the rows let go are freed rather than kept behind a slice; the
6373            // display frames alias the old buffer and go with it.
6374            self.locked_df = None;
6375            self.df = None;
6376            self.buffered_df = self
6377                .buffered_df
6378                .take()
6379                .map(|b| trim_rows(b, offset, end - start, None));
6380            self.buffered_start_row = start;
6381            self.buffered_end_row = end;
6382        }
6383        true
6384    }
6385
6386    /// Start row of the currently buffered range.
6387    pub fn buffered_start(&self) -> usize {
6388        self.buffered_start_row
6389    }
6390
6391    /// End row (exclusive) of the currently buffered range.
6392    pub fn buffered_end(&self) -> usize {
6393        self.buffered_end_row
6394    }
6395
6396    /// True for a scan of an object store in place.
6397    ///
6398    /// Polars fetches a Parquet row group whole for any slice that touches it and keeps
6399    /// nothing between collects, so the small, proximity-driven refills that suit a
6400    /// local file each download the same row group again: paging through one row group
6401    /// cost a fetch of it every few pages. A remote buffer is planned as a single window
6402    /// of `max_buffered_rows` around the view instead. Scrolling inside it costs
6403    /// nothing; leaving it, or a jump, costs one fetch.
6404    pub fn is_remote_source(&self) -> bool {
6405        self.remote_source
6406    }
6407
6408    /// The schema of the data as loaded, before any query or reshape: what a view's
6409    /// settings run on, and so what its schema rule records and matches.
6410    pub fn source_schema(&self) -> &Arc<Schema> {
6411        &self.original_schema
6412    }
6413
6414    /// Record the row groups of a remote Parquet object, `rows` in each, so a buffer
6415    /// fill is planned as whole groups (see `align_to_row_groups`). Also the row count.
6416    fn record_row_groups(&mut self, rows: &[usize]) {
6417        let mut offsets = Vec::with_capacity(rows.len() + 1);
6418        offsets.push(0);
6419        for n in rows {
6420            offsets.push(offsets.last().unwrap_or(&0) + n);
6421        }
6422        self.set_num_rows(*offsets.last().unwrap_or(&0));
6423        self.row_group_offsets = Some(offsets);
6424    }
6425
6426    /// Whether a Data Quality run over `scope` reads every row and every byte-bearing
6427    /// column of the source: the case where a copy of the whole objects costs no more
6428    /// than one of its passes. A filter or a hidden column may let a pass read less
6429    /// than the objects, and a binary column is never read at all.
6430    pub(crate) fn quality_reads_whole_source(
6431        &self,
6432        scope: &crate::data_quality::QualityScope,
6433    ) -> bool {
6434        use crate::data_quality::QualityScope;
6435        let columns = || {
6436            self.original_schema
6437                .iter()
6438                .filter(|(name, _)| name.as_str() != crate::schema_union::DRIFT_COLUMN)
6439        };
6440        if columns().any(|(_, dtype)| matches!(dtype, DataType::Binary)) {
6441            return false;
6442        }
6443        match scope {
6444            QualityScope::WholeSource => true,
6445            QualityScope::CurrentView => {
6446                let shown = self
6447                    .column_order
6448                    .iter()
6449                    .map(String::as_str)
6450                    .collect::<HashSet<_>>();
6451                !self.changes_rows() && columns().all(|(name, _)| shown.contains(name.as_str()))
6452            }
6453            _ => false,
6454        }
6455    }
6456
6457    /// Each remote object the dataset reads, in scan order: every file of a remote
6458    /// dataset, or the one object. `None` in place of one the open did not size.
6459    pub(crate) fn each_remote_object(
6460        &self,
6461    ) -> Option<Box<dyn Iterator<Item = Option<&RemoteObject>> + '_>> {
6462        let objects = self.remote_objects.as_ref()?;
6463        Some(match &self.remote_files {
6464            Some(remote) => Box::new(remote.urls.iter().map(|url| objects.get(url))),
6465            None => Box::new(objects.values().map(Some)),
6466        })
6467    }
6468
6469    /// The remote objects this dataset reads. `None` when any is unknown.
6470    pub(crate) fn remote_objects(&self) -> Option<Vec<RemoteObject>> {
6471        let objects = self
6472            .each_remote_object()?
6473            .map(|object| object.cloned())
6474            .collect::<Option<Vec<_>>>()?;
6475        (!objects.is_empty()).then_some(objects)
6476    }
6477
6478    /// The bytes and count of [`Self::remote_objects`], without copying them out:
6479    /// Setup asks on every frame.
6480    pub(crate) fn remote_objects_size(&self) -> Option<(u64, usize)> {
6481        let (bytes, count) = self
6482            .each_remote_object()?
6483            .try_fold((0u64, 0usize), |(bytes, count), object| {
6484                object.map(|object| (bytes + object.size, count + 1))
6485            })?;
6486        (count > 0).then_some((bytes, count))
6487    }
6488
6489    /// Record what the footers said about the dataset's columns. See `DatasetSchema`.
6490    /// `file_rows` is each file's row count, in scan order, and empty when they are not
6491    /// all known — the same condition under which the scan numbers its rows.
6492    fn record_dataset_schema(
6493        &mut self,
6494        schema: crate::schema_union::DatasetSchema,
6495        file_rows: &[usize],
6496        files: &[String],
6497    ) {
6498        self.drift_files = files.to_vec();
6499        // The scan numbers rows exactly when the files differ and every one is counted.
6500        self.drift_column_present = schema.drifts() && file_rows.len() == schema.file_group.len();
6501        self.drift_groups = Arc::new(schema.groups.clone());
6502        self.drift_file_group = schema.file_group.clone();
6503        self.drift_file_starts = Vec::with_capacity(file_rows.len());
6504        let mut row = 0usize;
6505        for rows in file_rows {
6506            self.drift_file_starts.push(row);
6507            row += rows;
6508        }
6509        self.drift_dataset_rows = row;
6510        self.drift_at_open = self.drift_column_present;
6511        self.groups_at_open = self.drift_groups.clone();
6512        self.notes = Self::notes_datui_can_act_on(&schema, self.drift_column_present);
6513        self.notes_at_open = self.notes.clone();
6514        self.notes_seen = false;
6515        self.read_as_text = Vec::new();
6516        self.dataset_at_open = Some(schema.clone());
6517        self.dataset_schema = Some(schema);
6518    }
6519
6520    /// The pass that is still reading this dataset's footers, if one is.
6521    pub fn footers_pending(&self) -> Option<FootersJoin> {
6522        self.footers_pending.clone()
6523    }
6524
6525    /// Whether *this frame's* row count is already on its way.
6526    ///
6527    /// The frame matters: what the pass is bringing is the dataset's count, which is
6528    /// not the count of a query's result. Asking this about the wrong frame is how a
6529    /// count nobody else was going to take gets declined.
6530    ///
6531    /// A staged open is still reading every footer, and those footers hold the count.
6532    /// Asking for it separately would read all of them a second time, so the dataset
6533    /// says it will have one shortly and the caller does not start a count of its own.
6534    pub fn counts_itself_later(&self) -> bool {
6535        // Lines still being indexed: any frame's count is of the lines so far, and the
6536        // indexing is bringing the rest.
6537        if self.indexing().is_some() && !self.num_rows_valid {
6538            return true;
6539        }
6540        // Only while it does not have one, and only while the frame is the scan. What
6541        // the pass is bringing is the *dataset's* count; a query's result has a count
6542        // of its own that nobody else is going to take. Declining it there means the
6543        // row count spins for as long as the query is open and `End` says it is
6544        // counting while nothing is — and it costs nothing to take, because a frame
6545        // that is not the scan does not read footers for it either.
6546        self.footers_pending.is_some() && !self.num_rows_valid && self.is_pristine()
6547    }
6548
6549    /// The lines being indexed behind the first rows, if they still are.
6550    pub fn indexing(&self) -> Option<&Arc<crate::lines::Lines>> {
6551        // Asked of the lines, so a dataset set aside while they finished (the quality
6552        // evidence view) does not wait for them for good.
6553        self.indexing.as_ref().filter(|lines| lines.indexing())
6554    }
6555
6556    /// The lines this dataset opened from in part, until it has been told they are all
6557    /// in, though their indexing is paused: what an indexing thread works on.
6558    pub fn lines_to_index(&self) -> Option<&Arc<crate::lines::Lines>> {
6559        self.indexing.as_ref()
6560    }
6561
6562    /// The dataset's row count from a sample of its footers, while the frame is the
6563    /// dataset as loaded and its count is not known. `pass` is the estimate of the
6564    /// footer pass still reading, which the dataset has not been given yet.
6565    pub fn row_estimate(
6566        &self,
6567        pass: Option<crate::schema_union::RowEstimate>,
6568    ) -> Option<crate::schema_union::RowEstimate> {
6569        if self.num_rows_valid || !self.is_pristine() {
6570            return None;
6571        }
6572        self.row_estimate
6573            .or_else(|| pass.filter(|_| self.footers_pending.is_some()))
6574    }
6575
6576    /// Where each of the dataset's files starts in the view, with the total last: while
6577    /// the view keeps the dataset's rows and every file's rows are known.
6578    pub fn file_row_starts(&self) -> Option<Vec<usize>> {
6579        if self.changes_rows() {
6580            return None;
6581        }
6582        self.remote_files.as_ref()?.offsets.clone()
6583    }
6584
6585    /// How many files a count of the dataset reads footers of, when it reads them.
6586    pub fn files_to_count(&self) -> Option<usize> {
6587        self.remote_files
6588            .as_ref()
6589            .filter(|f| f.offsets.is_none())
6590            .map(|f| f.urls.len())
6591    }
6592
6593    /// Whether `#` is on for this dataset when the config leaves it to the format:
6594    /// text and logs, whose rows carry their place in the file.
6595    pub fn numbered_by_default(&self) -> bool {
6596        matches!(
6597            self.read_as,
6598            Some(crate::FileFormat::Text | crate::FileFormat::Journal)
6599        )
6600    }
6601
6602    /// Whether `#` is on and numbers the rows by their place in the view, because
6603    /// the view's rows do not carry their place in the source: a sorted or filtered
6604    /// view of data in a store, of many files, or too large to number.
6605    pub fn row_numbers_count_the_view(&self) -> bool {
6606        self.row_numbers
6607            && !self.carries_source_rows()
6608            && self.scan_is_the_root()
6609            && (!self.filters.is_empty() || !self.sort_columns.is_empty() || !self.sort_ascending)
6610    }
6611
6612    /// Every line is indexed, `rows` of them: the count of the lines in order, and the
6613    /// notes that say what the whole file holds. The frames already read every line
6614    /// (their height waits for the indexing), so nothing read through them is stale.
6615    /// Returns whether the dataset was waiting for them.
6616    pub(crate) fn lines_indexed(&mut self, rows: usize) -> bool {
6617        let Some(lines) = self.indexing.take() else {
6618            return false;
6619        };
6620        let notes = crate::lines::notes(&lines, self.indexing_guessed);
6621        let opened = std::mem::take(&mut self.indexing_notes);
6622        self.open_notes.retain(|n| !opened.contains(n));
6623        self.open_notes.extend(notes);
6624        // A file that shrank has no count to give: the lines so far are not all of it.
6625        if lines.shrank() {
6626            self.open_notes.push(crate::text_formats::note(
6627                crate::lines::SHRANK.to_string(),
6628                "the file".to_string(),
6629            ));
6630            return true;
6631        }
6632        // The "of" in `417 of 1,000` under a filter.
6633        self.pristine_rows = Some(rows);
6634        if self.is_pristine() {
6635            self.set_num_rows(rows);
6636        }
6637        true
6638    }
6639
6640    /// Give up on the rest of the footers: the pass could not read them.
6641    ///
6642    /// The dataset stays as it opened — a working view of it, built from two footers —
6643    /// and stops waiting. That matters beyond the columns: while a pass is pending the
6644    /// dataset declines to count itself, because the pass was going to bring the count
6645    /// with it. One failed pass would otherwise cost it an exact row count, and its
6646    /// windowed reads, for the rest of the session.
6647    pub fn give_up_on_pending_footers(&mut self) {
6648        self.footers_pending = None;
6649    }
6650
6651    /// Every footer's answer, joined to the dataset already on screen.
6652    ///
6653    /// The open painted from the first file and the newest; this is what the rest of
6654    /// them say. Columns only ever join: a name the opening schema did not have goes on
6655    /// the end, and every name already there keeps its place — including the places a
6656    /// user has since moved them to — so nothing moves under the cursor except to make
6657    /// room for what arrived.
6658    ///
6659    /// The frame is rebuilt rather than widened in place, because the scan itself
6660    /// differs: it now knows which files hold a column in a type the dataset cannot
6661    /// keep, and with every file's row count it can number the rows, which is what
6662    /// tells an absent cell from a null.
6663    ///
6664    /// Gives them back as `Err` rather than taking them, while the user is looking at
6665    /// something built on top of the scan instead of the scan itself — see
6666    /// [`Self::scan_is_the_root`]. Rebuilding the root under a query takes away the
6667    /// columns the query named; handing them back lets the caller keep them and offer
6668    /// them again when the view comes back to the data.
6669    pub fn join_dataset_schema(
6670        &mut self,
6671        mut found: FootersFound,
6672    ) -> std::result::Result<(), Box<FootersFound>> {
6673        if !self.scan_is_the_root() {
6674            // The columns must wait; what the footers said about the files need not.
6675            // `record_file_row_groups` keeps the offsets without touching the count of a
6676            // frame that is a query's result rather than the dataset — so letting the
6677            // query go gets the total back without going and fetching it.
6678            // Taken, not borrowed: this runs on every event for as long as the view
6679            // stays off the scan, and applying the same row groups on each keystroke is
6680            // a walk of every file in the dataset for nothing.
6681            let row_groups = std::mem::take(&mut found.row_groups);
6682            if !row_groups.is_empty() {
6683                self.record_file_row_groups(&row_groups);
6684            }
6685            // Boxed because what comes back is most of a dataset's worth of schema, and
6686            // an `Err` that size would be carried by every call that succeeds too.
6687            return Err(Box::new(found));
6688        }
6689        let FootersFound {
6690            dataset,
6691            lf,
6692            file_rows,
6693            files,
6694            row_groups,
6695            remote,
6696            estimate,
6697        } = found;
6698        self.row_estimate = if row_groups.is_empty() {
6699            estimate
6700        } else {
6701            None
6702        };
6703        let (file_rows, files) = (file_rows.as_slice(), files.as_slice());
6704        let known: std::collections::HashSet<&str> =
6705            self.column_order.iter().map(String::as_str).collect();
6706        let joining: Vec<String> = dataset
6707            .schema
6708            .iter_names()
6709            .map(|name| name.to_string())
6710            .filter(|name| {
6711                name != crate::schema_union::DRIFT_COLUMN && !known.contains(name.as_str())
6712            })
6713            .collect();
6714        drop(known);
6715        self.column_order.extend(joining);
6716        // Columns only join — but a name can still go, if the footer that was the only
6717        // evidence for it would not parse this time round. Every read projects
6718        // `column_order`, so a name the new schema does not have is not a missing
6719        // column on screen, it is a scan that cannot run at all.
6720        self.column_order
6721            .retain(|name| dataset.schema.contains(name.as_str()));
6722        let schema = dataset.schema.clone();
6723        // The scan is built at a schema, and the one this dataset opened with has never
6724        // heard of the columns that just arrived. Left in place, the first windowed
6725        // page read asks it for a column it does not have and the table stops showing
6726        // rows at the moment it was supposed to show more of them.
6727        match (remote, self.remote_files.as_mut()) {
6728            (Some(found), Some(remote)) => {
6729                remote.urls = Arc::new(found.urls);
6730                remote.scan = found.scan;
6731                // The counter too, and for the same reason the scan is replaced: it
6732                // answers one entry per file it was given, and the one the dataset
6733                // opened with was given every file listed. Left beside a shorter `urls`
6734                // its answer is dropped on a length check without a word, and the
6735                // dataset spends the rest of the session re-counting itself and never
6736                // reaching an end to jump to.
6737                remote.count = found.count;
6738            }
6739            // A local directory reads by file only once every footer is known.
6740            (Some(found), None) => self.remote_files = Some(found.into()),
6741            (None, _) => {}
6742        }
6743        // Takes the notes, the drift groups and the row starts with it, and clears
6744        // `read_as_text` — sound only because the offer to read a column as text is
6745        // not made until the footers are all in, so there is nothing to clear.
6746        self.record_dataset_schema(dataset, file_rows, files);
6747        self.footers_pending = None;
6748        // The rows on screen were read through the old frame. Dropping the buffer has
6749        // the next collect read them through the new one, at the row the user is still
6750        // sitting on — `start_row` and the column scroll are left exactly as they are.
6751        self.replace_root(lf, schema);
6752        // The joined scan may hold rows the two-footer open never saw, so the count
6753        // remembered for the narrow root no longer describes the dataset.
6754        self.pristine_rows = None;
6755        // Measured on the frame that just went. A dataset that opened two columns wide
6756        // and gained thirty would plan its first page after the join from the two-column
6757        // width, which against a bucket is a read many times the budget the user set.
6758        self.observed_bytes_per_row = None;
6759        // Every file's row groups are known now, so this is the dataset's count. Set
6760        // before the rebuild so the count is in place the moment the frame is, rather
6761        // than for any ordering the lines below depend on.
6762        if !row_groups.is_empty() {
6763            self.record_file_row_groups(&row_groups);
6764        }
6765        // Rebuilt but not read. This runs on the thread drawing the screen, and
6766        // `apply_transformations` ends in a `collect` — against a dataset in a bucket
6767        // that is a page fetched, and with no count yet it is a `len()` over every file
6768        // in the dataset, which is the whole cost this staging exists to avoid. The
6769        // buffer is gone and the caller reads it back off the event loop. This line is
6770        // what keeps the collect off this thread; do not take it away.
6771        self.deferred(Self::apply_transformations);
6772        Ok(())
6773    }
6774
6775    /// The frame as the user sees it: `lf` without the hidden row-index column.
6776    ///
6777    /// Everything that exports, reshapes, groups or analyses the data reads this. The
6778    /// buffer reads `lf` itself and keeps the column, which is how `display_drift`
6779    /// traces a row back to its file; it stays invisible because the display is only
6780    /// ever a projection of `column_order`, which never names it.
6781    pub fn visible_lf(&self) -> LazyFrame {
6782        Self::without_drift(self.lf.clone())
6783    }
6784
6785    /// Whether the frame scans a temporary file this state holds (a decompressed
6786    /// archive). A view captured at exit must not reference it: the file is removed
6787    /// when the last state holding it drops, and the plan would scan a path that no
6788    /// longer exists.
6789    pub fn scans_a_temp_file(&self) -> bool {
6790        self.decompress_temp_file.is_some() || !self.converted.is_empty()
6791    }
6792
6793    /// Whether the frame scans a downloaded remote file, removed when datui lets go of
6794    /// it; see [`Self::scans_a_temp_file`].
6795    pub fn scans_a_download(&self) -> bool {
6796        self.download.is_some()
6797    }
6798
6799    /// How the open reads the data, when an open found it; `None` for a frame handed
6800    /// in whole, such as one from Python.
6801    pub fn read_mode(&self) -> Option<crate::ReadMode> {
6802        self.read_mode
6803    }
6804
6805    /// The format the open read the data as. See [`OpenFacts::read_as`].
6806    pub fn read_as(&self) -> Option<crate::FileFormat> {
6807        self.read_as
6808    }
6809
6810    /// Whether the data was downloaded from a remote source before it was read.
6811    pub fn fetched(&self) -> bool {
6812        self.fetched
6813    }
6814
6815    /// The temporary files this state holds, which an error from reading it may name:
6816    /// a decompressed copy, a download.
6817    pub(crate) fn temp_files(&self) -> Vec<&Path> {
6818        let files = self.decompress_temp_file.iter().map(|file| file.path());
6819        let files = files.chain(self.download.iter().map(|download| download.path()));
6820        let files = files.chain(self.converted.iter().map(|file| file.path()));
6821        files.collect()
6822    }
6823
6824    /// `lf` without the hidden drift column. A non-strict drop, so it is a no-op on a
6825    /// frame that never had one and no caller has to know which it holds.
6826    fn without_drift(lf: LazyFrame) -> LazyFrame {
6827        lf.drop(by_name([crate::schema_union::DRIFT_COLUMN], false, false))
6828    }
6829
6830    /// The frame a query, a SQL statement or a fuzzy search builds on. Never carries
6831    /// the drift column: a query's rows are its own, and its schema becomes the
6832    /// column order, so the column would otherwise become one of the data's.
6833    pub fn query_source(&self) -> LazyFrame {
6834        Self::without_drift(self.original_lf.clone())
6835    }
6836
6837    /// Whether rows still know which file they came from.
6838    pub fn drifts(&self) -> bool {
6839        self.drift_column_present
6840    }
6841
6842    /// What each drift group is missing, for the renderer. Empty when nothing drifts.
6843    pub fn drift_groups(&self) -> Arc<Vec<crate::schema_union::DriftGroup>> {
6844        self.drift_groups.clone()
6845    }
6846
6847    /// Whether an export can name each row's file: the frame has to still carry the
6848    /// scan's row index, and the dataset has to have files to name.
6849    pub fn can_name_source_files(&self) -> bool {
6850        self.drift_column_present
6851            && !self.drift_files.is_empty()
6852            && self.drift_files.len() == self.drift_file_starts.len()
6853    }
6854
6855    /// The rows an export writes, planned and not run: the view, and when
6856    /// `name_files` asks and the dataset can, a column naming each row's file in
6857    /// place of the scan's hidden row index. Otherwise the index is dropped, so
6858    /// datui's own bookkeeping never lands in the user's file.
6859    pub fn export_frame(&self, name_files: bool) -> ExportFrame {
6860        if name_files && self.can_name_source_files() {
6861            ExportFrame {
6862                lf: self.lf.clone(),
6863                files: Some(SourceFiles {
6864                    names: Arc::new(self.drift_files.clone()),
6865                    starts: Arc::new(self.drift_file_starts.clone()),
6866                }),
6867            }
6868        } else {
6869            ExportFrame {
6870                lf: self.visible_lf(),
6871                files: None,
6872            }
6873        }
6874    }
6875
6876    /// What datui noticed about the dataset itself, as its footers were read.
6877    ///
6878    /// Separate from [`Self::notes`] because this is the half that belongs to the
6879    /// data: a snapshot taken to roll a view back has to put back these and not
6880    /// the view's, which describe a filter and sort that the rollback is undoing.
6881    pub fn dataset_notes(&self) -> &[crate::notes::Note] {
6882        &self.notes
6883    }
6884
6885    /// What datui noticed: about the dataset when it opened, then about the view the
6886    /// filter and sort have made of it. Empty when there is nothing to say.
6887    pub fn notes(&self) -> Vec<crate::notes::Note> {
6888        // What the read did first, because it is the frame everything below is about:
6889        // a directory read as CSV with a JSON file left out, or a lake table read as its
6890        // plain files, changes what every other note is a note about. `merged` rather
6891        // than a plain chain, because the open and the footer walk each count the
6892        // files a mixed directory's read passed over, and this is the one place both
6893        // tallies are in hand.
6894        let mut notes = crate::notes::merged(
6895            &self.open_notes,
6896            &self.notes,
6897            &self.view_notes,
6898            self.dataset_schema.as_ref(),
6899        );
6900        // What reading a table in place has found, which may grow after the open.
6901        if let Some(pushdown) = &self.pushdown {
6902            notes.extend(pushdown.notes());
6903        }
6904        notes.extend(self.unfit_notes.iter().flatten().cloned());
6905        notes.extend(self.changes_dropped.iter().cloned());
6906        if let Some((version, unfit)) = &self.changes_unfit
6907            && *version == self.changes_version
6908        {
6909            notes.extend(unfit.iter().cloned());
6910        }
6911        notes
6912    }
6913
6914    /// The source a full quality run over `scope` reads every row of, for the checks
6915    /// that read a source whole (an audio file's signal): the records of the data as
6916    /// loaded, or a view with nothing applied.
6917    pub(crate) fn window_for_quality(
6918        &self,
6919        scope: &crate::data_quality::QualityScope,
6920    ) -> Option<Arc<dyn crate::pushdown::Windowed>> {
6921        use crate::data_quality::QualityScope;
6922        matches!(scope, QualityScope::WholeSource | QualityScope::CurrentView)
6923            .then(|| {
6924                self.fixed_window
6925                    .clone()
6926                    .filter(|_| self.is_pristine() && self.indexing().is_none())
6927            })
6928            .flatten()
6929    }
6930
6931    /// The lake format whose plain files this dataset is, if it is one.
6932    ///
6933    /// For the chip in the control bar. The note says the same at length; this is what
6934    /// keeps the row count from reading as the table's.
6935    pub fn not_the_table(&self) -> Option<&'static str> {
6936        self.not_the_table
6937    }
6938
6939    /// The file's other tables, as `--table` names them; empty for a file of one.
6940    pub fn other_tables(&self) -> &[String] {
6941        &self.other_tables
6942    }
6943
6944    /// What a read through a format spec found, when the dataset was read through one.
6945    pub fn format_read(&self) -> Option<&Arc<crate::formats::Read>> {
6946        self.format_read.as_ref()
6947    }
6948
6949    /// The source a window of the view is read straight from, when there is one: the
6950    /// records or audio frames of the data as loaded, or the view a source runs itself.
6951    fn window_now(&self) -> Option<Arc<dyn crate::pushdown::Windowed>> {
6952        if let Some(window) = self.follow_window() {
6953            return Some(Arc::new(window));
6954        }
6955        if let Some(records) = self.fixed_window.as_ref().filter(|_| self.is_pristine()) {
6956            return Some(records.clone());
6957        }
6958        self.pushed_view().map(|view| view.window)
6959    }
6960
6961    /// The view as the source runs it, when it runs it: the data as loaded is the root
6962    /// (no query, reshape or drill), and the source can say the sidebar's filters and
6963    /// sort. Derived from them each time, so it never disagrees with them.
6964    pub(crate) fn pushed_view(&self) -> Option<crate::pushdown::PushedView> {
6965        let pushdown = self.pushdown.as_ref()?;
6966        if !self.scan_is_the_root() || self.drift_column_present {
6967            return None;
6968        }
6969        let sort: Vec<(String, bool)> = self
6970            .sort_columns
6971            .iter()
6972            .cloned()
6973            .zip(self.sort_descending.iter().copied())
6974            .collect();
6975        pushdown.view(&self.filters, &sort, !self.sort_ascending)
6976    }
6977
6978    /// The view's own count, from a source that runs the view, or for a followed file
6979    /// whose view is known up to a row, that count and the rows after it.
6980    pub(crate) fn source_counter(&self) -> Option<crate::pushdown::Counter> {
6981        if let Some(counter) = self.follow_counter() {
6982            return Some(counter);
6983        }
6984        self.pushed_view().map(|view| view.counter)
6985    }
6986
6987    /// The points where a followed view's rows are known, when they hold for the view
6988    /// on screen.
6989    fn follow_known(&self) -> Option<&[(usize, usize)]> {
6990        self.follow_known
6991            .as_ref()
6992            .filter(|(generation, _)| *generation == self.len_generation)
6993            .map(|(_, known)| known.as_slice())
6994    }
6995
6996    /// A followed file's windows, read from the mark before each: the rows as they are,
6997    /// or filtered with no sort, read on from where the view's rows are known.
6998    fn follow_window(&self) -> Option<crate::follow::Window> {
6999        let follow = self.follow.as_ref()?;
7000        let known = if self.is_pristine() {
7001            None
7002        } else if self.scan_is_the_root() && self.sort_columns.is_empty() && self.sort_ascending {
7003            Some(self.follow_known()?.to_vec())
7004        } else {
7005            return None;
7006        };
7007        Some(crate::follow::Window {
7008            lf: self.lf.clone(),
7009            path: follow.path().to_path_buf(),
7010            marks: follow.marks().clone(),
7011            known,
7012        })
7013    }
7014
7015    /// The count of a followed view known up to a file row: what was known, and the
7016    /// rows of the view among those after it, read from the mark before them.
7017    fn follow_counter(&self) -> Option<crate::pushdown::Counter> {
7018        let follow = self.follow.as_ref()?;
7019        let &(before, row) = self.follow_known()?.last()?;
7020        let rest = crate::follow::from_marks(&self.lf, follow.path(), follow.marks(), row, None)?;
7021        let streaming = self.polars_streaming;
7022        Some(Arc::new(move || {
7023            let df = crate::statistics::collect_lazy(row_count_lf(&rest), streaming)?;
7024            let after = match df.get(0).and_then(|row| row.first().cloned()) {
7025                Some(AnyValue::UInt64(n)) => n as usize,
7026                _ => 0,
7027            };
7028            Ok(before + after)
7029        }))
7030    }
7031
7032    /// What a read through a delimited spec found, when the dataset was read through
7033    /// one.
7034    pub fn delimited_read(&self) -> Option<&Arc<crate::delimited_spec::DelimitedRead>> {
7035        self.delimited.as_ref()
7036    }
7037
7038    /// The unit of the column named `column`, from a delimited spec's unit row or the
7039    /// file itself. A filter, sort, drill or query that keeps the loaded column, renamed
7040    /// or not, keeps its unit; a column a query computes has none, whatever it is called.
7041    pub fn unit_of(&self, column: &str) -> Option<&str> {
7042        if self.delimited.is_none() && self.file_units.is_empty() {
7043            return None;
7044        }
7045        let loaded = match &self.lineage {
7046            None => column,
7047            Some(lineage) => lineage
7048                .iter()
7049                .find(|(shown, _)| shown == column)
7050                .map(|(_, loaded)| loaded.as_str())?,
7051        };
7052        match &self.delimited {
7053            Some(read) => read.unit_of(loaded),
7054            None => self
7055                .file_units
7056                .iter()
7057                .find(|(name, _)| name == loaded)
7058                .map(|(_, unit)| unit.as_str()),
7059        }
7060    }
7061
7062    /// Each column of the view that has a unit, with it.
7063    pub fn units(&self) -> Vec<(String, String)> {
7064        if self.delimited.is_none() && self.file_units.is_empty() {
7065            return Vec::new();
7066        }
7067        self.schema
7068            .iter_names()
7069            .filter_map(|name| Some((name.to_string(), self.unit_of(name)?.to_string())))
7070            .collect()
7071    }
7072
7073    /// What the file said besides its rows: its Info panel tab.
7074    pub fn format_detail(&self) -> Option<&crate::text_formats::Detail> {
7075        self.detail.as_deref()
7076    }
7077
7078    /// The Info panel tab of a followed pipe's journal, read again once it has ended
7079    /// and its rows are all on hand: the frame that reads every entry. `None` for any
7080    /// other dataset, or when it has been asked for already.
7081    pub(crate) fn ended_journal_to_describe(&mut self) -> Option<LazyFrame> {
7082        let follow = self.follow.as_mut()?;
7083        if follow.described
7084            || follow.live()
7085            || follow.behind()
7086            || follow.spool().is_none()
7087            || self.read_as != Some(crate::FileFormat::Journal)
7088        {
7089            return None;
7090        }
7091        follow.described = true;
7092        Some(self.original_lf.clone())
7093    }
7094
7095    pub(crate) fn set_format_detail(&mut self, detail: crate::text_formats::Detail) {
7096        self.detail = Some(Arc::new(detail));
7097    }
7098
7099    /// Whether datui noticed anything at all. Answers what `notes()` is usually asked
7100    /// — whether to offer the tab — without building the list to find out.
7101    pub fn has_notes(&self) -> bool {
7102        // A view note needs a column the files disagree on, and such a column always
7103        // draws a note of its own when the dataset opens. So the view half can never
7104        // be the only half, and the Notes tab does not appear and disappear as the
7105        // user sorts.
7106        //
7107        // The open's own notes count: a directory read as one format with another left
7108        // out may have nothing else worth saying, and that is exactly the dataset whose
7109        // reader the user most wants to know about.
7110        !self.notes.is_empty()
7111            || !self.open_notes.is_empty()
7112            || self.unfit_notes.as_ref().is_some_and(|n| !n.is_empty())
7113            || !self.changes_dropped.is_empty()
7114            || self
7115                .changes_unfit
7116                .as_ref()
7117                .is_some_and(|(v, n)| *v == self.changes_version && !n.is_empty())
7118            || self
7119                .pushdown
7120                .as_ref()
7121                .is_some_and(|p| !p.notes().is_empty())
7122    }
7123
7124    /// Whether there is something to say that has not been offered yet.
7125    pub fn notes_unseen(&self) -> bool {
7126        self.has_notes() && !self.notes_seen
7127    }
7128
7129    /// The rows a filter or sort on `column` has to leave out: every row of every file
7130    /// that holds the column in a type it is not read in.
7131    fn unread_row_runs(&self, column: &str) -> Vec<(usize, usize)> {
7132        let Some(dataset) = self.dataset_schema.as_ref() else {
7133            return Vec::new();
7134        };
7135        let conflicts: Vec<bool> = (0..self.drift_file_starts.len())
7136            .map(|file| {
7137                dataset
7138                    .file_group
7139                    .get(file)
7140                    .and_then(|group| dataset.groups.get(*group as usize))
7141                    .is_some_and(|group| group.unread.iter().any(|name| name == column))
7142            })
7143            .collect();
7144        conflicting_row_runs(&self.drift_file_starts, self.drift_dataset_rows, &conflicts)
7145    }
7146
7147    /// Columns the filter or sort names that some file holds in another type, in the
7148    /// dataset's own column order and each named once however many times the view
7149    /// mentions it.
7150    fn view_columns_with_conflicts(&self) -> Vec<crate::schema_union::ColumnDrift> {
7151        let Some(dataset) = self.dataset_schema.as_ref() else {
7152            return Vec::new();
7153        };
7154        let named: HashSet<&str> = self
7155            .filters
7156            .iter()
7157            .map(|filter| filter.column.as_str())
7158            .chain(self.sort_columns.iter().map(String::as_str))
7159            .collect();
7160        dataset
7161            .columns
7162            .iter()
7163            .filter(|column| column.conflicting_files > 0 && named.contains(column.name.as_str()))
7164            .cloned()
7165            .collect()
7166    }
7167
7168    /// Leave out the rows whose files do not hold a filtered or sorted column in the
7169    /// type it is read as, and say how many.
7170    ///
7171    /// Those rows read as null in that column, and a null is not a value the column
7172    /// can be compared or ordered by: a filter drops them already, and a sort would
7173    /// otherwise gather them at one end as though they belonged there. They are left
7174    /// out of both, and the note says how many so the smaller count is never a
7175    /// surprise.
7176    fn view_exclusions(&self) -> Vec<(Vec<(usize, usize)>, crate::notes::Note)> {
7177        if !self.drift_column_present {
7178            return Vec::new();
7179        }
7180        let Some(dataset) = self.dataset_schema.as_ref() else {
7181            return Vec::new();
7182        };
7183        let mut out = Vec::new();
7184        for column in self.view_columns_with_conflicts() {
7185            let runs = self.unread_row_runs(&column.name);
7186            let rows: usize = runs.iter().map(|(start, end)| end - start).sum();
7187            if rows == 0 {
7188                continue;
7189            }
7190            let filtered = self
7191                .filters
7192                .iter()
7193                .any(|filter| filter.column.as_str() == column.name.as_str());
7194            let sorted = self
7195                .sort_columns
7196                .iter()
7197                .any(|sorted| sorted.as_str() == column.name.as_str());
7198            out.push((
7199                runs,
7200                crate::notes::left_out_note(&column, dataset, rows, filtered, sorted),
7201            ));
7202        }
7203        out
7204    }
7205
7206    /// The notes for what the filter and sort on screen leave out, for a caller that
7207    /// is putting a frame back that already leaves those rows out rather than building
7208    /// one. Derived, never stored across a change of view: a note that outlives the
7209    /// sort that earned it is the fault this is shaped to avoid.
7210    fn view_notes_only(&self) -> Vec<crate::notes::Note> {
7211        self.view_exclusions()
7212            .into_iter()
7213            .map(|(_, note)| note)
7214            .collect()
7215    }
7216
7217    fn leave_out_unread_rows(&self, mut lf: LazyFrame) -> (LazyFrame, Vec<crate::notes::Note>) {
7218        let mut notes = Vec::new();
7219        for (runs, note) in self.view_exclusions() {
7220            let keep = runs
7221                .iter()
7222                .map(|(start, end)| {
7223                    col(crate::schema_union::DRIFT_COLUMN)
7224                        .lt(lit(*start as u32))
7225                        .or(col(crate::schema_union::DRIFT_COLUMN).gt_eq(lit(*end as u32)))
7226                })
7227                .reduce(Expr::and);
7228            if let Some(keep) = keep {
7229                lf = lf.filter(keep);
7230            }
7231            notes.push(note);
7232        }
7233        (lf, notes)
7234    }
7235
7236    /// Whether the notes have been offered. Exact, where `!notes_unseen()` would also
7237    /// be true of a dataset that has nothing to say.
7238    pub fn notes_seen(&self) -> bool {
7239        self.notes_seen
7240    }
7241
7242    /// The Info panel has been opened; the quiet accent has done its job.
7243    pub fn mark_notes_seen(&mut self) {
7244        self.notes_seen = true;
7245    }
7246
7247    /// Read `column` as text from every file, so the values a type conflict hid can be
7248    /// seen.
7249    ///
7250    /// Rebuilds the scan rather than re-opening the dataset: everything it needs is
7251    /// already here. The file list, each file's row count and — since the footers were
7252    /// read — the type each file holds each conflicting column in are all on hand, so
7253    /// this costs no directory listing, no footer read and no request. The view goes
7254    /// with it: a filter and sort in force are re-applied to the new frame.
7255    ///
7256    /// Returns whether anything happened. `false` for a column that is not on offer,
7257    /// which is what the panel only ever asks about, and for one already read this way.
7258    /// A scan that cannot be built is kept as the error showing, and returned.
7259    pub fn read_column_as_text(&mut self, column: &str) -> PolarsResult<bool> {
7260        let name = PlSmallStr::from(column);
7261        let Some(dataset) = self.dataset_at_open.clone() else {
7262            return Ok(false);
7263        };
7264        if !self.drift_column_present || self.read_as_text.contains(&name) {
7265            return Ok(false);
7266        }
7267        if !dataset
7268            .columns
7269            .iter()
7270            .any(|drift| drift.name == name && drift.can_read_as_text())
7271        {
7272            return Ok(false);
7273        }
7274
7275        let mut as_text = self.read_as_text.clone();
7276        as_text.push(name);
7277
7278        // Built from the dataset as its footers found it, never from the view below.
7279        // The view has the column as text and nothing conflicting, so it no longer
7280        // holds the one thing the scan needs: the type each file actually wrote.
7281        let drift =
7282            crate::schema_union::ScanDrift::new(&self.drift_files, &dataset, &self.file_rows());
7283        let scanned = match self.remote_files.as_ref() {
7284            Some(remote) => (remote.scan)(&remote.urls, &as_text),
7285            None => crate::schema_union::lenient_scan(
7286                &self.drift_files,
7287                dataset.schema.clone(),
7288                None,
7289                drift.as_ref(),
7290                &as_text,
7291            ),
7292        };
7293        let lf = match scanned {
7294            Ok(lf) => lf,
7295            Err(e) => {
7296                // Shown where any failed read is; the view is as it was.
7297                self.error = Some(e.clone());
7298                return Err(e);
7299            }
7300        };
7301        // The remote scan hoists inside its own closure, as it does for the frame the
7302        // dataset opened with; only the local branch has it left to do.
7303        let lf = if self.remote_files.is_some() {
7304            lf
7305        } else {
7306            crate::hoist_partition_columns(
7307                lf,
7308                &dataset.schema,
7309                self.partition_columns.as_deref().unwrap_or(&[]),
7310                drift.is_some(),
7311            )
7312        };
7313
7314        let view = dataset.reading_as_text(&as_text);
7315        self.read_as_text = as_text;
7316        // `text_schema` keeps the columns in their places, so the order the user
7317        // arranged still names every one of them and still means what it did.
7318        let schema = view.schema.clone();
7319        self.drift_groups = Arc::new(view.groups.clone());
7320        self.groups_at_open = self.drift_groups.clone();
7321        self.notes = Self::notes_datui_can_act_on(&view, self.drift_column_present);
7322        self.notes_at_open = self.notes.clone();
7323        // One note went and another arrived, and the new one is about how the column
7324        // now compares — which matters most to a user who has a filter on it.
7325        self.notes_seen = false;
7326        self.dataset_schema = Some(view);
7327        self.replace_root(lf, schema);
7328        // Re-applies the filter and sort over the new frame, and with them the note
7329        // about what they leave out — which is one note shorter now.
7330        self.apply_transformations();
7331        Ok(true)
7332    }
7333
7334    /// The dataset's notes, with the offer to read a column as text left on only where
7335    /// taking it would work.
7336    ///
7337    /// A note is written from the footers' schema, which says whether a column *could*
7338    /// be shown as text. Whether it can be read that way is a second question: the scan
7339    /// has to know where each file's rows begin, and it does not for a dataset too large
7340    /// to read every footer, or one where a footer would not parse. Those are the same
7341    /// datasets that cannot draw the marks. An offer the panel shows and the action
7342    /// then declines is worse than no offer, so it is taken off here rather than
7343    /// refused later.
7344    fn notes_datui_can_act_on(
7345        dataset: &crate::schema_union::DatasetSchema,
7346        counted: bool,
7347    ) -> Vec<crate::notes::Note> {
7348        let mut notes = crate::notes::from_dataset(dataset);
7349        if !counted {
7350            for note in &mut notes {
7351                note.read_as_text = None;
7352            }
7353        }
7354        notes
7355    }
7356
7357    /// Each file's row count, as the footers gave them. The starts are kept rather than
7358    /// the counts, so this is their differences with the dataset's total closing the
7359    /// last one.
7360    fn file_rows(&self) -> Vec<usize> {
7361        self.drift_file_starts
7362            .iter()
7363            .enumerate()
7364            .map(|(file, start)| {
7365                self.drift_file_starts
7366                    .get(file + 1)
7367                    .copied()
7368                    .unwrap_or(self.drift_dataset_rows)
7369                    .saturating_sub(*start)
7370            })
7371            .collect()
7372    }
7373
7374    /// The columns being read as text rather than as the type most rows have.
7375    pub fn read_as_text(&self) -> &[PlSmallStr] {
7376        &self.read_as_text
7377    }
7378
7379    /// What the footers said about the dataset's columns, when it is many files.
7380    pub fn dataset_schema(&self) -> Option<&crate::schema_union::DatasetSchema> {
7381        self.dataset_schema.as_ref()
7382    }
7383
7384    /// The counter for a remote dataset's files, while its count would be the data's:
7385    /// the frame is the scan as loaded, and the files have not been counted yet.
7386    ///
7387    /// Not while a pass is already reading every footer of this dataset. That pass
7388    /// brings the row groups back with it, and this counter reads the same footers a
7389    /// second time — for a prefix of 6,541 files, 6,541 ranged reads to learn what is
7390    /// already on its way. Staging the open to save round trips and then spending them
7391    /// here would be worse than not staging it at all.
7392    pub fn remote_files_counter(&self) -> Option<FileCounter> {
7393        if self.footers_pending.is_some() {
7394            return None;
7395        }
7396        self.remote_files
7397            .as_ref()
7398            .filter(|f| f.offsets.is_none() && self.is_pristine())
7399            .map(|f| f.count.clone())
7400    }
7401
7402    /// Record the rows in each row group of each file of a remote dataset: the total,
7403    /// the row groups a buffer is planned in, and which files hold which rows.
7404    ///
7405    /// A local dataset has no files to window over, and takes the total and the groups.
7406    fn record_file_row_groups(&mut self, groups: &[Vec<usize>]) {
7407        if let Some(files) = self.remote_files.as_mut() {
7408            if groups.len() != files.urls.len() {
7409                return;
7410            }
7411            let mut offsets = Vec::with_capacity(groups.len() + 1);
7412            offsets.push(0);
7413            for file in groups {
7414                offsets.push(offsets.last().unwrap_or(&0) + file.iter().sum::<usize>());
7415            }
7416            files.offsets = Some(offsets);
7417        }
7418        let flat: Vec<usize> = groups.iter().flatten().copied().collect();
7419        if self.is_pristine() {
7420            self.record_row_groups(&flat);
7421        } else {
7422            // Kept for when the frame is the scan again (`restore_footer_count`).
7423            let mut row_offsets = Vec::with_capacity(flat.len() + 1);
7424            row_offsets.push(0);
7425            for n in &flat {
7426                row_offsets.push(row_offsets.last().unwrap_or(&0) + n);
7427            }
7428            self.row_group_offsets = Some(row_offsets);
7429        }
7430    }
7431
7432    /// The frame for buffer rows `[start, start + len)`, columns in display order. For a
7433    /// remote dataset whose files are counted, a scan of only the files holding them.
7434    /// How many of the dataset's files a page at `start` would read.
7435    ///
7436    /// A windowed remote scan reads only the files holding those rows; everything else
7437    /// hands the whole scan to Polars, which reads what it decides to and does not say.
7438    /// `None` is that second case — not zero, which would claim a page came from
7439    /// nowhere.
7440    pub fn files_a_page_reads(&self, start: usize, len: usize) -> Option<usize> {
7441        let offsets = self.files_window().and_then(|f| f.offsets.as_ref())?;
7442        let (first, last) = files_holding(offsets, start, len)?;
7443        Some(files_with_rows(offsets, first, last).len())
7444    }
7445
7446    /// The frame for buffer rows `[start, start + len)`, columns in display order. For a
7447    /// remote dataset whose files are counted, a scan of only the files holding them.
7448    fn buffer_lf(&self, start: usize, len: usize) -> PolarsResult<LazyFrame> {
7449        let mut all_columns = self.binary_stub_exprs();
7450        if self.carries_source_rows() {
7451            all_columns.push(col(crate::schema_union::DRIFT_COLUMN));
7452        }
7453        self.window_lf(start, len, all_columns)
7454    }
7455
7456    /// Whether the frame's rows carry their place in the source, for `#`: a dataset's
7457    /// rows that know their file, or lines, while the frame is still the scan's. A
7458    /// query's rows, a reshape's and a group's stand for no row of the source.
7459    pub fn carries_source_rows(&self) -> bool {
7460        self.drift_column_present
7461            || (self.scan_is_the_root() && (self.source_rows_at_open || self.view_numbered))
7462    }
7463
7464    /// What `#` shows for `rows` rows from `start`: each row's place in the source
7465    /// where the rows carry it, else its place in the view, counted from
7466    /// `row_start_index`. A pristine view's places are the source's either way.
7467    pub fn row_numbers_from(&self, start: usize, rows: usize) -> Vec<usize> {
7468        let view = |i: usize| start + i + self.row_start_index;
7469        let places = self
7470            .buffered_df
7471            .as_ref()
7472            .filter(|_| self.carries_source_rows())
7473            .and_then(|df| df.column(crate::schema_union::DRIFT_COLUMN).ok())
7474            .and_then(|column| {
7475                let offset = start.checked_sub(self.buffered_start_row)?;
7476                let len = rows.min(column.len().saturating_sub(offset));
7477                let slice = column.slice(offset as i64, len);
7478                let places = slice.u32().ok()?;
7479                // Several files' lines are numbered in their own file.
7480                let place = |p: usize| {
7481                    self.numbering
7482                        .as_ref()
7483                        .and_then(|lines| lines.line_in_file(p))
7484                        .unwrap_or(p)
7485                };
7486                Some(
7487                    places
7488                        .iter()
7489                        .map(|p| p.map(|p| place(p as usize) + self.row_start_index))
7490                        .collect::<Vec<_>>(),
7491                )
7492            });
7493        (0..rows)
7494            .map(|i| {
7495                places
7496                    .as_ref()
7497                    .and_then(|p| p.get(i).copied().flatten())
7498                    .unwrap_or_else(|| view(i))
7499            })
7500            .collect()
7501    }
7502
7503    /// The frame for rows `[start, start + len)` of the view, as `all_columns`. For a
7504    /// remote dataset whose files are counted, a scan of only the files holding them.
7505    fn window_lf(
7506        &self,
7507        start: usize,
7508        len: usize,
7509        all_columns: Vec<Expr>,
7510    ) -> PolarsResult<LazyFrame> {
7511        window_of(
7512            &self.lf,
7513            self.files_window(),
7514            self.window_now().as_deref(),
7515            &self.read_as_text,
7516            start,
7517            len,
7518            all_columns,
7519        )
7520    }
7521
7522    /// The view's rows as a find reads them: a window at a time, the way a page is
7523    /// read, and the buffer already on hand.
7524    pub(crate) fn view_rows(&self) -> ViewRows {
7525        ViewRows {
7526            lf: self.lf.clone(),
7527            files: self.files_window().cloned(),
7528            // A find reads every row it can reach: lines still being indexed are read
7529            // through the frame, which waits for them, not the window of those so far.
7530            records: self.window_now().filter(|_| self.indexing().is_none()),
7531            read_as_text: self.read_as_text.clone(),
7532            buffer: self
7533                .buffered_df
7534                .as_ref()
7535                .filter(|_| self.buffer_on_hand())
7536                .map(|df| (df.clone(), self.buffered_start_row)),
7537            num_rows: self.num_rows_valid.then_some(self.num_rows),
7538            streaming: self.polars_streaming,
7539            whole: sees_every_row_first(&self.lf),
7540            reads_up_to: reads_up_to_a_window(&self.lf),
7541        }
7542    }
7543
7544    /// Put the cursor on view row `row`, centered, for a find that matched there.
7545    /// Returns true if a collect is needed. A row past a provisional total is one the
7546    /// find read, so the total reaches it until the count lands.
7547    pub(crate) fn go_to_found_row(&mut self, row: usize) -> bool {
7548        if !self.num_rows_valid && self.num_rows <= row {
7549            self.num_rows = row + 1;
7550        }
7551        self.scroll_to_row_centered(row)
7552    }
7553
7554    /// The view row the cursor is on.
7555    pub(crate) fn cursor_row(&self) -> usize {
7556        self.start_row + self.table_state.selected().unwrap_or(0)
7557    }
7558
7559    /// Bytes a buffered row takes: measured on the last buffer collected, or until
7560    /// then estimated from the schema.
7561    fn bytes_per_row(&self) -> usize {
7562        self.observed_bytes_per_row.unwrap_or_else(|| {
7563            estimate_bytes_per_row(&self.schema, &self.column_order, &self.column_bytes)
7564        })
7565    }
7566
7567    /// Best available in-memory width estimate for one logical row.
7568    ///
7569    /// Data Quality uses this only for a preflight estimate and labels the result as
7570    /// approximate. Buffer planning uses the same source so the two surfaces do not
7571    /// disagree about the shape of the current view.
7572    pub fn estimated_row_bytes(&self) -> usize {
7573        self.bytes_per_row()
7574    }
7575
7576    /// Number of source files known to participate in the pristine dataset scan.
7577    /// Returns `None` after a query or reshape has broken the row-to-file mapping.
7578    pub fn source_file_count(&self) -> Option<usize> {
7579        self.is_pristine().then(|| self.loaded_file_count())
7580    }
7581
7582    /// Files the dataset was loaded from, whatever the view does with their rows.
7583    pub(crate) fn loaded_file_count(&self) -> usize {
7584        if !self.drift_files.is_empty() {
7585            self.drift_files.len()
7586        } else if let Some(remote) = &self.remote_files {
7587            remote.urls.len()
7588        } else {
7589            1
7590        }
7591    }
7592
7593    /// Rows the `max_buffered_mb` budget allows a buffer, never fewer than a screen;
7594    /// 0 for no budget. Planning to this, rather than trimming the collected frame to
7595    /// it, keeps a wide window from being materialized only to be cut down.
7596    fn byte_cap_rows(&self) -> usize {
7597        if self.max_buffered_mb == 0 {
7598            return 0;
7599        }
7600        let max_bytes = self.max_buffered_mb * 1024 * 1024;
7601        (max_bytes / self.bytes_per_row()).max(self.visible_rows.max(1))
7602    }
7603
7604    /// True while the buffer is planned as a remote window: a scan of an object store
7605    /// that nothing has been applied to. A query, filter, sort or reshape reads the
7606    /// object through a predicate, and `slice(0, N)` then stops at the first N matches,
7607    /// so the page-based window costs a row group where the remote one would read forty.
7608    fn remote_window(&self) -> bool {
7609        self.remote_source && self.is_pristine()
7610    }
7611
7612    /// The files a page reads by, while the frame is the scan as loaded: a filter or
7613    /// sort reads every file before its window, so it goes through the whole scan.
7614    fn files_window(&self) -> Option<&RemoteFiles> {
7615        self.remote_files.as_ref().filter(|_| self.is_pristine())
7616    }
7617
7618    /// Rows the buffer reaches past the view in one direction: `pages` of it for a local
7619    /// file, a remote scan with something applied to it (see `remote_window`), or a
7620    /// remote dataset of many files; half the window for a pristine remote object
7621    /// (`fit_window` trims the two halves plus the view back to the cap).
7622    ///
7623    /// Many files are read a few at a time instead: what a read costs there is the
7624    /// files it opens, not its rows, and a wide window over a dataset of small files
7625    /// (a day of blocks in 2009 is a few rows) is hundreds of downloads.
7626    fn reach_rows(&self, pages: usize) -> usize {
7627        if !self.remote_window() || self.remote_files.is_some() {
7628            return pages * self.visible_rows.max(1);
7629        }
7630        let window = if self.max_buffered_rows > 0 {
7631            self.max_buffered_rows
7632        } else {
7633            DEFAULT_MAX_BUFFERED_ROWS
7634        };
7635        window / 2
7636    }
7637
7638    /// The smallest buffer worth filling: a page plus the reach either side.
7639    fn min_buffer_len(&self) -> usize {
7640        self.visible_rows.max(1)
7641            + self.reach_rows(self.pages_lookahead)
7642            + self.reach_rows(self.pages_lookback)
7643    }
7644
7645    /// True when the view already shows the last page, so End has nothing to load.
7646    pub fn at_end(&self) -> bool {
7647        self.start_row == self.num_rows.saturating_sub(self.visible_rows)
7648    }
7649
7650    /// Fit a planned buffer `[buffer_start, buffer_end)` to the caps: `max_buffered_rows`
7651    /// and the byte budget around the view, then for a remote object whose footer is
7652    /// known the row groups the view lies in, cut back to the caps inside them.
7653    fn fit_window(
7654        &self,
7655        view_start: usize,
7656        view_end: usize,
7657        buffer_start: &mut usize,
7658        buffer_end: &mut usize,
7659    ) {
7660        let byte_cap = self.byte_cap_rows();
7661        let cap = match (self.max_buffered_rows, byte_cap) {
7662            (0, cap) | (cap, 0) => cap,
7663            (rows, bytes) => rows.min(bytes),
7664        };
7665        if cap > 0 {
7666            shrink_around_view(
7667                view_start,
7668                view_end,
7669                cap,
7670                0,
7671                self.num_rows_bound(),
7672                buffer_start,
7673                buffer_end,
7674            );
7675        }
7676        let Some(offsets) = self
7677            .row_group_offsets
7678            .as_deref()
7679            .filter(|_| self.remote_window())
7680        else {
7681            return;
7682        };
7683        (*buffer_start, *buffer_end) = align_to_row_groups(
7684            offsets,
7685            view_start,
7686            view_end,
7687            *buffer_start,
7688            *buffer_end,
7689            cap,
7690        );
7691        // The caps hold inside a group too: a group over them is read one window at
7692        // a time, the window kept inside the group so it never pulls the next one
7693        // before the view reaches it.
7694        if cap > 0 {
7695            let (floor, ceil) = (*buffer_start, *buffer_end);
7696            shrink_around_view(
7697                view_start,
7698                view_end,
7699                cap,
7700                floor,
7701                ceil,
7702                buffer_start,
7703                buffer_end,
7704            );
7705        }
7706        // Over many files, at most a few of them, around the view's.
7707        if let Some(file_offsets) = self.remote_files.as_ref().and_then(|f| f.offsets.as_ref()) {
7708            (*buffer_start, *buffer_end) = limit_files(
7709                file_offsets,
7710                view_start,
7711                view_end,
7712                *buffer_start,
7713                *buffer_end,
7714                MAX_FILES_PER_BUFFER,
7715            );
7716        }
7717        // A view straddling two groups needs both, but one is on hand: fetch the other
7718        // alone and stitch it on (see `apply_async_collect`).
7719        if self.buffer_on_hand() {
7720            let (held_start, held_end) = (self.buffered_start_row, self.buffered_end_row);
7721            if held_start <= *buffer_start && *buffer_start < held_end && held_end < *buffer_end {
7722                *buffer_start = held_end;
7723            } else if *buffer_start < held_start
7724                && held_start < *buffer_end
7725                && *buffer_end <= held_end
7726            {
7727                *buffer_end = held_start;
7728            }
7729        }
7730    }
7731
7732    fn load_buffer(&mut self, buffer_start: usize, buffer_end: usize) {
7733        let buffer_size = buffer_end.saturating_sub(buffer_start);
7734        if buffer_size == 0 {
7735            return;
7736        }
7737
7738        let use_streaming = self.polars_streaming;
7739        let lf = match self.buffer_lf(buffer_start, buffer_size) {
7740            Ok(lf) => lf,
7741            Err(e) => {
7742                self.error = Some(e);
7743                return;
7744            }
7745        };
7746        let full_df = match collect_lazy(lf, use_streaming) {
7747            Ok(df) => df,
7748            Err(e) => {
7749                self.error = Some(e);
7750                return;
7751            }
7752        };
7753
7754        // Stitched and cut as a background fill is, here on the spot, with the old
7755        // rows let go first: the plan has taken any it stitches on to.
7756        let plan = self.fill_plan(buffer_start, buffer_end, self.num_rows, self.num_rows_valid);
7757        self.release_display_buffer();
7758        let fitted = plan.fit(full_df);
7759        if fitted.bytes_per_row.is_some() {
7760            self.observed_bytes_per_row = fitted.bytes_per_row;
7761        }
7762        let full_df = fitted.df;
7763        let effective_buffer_start = fitted.start;
7764        let effective_buffer_end = fitted.start + full_df.height();
7765
7766        if self.locked_columns_count > 0 {
7767            let locked_names: Vec<&str> = self
7768                .column_order
7769                .iter()
7770                .take(self.locked_columns_count)
7771                .map(|s| s.as_str())
7772                .collect();
7773            let locked_df = match full_df.select(locked_names) {
7774                Ok(df) => df,
7775                Err(e) => {
7776                    self.error = Some(e);
7777                    return;
7778                }
7779            };
7780            self.locked_df = Some(locked_df);
7781        } else {
7782            self.locked_df = None;
7783        }
7784
7785        let scroll_names: Vec<&str> = self
7786            .column_order
7787            .iter()
7788            .skip(self.frozen_shown() + self.termcol_index)
7789            .map(|s| s.as_str())
7790            .collect();
7791        if scroll_names.is_empty() {
7792            self.df = None;
7793        } else {
7794            let scroll_df = match full_df.select(scroll_names) {
7795                Ok(df) => df,
7796                Err(e) => {
7797                    self.error = Some(e);
7798                    return;
7799                }
7800            };
7801            self.df = Some(scroll_df);
7802        }
7803        if self.error.is_some() {
7804            self.error = None;
7805        }
7806        self.buffered_start_row = effective_buffer_start;
7807        self.buffered_end_row = effective_buffer_end;
7808        self.buffered_df = Some(full_df);
7809    }
7810
7811    /// Let go of the buffer being replaced and the display frames cut from it. A
7812    /// synchronous load does so before its cut, so the cut's copy is not made while
7813    /// the old rows are still held; a stitch has already taken the rows it keeps.
7814    /// The view's rows come next, so a relearn asked for takes effect.
7815    fn release_display_buffer(&mut self) {
7816        self.widths.rows_arrived();
7817        self.buffered_df = None;
7818        self.locked_df = None;
7819        self.df = None;
7820    }
7821
7822    /// Recompute locked_df and df from the cached full buffer. Used when only termcol_index (or locked columns) changed.
7823    fn slice_buffer_into_display(&mut self) {
7824        let full_df = match self.buffered_df.as_ref() {
7825            Some(df) => df,
7826            None => return,
7827        };
7828
7829        if self.locked_columns_count > 0 {
7830            let locked_names: Vec<&str> = self
7831                .column_order
7832                .iter()
7833                .take(self.locked_columns_count)
7834                .map(|s| s.as_str())
7835                .collect();
7836            if let Ok(locked_df) = full_df.select(locked_names) {
7837                self.locked_df = Some(locked_df);
7838            }
7839        } else {
7840            self.locked_df = None;
7841        }
7842
7843        let scroll_names: Vec<&str> = self
7844            .column_order
7845            .iter()
7846            .skip(self.frozen_shown() + self.termcol_index)
7847            .map(|s| s.as_str())
7848            .collect();
7849        if scroll_names.is_empty() {
7850            self.df = None;
7851        } else {
7852            if let Ok(scroll_df) = full_df.select(scroll_names) {
7853                self.df = Some(scroll_df);
7854            }
7855        }
7856    }
7857
7858    /// Whether the view is inside the buffer and within a page of one of its ends, with
7859    /// more data past that end: where a collect would grow the buffer, if one ran. A
7860    /// scroll that stays inside the buffer runs none, so the growing waited until the
7861    /// view had left it — and the page was blank while it happened.
7862    pub fn wants_to_load_ahead(&self) -> bool {
7863        if self.visible_rows == 0
7864            || self.buffered_df.is_none()
7865            || !self.page_on_hand(self.start_row)
7866        {
7867            return false;
7868        }
7869        let near = self.proximity();
7870        let view_end = self.start_row
7871            + self
7872                .visible_rows
7873                .min(self.num_rows_bound().saturating_sub(self.start_row));
7874        let behind =
7875            self.start_row - self.buffered_start_row <= near && self.buffered_start_row > 0;
7876        let ahead = self.buffered_end_row - view_end <= near
7877            && self.buffered_end_row < self.num_rows_bound();
7878        behind || ahead
7879    }
7880
7881    /// How close the view comes to an end of the buffer before the buffer grows past
7882    /// it: half the reach ahead, and never under a page. A page was the whole margin, and
7883    /// a cloud fetch takes longer than the next PageDown does to cross it.
7884    fn proximity(&self) -> usize {
7885        (self.reach_rows(self.pages_lookahead) / 2).max(self.visible_rows)
7886    }
7887
7888    /// Where the view and the buffer are, to tell one load-ahead attempt from the next.
7889    pub fn buffer_position(&self) -> (u64, usize, usize, usize) {
7890        (
7891            self.len_generation(),
7892            self.start_row,
7893            self.buffered_start_row,
7894            self.buffered_end_row,
7895        )
7896    }
7897
7898    /// Whether every row of the page starting at `start` is in the buffer.
7899    fn page_on_hand(&self, start: usize) -> bool {
7900        let bound = self.num_rows_bound();
7901        let end = start + self.visible_rows.min(bound.saturating_sub(start));
7902        self.buffered_df.is_some()
7903            && self.buffered_end_row > 0
7904            && start >= self.buffered_start_row
7905            && end <= self.buffered_end_row
7906    }
7907
7908    /// The first row to draw: the view's own once its rows are on hand, and until then
7909    /// the last page that was drawn whole. The view moves the moment a key asks, before
7910    /// its rows are fetched, and drawn from there it was half a page of rows over half a
7911    /// page of nothing until the fetch landed.
7912    fn start_to_draw(&mut self) -> usize {
7913        if self.page_on_hand(self.start_row) {
7914            self.drawn_start = self.start_row;
7915            self.start_row
7916        } else if self.page_on_hand(self.drawn_start) {
7917            self.drawn_start
7918        } else {
7919            self.start_row
7920        }
7921    }
7922
7923    fn slice_from_buffer(&mut self) {
7924        // Buffer contains the full range [buffered_start_row, buffered_end_row)
7925        // The displayed portion [start_row, start_row + visible_rows) is a subset
7926        // We'll slice the displayed portion when rendering based on offset
7927        // No action needed here - the buffer is stored, slicing happens at render time
7928    }
7929
7930    /// Returns true if a buffer collect is needed after the scroll.
7931    pub fn select_next(&mut self) -> bool {
7932        self.table_state.select_next();
7933        if let Some(selected) = self.table_state.selected()
7934            && selected >= self.visible_rows
7935            && self.visible_rows > 0
7936        {
7937            return self.slide_table(1);
7938        }
7939        false
7940    }
7941
7942    /// Returns true if a buffer collect is needed after the scroll.
7943    pub fn page_down(&mut self) -> bool {
7944        self.slide_table(self.visible_rows as i64)
7945    }
7946
7947    /// Returns true if a buffer collect is needed after the scroll.
7948    pub fn select_previous(&mut self) -> bool {
7949        if let Some(selected) = self.table_state.selected() {
7950            self.table_state.select_previous();
7951            if selected == 0 && self.start_row > 0 {
7952                return self.slide_table(-1);
7953            }
7954        } else {
7955            self.table_state.select(Some(0));
7956        }
7957        false
7958    }
7959
7960    /// Returns true if a buffer collect is needed.
7961    pub fn scroll_to(&mut self, index: usize) -> bool {
7962        if self.start_row == index {
7963            return false;
7964        }
7965        self.start_row = index;
7966        true // caller must collect
7967    }
7968
7969    /// Set scroll position for go-to-line (centered). Returns true if a collect is needed.
7970    pub fn scroll_to_row_centered(&mut self, row_index: usize) -> bool {
7971        if self.num_rows == 0 || self.visible_rows == 0 {
7972            return false;
7973        }
7974        let center_offset = self.visible_rows / 2;
7975        let mut start_row = row_index.saturating_sub(center_offset);
7976        let max_start = self.num_rows.saturating_sub(self.visible_rows);
7977        start_row = start_row.min(max_start);
7978
7979        if self.start_row == start_row {
7980            let display_idx = row_index
7981                .saturating_sub(start_row)
7982                .min(self.visible_rows.saturating_sub(1));
7983            self.table_state.select(Some(display_idx));
7984            return false;
7985        }
7986
7987        self.start_row = start_row;
7988        let display_idx = row_index
7989            .saturating_sub(start_row)
7990            .min(self.visible_rows.saturating_sub(1));
7991        self.table_state.select(Some(display_idx));
7992        true // caller must collect
7993    }
7994
7995    /// Jump to the first page. Returns true if a collect is needed.
7996    pub fn scroll_to_start(&mut self) -> bool {
7997        self.table_state.select(Some(0));
7998        self.scroll_to(0)
7999    }
8000
8001    /// Jump to the last page. Returns true if a collect is needed.
8002    pub fn scroll_to_end(&mut self) -> bool {
8003        if self.num_rows == 0 {
8004            self.start_row = 0;
8005            self.buffered_start_row = 0;
8006            self.buffered_end_row = 0;
8007            return false;
8008        }
8009        let end_start = self.num_rows.saturating_sub(self.visible_rows);
8010        if self.start_row == end_start {
8011            self.select_last_visible_row();
8012            return false;
8013        }
8014        self.start_row = end_start;
8015        self.select_last_visible_row();
8016        true // caller must collect
8017    }
8018
8019    /// Set table selection to the last row in the current view (for use after scroll_to_end).
8020    fn select_last_visible_row(&mut self) {
8021        if self.num_rows == 0 {
8022            return;
8023        }
8024        let last_row_display_idx = (self.num_rows - 1).saturating_sub(self.start_row);
8025        let sel = last_row_display_idx.min(self.visible_rows.saturating_sub(1));
8026        self.table_state.select(Some(sel));
8027    }
8028
8029    /// Returns true if a buffer collect is needed after the scroll.
8030    pub fn half_page_down(&mut self) -> bool {
8031        let half = (self.visible_rows / 2).max(1) as i64;
8032        self.slide_table(half)
8033    }
8034
8035    /// Returns true if a buffer collect is needed after the scroll.
8036    pub fn half_page_up(&mut self) -> bool {
8037        if self.start_row == 0 {
8038            return false;
8039        }
8040        let half = (self.visible_rows / 2).max(1) as i64;
8041        self.slide_table(-half)
8042    }
8043
8044    /// Returns true if a buffer collect is needed after the scroll.
8045    pub fn page_up(&mut self) -> bool {
8046        if self.start_row == 0 {
8047            return false;
8048        }
8049        self.slide_table(-(self.visible_rows as i64))
8050    }
8051
8052    pub fn scroll_right(&mut self) {
8053        self.scroll_columns(ColumnMove::StepRight);
8054    }
8055
8056    pub fn scroll_left(&mut self) {
8057        self.scroll_columns(ColumnMove::StepLeft);
8058    }
8059
8060    /// Which shown columns the table last drew, and the cursor's: what the control
8061    /// bar's column position says.
8062    pub fn columns_on_screen(&self) -> Option<OnScreen> {
8063        self.on_screen
8064    }
8065
8066    /// How many columns scroll: the shown ones right of those drawn frozen.
8067    fn scroll_count(&self) -> usize {
8068        self.column_order.len().saturating_sub(self.frozen_shown())
8069    }
8070
8071    /// Move the view sideways, leaving the cursor where it is unless the view leaves
8072    /// it behind on the left. Planned from the widths the columns were last drawn at;
8073    /// reads nothing. A page that needs a column not drawn yet waits for the next
8074    /// draw, which measures it from the rows on hand; a relative move typed behind it
8075    /// waits too and lands after it, in order, so no key is lost or planned on a guess.
8076    pub fn scroll_columns(&mut self, mv: ColumnMove) {
8077        if matches!(
8078            mv,
8079            ColumnMove::First | ColumnMove::Last | ColumnMove::Reveal(_)
8080        ) {
8081            // Where these go does not depend on where the moves before them went.
8082            self.column_moves.clear();
8083        }
8084        if self.column_moves.is_empty()
8085            && let Some(start) = self.plan_known(mv)
8086        {
8087            self.apply_column_move(mv, start);
8088        } else {
8089            self.wait(WaitingMove::View(mv));
8090        }
8091    }
8092
8093    /// Move the column cursor (`h` `l` `[` `]` `{` `}`), the view following only when
8094    /// the cursor would leave the screen. Reads nothing; a move that needs a column
8095    /// not drawn yet waits for the next draw, in order, as [`Self::scroll_columns`]
8096    /// says.
8097    pub fn move_cursor(&mut self, mv: CursorMove) {
8098        if matches!(mv, CursorMove::First | CursorMove::Last) {
8099            self.column_moves.clear();
8100        }
8101        if !self.column_moves.is_empty()
8102            || !self.land_cursor_move(mv, &mut |state: &mut Self, view| state.plan_known(view))
8103        {
8104            self.wait(WaitingMove::Cursor(mv));
8105        }
8106    }
8107
8108    /// Put the cursor on the shown column `name` and show it as `g` does: left where
8109    /// it is when already whole on screen, else first after the frozen columns, or on
8110    /// the last page when it is there. A frozen column is on screen already.
8111    pub fn go_to_column(&mut self, name: &str) {
8112        let Some(at) = self.column_order.iter().position(|c| c == name) else {
8113            return;
8114        };
8115        self.column_moves.clear();
8116        self.place_cursor_at(at);
8117        if let Some(index) = at.checked_sub(self.frozen_shown()) {
8118            self.scroll_columns(ColumnMove::Reveal(index));
8119        }
8120    }
8121
8122    /// Put the cursor on the shown column `name`, scrolling as little as it takes to
8123    /// show it.
8124    pub fn set_current_column(&mut self, name: &str) {
8125        let Some(at) = self.column_order.iter().position(|c| c == name) else {
8126            return;
8127        };
8128        self.column_moves.clear();
8129        self.place_cursor_at(at);
8130        self.follow_cursor(&mut |state: &mut Self, view| state.plan_known(view));
8131    }
8132
8133    /// The column cursor's column: the one the per-column keys act on (value counts,
8134    /// copying a cell, the sidebar and inspector opening on it, a find in one column).
8135    /// The first shown column until the cursor moves; `None` with no columns shown.
8136    pub fn current_column(&self) -> Option<&str> {
8137        self.cursor_index().map(|at| self.column_order[at].as_str())
8138    }
8139
8140    /// The cursor's place among the shown columns, from 0, frozen ones first.
8141    pub fn current_column_index(&self) -> Option<usize> {
8142        self.cursor_index()
8143    }
8144
8145    fn cursor_index(&self) -> Option<usize> {
8146        let last = self.column_order.len().checked_sub(1)?;
8147        Some(
8148            self.cursor_column
8149                .as_deref()
8150                .and_then(|name| self.column_order.iter().position(|c| c == name))
8151                .unwrap_or(self.cursor_at.min(last)),
8152        )
8153    }
8154
8155    fn place_cursor_at(&mut self, at: usize) {
8156        self.cursor_column = self.column_order.get(at).cloned();
8157        self.cursor_at = at;
8158    }
8159
8160    /// After the shown columns changed: the cursor stays on its column by name, or,
8161    /// where that was hidden, takes the one now in its place; the next draw shows it.
8162    fn settle_cursor(&mut self) {
8163        let at = self.cursor_index().unwrap_or(0);
8164        self.place_cursor_at(at);
8165        self.reveal_cursor = true;
8166    }
8167
8168    /// Queue a move for the next draw, behind any already waiting.
8169    fn wait(&mut self, mv: WaitingMove) {
8170        if self.column_moves.len() < MAX_WAITING_MOVES {
8171            self.column_moves.push(mv);
8172        }
8173    }
8174
8175    /// Scroll as little as it takes to show the cursor's column whole, with `plan`;
8176    /// a plan that needs a width not drawn yet waits for the next draw.
8177    fn follow_cursor(&mut self, plan: &mut impl FnMut(&mut Self, ColumnMove) -> Option<usize>) {
8178        let Some(index) = self
8179            .cursor_index()
8180            .and_then(|at| at.checked_sub(self.frozen_shown()))
8181        else {
8182            return;
8183        };
8184        let view = ColumnMove::Keep(index);
8185        match plan(self, view) {
8186            // On screen already: nothing moves, and the trail `[` retraces stays.
8187            Some(start) if start == self.termcol_index => {}
8188            Some(start) => self.apply_column_move(view, start),
8189            None => self.wait(WaitingMove::View(view)),
8190        }
8191    }
8192
8193    /// Land a cursor move, the view planned with `plan`. Returns false, changing
8194    /// nothing, when a page cannot be planned yet: where it lands decides the cursor.
8195    fn land_cursor_move(
8196        &mut self,
8197        mv: CursorMove,
8198        plan: &mut impl FnMut(&mut Self, ColumnMove) -> Option<usize>,
8199    ) -> bool {
8200        let Some(cursor) = self.cursor_index() else {
8201            return true;
8202        };
8203        let last = self.column_order.len() - 1;
8204        let frozen = self.frozen_shown();
8205        match mv {
8206            CursorMove::Left | CursorMove::Right => {
8207                let at = if mv == CursorMove::Left {
8208                    cursor.saturating_sub(1)
8209                } else {
8210                    (cursor + 1).min(last)
8211                };
8212                self.place_cursor_at(at);
8213                self.follow_cursor(plan);
8214            }
8215            CursorMove::First | CursorMove::Last => {
8216                let (at, view) = if mv == CursorMove::First {
8217                    (0, ColumnMove::First)
8218                } else {
8219                    (last, ColumnMove::Last)
8220                };
8221                self.place_cursor_at(at);
8222                match plan(self, view) {
8223                    Some(start) => self.apply_column_move(view, start),
8224                    None => self.wait(WaitingMove::View(view)),
8225                }
8226            }
8227            CursorMove::PageLeft | CursorMove::PageRight => {
8228                let view = if mv == CursorMove::PageLeft {
8229                    ColumnMove::PageLeft
8230                } else {
8231                    ColumnMove::PageRight
8232                };
8233                let Some(start) = plan(self, view) else {
8234                    return false;
8235                };
8236                let from = self.termcol_index;
8237                self.apply_column_move(view, start);
8238                let at = if self.termcol_index != from {
8239                    // The new page, from its first column.
8240                    frozen + self.termcol_index
8241                } else if mv == CursorMove::PageRight {
8242                    // On the last page already: its last column.
8243                    last
8244                } else if cursor > frozen {
8245                    // On the first page: its first column, then the first of all.
8246                    frozen
8247                } else {
8248                    0
8249                };
8250                self.place_cursor_at(at.min(last));
8251            }
8252        }
8253        true
8254    }
8255
8256    /// The scrolling columns, by name.
8257    fn scrolling_names(&self) -> &[String] {
8258        &self.column_order[self.frozen_shown().min(self.column_order.len())..]
8259    }
8260
8261    /// `[` straight after the `]` that came here goes back where that one started,
8262    /// whatever the widths say, so a page and back is the page left.
8263    fn retrace(&self, mv: ColumnMove) -> Option<usize> {
8264        let &(back, to) = self.page_trail.last()?;
8265        (mv == ColumnMove::PageLeft && to == self.termcol_index).then_some(back)
8266    }
8267
8268    /// Where `mv` lands on the widths drawn in this view, or `None` when it needs one
8269    /// not drawn yet (or the room, before the first draw).
8270    fn plan_known(&self, mv: ColumnMove) -> Option<usize> {
8271        if let Some(back) = self.retrace(mv) {
8272            return Some(back);
8273        }
8274        let needs_widths = match mv {
8275            ColumnMove::StepLeft | ColumnMove::StepRight | ColumnMove::First => false,
8276            // Back to a column at or left of the first shown needs no width.
8277            ColumnMove::Keep(column) => column > self.termcol_index,
8278            _ => true,
8279        };
8280        let room = match self.scroll_room {
8281            Some(room) => room,
8282            None if needs_widths => return None,
8283            None => Room::default(),
8284        };
8285        let names = self.scrolling_names();
8286        crate::widgets::column_paging::plan(mv, self.termcol_index, names.len(), room, |i| {
8287            self.drawn_width(&names[i])
8288        })
8289    }
8290
8291    /// Land `mv` at `start`, keeping the trail `[` retraces. A cursor the view leaves
8292    /// behind on the left comes along, to the first column shown.
8293    fn apply_column_move(&mut self, mv: ColumnMove, start: usize) {
8294        let from = self.termcol_index;
8295        let start = start.min(self.scroll_count().saturating_sub(1));
8296        match mv {
8297            ColumnMove::PageRight => {
8298                if start > from {
8299                    self.page_trail.push((from, start));
8300                }
8301            }
8302            ColumnMove::PageLeft if self.retrace(mv) == Some(start) => {
8303                self.page_trail.pop();
8304            }
8305            _ => self.page_trail.clear(),
8306        }
8307        self.scroll_columns_to(start);
8308        let frozen = self.frozen_shown();
8309        if let Some(cursor) = self.cursor_index()
8310            && cursor >= frozen
8311            && cursor < frozen + self.termcol_index
8312        {
8313            self.place_cursor_at(frozen + self.termcol_index);
8314        }
8315    }
8316
8317    /// Forget sideways moves waiting on a draw and the trail `[` retraces: the
8318    /// columns they were counted over are gone.
8319    fn clear_column_moves(&mut self) {
8320        self.column_moves.clear();
8321        self.page_trail.clear();
8322    }
8323
8324    /// Start the scrolling columns at `start`, re-slicing the buffer held.
8325    fn scroll_columns_to(&mut self, start: usize) {
8326        let start = start.min(self.scroll_count().saturating_sub(1));
8327        if start != self.termcol_index {
8328            self.termcol_index = start;
8329            self.rescroll_columns();
8330        }
8331    }
8332
8333    /// Record the scrolling side as the renderer lays it out, land the moves waiting
8334    /// on it, in order, and bring the cursor back on screen when it may have left,
8335    /// with `width`, which measures a column not drawn yet from the rows on hand.
8336    /// Called while drawing, before the scrolling columns are drawn; reads nothing,
8337    /// and measures only the columns a move crosses. With no rows on hand the moves
8338    /// wait for a draw that has them.
8339    fn land_column_moves(&mut self, room: Room, mut width: impl FnMut(&mut Self, &str) -> u16) {
8340        if self.scroll_room != Some(room) {
8341            // A resize, or a frozen column given back: the cursor may be off screen.
8342            self.reveal_cursor = true;
8343        }
8344        self.scroll_room = Some(room);
8345        if (self.column_moves.is_empty() && !self.reveal_cursor)
8346            || !self.buffer_on_hand()
8347            || self.defer_collect
8348        {
8349            return;
8350        }
8351        let mut plan = |state: &mut Self, mv: ColumnMove| -> Option<usize> {
8352            if let Some(back) = state.retrace(mv) {
8353                return Some(back);
8354            }
8355            let from = state.termcol_index;
8356            let count = state.scroll_count();
8357            Some(
8358                crate::widgets::column_paging::plan(mv, from, count, room, |i| {
8359                    let name = state.scrolling_names()[i].clone();
8360                    Some(width(state, &name))
8361                })
8362                .unwrap_or(from),
8363            )
8364        };
8365        for mv in std::mem::take(&mut self.column_moves) {
8366            match mv {
8367                WaitingMove::View(mv) => {
8368                    let start = plan(self, mv).unwrap_or(self.termcol_index);
8369                    self.apply_column_move(mv, start);
8370                }
8371                WaitingMove::Cursor(mv) => {
8372                    self.land_cursor_move(mv, &mut plan);
8373                }
8374            }
8375        }
8376        if std::mem::take(&mut self.reveal_cursor) {
8377            self.follow_cursor(&mut plan);
8378        }
8379    }
8380
8381    /// Show the new column window, reading nothing.
8382    ///
8383    /// A sideways move changes which columns are on screen, not which rows, so it
8384    /// re-slices the buffer already held. It must not go through [`collect`], which
8385    /// counts the rows when the count is not yet known: `App::handle` calls
8386    /// `scroll_right` inline on the thread that draws and reads keys, and
8387    /// `key_acts_while_busy` lets Left and Right through while other work runs. On a
8388    /// staged-open cloud hive that count is a metadata read per object, and taken
8389    /// there it is a freeze no keystroke can interrupt.
8390    ///
8391    /// Nothing is drawn when there is no buffer to re-slice, or when what is held
8392    /// does not match the range it claims. [`collect`] reloaded the page in that
8393    /// second case; this does not, because `load_buffer` is a collect of that page
8394    /// and on a cloud hive that is row groups over the wire — the same freeze in a
8395    /// smaller size.
8396    ///
8397    /// The index still moves, so presses before the first buffer lands are spent on
8398    /// a view that cannot show them yet, and the first frame drawn is already scrolled
8399    /// to wherever they left it. That is the pre-existing behaviour: the old path
8400    /// redrew each press, but only by paying the wait this exists to avoid.
8401    ///
8402    /// [`collect`]: Self::collect
8403    fn rescroll_columns(&mut self) {
8404        if self.defer_collect || !self.buffer_on_hand() {
8405            return;
8406        }
8407        self.slice_buffer_into_display();
8408        if self.table_state.selected().is_none() {
8409            self.table_state.select(Some(0));
8410        }
8411    }
8412
8413    pub fn headers(&self) -> Vec<String> {
8414        self.column_order.clone()
8415    }
8416
8417    pub fn set_column_order(&mut self, order: Vec<String>) {
8418        self.column_order = order;
8419        self.clear_column_moves();
8420        // Fewer columns shown may leave the scroll past the last; keep one on screen.
8421        self.termcol_index = self
8422            .termcol_index
8423            .min(self.scroll_count().saturating_sub(1));
8424        self.buffered_start_row = 0;
8425        self.buffered_end_row = 0;
8426        self.buffered_df = None;
8427        self.settle_cursor();
8428        self.collect();
8429    }
8430
8431    pub fn set_locked_columns(&mut self, count: usize) {
8432        self.locked_columns_count = count.min(self.column_order.len());
8433        self.clear_column_moves();
8434        self.settle_cursor();
8435        self.termcol_index = self
8436            .termcol_index
8437            .min(self.scroll_count().saturating_sub(1));
8438        self.buffered_start_row = 0;
8439        self.buffered_end_row = 0;
8440        self.buffered_df = None;
8441        self.collect();
8442    }
8443
8444    pub fn locked_columns_count(&self) -> usize {
8445        self.locked_columns_count
8446    }
8447
8448    /// How many columns are drawn frozen: the count asked for, or fewer while the
8449    /// last layout could not fit them all beside a usable scrolling column. The ones
8450    /// left out lead the scrolling columns, so every column stays reachable.
8451    pub fn frozen_shown(&self) -> usize {
8452        let (asked, shown) = self.frozen_fit;
8453        if asked == self.locked_columns_count {
8454            shown.min(asked)
8455        } else {
8456            self.locked_columns_count
8457        }
8458    }
8459
8460    /// Take the layout's word for how many frozen columns fit, and re-slice the
8461    /// scrolling columns to start after them. Unscrolled, the frozen columns left out
8462    /// lead the scrolling ones; scrolled, the column the scroll started at stays
8463    /// first where it can, so a resize does not also move the view. Reads nothing;
8464    /// called while drawing, and only re-selects columns of the buffer already held.
8465    fn fit_frozen(&mut self, shown: usize) {
8466        let before = self.frozen_shown();
8467        let shown = shown.min(self.locked_columns_count);
8468        if shown == before {
8469            self.frozen_fit = (self.locked_columns_count, shown);
8470            return;
8471        }
8472        if self.defer_collect || !self.buffer_on_hand() {
8473            return;
8474        }
8475        self.frozen_fit = (self.locked_columns_count, shown);
8476        // The scrolling indices the trail was kept in shift with the frozen count.
8477        self.page_trail.clear();
8478        if self.termcol_index > 0 {
8479            let first = before + self.termcol_index;
8480            let last = self.column_order.len().saturating_sub(1);
8481            self.termcol_index = first.min(last).saturating_sub(shown);
8482        }
8483        self.slice_buffer_into_display();
8484    }
8485
8486    /// The type a column has in the frame on screen: with its name, the identity its
8487    /// width is kept under.
8488    fn width_dtype(&self, name: &str) -> DataType {
8489        self.schema.get(name).cloned().unwrap_or(DataType::Null)
8490    }
8491
8492    /// How a column's width is chosen.
8493    pub fn width_choice(&self, name: &str) -> WidthChoice {
8494        self.widths.choice(name, &self.width_dtype(name))
8495    }
8496
8497    /// The width a column was last drawn at, if it has been drawn.
8498    pub fn shown_width(&self, name: &str) -> Option<u16> {
8499        self.widths.shown(name, &self.width_dtype(name))
8500    }
8501
8502    /// The width the column takes on screen, the room it filled at the right edge
8503    /// included.
8504    pub fn on_screen_width(&self, name: &str) -> Option<u16> {
8505        self.widths.on_screen(name, &self.width_dtype(name))
8506    }
8507
8508    /// The width a column draws at in this view, if it has been drawn since the
8509    /// widths were last relearned. What a sideways page is planned with.
8510    fn drawn_width(&self, name: &str) -> Option<u16> {
8511        self.widths.drawn(name, &self.width_dtype(name))
8512    }
8513
8514    /// Set how each named column's width is chosen. Reads nothing: a fit is taken
8515    /// from the rows on screen when the table is next drawn.
8516    pub fn set_width_choices(&mut self, choices: impl IntoIterator<Item = (String, WidthChoice)>) {
8517        for (name, choice) in choices {
8518            let dtype = self.width_dtype(&name);
8519            self.widths.set_choice(&name, &dtype, choice);
8520        }
8521    }
8522
8523    /// One column's rows on screen, from the buffer already held, as the table draws
8524    /// them. For fitting a column that may be scrolled out of view.
8525    fn page_column(&self, name: &str, offset: usize, len: usize) -> Option<DataFrame> {
8526        let column = self.buffered_df.as_ref()?.select([name]).ok()?;
8527        visible_slice(&column, offset, len)
8528    }
8529
8530    // Getter methods for view creation
8531    /// Filters for a view: while drilled into a group these are the grouped view's,
8532    /// which is what a view reproduces (it cannot express a drill-down).
8533    pub fn get_filters(&self) -> &[FilterStatement] {
8534        match &self.grouped {
8535            Some(view) => &view.filters,
8536            None => &self.filters,
8537        }
8538    }
8539
8540    pub fn get_sort_columns(&self) -> &[String] {
8541        match &self.grouped {
8542            Some(view) => &view.sort_columns,
8543            None => &self.sort_columns,
8544        }
8545    }
8546
8547    pub fn get_sort_ascending(&self) -> bool {
8548        match &self.grouped {
8549            Some(view) => view.sort_ascending,
8550            None => self.sort_ascending,
8551        }
8552    }
8553
8554    pub fn get_sort_descending(&self) -> &[bool] {
8555        match &self.grouped {
8556            Some(view) => &view.sort_descending,
8557            None => &self.sort_descending,
8558        }
8559    }
8560
8561    /// Filters applied to the frame on screen (inside the group while drilled). This is
8562    /// what the Sort & Filter sidebar shows and edits.
8563    pub fn view_filters(&self) -> &[FilterStatement] {
8564        &self.filters
8565    }
8566
8567    pub fn view_sort_columns(&self) -> &[String] {
8568        &self.sort_columns
8569    }
8570
8571    pub fn view_sort_ascending(&self) -> bool {
8572        self.sort_ascending
8573    }
8574
8575    pub fn view_sort_descending(&self) -> &[bool] {
8576        &self.sort_descending
8577    }
8578
8579    /// The header's sort marks: the sidebar's sort, or else the ORDER BY of the SQL
8580    /// in effect, while its own rows are on screen (not a group drilled into).
8581    pub fn header_sort(&self) -> (Vec<String>, Vec<bool>) {
8582        if self.sort_columns.is_empty() && self.grouped.is_none() {
8583            self.query_order.iter().cloned().unzip()
8584        } else {
8585            (self.sort_columns.clone(), self.sort_descending.clone())
8586        }
8587    }
8588
8589    /// The pivot/melt result in effect, for a snapshot that may need to put it back.
8590    pub fn reshaped_lf_clone(&self) -> Option<LazyFrame> {
8591        self.reshaped_lf.clone()
8592    }
8593
8594    pub fn get_column_order(&self) -> &[String] {
8595        &self.column_order
8596    }
8597
8598    /// Whether the table shows its defaults: no query, filters, sort or
8599    /// reshape, every column in file order, nothing locked. A view saved
8600    /// from this state would carry nothing — and, matching by schema, it
8601    /// would shadow real views in the apply gate as a well-used no-op.
8602    pub fn is_at_defaults(&self) -> bool {
8603        self.sampled.is_none()
8604            && self.column_changes.is_empty()
8605            && self.active_query.is_empty()
8606            && self.active_sql_query.is_empty()
8607            && self.active_fuzzy_query.is_empty()
8608            && self.filters.is_empty()
8609            && self.sort_columns.is_empty()
8610            && self.last_pivot_spec.is_none()
8611            && self.last_melt_spec.is_none()
8612            && self.locked_columns_count() == 0
8613            && self
8614                .column_order
8615                .iter()
8616                .map(String::as_str)
8617                .eq(self.schema.iter_names().map(|s| s.as_str()))
8618    }
8619
8620    pub fn get_active_query(&self) -> &str {
8621        &self.active_query
8622    }
8623
8624    pub fn get_active_sql_query(&self) -> &str {
8625        &self.active_sql_query
8626    }
8627
8628    /// Whether the rows on screen can be read at all: a sort, a filter or a column
8629    /// named in the layout that the frame does not have fails here. Resolves the plan
8630    /// and reads no rows.
8631    pub fn check_plan(&self) -> PolarsResult<()> {
8632        self.lf
8633            .clone()
8634            .select(self.binary_stub_exprs())
8635            .collect_schema()
8636            .map(|_| ())
8637    }
8638
8639    /// The view as it is now, to go back to if a query fails while running.
8640    pub fn rollback_point(&self) -> ViewRollback {
8641        ViewRollback {
8642            root_generation: self.root_generation,
8643            counted: None,
8644            drawn_start: self.drawn_start,
8645            lf: self.lf.clone(),
8646            unsorted_lf: self.unsorted_lf.clone(),
8647            base_lf: self.base_lf.clone(),
8648            df: self.df.clone(),
8649            locked_df: self.locked_df.clone(),
8650            table_state: self.table_state,
8651            start_row: self.start_row,
8652            termcol_index: self.termcol_index,
8653            cursor_column: self.cursor_column.clone(),
8654            cursor_at: self.cursor_at,
8655            schema: self.schema.clone(),
8656            num_rows: self.num_rows,
8657            num_rows_valid: self.num_rows_valid,
8658            len_generation: self.len_generation,
8659            filters: self.filters.clone(),
8660            sort_columns: self.sort_columns.clone(),
8661            sort_descending: self.sort_descending.clone(),
8662            sort_ascending: self.sort_ascending,
8663            active_query: self.active_query.clone(),
8664            active_sql_query: self.active_sql_query.clone(),
8665            query_order: self.query_order.clone(),
8666            active_fuzzy_query: self.active_fuzzy_query.clone(),
8667            column_order: self.column_order.clone(),
8668            locked_columns_count: self.locked_columns_count,
8669            frozen_fit: self.frozen_fit,
8670            grouped: self.grouped.clone(),
8671            reshaped_lf: self.reshaped_lf.clone(),
8672            last_pivot_spec: self.last_pivot_spec.clone(),
8673            last_melt_spec: self.last_melt_spec.clone(),
8674            reshape_source: self.reshape_source.clone(),
8675            base_steps: self.base_steps.clone(),
8676            reshape_steps: self.reshape_steps.clone(),
8677            lineage: self.lineage.clone(),
8678            reshape_lineage: self.reshape_lineage.clone(),
8679            group_source: self.group_source.clone(),
8680            drilled_down_group_index: self.drilled_down_group_index,
8681            drilled_down_group_key: self.drilled_down_group_key.clone(),
8682            drilled_down_group_key_columns: self.drilled_down_group_key_columns.clone(),
8683            drift_column_present: self.drift_column_present,
8684            view_numbered: self.view_numbered,
8685            drift_groups: self.drift_groups.clone(),
8686            notes: self.notes.clone(),
8687            notes_seen: self.notes_seen,
8688            view_notes: self.view_notes.clone(),
8689            column_changes: self.column_changes.clone(),
8690            changes_version: self.changes_version,
8691            changes_dropped: self.changes_dropped.clone(),
8692            observed_bytes_per_row: self.observed_bytes_per_row,
8693            buffered_start_row: self.buffered_start_row,
8694            buffered_end_row: self.buffered_end_row,
8695            buffered_df: self.buffered_df.clone(),
8696        }
8697    }
8698
8699    /// Put back the view `rollback_point` saved, with no error showing. Its row count
8700    /// and buffer come back with it, and a count of it that landed meanwhile (see
8701    /// [`ViewRollback::count_landed`]), so nothing is read again. Nothing is read here:
8702    /// a view with no rows on hand has them read by the caller's next collect.
8703    ///
8704    /// A checkpoint taken over data since replaced — a join of the remaining footers,
8705    /// a column read as text — holds frames built on data no longer loaded. Putting
8706    /// those back would mix two roots, so the view returns to the data as loaded instead.
8707    pub fn roll_back(&mut self, saved: ViewRollback) {
8708        if saved.root_generation != self.root_generation {
8709            self.return_to_root();
8710            return;
8711        }
8712        self.widths.keep_learned();
8713        self.drawn_start = saved.drawn_start;
8714        self.lf = saved.lf;
8715        self.unsorted_lf = saved.unsorted_lf;
8716        self.base_lf = saved.base_lf;
8717        self.df = saved.df;
8718        self.locked_df = saved.locked_df;
8719        self.table_state = saved.table_state;
8720        self.start_row = saved.start_row;
8721        self.termcol_index = saved.termcol_index;
8722        self.clear_column_moves();
8723        self.schema = saved.schema;
8724        self.num_rows = saved.num_rows;
8725        self.num_rows_valid = saved.num_rows_valid;
8726        self.len_generation = saved.len_generation;
8727        self.filters = saved.filters;
8728        self.sort_columns = saved.sort_columns;
8729        self.sort_descending = saved.sort_descending;
8730        self.sort_ascending = saved.sort_ascending;
8731        self.active_query = saved.active_query;
8732        self.active_sql_query = saved.active_sql_query;
8733        self.query_order = saved.query_order;
8734        self.active_fuzzy_query = saved.active_fuzzy_query;
8735        self.column_order = saved.column_order;
8736        self.locked_columns_count = saved.locked_columns_count;
8737        self.frozen_fit = saved.frozen_fit;
8738        self.cursor_column = saved.cursor_column;
8739        self.cursor_at = saved.cursor_at;
8740        self.reveal_cursor = true;
8741        self.grouped = saved.grouped;
8742        // A q query or a search forgets the pivot or melt it replaces.
8743        self.reshaped_lf = saved.reshaped_lf;
8744        self.last_pivot_spec = saved.last_pivot_spec;
8745        self.last_melt_spec = saved.last_melt_spec;
8746        self.reshape_source = saved.reshape_source;
8747        self.base_steps = saved.base_steps;
8748        self.reshape_steps = saved.reshape_steps;
8749        self.lineage = saved.lineage;
8750        self.reshape_lineage = saved.reshape_lineage;
8751        self.group_source = saved.group_source;
8752        self.drilled_down_group_index = saved.drilled_down_group_index;
8753        self.drilled_down_group_key = saved.drilled_down_group_key;
8754        self.drilled_down_group_key_columns = saved.drilled_down_group_key_columns;
8755        self.drift_column_present = saved.drift_column_present;
8756        self.view_numbered = saved.view_numbered;
8757        self.drift_groups = saved.drift_groups;
8758        self.column_changes = saved.column_changes;
8759        self.changes_version = saved.changes_version;
8760        self.changes_dropped = saved.changes_dropped;
8761        self.notes = saved.notes;
8762        self.notes_seen = saved.notes_seen;
8763        self.view_notes = saved.view_notes;
8764        self.observed_bytes_per_row = saved.observed_bytes_per_row;
8765        self.buffered_start_row = saved.buffered_start_row;
8766        self.buffered_end_row = saved.buffered_end_row;
8767        self.buffered_df = saved.buffered_df;
8768        self.error = None;
8769        // After the frame, so the count is taken as this frame's.
8770        if let Some(counted) = saved.counted {
8771            self.take_count(counted.rows, counted.file_row_groups.as_deref());
8772        }
8773    }
8774
8775    /// Run `steps` as one transition of the view: planned, never read (no collect
8776    /// runs while they do), starting with no error showing. If a step fails, the view
8777    /// before them is put back and the error returned. If they all plan, the view
8778    /// before them comes back with the result, for the caller to restore with
8779    /// [`Self::roll_back`] should reading the new view's rows fail.
8780    pub fn try_transition<T, E>(
8781        &mut self,
8782        steps: impl FnOnce(&mut Self) -> std::result::Result<T, E>,
8783    ) -> std::result::Result<(T, ViewRollback), E> {
8784        let saved = self.rollback_point();
8785        self.error = None;
8786        match self.deferred(steps) {
8787            Ok(value) => Ok((value, saved)),
8788            Err(e) => {
8789                self.roll_back(saved);
8790                Err(e)
8791            }
8792        }
8793    }
8794
8795    /// Run `steps` with every collect they would make left to the caller, who reads
8796    /// the rows off the UI thread (`prepare_async_collect`).
8797    pub fn deferred<R>(&mut self, steps: impl FnOnce(&mut Self) -> R) -> R {
8798        let deferred = std::mem::replace(&mut self.defer_collect, true);
8799        let result = steps(self);
8800        self.defer_collect = deferred;
8801        result
8802    }
8803
8804    /// A background count of frame `len_generation` came back. Taken when that frame is
8805    /// the one on screen; returns whether it was.
8806    pub fn count_landed(
8807        &mut self,
8808        len_generation: u64,
8809        rows: usize,
8810        file_row_groups: Option<&[Vec<usize>]>,
8811    ) -> bool {
8812        let current = len_generation == self.len_generation;
8813        if current {
8814            self.take_count(rows, file_row_groups);
8815        }
8816        current
8817    }
8818
8819    /// What a staged open leaves: a row total from however far the buffer reached,
8820    /// with no count taken.
8821    #[cfg(test)]
8822    pub(crate) fn set_provisional_rows(&mut self, n: usize) {
8823        self.num_rows = n;
8824    }
8825
8826    /// The count of the frame on screen: from the files' row groups when there are
8827    /// some, else the total.
8828    fn take_count(&mut self, rows: usize, file_row_groups: Option<&[Vec<usize>]>) {
8829        match file_row_groups {
8830            Some(groups) => self.record_file_row_groups(groups),
8831            None => self.set_num_rows(rows),
8832        }
8833    }
8834
8835    /// The follow of the file this dataset reads, while it is followed.
8836    pub fn follow(&self) -> Option<&crate::follow::Follow> {
8837        self.follow.as_ref()
8838    }
8839
8840    pub fn follow_mut(&mut self) -> Option<&mut crate::follow::Follow> {
8841        self.follow.as_mut()
8842    }
8843
8844    /// Join `fields`, which arrived in a followed pipe's NDJSON after the open, to the
8845    /// dataset: its scan reads them, and they go on the end of the column order, as a
8846    /// dataset's footers join theirs. `Err` while the view is a query, a reshape or a
8847    /// group, which would lose the columns it is built from: the caller holds them
8848    /// until the view is back on the data. `Ok(false)` when there is nothing to join.
8849    pub(crate) fn join_followed_fields(
8850        &mut self,
8851        fields: &[Field],
8852    ) -> std::result::Result<bool, ()> {
8853        if !self.scan_is_the_root() {
8854            return Err(());
8855        }
8856        let (Some(follow), Some(format)) = (self.follow.as_ref(), self.read_as) else {
8857            return Ok(false);
8858        };
8859        let (path, rows) = (follow.path().to_path_buf(), follow.shown());
8860        let Some(mut lf) = crate::follow::widen(&self.original_lf, &path, format, fields, rows)
8861        else {
8862            return Ok(false);
8863        };
8864        let Ok(schema) = lf.collect_schema() else {
8865            return Ok(false);
8866        };
8867        let known: std::collections::HashSet<&str> =
8868            self.column_order.iter().map(String::as_str).collect();
8869        let joining: Vec<String> = schema
8870            .iter_names()
8871            .map(|name| name.to_string())
8872            .filter(|name| !known.contains(name.as_str()))
8873            .collect();
8874        drop(known);
8875        self.column_order.extend(joining);
8876        self.replace_root(lf, schema);
8877        if self.is_pristine() {
8878            // The same rows the watcher counted, with more columns.
8879            self.set_num_rows(rows);
8880        }
8881        // Rebuilt but not read, as a footer join is: the caller reads the rows on
8882        // screen off the event loop.
8883        self.deferred(Self::apply_transformations);
8884        Ok(true)
8885    }
8886
8887    /// Follow the file this dataset reads with `follow`, whose watcher is running.
8888    pub fn start_following(&mut self, follow: crate::follow::Follow) {
8889        self.follow = Some(follow);
8890    }
8891
8892    /// Stop following. The rows read so far stay.
8893    pub fn stop_following(&mut self) {
8894        if let Some(mut follow) = self.follow.take() {
8895            follow.end();
8896        }
8897    }
8898
8899    /// Put the view on the last page, leaving the cursor where it is until the rows of
8900    /// that page are read: the next read is of that page alone.
8901    pub(crate) fn aim_at_end(&mut self) {
8902        if self.num_rows_valid && self.visible_rows > 0 {
8903            self.start_row = self.num_rows.saturating_sub(self.visible_rows);
8904        }
8905    }
8906
8907    /// Whether the cursor is on the last row of a view whose length is known.
8908    pub fn on_last_row(&self) -> bool {
8909        self.num_rows_valid
8910            && (self.num_rows == 0
8911                || self.start_row + self.table_state.selected().unwrap_or(0) + 1 >= self.num_rows)
8912    }
8913
8914    /// Every frame the view holds that carries the scan of the data as loaded.
8915    fn each_frame(&mut self, mut f: impl FnMut(&mut LazyFrame)) {
8916        f(&mut self.original_lf);
8917        f(&mut self.base_lf);
8918        f(&mut self.lf);
8919        if let Some(lf) = self.unsorted_lf.as_mut() {
8920            f(lf);
8921        }
8922        if let Some(lf) = self.reshaped_lf.as_mut() {
8923            f(lf);
8924        }
8925        if let Some(source) = self.group_source.as_mut() {
8926            f(&mut source.rows);
8927        }
8928        if let Some(grouped) = self.grouped.as_mut() {
8929            f(&mut grouped.lf);
8930            f(&mut grouped.base_lf);
8931            if let Some(source) = grouped.group_source.as_mut() {
8932                f(&mut source.rows);
8933            }
8934        }
8935    }
8936
8937    /// The followed file holds `rows` complete rows now: every frame reads that many,
8938    /// so the query, filters and sort run over the new ones too. `restarted` when the
8939    /// file was read again from its start. Returns whether the rows on hand still
8940    /// stand: a view that only filters, with nothing reordered, keeps the rows it had,
8941    /// since rows only arrive after them.
8942    pub(crate) fn follow_to(&mut self, rows: usize, restarted: bool) -> bool {
8943        let Some(path) = self.follow.as_ref().map(|f| f.path().to_path_buf()) else {
8944            return true;
8945        };
8946        let rows_stand = !restarted
8947            && self.sort_columns.is_empty()
8948            && self.sort_ascending
8949            && self.scan_is_the_root();
8950        let known = self.known_before_follow(&path, restarted);
8951        self.each_frame(|lf| crate::follow::bound(lf, &path, rows));
8952        self.invalidate_num_rows();
8953        self.follow_known = known.map(|known| (self.len_generation, known));
8954        if self.is_pristine() {
8955            // The watcher counted them as the scan reads them: nothing to count again.
8956            self.set_num_rows(rows);
8957        } else if self.scan_is_the_root() {
8958            self.pristine_rows = Some(rows);
8959        }
8960        if restarted {
8961            self.start_row = 0;
8962            self.table_state.select(Some(0));
8963        }
8964        if !rows_stand {
8965            self.drop_buffer();
8966        }
8967        rows_stand
8968    }
8969
8970    /// Where the view's rows are known, before the frames read more of the followed file
8971    /// at `path`: what was known for the count on screen, and the count itself when it
8972    /// is exact. Only for a view whose rows are each kept or not by itself (filters and
8973    /// a sort over the file's rows), so the rows that arrive are counted alone.
8974    fn known_before_follow(&mut self, path: &Path, restarted: bool) -> Option<Vec<(usize, usize)>> {
8975        if restarted || self.is_pristine() || !self.scan_is_the_root() {
8976            return None;
8977        }
8978        let mut known = self
8979            .follow_known
8980            .take()
8981            .filter(|(generation, _)| *generation == self.len_generation)
8982            .map(|(_, known)| known);
8983        if self.num_rows_valid
8984            && let Some(row) = crate::follow::bound_of(&self.lf, path)
8985        {
8986            let known = known.get_or_insert_with(Vec::new);
8987            // One point per stretch of marks is enough to read on from.
8988            if let [.., before, last] = known.as_slice()
8989                && last.1 - before.1 < crate::follow::MARK_ROWS as usize
8990            {
8991                known.pop();
8992            }
8993            if known.last().is_none_or(|&(_, at)| at < row) {
8994                known.push((self.num_rows, row));
8995            }
8996        }
8997        known
8998    }
8999
9000    /// The followed file was deleted: every frame reads it through `file`, a handle
9001    /// held on it, which still reads what it held.
9002    pub(crate) fn read_followed_through(&mut self, file: &std::fs::File) {
9003        let Some(path) = self.follow.as_ref().map(|f| f.path().to_path_buf()) else {
9004            return;
9005        };
9006        self.each_frame(|lf| crate::follow::read_through(lf, &path, file));
9007    }
9008
9009    /// The frame on screen: the root, then the query or reshape, the filters and the
9010    /// sort. Column order is applied when rows are read.
9011    pub fn lf(&self) -> &LazyFrame {
9012        &self.lf
9013    }
9014
9015    /// Hand back the frame on screen, for a caller that built a state only to load it.
9016    pub fn into_lf(self) -> LazyFrame {
9017        self.lf
9018    }
9019
9020    /// The schema of the frame on screen.
9021    pub fn schema(&self) -> &Arc<Schema> {
9022        &self.schema
9023    }
9024
9025    /// The rows the frame holds: exact when [`Self::is_num_rows_valid`], else as far as
9026    /// the reads so far have reached.
9027    pub fn num_rows(&self) -> usize {
9028        self.num_rows
9029    }
9030
9031    /// Why the last query, step or read failed, while it is still showing.
9032    pub fn error(&self) -> Option<&PolarsError> {
9033        self.error.as_ref()
9034    }
9035
9036    /// Stop showing the last failure. The view is as it was; only the message goes.
9037    pub fn dismiss_error(&mut self) {
9038        self.error = None;
9039    }
9040
9041    /// The first row of the page on screen.
9042    pub fn start_row(&self) -> usize {
9043        self.start_row
9044    }
9045
9046    /// The hive partition columns the dataset was loaded with.
9047    pub fn partition_columns(&self) -> Option<&[String]> {
9048        self.partition_columns.as_deref()
9049    }
9050
9051    /// Whether reads use Polars' streaming engine.
9052    pub fn polars_streaming(&self) -> bool {
9053        self.polars_streaming
9054    }
9055
9056    /// The group drilled into, as its key columns and their values.
9057    pub fn drilled_group_key(&self) -> Option<(&[String], &[String])> {
9058        let values = self.drilled_down_group_key.as_deref()?;
9059        let columns = self
9060            .drilled_down_group_key_columns
9061            .as_deref()
9062            .unwrap_or_default();
9063        Some((columns, values))
9064    }
9065
9066    /// The columns of `df`, the table SQL runs against, with their types. From the
9067    /// schema already known for the data as loaded; a drilled group or a reshape only
9068    /// has its plan resolved, which reads nothing.
9069    pub fn sql_table_columns(&self) -> Vec<(String, DataType)> {
9070        let schema = if self.grouped.is_none() && self.reshaped_lf.is_none() {
9071            Some(self.original_schema.clone())
9072        } else {
9073            self.query_root().collect_schema().ok()
9074        };
9075        schema
9076            .map(|schema| {
9077                schema
9078                    .iter()
9079                    .filter(|(name, _)| name.as_str() != crate::schema_union::DRIFT_COLUMN)
9080                    .map(|(name, dtype)| (name.to_string(), dtype.clone()))
9081                    .collect()
9082            })
9083            .unwrap_or_default()
9084    }
9085
9086    /// Rows `df` holds, when that is known without counting: the data as loaded,
9087    /// once its count has come back.
9088    pub fn sql_table_rows(&self) -> Option<usize> {
9089        if self.grouped.is_some() || self.reshaped_lf.is_some() {
9090            return None;
9091        }
9092        self.pristine_rows
9093    }
9094
9095    pub fn get_active_fuzzy_query(&self) -> &str {
9096        &self.active_fuzzy_query
9097    }
9098
9099    pub fn last_pivot_spec(&self) -> Option<&PivotSpec> {
9100        self.last_pivot_spec.as_ref()
9101    }
9102
9103    pub fn last_melt_spec(&self) -> Option<&MeltSpec> {
9104        self.last_melt_spec.as_ref()
9105    }
9106
9107    /// What the pivot or melt in effect ran over. See the field.
9108    pub fn reshape_source(&self) -> Option<&ReshapeSource> {
9109        self.reshape_source.as_ref()
9110    }
9111
9112    /// Whether the view is a grouping's result (a `by` query, a SQL GROUP BY), so its
9113    /// rows drill into groups. Recorded by the query, never inferred from list columns:
9114    /// a table loaded with one is not grouped.
9115    pub fn is_grouped(&self) -> bool {
9116        self.group_source.is_some()
9117    }
9118
9119    /// Whether any column holds lists.
9120    fn has_list_columns(&self) -> bool {
9121        self.schema
9122            .iter()
9123            .any(|(_, dtype)| matches!(dtype, DataType::List(_)))
9124    }
9125
9126    pub fn group_key_columns(&self) -> Vec<String> {
9127        self.schema
9128            .iter()
9129            .filter(|(_, dtype)| !matches!(dtype, DataType::List(_)))
9130            .map(|(name, _)| name.to_string())
9131            .collect()
9132    }
9133
9134    pub fn group_value_columns(&self) -> Vec<String> {
9135        self.schema
9136            .iter()
9137            .filter(|(_, dtype)| matches!(dtype, DataType::List(_)))
9138            .map(|(name, _)| name.to_string())
9139            .collect()
9140    }
9141
9142    /// Names of binary columns in the source schema. Their values are not read into the display
9143    /// buffer (the `‹binary›` stub stands in); the renderer uses this to style those cells.
9144    pub fn binary_column_names(&self) -> std::collections::HashSet<String> {
9145        self.schema
9146            .iter()
9147            .filter(|(_, dtype)| matches!(dtype, DataType::Binary))
9148            .map(|(name, _)| name.to_string())
9149            .collect()
9150    }
9151
9152    /// Estimated heap size in bytes of the currently buffered slice (locked + scrollable), if collected.
9153    pub fn buffered_memory_bytes(&self) -> Option<usize> {
9154        let locked = self
9155            .locked_df
9156            .as_ref()
9157            .map(|df| df.estimated_size())
9158            .unwrap_or(0);
9159        let scroll = self.df.as_ref().map(|df| df.estimated_size()).unwrap_or(0);
9160        if locked == 0 && scroll == 0 {
9161            None
9162        } else {
9163            Some(locked + scroll)
9164        }
9165    }
9166
9167    /// Number of rows currently in the buffer. 0 if no buffer loaded.
9168    pub fn buffered_rows(&self) -> usize {
9169        self.buffered_end_row
9170            .saturating_sub(self.buffered_start_row)
9171    }
9172
9173    /// The first `limit` distinct non-null values of `column` among the rows already
9174    /// buffered for display, as text. Reads nothing: a column the buffer lacks gives
9175    /// none.
9176    pub(crate) fn buffered_values(&self, column: &str, limit: usize) -> Vec<String> {
9177        let Some(series) = [self.df.as_ref(), self.locked_df.as_ref()]
9178            .into_iter()
9179            .flatten()
9180            .find_map(|df| df.column(column).ok())
9181        else {
9182            return Vec::new();
9183        };
9184        let series = series.as_materialized_series();
9185        let mut values = Vec::new();
9186        for value in (0..series.len()).filter_map(|index| series.get(index).ok()) {
9187            if values.len() == limit {
9188                break;
9189            }
9190            let text = match value {
9191                AnyValue::Null => continue,
9192                AnyValue::String(text) => text.to_string(),
9193                AnyValue::List(items) => crate::exact::list_preview(&items),
9194                // Drawn on the UI thread: Polars' display panics on a date past
9195                // the calendar.
9196                value => {
9197                    crate::exact::past_calendar_text(&value).unwrap_or_else(|| value.to_string())
9198                }
9199            };
9200            if !values.contains(&text) {
9201                values.push(text);
9202            }
9203        }
9204        values
9205    }
9206
9207    /// Current scrollable display buffer. None until first collect().
9208    pub fn display_df(&self) -> Option<&DataFrame> {
9209        self.df.as_ref()
9210    }
9211
9212    /// Visible-window slice of the display buffer (same as passed to render_dataframe).
9213    pub fn display_slice_df(&self) -> Option<DataFrame> {
9214        let df = self.df.as_ref()?;
9215        let offset = self.start_row.saturating_sub(self.buffered_start_row);
9216        let slice_len = self.visible_rows.min(df.height().saturating_sub(offset));
9217        if offset < df.height() && slice_len > 0 {
9218            Some(df.slice(offset as i64, slice_len))
9219        } else {
9220            None
9221        }
9222    }
9223
9224    /// The selected row with every display column, raw and in display order:
9225    /// what a row copy carries.
9226    pub fn copy_row_df(&self) -> Option<DataFrame> {
9227        let df = self.buffered_df.as_ref()?;
9228        let absolute = self.start_row + self.table_state.selected()?;
9229        let offset = absolute.checked_sub(self.buffered_start_row)?;
9230        if offset >= df.height() {
9231            return None;
9232        }
9233        let names: Vec<&str> = self.column_order.iter().map(|s| s.as_str()).collect();
9234        df.select(names).ok().map(|d| d.slice(offset as i64, 1))
9235    }
9236
9237    /// The rows on screen with every display column, raw, in display order and
9238    /// untouched by the column scroll: a copy that lost the columns scrolled
9239    /// past would deny exactly the identifiers that make the rows readable.
9240    pub fn copy_view_df(&self) -> Option<DataFrame> {
9241        let df = self.buffered_df.as_ref()?;
9242        let names: Vec<&str> = self.column_order.iter().map(|s| s.as_str()).collect();
9243        let selected = df.select(names).ok()?;
9244        let offset = self.start_row.saturating_sub(self.buffered_start_row);
9245        let len = self
9246            .visible_rows
9247            .min(selected.height().saturating_sub(offset));
9248        (len > 0).then(|| selected.slice(offset as i64, len))
9249    }
9250
9251    /// The selected row's value in one column, exactly as stored (see
9252    /// [`crate::exact`]): a float as the decimal that reads back to it, never
9253    /// the table's rounded preview. A null is an empty string, like a null in
9254    /// an export — never the UI's glyph. A list or struct is JSON, as in a CSV
9255    /// export.
9256    pub fn copy_cell_value(&self, column: &str) -> Option<String> {
9257        let row = self.copy_row_df()?;
9258        crate::exact::copy_text(row.column(column).ok()?).ok()
9259    }
9260
9261    /// The selected row's number as the row-numbers column would print it.
9262    pub fn selected_display_row(&self) -> Option<usize> {
9263        Some(self.start_row + self.table_state.selected()? + self.row_start_index)
9264    }
9265
9266    /// Rows times estimated row width, for the copy guard: what collecting the whole
9267    /// view would hold, with binary at the base64 size a copy writes. None until the
9268    /// count has run, and while a binary column's width is unknown: the buffer holds a
9269    /// stub for it, so only a footer says how wide it is.
9270    pub fn estimated_copy_bytes(&self) -> Option<usize> {
9271        let rows = self.num_rows_if_valid()?;
9272        if rows == 0 {
9273            return Some(0);
9274        }
9275        let base64 = |bytes: usize| bytes.div_ceil(3) * 4;
9276        let footer_width = |name: &str| {
9277            self.column_bytes
9278                .iter()
9279                .find(|(n, _)| n == name)
9280                .map(|(_, w)| *w)
9281        };
9282        let mut row = self.bytes_per_row();
9283        for name in &self.column_order {
9284            match self.schema.get(name.as_str()) {
9285                Some(DataType::Binary) => row += base64(footer_width(name)?),
9286                // Buffered whole, so the buffer measured it with the row; base64 adds
9287                // a third on top.
9288                Some(dtype) if crate::nested_json::has_binary(dtype) => {
9289                    let buffered = self.buffered_df.as_ref().and_then(|df| {
9290                        let column = df.column(name).ok()?;
9291                        (df.height() > 0)
9292                            .then(|| column.as_materialized_series().estimated_size() / df.height())
9293                    });
9294                    row += buffered.or_else(|| footer_width(name)).unwrap_or(0) / 3;
9295                }
9296                _ => {}
9297            }
9298        }
9299        Some(rows.saturating_mul(row))
9300    }
9301
9302    /// The drift group of each row from the top of the view down, for a frame
9303    /// `frame_rows` tall, when the dataset's files differ.
9304    ///
9305    /// Sized by the frame about to be drawn rather than by `visible_rows`, which is
9306    /// what the *last* frame drew and is 0 before there has been one. Taking it from
9307    /// `visible_rows` left the opening frame of every drifting dataset — the one the
9308    /// user is looking at when nothing has been pressed yet — with no groups at all,
9309    /// so every absent cell fell back to the plain null glyph.
9310    ///
9311    /// Callers pass the whole frame's height, a header more than the rows it draws.
9312    /// Deliberately: the table reads this by row index and ignores what it does not
9313    /// reach, so a group too many costs a `u32` and a group too few costs a mark.
9314    ///
9315    /// Empty once a query or reshape has replaced the frame: those rows stand for no
9316    /// file, so their nulls are ordinary nulls. A frame is a screen tall, so this is
9317    /// a few dozen values.
9318    pub fn display_drift(&self, frame_rows: usize) -> Vec<u32> {
9319        if !self.drift_column_present {
9320            return Vec::new();
9321        }
9322        let Some(df) = self.buffered_df.as_ref() else {
9323            return Vec::new();
9324        };
9325        let Ok(column) = df.column(crate::schema_union::DRIFT_COLUMN) else {
9326            return Vec::new();
9327        };
9328        let offset = self.start_row.saturating_sub(self.buffered_start_row);
9329        let len = frame_rows.min(column.len().saturating_sub(offset));
9330        if len == 0 {
9331            return Vec::new();
9332        }
9333        let slice = column.slice(offset as i64, len);
9334        let Ok(rows) = slice.u32() else {
9335            return Vec::new();
9336        };
9337        // The column holds each row's place in the dataset. The file it came from is
9338        // the last one starting at or before it, and the file says what it is missing.
9339        rows.iter()
9340            .map(|row| self.file_group_of(row.unwrap_or(0) as usize))
9341            .collect()
9342    }
9343
9344    /// Maximum buffer size in rows (0 = no limit).
9345    pub fn max_buffered_rows(&self) -> usize {
9346        self.max_buffered_rows
9347    }
9348
9349    /// Maximum buffer size in MiB (0 = no limit).
9350    pub fn max_buffered_mb(&self) -> usize {
9351        self.max_buffered_mb
9352    }
9353
9354    /// Whether Enter on a row drills into its group: the view is a grouped result and
9355    /// not already a group's rows.
9356    pub fn can_drill_down(&self) -> bool {
9357        !self.is_drilled_down() && self.is_grouped()
9358    }
9359
9360    /// Whether a drill shows the lists of a row as the group's rows rather than
9361    /// filtering the source: a `by` result that holds its groups as lists.
9362    fn drills_lists(&self) -> bool {
9363        self.has_list_columns() && self.group_source.as_ref().is_some_and(|s| s.rows_in_lists)
9364    }
9365
9366    /// The columns of a row that a drill into its group reads: every column of a result
9367    /// holding its groups as lists, the keys of one holding aggregates.
9368    fn drill_columns(&self) -> Vec<String> {
9369        if self.drills_lists() {
9370            return self.schema.iter_names().map(|n| n.to_string()).collect();
9371        }
9372        self.group_source
9373            .iter()
9374            .flat_map(|source| source.keys.iter().map(|(name, _)| name.to_string()))
9375            .collect()
9376    }
9377
9378    /// The columns the inspector lists for a row: the table's, in its order, then
9379    /// the ones hidden from it, in schema order. Never the scan's own row index.
9380    pub fn inspect_fields(&self) -> Vec<InspectField> {
9381        let shown = self.column_order.iter().filter_map(|name| {
9382            Some(InspectField {
9383                name: name.clone(),
9384                dtype: self.schema.get(name.as_str())?.clone(),
9385                hidden: false,
9386            })
9387        });
9388        let hidden = self
9389            .schema
9390            .iter()
9391            .filter(|(name, _)| {
9392                name.as_str() != crate::schema_union::DRIFT_COLUMN
9393                    && !self.column_order.iter().any(|c| c == name.as_str())
9394            })
9395            .map(|(name, dtype)| InspectField {
9396                name: name.to_string(),
9397                dtype: dtype.clone(),
9398                hidden: true,
9399            });
9400        shown.chain(hidden).collect()
9401    }
9402
9403    /// The selected row as the buffer holds it, with the file group that says what
9404    /// its nulls are. Reads nothing; `None` while the row is not on hand.
9405    pub fn inspect_row(&self) -> Option<InspectRow> {
9406        self.inspect_row_at(self.start_row + self.table_state.selected()?)
9407    }
9408
9409    /// Row `row` of the view as the buffer holds it, as [`Self::inspect_row`] does
9410    /// the selected one: Compare's next row. `None` while it is not on hand.
9411    pub fn inspect_row_at(&self, row: usize) -> Option<InspectRow> {
9412        let df = self.buffered_df.as_ref()?;
9413        let offset = row.checked_sub(self.buffered_start_row)?;
9414        if offset >= df.height() {
9415            return None;
9416        }
9417        let names: Vec<&str> = self.column_order.iter().map(|s| s.as_str()).collect();
9418        let values = df.select(names).ok()?.slice(offset as i64, 1);
9419        let drift_group = self
9420            .drift_column_present
9421            .then(|| df.column(crate::schema_union::DRIFT_COLUMN).ok())
9422            .flatten()
9423            .and_then(|c| c.get(offset).ok())
9424            .and_then(|v| v.extract::<usize>())
9425            .map(|place| self.file_group_of(place));
9426        Some(InspectRow {
9427            row,
9428            frame: self.len_generation,
9429            display_row: row + self.row_start_index,
9430            values,
9431            drift_group,
9432        })
9433    }
9434
9435    /// The drift group of the file holding the dataset's row `place`.
9436    fn file_group_of(&self, place: usize) -> u32 {
9437        let file = self
9438            .drift_file_starts
9439            .partition_point(|&start| start <= place)
9440            .saturating_sub(1);
9441        self.drift_file_group.get(file).copied().unwrap_or(0)
9442    }
9443
9444    /// What a null in `column` is, for a row of file group `group`: the data's own,
9445    /// a file without the column, or a file holding it in another type.
9446    pub fn null_kind(&self, column: &str, group: Option<u32>) -> NullKind {
9447        let Some(group) = group.and_then(|g| self.drift_groups.get(g as usize)) else {
9448            return NullKind::Null;
9449        };
9450        if group.absent.iter().any(|c| c == column) {
9451            NullKind::Absent
9452        } else if group.unread.iter().any(|c| c == column) {
9453            NullKind::Conflict
9454        } else {
9455            NullKind::Null
9456        }
9457    }
9458
9459    /// The frame that reads `columns` of row `row` of the view: for the inspector's
9460    /// fields the buffer does not hold. One row, through the same window the buffer
9461    /// reads, so a remote dataset reads only the file holding it. Run off this thread.
9462    pub fn inspect_read_lf(&self, row: usize, columns: &[String]) -> PolarsResult<LazyFrame> {
9463        let exprs = columns.iter().map(|c| col(c.as_str())).collect();
9464        self.window_lf(row, 1, exprs)
9465    }
9466
9467    /// What drilling into the group on row `group_index` of the view reads, or `None`
9468    /// when the view is not grouped. The row is on screen, so the buffer holds it and
9469    /// nothing is computed again. A column the buffer lacks (hidden, or a binary stub)
9470    /// means reading the row, and one row of an aggregate is the whole aggregate, so the
9471    /// caller reads it off the UI thread.
9472    pub fn drill_row(&self, group_index: usize) -> Option<DrillRow> {
9473        if !self.can_drill_down() {
9474            return None;
9475        }
9476        let columns = self.drill_columns();
9477        let buffered = self
9478            .buffered_df
9479            .as_ref()
9480            .filter(|_| (self.buffered_start_row..self.buffered_end_row).contains(&group_index))
9481            .filter(|_| {
9482                columns
9483                    .iter()
9484                    .all(|c| !matches!(self.schema.get(c.as_str()), Some(DataType::Binary)))
9485            })
9486            .and_then(|df| df.select(columns.iter().map(|c| c.as_str())).ok())
9487            .map(|df| df.slice((group_index - self.buffered_start_row) as i64, 1))
9488            .filter(|row| row.height() == 1);
9489        Some(match buffered {
9490            Some(row) => DrillRow::Buffered(row),
9491            None => DrillRow::Read(Box::new(
9492                self.visible_lf()
9493                    .select(columns.iter().map(|c| col(c.as_str())).collect::<Vec<_>>())
9494                    .slice(group_index as i64, 1),
9495            )),
9496        })
9497    }
9498
9499    /// Show the rows of the group on row `group_index` of the view, reading the row on
9500    /// this thread if the buffer does not hold it. The app goes through
9501    /// [`Self::drill_row`] instead, so that read never holds up a key.
9502    pub fn drill_down_into_group(&mut self, group_index: usize) -> Result<()> {
9503        let row = match self.drill_row(group_index) {
9504            None => return Ok(()),
9505            Some(DrillRow::Buffered(row)) => row,
9506            Some(DrillRow::Read(lf)) => collect_lazy(*lf, self.polars_streaming)?,
9507        };
9508        self.drill_down_with_row(group_index, &row)
9509    }
9510
9511    /// Show the rows of the group whose row `group_index` of the view is `row`, as
9512    /// [`Self::drill_row`] gave it. A result that holds each group as lists shows those
9513    /// lists as rows; one that holds only aggregates shows the source rows sharing the
9514    /// group's keys, key columns first.
9515    pub fn drill_down_with_row(&mut self, group_index: usize, row: &DataFrame) -> Result<()> {
9516        if !self.can_drill_down() {
9517            return Ok(());
9518        }
9519        if row.height() == 0 {
9520            return Err(color_eyre::eyre::eyre!("Group index out of bounds"));
9521        }
9522        let mut group = if self.drills_lists() {
9523            Self::group_from_lists(
9524                row,
9525                self.group_key_columns(),
9526                self.group_value_columns(),
9527                self.lineage.clone(),
9528            )?
9529        } else if let Some(source) = &self.group_source {
9530            Self::group_from_source(source, row)?
9531        } else {
9532            return Ok(());
9533        };
9534        // A list form that also aggregates (`select a, n: count a by k`) holds its
9535        // aggregates beside the keys; the query knows which columns are keys.
9536        if let Some(source) = self.group_source.as_ref().filter(|_| self.drills_lists()) {
9537            let keys: Vec<&str> = source.keys.iter().map(|(n, _)| n.as_str()).collect();
9538            (group.key_columns, group.key_values) = group
9539                .key_columns
9540                .into_iter()
9541                .zip(group.key_values)
9542                .filter(|(name, _)| keys.contains(&name.as_str()))
9543                .unzip();
9544        }
9545        self.enter_group(group, group_index, false)
9546    }
9547
9548    /// Show the rows of the view holding `value` in `column` (null matching nulls),
9549    /// as a drill into a group does: the breadcrumb names the value, and Esc comes
9550    /// back to the view. Inside a group already, it narrows that group, and Esc goes
9551    /// back to the view the group was drilled from.
9552    pub fn drill_into_value(&mut self, column: &str, value: AnyValue<'static>) -> Result<()> {
9553        let dtype = self
9554            .schema
9555            .get(column)
9556            .cloned()
9557            .ok_or_else(|| color_eyre::eyre::eyre!("no column {column}"))?;
9558        let label = crate::exact::str_value(&value).to_string();
9559        let mut steps = self.view_steps();
9560        steps.push(match crate::python_script::py_value(&value) {
9561            Some(literal) => Step::Matching(vec![(
9562                format!("pl.col({})", crate::python_script::py_str(column)),
9563                literal,
9564            )]),
9565            None => Step::Unreproducible(format!(
9566                "drilled down to the rows where {column} is {label}, a value of a type not written as Python"
9567            )),
9568        });
9569        let matches = col(column).eq_missing(lit(Scalar::new(dtype, value)));
9570        let group = GroupRows {
9571            lf: self.visible_lf().filter(matches),
9572            key_columns: vec![column.to_string()],
9573            key_values: vec![label],
9574            lead: vec![column.to_string()],
9575            steps,
9576            lineage: self.lineage.clone(),
9577        };
9578        if !self.is_drilled_down() {
9579            let index = self.start_row + self.table_state.selected().unwrap_or(0);
9580            return self.enter_group(group, index, true);
9581        }
9582        let schema = group.lf.clone().collect_schema()?;
9583        let order = std::mem::take(&mut self.column_order);
9584        if let Some(keys) = self.drilled_down_group_key_columns.as_mut() {
9585            keys.extend(group.key_columns);
9586        }
9587        if let Some(values) = self.drilled_down_group_key.as_mut() {
9588            values.extend(group.key_values);
9589        }
9590        // The group's filters and sort are in the frame now.
9591        self.filters.clear();
9592        self.sort_columns.clear();
9593        self.sort_descending.clear();
9594        self.sort_ascending = true;
9595        self.install_base(group.lf, schema);
9596        self.base_steps = group.steps;
9597        self.lineage = group.lineage;
9598        self.column_order = order;
9599        self.start_row = 0;
9600        self.termcol_index = 0;
9601        self.clear_column_moves();
9602        self.settle_cursor();
9603        self.table_state.select(Some(0));
9604        self.collect();
9605        Ok(())
9606    }
9607
9608    /// Whether the drill on screen came from Value Counts.
9609    pub fn drilled_into_value(&self) -> bool {
9610        self.grouped.as_ref().is_some_and(|view| view.by_value)
9611    }
9612
9613    /// Show `group`, the group on row `group_index` of the view, keeping the view to
9614    /// come back to.
9615    fn enter_group(&mut self, group: GroupRows, group_index: usize, by_value: bool) -> Result<()> {
9616        let schema = group.lf.clone().collect_schema()?;
9617        self.drilled_down_group_key = Some(group.key_values);
9618        self.drilled_down_group_key_columns = Some(group.key_columns);
9619
9620        // The group becomes the pipeline root while drilled in, so a sidebar filter or
9621        // sort applies within it instead of rebuilding the grouped view underneath.
9622        self.grouped = Some(GroupedView {
9623            lf: self.lf.clone(),
9624            base_lf: self.base_lf.clone(),
9625            filters: std::mem::take(&mut self.filters),
9626            sort_columns: std::mem::take(&mut self.sort_columns),
9627            sort_descending: std::mem::take(&mut self.sort_descending),
9628            sort_ascending: self.sort_ascending,
9629            drift: self.drift_column_present,
9630            drift_groups: self.drift_groups.clone(),
9631            view_numbered: self.view_numbered,
9632            notes: self.notes.clone(),
9633            group_source: self.group_source.take(),
9634            column_order: self.column_order.clone(),
9635            locked_columns_count: self.locked_columns_count,
9636            start_row: self.start_row,
9637            termcol_index: self.termcol_index,
9638            cursor_column: self.cursor_column.clone(),
9639            selected: self.table_state.selected(),
9640            by_value,
9641            base_steps: std::mem::take(&mut self.base_steps),
9642            lineage: self.lineage.clone(),
9643        });
9644        self.sort_ascending = true;
9645        self.install_base(group.lf, schema);
9646        self.base_steps = group.steps;
9647        self.lineage = group.lineage;
9648        // Led by the keys, as a group drilled from lists is.
9649        let rest: Vec<String> = std::mem::take(&mut self.column_order)
9650            .into_iter()
9651            .filter(|c| !group.lead.contains(c))
9652            .collect();
9653        self.column_order = group.lead.into_iter().chain(rest).collect();
9654        self.drilled_down_group_index = Some(group_index);
9655        self.start_row = 0;
9656        self.termcol_index = 0;
9657        self.clear_column_moves();
9658        self.locked_columns_count = 0;
9659        self.settle_cursor();
9660        self.table_state.select(Some(0));
9661        self.collect();
9662
9663        Ok(())
9664    }
9665
9666    /// A group held as lists in `row`: the keys repeated beside the lists as columns.
9667    fn group_from_lists(
9668        row: &DataFrame,
9669        key_columns: Vec<String>,
9670        value_columns: Vec<String>,
9671        lineage: Lineage,
9672    ) -> Result<GroupRows> {
9673        if value_columns.is_empty() {
9674            return Err(color_eyre::eyre::eyre!("No value columns in grouped data"));
9675        }
9676        let row_count = match row.column(&value_columns[0])?.get(0)? {
9677            AnyValue::List(list_series) => list_series.len(),
9678            _ => 0,
9679        };
9680
9681        let mut columns = Vec::new();
9682        let mut key_values = Vec::new();
9683        for col_name in &key_columns {
9684            let key = row.column(col_name)?;
9685            key_values.push(crate::exact::str_value(&key.get(0)?).to_string());
9686            // Repeated in its own type, so a date stays a date and a null key null.
9687            columns.push(key.new_from_index(0, row_count));
9688        }
9689        for col_name in &value_columns {
9690            if let AnyValue::List(list_series) = row.column(col_name)?.get(0)? {
9691                columns.push(list_series.with_name(col_name.as_str().into()).into());
9692            }
9693        }
9694        let group = key_columns
9695            .iter()
9696            .zip(&key_values)
9697            .map(|(c, v)| format!("{c} = {v}"))
9698            .collect::<Vec<_>>()
9699            .join(", ");
9700        Ok(GroupRows {
9701            lf: DataFrame::new_infer_height(columns)?.lazy(),
9702            key_columns,
9703            key_values,
9704            // Already first.
9705            lead: Vec::new(),
9706            steps: vec![Step::Unreproducible(format!(
9707                "drilled down into the group {group}, read from the grouped result's lists: \
9708                 not written as Python"
9709            ))],
9710            // The lists keep the result's names.
9711            lineage,
9712        })
9713    }
9714
9715    /// A group of an aggregated result in `row`: the source rows whose keys equal the
9716    /// row's, a null key matching nulls.
9717    fn group_from_source(source: &GroupSource, row: &DataFrame) -> Result<GroupRows> {
9718        let mut predicate: Option<Expr> = None;
9719        let mut key_columns = Vec::new();
9720        let mut key_values = Vec::new();
9721        let mut lead = Vec::new();
9722        let mut matching = Vec::new();
9723        for (i, (name, expr)) in source.keys.iter().enumerate() {
9724            let column = row.column(name)?;
9725            let value = column.get(0)?.into_static();
9726            matching.push(
9727                source
9728                    .python_keys
9729                    .get(i)
9730                    .cloned()
9731                    .flatten()
9732                    .zip(crate::python_script::py_value(&value)),
9733            );
9734            key_columns.push(name.to_string());
9735            key_values.push(crate::exact::str_value(&value).to_string());
9736            // The key as the query computed it, against the value it produced; the alias
9737            // only named the result's column.
9738            let key = expr.clone().meta().undo_aliases();
9739            if let Expr::Column(source_column) = &key
9740                && !source.scratch.contains(source_column)
9741            {
9742                lead.push(source_column.to_string());
9743            }
9744            let matches = key.eq_missing(lit(Scalar::new(column.dtype().clone(), value)));
9745            predicate = Some(match predicate {
9746                Some(all) => all.and(matches),
9747                None => matches,
9748            });
9749        }
9750        let rows = source.rows.clone();
9751        let mut lf = match predicate {
9752            Some(predicate) => rows.filter(predicate),
9753            None => rows,
9754        };
9755        if !source.scratch.is_empty() {
9756            lf = lf.drop(by_name(source.scratch.iter().cloned(), true, false));
9757        }
9758        let matching: Option<Vec<(String, String)>> = matching.into_iter().collect();
9759        let steps = match (&source.python_rows, matching) {
9760            (Some(rows), Some(matching)) => {
9761                let mut steps = rows.clone();
9762                steps.push(Step::Matching(matching));
9763                if !source.scratch.is_empty() {
9764                    steps.push(Step::Drop(
9765                        source.scratch.iter().map(|c| c.to_string()).collect(),
9766                    ));
9767                }
9768                steps
9769            }
9770            _ => vec![Step::Unreproducible(format!(
9771                "drilled down into the group {}: not written as Python",
9772                key_columns
9773                    .iter()
9774                    .zip(&key_values)
9775                    .map(|(c, v)| format!("{c} = {v}"))
9776                    .collect::<Vec<_>>()
9777                    .join(", ")
9778            ))],
9779        };
9780        Ok(GroupRows {
9781            lf,
9782            key_columns,
9783            key_values,
9784            lead,
9785            steps,
9786            lineage: source.lineage.clone(),
9787        })
9788    }
9789
9790    pub fn drill_up(&mut self) -> Result<()> {
9791        let Some(view) = self.grouped.take() else {
9792            return Err(color_eyre::eyre::eyre!("Not in drill-down mode"));
9793        };
9794        let schema = Self::without_drift(view.lf.clone()).collect_schema()?;
9795        self.invalidate_num_rows();
9796        // The buffer holds the group's rows; kept, it would stand in for the grouped
9797        // view wherever the view fits inside it.
9798        self.drop_buffer();
9799        self.observed_bytes_per_row = None;
9800        self.widths.relearn();
9801        self.lf = view.lf;
9802        self.unsorted_lf = None;
9803        self.base_lf = view.base_lf;
9804        self.base_steps = view.base_steps;
9805        self.lineage = view.lineage;
9806        self.filters = view.filters;
9807        self.sort_columns = view.sort_columns;
9808        self.sort_descending = view.sort_descending;
9809        self.sort_ascending = view.sort_ascending;
9810        self.drift_column_present = view.drift;
9811        self.drift_groups = view.drift_groups;
9812        self.view_numbered = view.view_numbered;
9813        self.notes = view.notes;
9814        self.group_source = view.group_source;
9815        // The frame put back here already leaves out whatever its filter and sort left
9816        // out, so the notes saying so have to come back with it. They are derived rather
9817        // than saved, so they cannot go stale against a frame that changed while it was
9818        // drilled into.
9819        self.view_notes = self.view_notes_only();
9820        self.schema = schema;
9821        self.column_order = view.column_order;
9822        self.locked_columns_count = view.locked_columns_count;
9823        self.drilled_down_group_index = None;
9824        self.drilled_down_group_key = None;
9825        self.drilled_down_group_key_columns = None;
9826        self.start_row = view.start_row;
9827        self.termcol_index = view.termcol_index;
9828        self.clear_column_moves();
9829        self.cursor_column = view.cursor_column;
9830        self.settle_cursor();
9831        self.table_state.select(view.selected);
9832        self.collect();
9833        Ok(())
9834    }
9835
9836    pub fn get_analysis_dataframe(&self) -> Result<DataFrame> {
9837        Ok(collect_lazy(self.visible_lf(), self.polars_streaming)?)
9838    }
9839
9840    pub fn get_analysis_context(&self) -> crate::statistics::AnalysisContext {
9841        crate::statistics::AnalysisContext {
9842            has_query: !self.active_query.is_empty(),
9843            query: self.active_query.clone(),
9844            has_filters: !self.filters.is_empty(),
9845            filter_count: self.filters.len(),
9846            is_drilled_down: self.is_drilled_down(),
9847            group_key: self.drilled_down_group_key.clone(),
9848            group_columns: self.drilled_down_group_key_columns.clone(),
9849        }
9850    }
9851
9852    /// Plan a pivot of the view (long → wide). Nothing is read until the job runs.
9853    /// Never uses `original_lf`.
9854    pub fn plan_pivot(&self, spec: &PivotSpec) -> PivotJob {
9855        PivotJob {
9856            view: self.visible_lf(),
9857            spec: spec.clone(),
9858            streaming: self.polars_streaming,
9859        }
9860    }
9861
9862    /// Show `pivoted`, the result of `spec`'s [`PivotJob`], as the new pipeline root.
9863    pub fn install_pivot(&mut self, spec: &PivotSpec, pivoted: DataFrame) -> Result<()> {
9864        let index = if spec.index.is_empty() {
9865            // What the pivot itself took as the index: every other column of the view.
9866            self.schema
9867                .iter_names()
9868                .map(|n| n.to_string())
9869                .filter(|n| {
9870                    n != &spec.pivot_column
9871                        && n != &spec.value_column
9872                        && n != crate::schema_union::DRIFT_COLUMN
9873                })
9874                .collect()
9875        } else {
9876            spec.index.clone()
9877        };
9878        let kept = index.clone();
9879        let step = Step::Pivot {
9880            index,
9881            on: spec.pivot_column.clone(),
9882            values: spec.value_column.clone(),
9883            aggregation: spec.aggregation,
9884        };
9885        self.last_pivot_spec = Some(spec.clone());
9886        self.last_melt_spec = None;
9887        self.replace_lf_after_reshape(pivoted.lazy(), step, &kept)
9888    }
9889
9890    /// Pivot the view here and now, reading it on this thread. The Pivot & Melt builder
9891    /// and views run the [`PivotJob`] in the background instead.
9892    pub fn pivot(&mut self, spec: &PivotSpec) -> Result<()> {
9893        let pivoted = self.plan_pivot(spec).run()?;
9894        self.install_pivot(spec, pivoted)
9895    }
9896
9897    /// `view` melted by `spec`, planned only: the table's melt and the builder's
9898    /// preview build it the same way.
9899    pub(crate) fn melt_lf(view: LazyFrame, spec: &MeltSpec) -> Result<LazyFrame> {
9900        let on = cols(spec.value_columns.iter().map(|s| s.as_str()));
9901        let index = cols(spec.index.iter().map(|s| s.as_str()));
9902        let args = UnpivotArgsDSL {
9903            on: Some(on),
9904            index,
9905            variable_name: Some(PlSmallStr::from(spec.variable_name.as_str())),
9906            value_name: Some(PlSmallStr::from(spec.value_name.as_str())),
9907        };
9908        Ok(Self::melt_dates_as_text(view, spec, &args)?.unpivot(args))
9909    }
9910
9911    /// Melt the current `LazyFrame` (wide → long). Never uses `original_lf`.
9912    pub fn melt(&mut self, spec: &MeltSpec) -> Result<()> {
9913        let lf = Self::melt_lf(self.visible_lf(), spec)?;
9914        let step = Step::Melt {
9915            index: spec.index.clone(),
9916            on: spec.value_columns.clone(),
9917            variable_name: spec.variable_name.clone(),
9918            value_name: spec.value_name.clone(),
9919        };
9920        self.last_melt_spec = Some(spec.clone());
9921        self.last_pivot_spec = None;
9922        self.replace_lf_after_reshape(lf, step, &spec.index)?;
9923        Ok(())
9924    }
9925
9926    /// A melt of dates with text casts the dates to text, which panics on one past
9927    /// the calendar. Those columns become text first, such a date its stored
9928    /// number, as the rows are read.
9929    fn melt_dates_as_text(
9930        view: LazyFrame,
9931        spec: &MeltSpec,
9932        args: &UnpivotArgsDSL,
9933    ) -> Result<LazyFrame> {
9934        let schema = view.clone().collect_schema()?;
9935        let melted = view.clone().unpivot(args.clone()).collect_schema()?;
9936        if melted.get(spec.value_name.as_str()) != Some(&DataType::String) {
9937            return Ok(view);
9938        }
9939        let texts: Vec<Expr> = spec
9940            .value_columns
9941            .iter()
9942            .filter(|name| {
9943                schema
9944                    .get(name.as_str())
9945                    .is_some_and(crate::past_calendar::can_leave_calendar)
9946            })
9947            .map(|name| {
9948                crate::past_calendar::text_expr(
9949                    Expr::Column(PlSmallStr::from(name.as_str())),
9950                    polars::chunked_array::cast::CastOptions::NonStrict,
9951                )
9952            })
9953            .collect();
9954        Ok(if texts.is_empty() {
9955            view
9956        } else {
9957            view.with_columns(texts)
9958        })
9959    }
9960
9961    /// Show `lf`, the view reshaped, as the new pipeline root. `kept` are the view's
9962    /// columns it carries as they were: a pivot's index, a melt's id columns.
9963    fn replace_lf_after_reshape(
9964        &mut self,
9965        lf: LazyFrame,
9966        step: Step,
9967        kept: &[String],
9968    ) -> Result<()> {
9969        let schema = lf.clone().collect_schema()?;
9970        let lineage = traced(
9971            &self.lineage,
9972            kept.iter().map(|c| (c.clone(), c.clone())).collect(),
9973        );
9974        let mut steps = self.view_steps();
9975        steps.push(step);
9976        // Taken before the view state below is reset. Over an earlier reshape there is
9977        // no source a view could replay, so none is kept.
9978        let text = |q: &str| Some(q.trim().to_string()).filter(|q| !q.is_empty());
9979        let source = ReshapeSource {
9980            query: text(&self.active_query),
9981            sql_query: text(&self.active_sql_query),
9982            fuzzy_query: text(&self.active_fuzzy_query),
9983            filters: self.filters.clone(),
9984            sort_columns: self.sort_columns.clone(),
9985            sort_descending: self.sort_descending.clone(),
9986        };
9987        self.reshape_source = (self.reshaped_lf.is_none() && !source.is_empty()).then_some(source);
9988        self.reshaped_lf = Some(lf.clone());
9989        self.install_base(lf, schema);
9990        self.base_steps = steps.clone();
9991        self.reshape_steps = Some(steps);
9992        self.lineage = lineage.clone();
9993        self.reshape_lineage = lineage;
9994        self.reset_view_state(0);
9995        self.error = None;
9996        self.df = None;
9997        self.locked_df = None;
9998        self.collect();
9999        Ok(())
10000    }
10001
10002    /// The sidebar filters with their values typed against the columns they test.
10003    fn typed_filters(&self) -> Vec<SidebarFilter> {
10004        self.filters
10005            .iter()
10006            .map(|f| SidebarFilter::typed_in(f, &self.schema, &self.column_order))
10007            .collect()
10008    }
10009
10010    /// What the open did to the rows its reader gave, as Python method calls.
10011    pub fn read_python(&self) -> &[String] {
10012        &self.read_python
10013    }
10014
10015    /// See the field: the notes the read made, for the open to carry to the dataset.
10016    pub fn read_notes(&self) -> &[crate::notes::Note] {
10017        &self.read_notes
10018    }
10019
10020    /// See the field.
10021    pub fn read_units(&self) -> Option<&[(String, String)]> {
10022        self.read_units.as_deref()
10023    }
10024
10025    /// The read's typing: its notes go with the read's, its columns are counted later.
10026    fn take_typing(&mut self, mut typing: Typing) {
10027        self.read_notes.append(&mut typing.notes);
10028        self.typing = typing;
10029    }
10030
10031    /// See [`Self::take_typing`]: what the scan hands the open, for the dataset.
10032    pub(crate) fn typing(&self) -> &Typing {
10033        &self.typing
10034    }
10035
10036    /// What is left to count of the values the read's types made null: the frame
10037    /// before the types, and the columns. `None` once counted, or with nothing typed.
10038    pub(crate) fn unfit_to_count(&self) -> Option<(LazyFrame, Vec<crate::column_types::Typed>)> {
10039        if self.unfit_notes.is_some() || self.typing.typed.is_empty() {
10040            return None;
10041        }
10042        Some((self.typing.source.clone()?, self.typing.typed.clone()))
10043    }
10044
10045    /// The view's column types and made columns, in the order asked.
10046    pub fn column_changes(&self) -> &[crate::column_types::ColumnChange] {
10047        &self.column_changes
10048    }
10049
10050    /// The columns the view gave a type: the type row draws them in the accent.
10051    pub fn retyped_columns(&self) -> Vec<String> {
10052        self.column_changes
10053            .iter()
10054            .filter(|c| matches!(c.change, crate::column_types::Change::Typed(_)))
10055            .map(|c| c.name.clone())
10056            .collect()
10057    }
10058
10059    /// `column`'s type before the view's: as the read gave it, or as the view made it.
10060    pub fn type_as_read(&self, column: &str) -> Option<DataType> {
10061        let base = self.base_lf.clone().collect_schema().ok()?;
10062        base.get(column)
10063            .or_else(|| self.schema.get(column))
10064            .cloned()
10065    }
10066
10067    /// Up to `n` of `column`'s values as read that are not blank, as text: from the
10068    /// rows on hand, or from the first rows when the view has typed the column.
10069    pub fn values_on_screen(&self, column: &str, n: usize) -> Vec<String> {
10070        let from_buffer = self.column_type_of(column).is_none();
10071        let df = if from_buffer {
10072            self.buffered_df.clone()
10073        } else {
10074            self.base_lf
10075                .clone()
10076                .select([col(column)])
10077                .limit(n as IdxSize * 10)
10078                .collect()
10079                .ok()
10080        };
10081        let Some(values) = df.and_then(|df| df.column(column).ok().cloned()) else {
10082            return Vec::new();
10083        };
10084        let Ok(text) = values.cast(&DataType::String) else {
10085            return Vec::new();
10086        };
10087        let Ok(text) = text.str().cloned() else {
10088            return Vec::new();
10089        };
10090        text.iter()
10091            .flatten()
10092            .map(str::trim)
10093            .filter(|v| !v.is_empty())
10094            .take(n)
10095            .map(str::to_string)
10096            .collect()
10097    }
10098
10099    /// The type the view gives `column`, if it gives one.
10100    pub fn column_type_of(&self, column: &str) -> Option<&crate::column_types::ColumnType> {
10101        self.column_changes.iter().find_map(|c| match &c.change {
10102            crate::column_types::Change::Typed(ty) if c.name == column => Some(ty),
10103            _ => None,
10104        })
10105    }
10106
10107    /// `column` as `ty`, or as read again with `None`. The view's own type wins over
10108    /// what the read gave the column. Lazy: the next rows read are typed.
10109    pub fn set_column_type(&mut self, column: &str, ty: Option<crate::column_types::ColumnType>) {
10110        use crate::column_types::{Change, ColumnChange};
10111        self.column_changes
10112            .retain(|c| !(c.name == column && matches!(c.change, Change::Typed(_))));
10113        if let Some(ty) = ty {
10114            self.column_changes.push(ColumnChange {
10115                name: column.to_string(),
10116                change: Change::Typed(ty),
10117            });
10118        }
10119        self.column_changes_changed();
10120    }
10121
10122    /// A column made from others, as a spec's derived column is, before the first
10123    /// column it is made from, which stays. Its name may not be taken.
10124    pub fn add_made_column(
10125        &mut self,
10126        derived: crate::column_types::Derived,
10127    ) -> std::result::Result<(), String> {
10128        use crate::column_types::{Change, ColumnChange};
10129        if self.schema.contains(&derived.name) {
10130            return Err(format!("a column is named {} already", derived.name));
10131        }
10132        for from in &derived.from {
10133            if !self.schema.contains(from) {
10134                return Err(format!("no column {from}"));
10135            }
10136        }
10137        let first = derived.from[0].clone();
10138        let at = self
10139            .column_order
10140            .iter()
10141            .position(|c| *c == first)
10142            .unwrap_or(self.column_order.len());
10143        self.column_order.insert(at, derived.name.clone());
10144        self.column_changes.push(ColumnChange {
10145            name: derived.name,
10146            change: Change::Made {
10147                from: derived.from,
10148                kind: derived.kind.name().to_string(),
10149                format: derived.format,
10150            },
10151        });
10152        self.column_changes_changed();
10153        Ok(())
10154    }
10155
10156    /// A saved view's column changes, in place of the view's own. A change whose column
10157    /// this data does not have is left out, with a note; the names left out are
10158    /// returned.
10159    pub fn set_column_changes(
10160        &mut self,
10161        changes: &[crate::column_types::ColumnChange],
10162    ) -> Vec<String> {
10163        self.column_changes = Vec::new();
10164        let base = self
10165            .base_lf
10166            .clone()
10167            .collect_schema()
10168            .unwrap_or_else(|_| self.schema.clone());
10169        let mut known: Vec<String> = base.iter_names().map(|n| n.to_string()).collect();
10170        let mut dropped = Vec::new();
10171        for change in changes {
10172            let fits = match &change.change {
10173                crate::column_types::Change::Typed(_) => known.contains(&change.name),
10174                crate::column_types::Change::Made { from, .. } => {
10175                    from.iter().all(|f| known.contains(f)) && change.derived().is_some()
10176                }
10177            };
10178            if fits {
10179                if !known.contains(&change.name) {
10180                    known.push(change.name.clone());
10181                }
10182                self.column_changes.push(change.clone());
10183            } else {
10184                dropped.push(change.name.clone());
10185            }
10186        }
10187        self.changes_dropped = if dropped.is_empty() {
10188            Vec::new()
10189        } else {
10190            vec![crate::notes::Note {
10191                summary: format!(
10192                    "view steps left out, no such column: {}",
10193                    crate::notes::some_names(&dropped)
10194                ),
10195                scope: "the view's column types".to_string(),
10196                read_as_text: None,
10197                passed_over: None,
10198            }]
10199        };
10200        // The made columns go before their first source, as they did when made.
10201        for change in &self.column_changes {
10202            if let crate::column_types::Change::Made { from, .. } = &change.change
10203                && !self.column_order.contains(&change.name)
10204            {
10205                let at = self
10206                    .column_order
10207                    .iter()
10208                    .position(|c| *c == from[0])
10209                    .unwrap_or(self.column_order.len());
10210                self.column_order.insert(at, change.name.clone());
10211            }
10212        }
10213        self.column_changes_changed();
10214        dropped
10215    }
10216
10217    /// Drop the view's column changes and their notes, as a new pipeline root does.
10218    fn forget_column_changes(&mut self) {
10219        if self.column_changes.is_empty() && self.changes_dropped.is_empty() {
10220            return;
10221        }
10222        self.column_changes.clear();
10223        self.changes_dropped.clear();
10224        self.changes_version += 1;
10225        self.changes_unfit = None;
10226    }
10227
10228    /// After the column changes change: the schema shows them, a made column gone
10229    /// leaves the column order, and the rows are read again.
10230    fn column_changes_changed(&mut self) {
10231        self.changes_version += 1;
10232        let (changed, _) = self.with_column_changes(self.base_lf.clone());
10233        if let Ok(schema) = changed.clone().collect_schema() {
10234            self.schema = schema;
10235        }
10236        let schema = self.schema.clone();
10237        self.column_order.retain(|c| schema.contains(c));
10238        for name in schema.iter_names() {
10239            if !self.column_order.iter().any(|c| c == name.as_str()) {
10240                self.column_order.push(name.to_string());
10241            }
10242        }
10243        self.widths.relearn();
10244        self.drop_buffer();
10245        self.apply_transformations();
10246    }
10247
10248    /// `lf` with the view's column changes, in order, and what the count of the values
10249    /// they made null needs: the frame with only the made columns, and the typed
10250    /// columns with the types they had there. A change whose column is not in `lf` is
10251    /// passed over.
10252    fn with_column_changes(
10253        &self,
10254        mut lf: LazyFrame,
10255    ) -> (
10256        LazyFrame,
10257        Option<(LazyFrame, Vec<crate::column_types::Typed>)>,
10258    ) {
10259        use crate::column_types::Change;
10260        if self.column_changes.is_empty() {
10261            return (lf, None);
10262        }
10263        let Ok(schema) = lf.collect_schema() else {
10264            return (lf, None);
10265        };
10266        let mut schema = (*schema).clone();
10267        let mut made = lf.clone();
10268        let mut typed = Vec::new();
10269        for change in &self.column_changes {
10270            let name = PlSmallStr::from(change.name.as_str());
10271            match &change.change {
10272                Change::Typed(ty) => {
10273                    let Some(from) = schema.get(&name).cloned() else {
10274                        continue;
10275                    };
10276                    lf = lf.with_column(ty.expr(&change.name, &from).alias(name.clone()));
10277                    typed.push(crate::column_types::Typed {
10278                        column: change.name.clone(),
10279                        ty: ty.clone(),
10280                        from,
10281                    });
10282                    schema.with_column(name, ty.dtype.clone());
10283                }
10284                Change::Made { from, .. } => {
10285                    let Some(derived) = change.derived() else {
10286                        continue;
10287                    };
10288                    if !from.iter().all(|f| schema.contains(f.as_str())) {
10289                        continue;
10290                    }
10291                    lf = lf.with_column(derived.expr().alias(name.clone()));
10292                    made = made.with_column(derived.expr().alias(name.clone()));
10293                    schema.with_column(name, DataType::Null);
10294                }
10295            }
10296        }
10297        let count = (!typed.is_empty()).then_some((made, typed));
10298        (lf, count)
10299    }
10300
10301    /// What is left to count of the values the view's column types made null: the
10302    /// frame, the columns and the version of the changes it is for.
10303    pub(crate) fn changes_unfit_to_count(
10304        &self,
10305    ) -> Option<(LazyFrame, Vec<crate::column_types::Typed>, u64)> {
10306        if self
10307            .changes_unfit
10308            .as_ref()
10309            .is_some_and(|(version, _)| *version == self.changes_version)
10310        {
10311            return None;
10312        }
10313        let (_, count) = self.with_column_changes(self.base_lf.clone());
10314        let (source, typed) = count?;
10315        Some((source, typed, self.changes_version))
10316    }
10317
10318    /// The counts for the view's column types at `version`, as notes.
10319    pub(crate) fn changes_unfit_counted(
10320        &mut self,
10321        version: u64,
10322        unfit: &[crate::column_types::Unfit],
10323    ) {
10324        if version == self.changes_version {
10325            // Something new to say: the `i` chip lights again.
10326            if !unfit.is_empty() {
10327                self.notes_seen = false;
10328            }
10329            self.changes_unfit = Some((
10330                version,
10331                crate::column_types::unfit_notes(unfit, "the view's column types"),
10332            ));
10333        }
10334    }
10335
10336    /// The counts of the values the types made null, as notes.
10337    pub(crate) fn unfit_counted(&mut self, unfit: &[crate::column_types::Unfit]) {
10338        if !unfit.is_empty() {
10339            self.notes_seen = false;
10340        }
10341        self.unfit_notes = Some(crate::column_types::unfit_notes(
10342            unfit,
10343            "counted over every row",
10344        ));
10345    }
10346
10347    /// How `lf` was built: the base's steps, then the filters and the sort.
10348    fn view_steps(&self) -> Vec<Step> {
10349        let mut steps = self.base_steps.clone();
10350        if !self.column_changes.is_empty() {
10351            let said: Vec<String> = self
10352                .column_changes
10353                .iter()
10354                .map(crate::column_types::ColumnChange::to_toml)
10355                .collect();
10356            steps.push(Step::Unreproducible(format!(
10357                "datui typed columns as a format spec would: {}",
10358                said.join("; ")
10359            )));
10360        }
10361        if !self.filters.is_empty() {
10362            let typed = self.typed_filters();
10363            let durations: Vec<String> = typed
10364                .iter()
10365                .flat_map(|f| f.unscriptable_columns())
10366                .collect();
10367            // Said before the filter: from there the script cannot keep the rows
10368            // datui keeps.
10369            if !durations.is_empty() {
10370                steps.push(Step::Unreproducible(format!(
10371                    "a kept find matches {} as datui writes durations",
10372                    durations.join(", ")
10373                )));
10374            }
10375            steps.push(Step::Filter(typed));
10376        }
10377        // Rows of files that hold a filtered or sorted column as another type: datui
10378        // leaves them out by where they were read, which a script cannot know.
10379        let left_out: Vec<String> = self
10380            .view_exclusions()
10381            .into_iter()
10382            .map(|(_, note)| note.summary)
10383            .collect();
10384        if !left_out.is_empty() {
10385            steps.push(Step::Unreproducible(left_out.join("; ")));
10386        }
10387        if !self.sort_columns.is_empty() {
10388            steps.push(Step::Sort {
10389                columns: self.sort_columns.clone(),
10390                descending: self.sort_descending.clone(),
10391            });
10392        } else if !self.sort_ascending {
10393            steps.push(Step::Reverse);
10394        }
10395        steps
10396    }
10397
10398    /// The view as Copy as Python writes it: every step from the data as loaded to
10399    /// the columns shown, in their order.
10400    pub fn python_steps(&self) -> Vec<Step> {
10401        let mut steps = self.view_steps();
10402        let in_order = self.column_order.iter().map(String::as_str).eq(self
10403            .schema
10404            .iter_names()
10405            .map(|s| s.as_str())
10406            .filter(|s| *s != crate::schema_union::DRIFT_COLUMN));
10407        if !in_order {
10408            steps.push(Step::Select(self.column_order.clone()));
10409        }
10410        steps
10411    }
10412
10413    pub fn is_drilled_down(&self) -> bool {
10414        self.drilled_down_group_index.is_some()
10415    }
10416
10417    /// Rebuild `lf` as `base_lf` → filters → sort. Column order is applied at collect.
10418    /// A source that runs the filters and sort itself gives the frame instead.
10419    fn apply_transformations(&mut self) {
10420        if let Some(view) = self.pushed_view() {
10421            let sorted = !self.sort_columns.is_empty() || !self.sort_ascending;
10422            self.unsorted_lf = sorted
10423                .then(|| {
10424                    self.pushdown
10425                        .as_ref()
10426                        .and_then(|p| p.view(&self.filters, &[], false))
10427                        .map(|unsorted| unsorted.lf)
10428                })
10429                .flatten();
10430            self.view_notes = Vec::new();
10431            self.view_numbered = false;
10432            self.invalidate_num_rows();
10433            self.lf = self.with_column_changes(view.lf).0;
10434            self.restore_footer_count();
10435            self.collect();
10436            return;
10437        }
10438        let mut lf = self.with_column_changes(self.base_lf.clone()).0;
10439        self.view_numbered = self.row_numbers && self.wants_view_numbers();
10440        if self.view_numbered {
10441            lf = lf.with_row_index(crate::schema_union::DRIFT_COLUMN, None);
10442        }
10443        if let Some(e) = crate::python_script::filters_expr(&self.typed_filters()) {
10444            lf = lf.filter(e);
10445        }
10446
10447        // Before the sort, so the rows it would have placed among the ordered ones are
10448        // already gone rather than ordered and then dropped.
10449        let (excluded, view_notes) = self.leave_out_unread_rows(lf);
10450        lf = excluded;
10451        // A new thing to say, so the quiet accent on `i` earns its place again.
10452        if !view_notes.is_empty() && view_notes != self.view_notes {
10453            self.notes_seen = false;
10454        }
10455        self.view_notes = view_notes;
10456
10457        // What an analysis reads: the view before its order, which no statistic
10458        // depends on and every sampled read of a sorted frame would pay for.
10459        self.unsorted_lf =
10460            (!self.sort_columns.is_empty() || !self.sort_ascending).then(|| lf.clone());
10461        if !self.sort_columns.is_empty() {
10462            lf = lf.sort_by_exprs(
10463                self.sort_columns.iter().map(col).collect::<Vec<_>>(),
10464                sort_options(self.sort_descending.clone()),
10465            );
10466        } else if !self.sort_ascending {
10467            lf = lf.reverse();
10468        }
10469
10470        self.invalidate_num_rows();
10471        self.lf = lf;
10472        self.restore_footer_count();
10473        self.collect();
10474    }
10475
10476    /// Sort with one direction for every column. `ascending` also sets the natural
10477    /// order when `columns` is empty.
10478    pub fn sort(&mut self, columns: Vec<String>, ascending: bool) {
10479        let descending = vec![!ascending; columns.len()];
10480        self.sort_ascending = ascending;
10481        self.sort_by(columns, descending);
10482    }
10483
10484    /// Sort with a direction per column.
10485    pub fn sort_by(&mut self, columns: Vec<String>, descending: Vec<bool>) {
10486        debug_assert_eq!(columns.len(), descending.len());
10487        // The one-direction flag lives on as the primary column's, for the places
10488        // that still speak it: views written for older readers, and `r`'s
10489        // natural-order fallback (which an empty sort leaves alone).
10490        if let Some(first) = descending.first() {
10491            self.sort_ascending = !first;
10492        }
10493        // Other rows come first. The sidebar sends the sort again on any apply, so
10494        // only a sort that changed counts.
10495        if columns != self.sort_columns || descending != self.sort_descending {
10496            self.widths.relearn();
10497        }
10498        self.sort_columns = columns;
10499        self.sort_descending = descending;
10500        self.buffered_start_row = 0;
10501        self.buffered_end_row = 0;
10502        self.buffered_df = None;
10503        self.apply_transformations();
10504    }
10505
10506    pub fn reverse(&mut self) {
10507        // The order is laid on top of what is there, so what is there is the frame
10508        // before it — unless an order was already laid, whose own frame is kept.
10509        if self.unsorted_lf.is_none() {
10510            self.unsorted_lf = Some(self.lf.clone());
10511        }
10512        self.sort_ascending = !self.sort_ascending;
10513        self.widths.relearn();
10514        // Reversing a sorted view flips every column's direction, so `r` twice is
10515        // always the identity whatever mix of directions was applied.
10516        for direction in &mut self.sort_descending {
10517            *direction = !*direction;
10518        }
10519
10520        self.buffered_start_row = 0;
10521        self.buffered_end_row = 0;
10522        self.buffered_df = None;
10523
10524        // A source that runs the order runs it backward too.
10525        if self.pushed_view().is_some() {
10526            self.apply_transformations();
10527            return;
10528        }
10529        if !self.sort_columns.is_empty() {
10530            self.invalidate_num_rows();
10531            self.lf = self.lf.clone().sort_by_exprs(
10532                self.sort_columns.iter().map(col).collect::<Vec<_>>(),
10533                sort_options(self.sort_descending.clone()),
10534            );
10535            self.collect();
10536        } else {
10537            self.invalidate_num_rows();
10538            self.lf = self.lf.clone().reverse();
10539            self.collect();
10540        }
10541    }
10542
10543    pub fn filter(&mut self, filters: Vec<FilterStatement>) {
10544        // The sidebar sends the filters again on any apply; only a change is new rows.
10545        if filters != self.filters {
10546            self.widths.relearn();
10547        }
10548        self.filters = filters;
10549        // A new result set, viewed from the top: a position deep in the old one would
10550        // plan a slice past a smaller result, which reads nothing.
10551        self.start_row = 0;
10552        self.buffered_start_row = 0;
10553        self.buffered_end_row = 0;
10554        self.buffered_df = None;
10555        self.apply_transformations();
10556    }
10557
10558    pub fn query(&mut self, query: String) {
10559        self.error = None;
10560
10561        let trimmed_query = query.trim();
10562        if trimmed_query.is_empty() {
10563            self.reset_lf_to_original();
10564            self.collect();
10565            return;
10566        }
10567
10568        let source_schema = self.query_source().collect_schema().ok();
10569        let parsed = parse_query_over(&query, source_schema.as_deref())
10570            .map(|parsed| parsed.past_calendar_safe(source_schema.as_deref()));
10571        match parsed {
10572            Ok(ParsedQuery {
10573                cols,
10574                filter,
10575                group_by: group_by_cols,
10576                group_by_names: group_by_col_names,
10577                distinct,
10578            }) => {
10579                let mut lf = self.query_source();
10580                let mut schema_opt: Option<Arc<Schema>> = None;
10581                // The query runs over the data as loaded: a column selected as it is, or
10582                // renamed, is still a loaded one; a computed one is not.
10583                let lineage = if cols.is_empty() && group_by_cols.is_empty() {
10584                    None
10585                } else {
10586                    let mut kept = passed_through(&group_by_cols);
10587                    if cols.is_empty() {
10588                        // Every other column, as each group's list of its values.
10589                        kept.extend(
10590                            source_schema
10591                                .iter()
10592                                .flat_map(|schema| schema.iter_names())
10593                                .filter(|n| !group_by_col_names.iter().any(|g| g == n.as_str()))
10594                                .map(|n| (n.to_string(), n.to_string())),
10595                        );
10596                    } else {
10597                        kept.extend(passed_through(&cols));
10598                    }
10599                    Some(Arc::new(kept))
10600                };
10601
10602                // Apply filter first (where clause)
10603                if let Some(f) = filter {
10604                    lf = lf.filter(f);
10605                }
10606                // What a drill-down into one of the groups shows.
10607                let group_rows = lf.clone();
10608
10609                if !group_by_cols.is_empty() {
10610                    if !cols.is_empty() {
10611                        lf = lf.group_by(group_by_cols.clone()).agg(cols);
10612                    } else {
10613                        let schema = match lf.clone().collect_schema() {
10614                            Ok(s) => s,
10615                            Err(e) => {
10616                                self.error = Some(e);
10617                                return; // Don't modify state on error
10618                            }
10619                        };
10620                        let all_columns: Vec<String> =
10621                            schema.iter_names().map(|s| s.to_string()).collect();
10622
10623                        // In Polars, when you group_by and aggregate columns without explicit aggregation functions,
10624                        // Polars automatically collects the values as lists. We need to aggregate all columns
10625                        // except the group columns to avoid duplicates.
10626                        let mut agg_exprs = Vec::new();
10627                        for col_name in &all_columns {
10628                            if !group_by_col_names.contains(col_name) {
10629                                agg_exprs.push(col(col_name));
10630                            }
10631                        }
10632
10633                        lf = lf.group_by(group_by_cols.clone()).agg(agg_exprs);
10634                    }
10635                    // Sort by the result's group-key column names (first N columns after agg).
10636                    // Works for aliased or plain names without relying on parser-derived names.
10637                    let schema = match lf.collect_schema() {
10638                        Ok(s) => s,
10639                        Err(e) => {
10640                            self.error = Some(e);
10641                            return;
10642                        }
10643                    };
10644                    schema_opt = Some(schema.clone());
10645                    let sort_exprs: Vec<Expr> = schema
10646                        .iter_names()
10647                        .take(group_by_cols.len())
10648                        .map(|n| col(n.as_str()))
10649                        .collect();
10650                    let options = sort_options(vec![false; sort_exprs.len()]);
10651                    lf = lf.sort_by_exprs(sort_exprs, options);
10652                } else if !cols.is_empty() {
10653                    lf = lf.select(cols);
10654                }
10655                if distinct {
10656                    // Stable, so a grouped result keeps its sorted order.
10657                    lf = lf.unique_stable(None, UniqueKeepStrategy::First);
10658                }
10659
10660                let schema = match schema_opt {
10661                    Some(s) => s,
10662                    None => match lf.collect_schema() {
10663                        Ok(s) => s,
10664                        Err(e) => {
10665                            self.error = Some(e);
10666                            return;
10667                        }
10668                    },
10669                };
10670
10671                // Group columns come first in the result; lock that leading run.
10672                let locked = schema
10673                    .iter_names()
10674                    .take_while(|c| group_by_col_names.iter().any(|g| g.as_str() == c.as_str()))
10675                    .count();
10676                // The keys lead the result in `by` order, whatever they were named.
10677                let keys: Vec<(PlSmallStr, Expr)> =
10678                    schema.iter_names().cloned().zip(group_by_cols).collect();
10679                // Python's division depends on the types the query read.
10680                let input = source_schema.unwrap_or_default();
10681                let steps = vec![Step::Query {
10682                    query: query.clone(),
10683                    input: input.clone(),
10684                    keys: keys.iter().map(|(name, _)| name.to_string()).collect(),
10685                }];
10686                // The same keys as Python, for a drill into one of the groups.
10687                let python_keys: Vec<Option<String>> = match crate::query::parse_nodes(&query) {
10688                    Ok(mut nodes) => {
10689                        nodes.resolve_division(&input);
10690                        nodes
10691                            .group_by
10692                            .iter()
10693                            .map(|key| Some(key.without_aliases().python()))
10694                            .collect()
10695                    }
10696                    Err(_) => vec![None; keys.len()],
10697                };
10698                let python_rows = Some(vec![Step::QueryRows {
10699                    query: query.clone(),
10700                    input,
10701                }]);
10702                self.install_query_result(lf, schema, ActiveQuery::Dsl(query), locked, steps);
10703                self.lineage = lineage;
10704                if !keys.is_empty() {
10705                    self.group_source = Some(GroupSource {
10706                        rows: group_rows,
10707                        keys,
10708                        scratch: Vec::new(),
10709                        rows_in_lists: true,
10710                        python_rows,
10711                        python_keys,
10712                        // The data as loaded, filtered.
10713                        lineage: None,
10714                    });
10715                }
10716                self.forget_reshape();
10717                // Collect will clamp start_row to valid range, but we want to ensure it's 0
10718                // So we set it to 0, collect (which may clamp it), then ensure it's 0 again
10719                self.collect();
10720                // After collect(), ensure we're at the top (collect() may have clamped if num_rows was wrong)
10721                // But if num_rows > 0, we want start_row = 0 to show the first row
10722                if self.num_rows > 0 {
10723                    self.start_row = 0;
10724                }
10725            }
10726            Err(e) => {
10727                // Parse errors are already user-facing strings; store as ComputeError
10728                self.error = Some(PolarsError::ComputeError(e.into()));
10729            }
10730        }
10731    }
10732
10733    /// The data a query runs against: the drilled group while drilled into one, else the
10734    /// pivot/melt result while one is in effect, otherwise the data as loaded. Never the
10735    /// sidebar filters or sort, which go on top, and never a previous SQL result.
10736    pub fn query_root(&self) -> LazyFrame {
10737        if self.grouped.is_some() {
10738            // While drilled, `base_lf` is the group (see `drill_down_into_group`).
10739            return self.base_lf.clone();
10740        }
10741        Self::without_drift(
10742            self.reshaped_lf
10743                .clone()
10744                .unwrap_or_else(|| self.original_lf.clone()),
10745        )
10746    }
10747
10748    /// Which loaded column each column of [`Self::query_root`] is.
10749    #[cfg(feature = "sql")]
10750    fn root_lineage(&self) -> Lineage {
10751        if self.grouped.is_some() {
10752            self.lineage.clone()
10753        } else if self.reshaped_lf.is_some() {
10754            self.reshape_lineage.clone()
10755        } else {
10756            None
10757        }
10758    }
10759
10760    /// How [`Self::query_root`] was built, as Copy as Python steps.
10761    #[cfg(feature = "sql")]
10762    fn query_root_steps(&self) -> Vec<Step> {
10763        if self.grouped.is_some() {
10764            return self.base_steps.clone();
10765        }
10766        match (&self.reshaped_lf, &self.reshape_steps) {
10767            (None, _) => Vec::new(),
10768            (Some(_), Some(steps)) => steps.clone(),
10769            (Some(_), None) => vec![Step::Unreproducible(
10770                "datui reshaped the data in a way it cannot write as Python".to_string(),
10771            )],
10772        }
10773    }
10774
10775    /// Execute a SQL query against `query_root` (registered as table "df"): the drilled
10776    /// group or the reshaped data when one is in effect, otherwise the data as loaded —
10777    /// never the sidebar filters or a previous SQL result. Running a query starts a
10778    /// fresh view: `install_query_result` clears the sidebar filters and sort, and they
10779    /// are applied after it. Empty SQL resets to original state. Does not call
10780    /// collect(); the event loop does that via AppEvent::Collect.
10781    pub fn sql_query(&mut self, sql: String) {
10782        self.error = None;
10783        let trimmed = sql.trim();
10784        if trimmed.is_empty() {
10785            self.reset_lf_to_original();
10786            return;
10787        }
10788
10789        #[cfg(feature = "sql")]
10790        {
10791            use polars_sql::SQLContext;
10792            let mut ctx = SQLContext::new();
10793            let root = self.query_root();
10794            let root_steps = self.query_root_steps();
10795            ctx.register("df", root.clone());
10796            match ctx.execute(trimmed) {
10797                Ok(mut result_lf) => {
10798                    // First, so the schema and the group source read the plan that
10799                    // runs. It changes expressions in place and only adds a projection
10800                    // over a union's inputs: the nodes stable_order orders and the
10801                    // filter count_subquery_values_once rewrites keep their shape.
10802                    crate::past_calendar::guard_plan(&mut result_lf.logical_plan);
10803                    // Read before datui orders the plan stably or by group keys: the
10804                    // marks say what the statement asked for.
10805                    let order = ordered_by(&result_lf.logical_plan);
10806                    let mut schema = match result_lf.clone().collect_schema() {
10807                        Ok(s) => s,
10808                        Err(e) => {
10809                            self.error = Some(e);
10810                            return;
10811                        }
10812                    };
10813                    let leftover =
10814                        leftover_subquery_value_columns(&mut result_lf.logical_plan, &schema);
10815                    if !leftover.is_empty() {
10816                        let shown = Arc::make_mut(&mut schema);
10817                        for name in &leftover {
10818                            shown.shift_remove(name);
10819                        }
10820                        result_lf = result_lf.drop(Selector::ByName {
10821                            names: leftover.into(),
10822                            strict: true,
10823                        });
10824                    }
10825                    let root_lineage = self.root_lineage();
10826                    let lineage = {
10827                        let columns = root.clone().collect_schema().unwrap_or_default();
10828                        let names: Vec<&str> = columns.iter_names().map(|n| n.as_str()).collect();
10829                        let shown: Vec<&str> = schema.iter_names().map(|n| n.as_str()).collect();
10830                        traced(
10831                            &root_lineage,
10832                            crate::sql_group::passed_through(trimmed, &names, &shown),
10833                        )
10834                    };
10835                    let group_source = Self::sql_group_source(
10836                        &mut ctx,
10837                        trimmed,
10838                        root,
10839                        &root_steps,
10840                        &mut result_lf,
10841                        &schema,
10842                        root_lineage,
10843                    );
10844                    // Groups sorted by their keys have no ties, and a statement simple
10845                    // enough to trace holds nothing else that gives rows in any order.
10846                    // Keeping the groups' order too would double the grouping's time.
10847                    if !group_source.as_ref().is_some_and(|(_, by_keys)| *by_keys) {
10848                        stable_order(&mut result_lf.logical_plan);
10849                    }
10850                    count_subquery_values_once(&mut result_lf.logical_plan);
10851                    let ordered_by = match &group_source {
10852                        Some((source, true)) => schema
10853                            .iter_names()
10854                            .filter(|name| source.keys.iter().any(|(key, _)| key == *name))
10855                            .map(|name| name.to_string())
10856                            .collect(),
10857                        _ => Vec::new(),
10858                    };
10859                    let mut steps = root_steps;
10860                    steps.push(Step::Sql {
10861                        sql: trimmed.to_string(),
10862                        ordered_by,
10863                    });
10864                    let query_order = order
10865                        .into_iter()
10866                        .take_while(|(name, _)| schema.contains(name))
10867                        .collect();
10868                    self.install_query_result(result_lf, schema, ActiveQuery::Sql(sql), 0, steps);
10869                    self.query_order = query_order;
10870                    self.lineage = lineage;
10871                    self.install_sql_group_source(group_source.map(|(source, _)| source));
10872                }
10873                Err(e) => {
10874                    self.error = Some(e);
10875                }
10876            }
10877        }
10878
10879        #[cfg(not(feature = "sql"))]
10880        {
10881            self.error = Some(PolarsError::ComputeError(
10882                "SQL is not supported in this build. Rebuild with default features.".into(),
10883            ));
10884        }
10885    }
10886
10887    /// What a SQL `GROUP BY` result was grouped from, when the statement is a grouping
10888    /// of `df` simple enough to trace (see [`crate::sql_group`]). Plans without reading.
10889    /// A grouping with no ORDER BY or LIMIT has its rows sorted by key, as a `by`
10890    /// query's are: Polars returns groups in any order, and every read of the result
10891    /// (each page, the row count, coming back from a drill) would otherwise be free to
10892    /// shuffle them. Also says whether it sorted them.
10893    #[cfg(feature = "sql")]
10894    fn sql_group_source(
10895        ctx: &mut polars_sql::SQLContext,
10896        sql: &str,
10897        root: LazyFrame,
10898        root_steps: &[Step],
10899        result_lf: &mut LazyFrame,
10900        result: &Schema,
10901        lineage: Lineage,
10902    ) -> Option<(GroupSource, bool)> {
10903        use crate::sql_group::KeySource;
10904        let columns = root.clone().collect_schema().ok()?;
10905        let names: Vec<&str> = columns.iter_names().map(|n| n.as_str()).collect();
10906        let plan = crate::sql_group::plan(sql, &names, result.len())?;
10907        let mut rows = ctx.execute(&plan.source_sql).ok()?;
10908        crate::past_calendar::guard_plan(&mut rows.logical_plan);
10909        let source_schema = rows.clone().collect_schema().ok()?;
10910        let mut scratch = Vec::new();
10911        let mut keys = Vec::with_capacity(plan.keys.len());
10912        let mut python_keys = Vec::with_capacity(plan.keys.len());
10913        for key in plan.keys {
10914            let (name, dtype) = result.get_at_index(key.result_index)?;
10915            let column = match key.source {
10916                KeySource::Column(c) => PlSmallStr::from(c),
10917                KeySource::Computed(c) => {
10918                    let c = PlSmallStr::from(c);
10919                    scratch.push(c.clone());
10920                    c
10921                }
10922            };
10923            // A key the grouping changed the type of would never compare equal.
10924            if source_schema.get(&column) != Some(dtype) {
10925                return None;
10926            }
10927            python_keys.push(Some(format!(
10928                "pl.col({})",
10929                crate::python_script::py_str(&column)
10930            )));
10931            keys.push((name.clone(), col(column)));
10932        }
10933        if !plan.ordered {
10934            // In the order they are shown.
10935            let by: Vec<Expr> = result
10936                .iter_names()
10937                .filter(|name| keys.iter().any(|(key, _)| key == *name))
10938                .map(|name| col(name.clone()))
10939                .collect();
10940            let options = sort_options(vec![false; by.len()]);
10941            *result_lf = result_lf.clone().sort_by_exprs(by, options);
10942        }
10943        let mut python_rows = root_steps.to_vec();
10944        python_rows.push(Step::Sql {
10945            sql: plan.source_sql.clone(),
10946            ordered_by: Vec::new(),
10947        });
10948        let source = GroupSource {
10949            rows,
10950            keys,
10951            scratch,
10952            rows_in_lists: false,
10953            python_rows: Some(python_rows),
10954            python_keys,
10955            // `SELECT *` of the root, the scratch keys left out of a drill.
10956            lineage,
10957        };
10958        Some((source, !plan.ordered))
10959    }
10960
10961    /// Record a SQL result's group source and freeze the keys that lead it, as a `by`
10962    /// query's are.
10963    #[cfg(feature = "sql")]
10964    fn install_sql_group_source(&mut self, source: Option<GroupSource>) {
10965        let Some(source) = source else {
10966            return;
10967        };
10968        self.locked_columns_count = self
10969            .schema
10970            .iter_names()
10971            .take_while(|c| source.keys.iter().any(|(k, _)| k == *c))
10972            .count();
10973        self.group_source = Some(source);
10974    }
10975
10976    /// Fuzzy search: filter rows where any string column matches the query.
10977    /// Query is split on whitespace; each token must match (in order, case-insensitive) in some string column.
10978    /// Empty query resets to original_lf.
10979    pub fn fuzzy_search(&mut self, query: String) {
10980        self.error = None;
10981        let trimmed = query.trim();
10982        if trimmed.is_empty() {
10983            self.reset_lf_to_original();
10984            self.collect();
10985            return;
10986        }
10987        // The search runs over the data as loaded, so its columns come from there too,
10988        // not from a DSL query's possibly renamed schema.
10989        let schema = match self.query_source().collect_schema() {
10990            Ok(schema) => schema,
10991            Err(e) => {
10992                self.error = Some(e);
10993                return;
10994            }
10995        };
10996        let string_cols: Vec<String> = schema
10997            .iter()
10998            .filter(|(_, dtype)| dtype.is_string())
10999            .map(|(name, _)| name.to_string())
11000            .collect();
11001        if string_cols.is_empty() {
11002            self.error = Some(PolarsError::ComputeError(
11003                "A text match needs at least one text column".into(),
11004            ));
11005            return;
11006        }
11007        let tokens: Vec<&str> = trimmed
11008            .split_whitespace()
11009            .filter(|s| !s.is_empty())
11010            .collect();
11011        let token_exprs: Vec<Expr> = tokens
11012            .iter()
11013            .map(|token| {
11014                let pattern = fuzzy_token_regex(token);
11015                string_cols
11016                    .iter()
11017                    .map(|c| col(c.as_str()).str().contains(lit(pattern.as_str()), false))
11018                    .reduce(|a, b| a.or(b))
11019                    .unwrap()
11020            })
11021            .collect();
11022        let combined = token_exprs.into_iter().reduce(|a, b| a.and(b)).unwrap();
11023        let lf = self.query_source().filter(combined);
11024        let steps = vec![Step::Search {
11025            patterns: tokens.iter().map(|t| fuzzy_token_regex(t)).collect(),
11026            columns: string_cols.clone(),
11027        }];
11028        self.install_query_result(lf, schema, ActiveQuery::Fuzzy(query), 0, steps);
11029        // Rows of the data as loaded, every column as it is.
11030        self.lineage = None;
11031        self.forget_reshape();
11032        self.collect();
11033    }
11034}
11035
11036/// Case-insensitive regex for one token: chars in order with `.*` between.
11037pub(crate) fn fuzzy_token_regex(token: &str) -> String {
11038    let inner: String =
11039        token
11040            .chars()
11041            .map(|c| regex::escape(&c.to_string()))
11042            .fold(String::new(), |mut s, e| {
11043                if !s.is_empty() {
11044                    s.push_str(".*");
11045                }
11046                s.push_str(&e);
11047                s
11048            });
11049    format!("(?i).*{}.*", inner)
11050}
11051
11052pub struct DataTable {
11053    pub header_bg: Color,
11054    pub header_fg: Color,
11055    pub row_numbers_fg: Color,
11056    pub separator_fg: Color,
11057    pub table_cell_padding: u16,
11058    pub alternate_row_bg: Option<Color>,
11059    /// When true, colorize cells by column type using the optional colors below.
11060    pub column_colors: bool,
11061    pub str_col: Option<Color>,
11062    pub int_col: Option<Color>,
11063    pub float_col: Option<Color>,
11064    pub bool_col: Option<Color>,
11065    pub temporal_col: Option<Color>,
11066    /// Color for binary-column placeholder cells (the `‹binary›` stub). Applied with italic,
11067    /// independent of `column_colors`, so stubs always read as "placeholder, not data".
11068    pub binary_col: Option<Color>,
11069    /// Names of columns that are binary in the source schema. Their cells hold the `‹binary›`
11070    /// stub (see [`binary_stub`]) and are styled with `binary_col` + italic.
11071    pub binary_cols: std::collections::HashSet<String>,
11072    /// Display-time number formatting (digit grouping, separators, alignment).
11073    pub number_format: NumberFormatSettings,
11074    /// Draw a second header row naming each column's type.
11075    pub dtype_row: bool,
11076    /// Tint under the row the cursor is on. `None` falls back to reversed video.
11077    pub selected_bg: Option<Color>,
11078    /// The full selected-row style, from the theme's `highlight_style` helper.
11079    pub selection_style: Style,
11080    /// The rail beside the selected row and the off-screen column hints.
11081    pub accent: Color,
11082    /// Null cells and the type row.
11083    pub dimmed: Color,
11084    /// Per row on screen, its file's drift group. Empty when the dataset's files agree,
11085    /// or when the rows no longer stand for rows of a file.
11086    pub drift_rows: Vec<u32>,
11087    /// What each drift group is missing. Indexed by the values in `drift_rows`.
11088    pub drift_groups: Arc<Vec<crate::schema_union::DriftGroup>>,
11089    /// Columns the view is sorted by; each carries a direction mark in the header.
11090    /// Filled from the state at render, so the marks always describe the frame drawn.
11091    pub sort_columns: Vec<String>,
11092    /// Which way each of them runs, per column, as it is applied.
11093    pub sort_descending: Vec<bool>,
11094    /// The column cursor's column: its header and cells are tinted.
11095    pub current_column: Option<String>,
11096    /// The column cursor's cells, from the theme's `column_cursor_style` helper.
11097    pub column_cursor_style: Style,
11098    /// The column cursor's header and the current cell, from the theme's
11099    /// `cell_cursor_style` helper.
11100    pub cell_cursor_style: Style,
11101    /// The glyph set the table draws with: the terminal's, unless a test asks for one.
11102    pub glyphs: &'static crate::glyphs::Glyphs,
11103    /// The terminal's width, which bounds automatic text widths (see
11104    /// [`crate::widgets::column_widths::text_cap`]). 0 takes the table's own width.
11105    pub screen_width: u16,
11106    /// The cell a find landed on: its view row and column.
11107    pub find_cell: Option<(usize, String)>,
11108    /// How that cell is drawn, from the theme's `find_match_style`.
11109    pub find_style: Style,
11110    /// The found cell's column, while the cursor is on its row: set at render.
11111    find_column: Option<String>,
11112    /// The cells a find being typed matches, by view row and column, drawn as
11113    /// found.
11114    pub match_cells: Option<std::sync::Arc<crate::find::MatchCells>>,
11115    /// The view row the first row drawn is: set at render.
11116    drawn_from: usize,
11117    /// Each column's unit from a delimited spec's unit row, for the type row: set at
11118    /// render.
11119    units: Vec<(String, String)>,
11120    /// The columns the view gave a type: their type row is in the accent.
11121    retyped: Vec<String>,
11122}
11123
11124impl Default for DataTable {
11125    fn default() -> Self {
11126        Self {
11127            header_bg: Color::Reset,
11128            header_fg: Color::Reset,
11129            row_numbers_fg: Color::Reset,
11130            separator_fg: Color::Reset,
11131            table_cell_padding: 1,
11132            alternate_row_bg: None,
11133            column_colors: false,
11134            str_col: None,
11135            int_col: None,
11136            float_col: None,
11137            bool_col: None,
11138            temporal_col: None,
11139            binary_col: None,
11140            binary_cols: std::collections::HashSet::new(),
11141            number_format: NumberFormatSettings::default(),
11142            dtype_row: false,
11143            selected_bg: None,
11144            selection_style: Style::default(),
11145            accent: Color::Reset,
11146            dimmed: Color::Reset,
11147            drift_rows: Vec::new(),
11148            drift_groups: Arc::new(Vec::new()),
11149            sort_columns: Vec::new(),
11150            sort_descending: Vec::new(),
11151            current_column: None,
11152            column_cursor_style: Style::default(),
11153            cell_cursor_style: Style::default(),
11154            glyphs: crate::glyphs::get(),
11155            screen_width: 0,
11156            find_cell: None,
11157            find_style: Style::default(),
11158            find_column: None,
11159            match_cells: None,
11160            drawn_from: 0,
11161            units: Vec::new(),
11162            retyped: Vec::new(),
11163        }
11164    }
11165}
11166
11167/// The frame that counts `lf`'s rows.
11168///
11169/// `len()` is `UInt32`, and summing it over the union a many-file scan builds widens to
11170/// `UInt128`, which Polars 0.55 cannot reduce: the in-memory engine errors and the
11171/// streaming one panics, taking the whole app with it. Counting in `UInt64` stays
11172/// inside what both engines implement.
11173pub(crate) fn row_count_lf(lf: &LazyFrame) -> LazyFrame {
11174    lf.clone().select([len().cast(DataType::UInt64)])
11175}
11176
11177pub use crate::column_types::dtype_label;
11178
11179/// The columns a read gave a type, the frame before it did, and its notes.
11180#[derive(Clone, Default)]
11181pub struct Typing {
11182    pub(crate) source: Option<LazyFrame>,
11183    pub(crate) typed: Vec<crate::column_types::Typed>,
11184    pub(crate) notes: Vec<crate::notes::Note>,
11185    /// The columns the scan read as text, by the names it read them under, for Copy
11186    /// as Python's `schema_overrides`.
11187    pub(crate) text: Vec<String>,
11188}
11189
11190impl std::fmt::Debug for Typing {
11191    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
11192        f.debug_struct("Typing")
11193            .field("typed", &self.typed)
11194            .field("notes", &self.notes)
11195            .finish_non_exhaustive()
11196    }
11197}
11198
11199/// Parameters for rendering the row numbers column.
11200struct RowNumbersParams {
11201    start_row: usize,
11202    visible_rows: usize,
11203    num_rows: usize,
11204    /// The number each row on screen shows: see [`DataTableState::row_numbers_from`].
11205    numbers: Vec<usize>,
11206    selected_row: Option<usize>,
11207}
11208
11209/// Placeholder shown in the table for binary columns. Their values (often large blobs, e.g.
11210/// raw document bytes) are never read into the display buffer — only this stub is — which keeps
11211/// scrolling and jump-to-end fast. The real bytes remain in `lf` for export/analysis.
11212///
11213/// The text comes from the active glyph set (`binary_stub`), so ASCII terminals get a
11214/// readable `<binary>` instead of mojibake.
11215pub(crate) fn binary_stub() -> &'static str {
11216    crate::glyphs::get().binary_stub
11217}
11218
11219/// One column of the rows on screen, formatted once: what the layout measures and
11220/// what the table draws.
11221struct ColumnSlice {
11222    name: String,
11223    drift_mark: &'static str,
11224    sort_mark: &'static str,
11225    /// Cells for the name and both marks.
11226    header_width: u16,
11227    type_label: Option<String>,
11228    type_width: u16,
11229    cells: Vec<SliceCell>,
11230    /// Cells for the widest value on screen.
11231    value_width: u16,
11232    /// Whether any row on screen holds a value rather than a null.
11233    has_values: bool,
11234    /// The width the column is laid out at: its stable width once sized (see
11235    /// [`ColumnWidths`]), until then what shows this page whole.
11236    width: u16,
11237    right_align: bool,
11238    /// Whether a value may be shown clipped beside other columns (see
11239    /// [`is_truncatable_dtype`]).
11240    clips: bool,
11241    cell_style: Option<Style>,
11242    /// The column's type colour, for its heading.
11243    colour: Option<Color>,
11244}
11245
11246impl ColumnSlice {
11247    /// The width the column asks for: whole within it, or clipped behind the marker.
11248    fn natural_width(&self) -> u16 {
11249        self.width
11250    }
11251
11252    fn measure(&self) -> PageMeasure {
11253        PageMeasure {
11254            header: self.header_width,
11255            type_label: self.type_width,
11256            values: self.value_width,
11257            has_values: self.has_values,
11258            clips: self.clips,
11259        }
11260    }
11261}
11262
11263/// What sizes columns for one draw: the stable widths, the types that key them, and
11264/// the cap on automatic text.
11265struct Sizing<'a> {
11266    widths: &'a mut ColumnWidths,
11267    schema: &'a Schema,
11268    cap: u16,
11269}
11270
11271enum SliceCell {
11272    /// A null, drawn as the glyph for its kind of empty.
11273    Null(&'static str),
11274    Value(String),
11275}
11276
11277/// The most sideways moves held for a draw: a held key typed faster than frames
11278/// can land, bounded so the draw that lands them stays a frame's work.
11279const MAX_WAITING_MOVES: usize = 32;
11280
11281/// A sideways move waiting on a draw: the view's own, or the column cursor's.
11282#[derive(Debug, Clone, Copy, PartialEq, Eq)]
11283enum WaitingMove {
11284    View(ColumnMove),
11285    Cursor(CursorMove),
11286}
11287
11288/// Where the scrolling columns are, for the off-screen hints over the header.
11289struct ScrollCue {
11290    area: Rect,
11291    more_left: bool,
11292    /// Columns not drawn at all, right of the last drawn.
11293    more_right: usize,
11294}
11295
11296/// Columns laid out for one side of the table: the columns that fit, the width each
11297/// gets, and the rows they are drawn for.
11298struct FittedColumns {
11299    cols: Vec<ColumnSlice>,
11300    widths: Vec<u16>,
11301    rows: usize,
11302    /// The last column ends at the table's right edge with more columns after it:
11303    /// its heading leaves its last cell to the off-screen hint, which would
11304    /// otherwise cover the heading's last character or clip marker.
11305    hint_cell: bool,
11306}
11307
11308/// The least the scrolling side keeps beside frozen columns, and the most: a third of
11309/// the table in between, and never over half. Enough for any number or timestamp whole,
11310/// and for the start of a text column.
11311const MIN_SCROLL_RESERVE: u16 = 12;
11312const MAX_SCROLL_RESERVE: u16 = 40;
11313
11314/// Narrower than this, a column is not drawn as a clipped sliver: the clip marker plus
11315/// at least one cell of what it marks.
11316fn min_partial_width(g: &crate::glyphs::Glyphs) -> u16 {
11317    let marker = u16::try_from(crate::glyphs::cell_width(g.ellipsis)).unwrap_or(u16::MAX);
11318    marker.saturating_add(1).max(3)
11319}
11320
11321/// Which side of the frozen separator a layout is for. Both follow one sizing rule;
11322/// they differ in what they do with a column that does not fit whole.
11323#[derive(Clone, Copy, PartialEq, Eq)]
11324enum Side {
11325    /// Only the first column may be clipped, and never as a numeric preview: any other
11326    /// column that does not fit whole scrolls instead, where it can be read whole.
11327    Frozen,
11328    /// The last column shown may be clipped, and a first column with no room for its
11329    /// values shows a marked preview rather than nothing.
11330    Scrolling,
11331}
11332
11333/// The width a column gets with `remaining` cells left, or `None` to leave it for the
11334/// next scroll. The heading and the values are fitted separately: a long heading is
11335/// clipped over values that fit whole, and never hides them. Text may be clipped, since
11336/// a clipped string still reads as its start. A number or timestamp that does not fit
11337/// is left for the scroll, as a cut one reads as a different value, unless it is the
11338/// first scrolling column and nothing else would show: then it is drawn clipped,
11339/// behind the clip marker.
11340fn fit_column(
11341    col: &ColumnSlice,
11342    remaining: u16,
11343    min_partial: u16,
11344    first: bool,
11345    side: Side,
11346) -> Option<u16> {
11347    if col.natural_width() <= remaining {
11348        return Some(col.natural_width());
11349    }
11350    if side == Side::Frozen && !first {
11351        return None;
11352    }
11353    if remaining >= min_partial && (col.value_width <= remaining || col.clips) {
11354        return Some(remaining);
11355    }
11356    (side == Side::Scrolling && first && remaining > 0).then_some(remaining)
11357}
11358
11359/// Fitted `spans` as one line of a `width`-cell column, flush right when `right`.
11360/// ratatui places a span by its whole string's width, which can differ from the cells
11361/// it draws (`لا` draws two, a halfwidth sound mark one): its right alignment then
11362/// pushes the last grapheme off the cell, and a span after such a one overwrites it.
11363/// So the padding is counted in drawn cells, and spans that disagree are drawn as one.
11364fn cell_line(mut spans: Vec<Span<'static>>, width: u16, right: bool) -> Line<'static> {
11365    use unicode_width::UnicodeWidthStr;
11366    let drawn = |s: &Span| crate::glyphs::cell_width(&s.content);
11367    if spans.len() > 1 && spans.iter().any(|s| drawn(s) != s.content.width()) {
11368        let style = spans[0].style;
11369        let joined: String = spans.iter().map(|s| s.content.as_ref()).collect();
11370        spans = vec![Span::styled(joined, style)];
11371    }
11372    let used: usize = spans.iter().map(drawn).sum();
11373    let pad = usize::from(width).saturating_sub(used);
11374    if right && pad > 0 {
11375        spans.insert(0, Span::raw(" ".repeat(pad)));
11376    }
11377    Line::from(spans)
11378}
11379
11380/// The rows of `df` on screen: `len` of them from `offset`, or `None` past its end.
11381/// As [`visible_slice`], or the frame's columns with no rows when none of its rows is
11382/// on screen, so a table of no rows still draws its header.
11383fn visible_or_header(df: &DataFrame, offset: usize, len: usize) -> Option<DataFrame> {
11384    if df.width() == 0 {
11385        return None;
11386    }
11387    Some(visible_slice(df, offset, len).unwrap_or_else(|| df.clear()))
11388}
11389
11390fn visible_slice(df: &DataFrame, offset: usize, len: usize) -> Option<DataFrame> {
11391    let len = len.min(df.height().saturating_sub(offset));
11392    (offset < df.height() && len > 0).then(|| df.slice(offset as i64, len))
11393}
11394
11395/// Whether a column whose value doesn't fully fit may be shown truncated. Textual columns
11396/// (strings, raw bytes, categorical/enum labels) and nested previews (structs, lists,
11397/// arrays) are fine to clip: a partial value still reads as a clipped string, and a
11398/// nested value left whole would hold one long value's width on every page after it.
11399/// Numeric, temporal and boolean columns are excluded: a truncated number or
11400/// timestamp reads as a different (wrong) value, so those are dropped until scrolled into view.
11401fn is_truncatable_dtype(dtype: &DataType) -> bool {
11402    match dtype {
11403        DataType::String | DataType::Binary => true,
11404        other => other.is_categorical() || other.is_enum() || other.is_nested(),
11405    }
11406}
11407
11408impl DataTable {
11409    pub fn new() -> Self {
11410        Self::default()
11411    }
11412
11413    pub fn with_colors(
11414        mut self,
11415        header_bg: Color,
11416        header_fg: Color,
11417        row_numbers_fg: Color,
11418        separator_fg: Color,
11419    ) -> Self {
11420        self.header_bg = header_bg;
11421        self.header_fg = header_fg;
11422        self.row_numbers_fg = row_numbers_fg;
11423        self.separator_fg = separator_fg;
11424        self
11425    }
11426
11427    pub fn with_cell_padding(mut self, padding: u16) -> Self {
11428        self.table_cell_padding = padding;
11429        self
11430    }
11431
11432    /// The terminal's width, so automatic widths do not change when a sidebar opens.
11433    pub fn with_screen_width(mut self, width: u16) -> Self {
11434        self.screen_width = width;
11435        self
11436    }
11437
11438    pub fn with_alternate_row_bg(mut self, color: Option<Color>) -> Self {
11439        self.alternate_row_bg = color;
11440        self
11441    }
11442
11443    /// Enable column-type coloring and set colors for string, int, float, bool, and temporal columns.
11444    pub fn with_column_type_colors(
11445        mut self,
11446        str_col: Color,
11447        int_col: Color,
11448        float_col: Color,
11449        bool_col: Color,
11450        temporal_col: Color,
11451    ) -> Self {
11452        self.column_colors = true;
11453        self.str_col = Some(str_col);
11454        self.int_col = Some(int_col);
11455        self.float_col = Some(float_col);
11456        self.bool_col = Some(bool_col);
11457        self.temporal_col = Some(temporal_col);
11458        self
11459    }
11460
11461    /// Set the color used for binary-column placeholder cells.
11462    pub fn with_binary_col(mut self, color: Color) -> Self {
11463        self.binary_col = Some(color);
11464        self
11465    }
11466
11467    /// Set the names of binary columns, whose cells render the `‹binary›` stub.
11468    pub fn with_binary_columns(mut self, names: std::collections::HashSet<String>) -> Self {
11469        self.binary_cols = names;
11470        self
11471    }
11472
11473    /// Set display-time number formatting (digit grouping and alignment).
11474    pub fn with_number_format(mut self, settings: NumberFormatSettings) -> Self {
11475        self.number_format = settings;
11476        self
11477    }
11478
11479    /// Show or hide the second header row of column types.
11480    pub fn with_dtype_row(mut self, on: bool) -> Self {
11481        self.dtype_row = on;
11482        self
11483    }
11484
11485    /// Tell the table which rows came from files missing which columns, so a cell the
11486    /// file never had draws differently from a null the data holds.
11487    pub fn with_drift(
11488        mut self,
11489        rows: Vec<u32>,
11490        groups: Arc<Vec<crate::schema_union::DriftGroup>>,
11491    ) -> Self {
11492        self.drift_rows = rows;
11493        self.drift_groups = groups;
11494        self
11495    }
11496
11497    /// The columns the view is sorted by, and which way each runs, for the header
11498    /// marks. The stateful render fills this from the state itself; the builder is
11499    /// for direct callers of `render_dataframe`, such as tests.
11500    pub fn with_sort(mut self, columns: Vec<String>, descending: Vec<bool>) -> Self {
11501        debug_assert_eq!(columns.len(), descending.len());
11502        self.sort_columns = columns;
11503        self.sort_descending = descending;
11504        self
11505    }
11506
11507    /// The direction mark after a column's name when the view is sorted by it. Every
11508    /// column of a multi-sort carries one — the mark alone, no position number — and
11509    /// each shows its own column's direction. Empty for unsorted columns.
11510    fn sort_mark_for(&self, column: &str) -> &'static str {
11511        if let Some(i) = self.sort_columns.iter().position(|c| c == column) {
11512            let g = self.glyphs;
11513            if self.sort_descending.get(i).copied().unwrap_or(false) {
11514                g.sort_desc
11515            } else {
11516                g.sort_asc
11517            }
11518        } else {
11519            ""
11520        }
11521    }
11522
11523    /// The footnote mark after a column's name, when it is not in every file or the
11524    /// files disagree on its type. Empty otherwise.
11525    fn drift_mark_for(&self, column: &str, drifting: &HashSet<&str>) -> &'static str {
11526        if drifting.contains(column) {
11527            self.glyphs.drift_mark
11528        } else {
11529            ""
11530        }
11531    }
11532
11533    /// Every column some file is missing, gathered once a frame. Most columns are in
11534    /// every file, and this keeps them to one hash lookup rather than a walk of every
11535    /// group's lists.
11536    fn drifting_columns(&self) -> HashSet<&str> {
11537        self.drift_groups
11538            .iter()
11539            .flat_map(|group| group.absent.iter().chain(group.unread.iter()))
11540            .map(|name| name.as_str())
11541            .collect()
11542    }
11543
11544    /// What a null in `column` draws as, per drift group: the plain null glyph, the
11545    /// absent glyph for a group whose files never had the column, or the conflict
11546    /// glyph for one whose files hold it in another type. Empty when nothing drifts.
11547    fn null_glyphs_for(
11548        &self,
11549        column: &str,
11550        g: &'static crate::glyphs::Glyphs,
11551        drifting: &HashSet<&str>,
11552    ) -> Vec<&'static str> {
11553        if self.drift_rows.is_empty() || !drifting.contains(column) {
11554            return Vec::new();
11555        }
11556        self.drift_groups
11557            .iter()
11558            .map(|group| {
11559                if group.absent.iter().any(|c| c == column) {
11560                    g.absent
11561                } else if group.unread.iter().any(|c| c == column) {
11562                    g.conflict
11563                } else {
11564                    g.null
11565                }
11566            })
11567            .collect()
11568    }
11569
11570    /// The selected row's style and tint, the rail colour, and the dim colour
11571    /// for nulls. The style comes from the theme's `highlight_style` helper,
11572    /// so this widget never invents a fallback of its own.
11573    pub fn with_selection_colors(
11574        mut self,
11575        selection_style: Style,
11576        selected_bg: Option<Color>,
11577        accent: Color,
11578        dimmed: Color,
11579    ) -> Self {
11580        self.selection_style = selection_style;
11581        self.selected_bg = selected_bg;
11582        self.accent = accent;
11583        self.dimmed = dimmed;
11584        self
11585    }
11586
11587    /// Mark the cell a find landed on, drawn in `style` in place of the current cell's
11588    /// style while the cursor is on it.
11589    /// Draw these cells (view row, column) as found: a find's matches as it is typed.
11590    pub fn with_match_cells(
11591        mut self,
11592        cells: Option<std::sync::Arc<crate::find::MatchCells>>,
11593    ) -> Self {
11594        self.match_cells = cells;
11595        self
11596    }
11597
11598    pub fn with_find_cell(mut self, cell: Option<(usize, String)>, style: Style) -> Self {
11599        self.find_cell = cell;
11600        self.find_style = style;
11601        self
11602    }
11603
11604    /// The column cursor's styles, from the theme's helpers: its cells, and its
11605    /// header and the current cell.
11606    pub fn with_cursor_styles(mut self, column: Style, cell: Style) -> Self {
11607        self.column_cursor_style = column;
11608        self.cell_cursor_style = cell;
11609        self
11610    }
11611
11612    /// How many rows the header takes: the names, plus the type row when it is on.
11613    pub fn header_height(&self) -> u16 {
11614        if self.dtype_row { 2 } else { 1 }
11615    }
11616
11617    /// Style of the highlighted row, from the theme's `highlight_style` helper.
11618    fn highlight_style(&self) -> Style {
11619        self.selection_style
11620    }
11621
11622    /// Return the color for a column dtype when column_colors is enabled.
11623    fn column_type_color(&self, dtype: &DataType) -> Option<Color> {
11624        if !self.column_colors {
11625            return None;
11626        }
11627        match dtype {
11628            DataType::String => self.str_col,
11629            DataType::Int8
11630            | DataType::Int16
11631            | DataType::Int32
11632            | DataType::Int64
11633            | DataType::UInt8
11634            | DataType::UInt16
11635            | DataType::UInt32
11636            | DataType::UInt64 => self.int_col,
11637            DataType::Float32 | DataType::Float64 => self.float_col,
11638            DataType::Boolean => self.bool_col,
11639            DataType::Date | DataType::Datetime(_, _) | DataType::Time | DataType::Duration(_) => {
11640                self.temporal_col
11641            }
11642            _ => None,
11643        }
11644    }
11645
11646    /// Render `df` into `area` on its own, with widths learned from this page alone,
11647    /// returning how many columns were shown. For tests of the layout rules.
11648    ///
11649    /// `leading_gap` keeps the first column one cell off the left edge: the columns right
11650    /// of the frozen separator, which would otherwise touch it.
11651    #[cfg(test)]
11652    fn render_dataframe(
11653        &self,
11654        df: &DataFrame,
11655        area: Rect,
11656        buf: &mut Buffer,
11657        state: &mut TableState,
11658        leading_gap: bool,
11659        _start_row_offset: usize,
11660    ) -> usize {
11661        let mut widths = ColumnWidths::default();
11662        let sizing = Sizing {
11663            widths: &mut widths,
11664            schema: df.schema(),
11665            cap: self.text_cap(area.width),
11666        };
11667        self.render_scrolling(df, area, buf, state, leading_gap, sizing)
11668            .0
11669    }
11670
11671    /// The cap on automatic text widths, from the terminal's width when known.
11672    fn text_cap(&self, table_width: u16) -> u16 {
11673        let basis = if self.screen_width > 0 {
11674            self.screen_width
11675        } else {
11676            table_width
11677        };
11678        crate::widgets::column_widths::text_cap(basis)
11679    }
11680
11681    /// Lay out and draw the scrolling columns, at their stable widths. Returns how
11682    /// many were drawn.
11683    fn render_scrolling(
11684        &self,
11685        df: &DataFrame,
11686        area: Rect,
11687        buf: &mut Buffer,
11688        state: &mut TableState,
11689        leading_gap: bool,
11690        mut sizing: Sizing,
11691    ) -> (usize, Vec<(u16, u16, String)>, usize) {
11692        let rows = df
11693            .height()
11694            .min((area.height as usize).saturating_sub(self.header_height() as usize));
11695        let lead = u16::from(leading_gap);
11696        let mut fitted = self.fit_columns(df, rows, area.width, lead, Side::Scrolling, &mut sizing);
11697        let shown = fitted.cols.len();
11698        let mut used = fitted
11699            .widths
11700            .iter()
11701            .fold(lead, |used, &w| used.saturating_add(w))
11702            + self
11703                .table_cell_padding
11704                .saturating_mul(u16::try_from(shown.saturating_sub(1)).unwrap_or(u16::MAX));
11705        if let Some(filled) =
11706            self.fill_last_column(&mut fitted, area.width.saturating_sub(used), &mut sizing)
11707        {
11708            used = used.saturating_add(filled);
11709        }
11710        fitted.hint_cell = shown > 0 && shown < df.width() && used >= area.width;
11711        let (columns, rows) = self.draw_columns(&fitted, area, buf, state, leading_gap);
11712        (shown, columns, rows)
11713    }
11714
11715    /// Widen the last column drawn by the `room` left at the table's right edge, so a
11716    /// long text on the far right runs to the edge rather than stopping at its cap.
11717    /// Only a column drawn whole at an automatic width, and not a right-aligned number,
11718    /// which would only move away from its heading. Returns the cells it took.
11719    fn fill_last_column(
11720        &self,
11721        fitted: &mut FittedColumns,
11722        room: u16,
11723        sizing: &mut Sizing,
11724    ) -> Option<u16> {
11725        let (col, width) = fitted.cols.last().zip(fitted.widths.last_mut())?;
11726        if room == 0 || col.right_align || *width < col.natural_width() {
11727            return None;
11728        }
11729        let dtype = sizing.schema.get(col.name.as_str())?;
11730        if sizing.widths.choice(&col.name, dtype) != WidthChoice::Auto {
11731            return None;
11732        }
11733        *width = width.saturating_add(room);
11734        sizing.widths.fill(&col.name, dtype, *width);
11735        Some(room)
11736    }
11737
11738    /// The frozen columns that fit beside a usable scrolling column, with their widths.
11739    ///
11740    /// `width` is the room right of the row numbers. Whenever anything scrolls, the
11741    /// scrolling side keeps a third of the table (within bounds), so a wide frozen prefix
11742    /// can never leave it a sliver; the frozen columns that do not fit in the rest are
11743    /// the caller's to hand to the scrolling side, where each can be read whole. Only
11744    /// the first frozen column is ever clipped to stay frozen, and never a number.
11745    fn fit_frozen_columns(
11746        &self,
11747        locked: &DataFrame,
11748        rows: usize,
11749        width: u16,
11750        nothing_else_scrolls: bool,
11751        table_width: u16,
11752        sizing: &mut Sizing,
11753    ) -> FittedColumns {
11754        // The space before the separator, and the separator. The gap after it is the
11755        // scrolling side's.
11756        let room = width.saturating_sub(2);
11757        if nothing_else_scrolls {
11758            let fitted = self.fit_columns(locked, rows, room, 0, Side::Frozen, sizing);
11759            if fitted.cols.len() == locked.width() {
11760                return fitted;
11761            }
11762        }
11763        let reserve = (table_width / 3)
11764            .clamp(MIN_SCROLL_RESERVE, MAX_SCROLL_RESERVE)
11765            .min(table_width / 2);
11766        self.fit_columns(
11767            locked,
11768            rows,
11769            room.saturating_sub(reserve),
11770            0,
11771            Side::Frozen,
11772            sizing,
11773        )
11774    }
11775
11776    /// Lay `df`'s columns out left to right in `width` cells, `lead` of them taken first,
11777    /// formatting a column's first `rows` values only once it is reached. The one sizing
11778    /// rule for frozen and scrolling columns alike: each column at its stable width,
11779    /// learned from the rows on screen in terminal cells the first time it is drawn.
11780    fn fit_columns(
11781        &self,
11782        df: &DataFrame,
11783        rows: usize,
11784        width: u16,
11785        lead: u16,
11786        side: Side,
11787        sizing: &mut Sizing,
11788    ) -> FittedColumns {
11789        let drifting = self.drifting_columns();
11790        let min_partial = min_partial_width(self.glyphs);
11791        // Reused across every cell so formatting allocates only the string each cell keeps.
11792        let mut scratch = String::new();
11793        let mut fitted = FittedColumns {
11794            cols: Vec::new(),
11795            widths: Vec::new(),
11796            rows,
11797            hint_cell: false,
11798        };
11799        let mut used = lead;
11800        for col_index in 0..df.width() {
11801            let remaining = width.saturating_sub(used);
11802            if remaining == 0 {
11803                break;
11804            }
11805            let mut col = self.slice_column(df, col_index, rows, &drifting, &mut scratch);
11806            let dtype = sizing
11807                .schema
11808                .get(col.name.as_str())
11809                .unwrap_or_else(|| df[col_index].dtype());
11810            col.width = sizing
11811                .widths
11812                .width(&col.name, dtype, col.measure(), sizing.cap);
11813            let first = fitted.cols.is_empty();
11814            let Some(w) = fit_column(&col, remaining, min_partial, first, side) else {
11815                break;
11816            };
11817            let whole = w >= col.natural_width();
11818            used = used
11819                .saturating_add(w)
11820                .saturating_add(self.table_cell_padding);
11821            fitted.cols.push(col);
11822            fitted.widths.push(w);
11823            if !whole {
11824                // A clipped column took everything left; nothing after it can fit.
11825                break;
11826            }
11827        }
11828        fitted
11829    }
11830
11831    /// One column's heading, type and first `rows` values, formatted and measured.
11832    fn slice_column(
11833        &self,
11834        df: &DataFrame,
11835        col_index: usize,
11836        rows: usize,
11837        drifting: &HashSet<&str>,
11838        scratch: &mut String,
11839    ) -> ColumnSlice {
11840        let g = self.glyphs;
11841        let col_data = &df[col_index];
11842        let name = col_data.name().as_str();
11843        let dtype = col_data.dtype();
11844        // Binary columns hold the `‹binary›` stub: style them with binary_col + italic so they
11845        // read as a placeholder rather than data, regardless of the column_colors setting.
11846        let is_binary = self.binary_cols.contains(name);
11847        let cell_style = if is_binary {
11848            let mut s = Style::default().add_modifier(Modifier::ITALIC);
11849            if let Some(c) = self.binary_col {
11850                s = s.fg(c);
11851            }
11852            Some(s)
11853        } else {
11854            self.column_type_color(dtype)
11855                .map(|c| Style::default().fg(c))
11856        };
11857        // Resolved once per column: dtype eligibility and the include/exclude globs never
11858        // touch the per-cell path. A binary column holds the stub, not a number, so it is
11859        // always passthrough.
11860        let col_fmt = if is_binary {
11861            CellFormatter::Passthrough
11862        } else {
11863            self.number_format.formatter_for(name, dtype)
11864        };
11865        // Numeric columns render flush-right so magnitudes line up; strings, booleans,
11866        // temporals and binary stubs stay left.
11867        let right_align = self.number_format.align_numeric_right
11868            && !is_binary
11869            && numfmt::is_right_aligned_dtype(dtype);
11870        // A null in this column means different things in different files: the data's
11871        // own null, a file written without the column, or a file that stores it in
11872        // another type. Resolved once per column, by group.
11873        let null_glyph_by_group = self.null_glyphs_for(name, g, drifting);
11874
11875        let mut cells = Vec::with_capacity(rows);
11876        let mut value_width = 0usize;
11877        for row_index in 0..rows.min(col_data.len()) {
11878            let value = col_data.get(row_index).unwrap();
11879            if matches!(value, AnyValue::Null) {
11880                let glyph = self
11881                    .drift_rows
11882                    .get(row_index)
11883                    .and_then(|group| null_glyph_by_group.get(*group as usize))
11884                    .copied()
11885                    .unwrap_or(g.null);
11886                value_width = value_width.max(crate::glyphs::cell_width(glyph));
11887                cells.push(SliceCell::Null(glyph));
11888                continue;
11889            }
11890            // A list is previewed here, for the cells on screen only: the buffer keeps
11891            // it a list, as formatting a whole row group's lists stalled every scroll.
11892            let text = match &value {
11893                AnyValue::List(items) => Cow::Owned(crate::exact::list_preview(items)),
11894                value => numfmt::format_any_value(&col_fmt, value, scratch),
11895            };
11896            // A break or a tab would vanish from a cell and run the text together.
11897            // Only a cell's start can be drawn: measuring a huge value whole would
11898            // cost every frame what the value costs.
11899            let text = crate::exact::cell_preview(&text, g);
11900            value_width = value_width.max(crate::glyphs::cell_width(&text));
11901            cells.push(SliceCell::Value(text));
11902        }
11903
11904        let drift_mark = self.drift_mark_for(name, drifting);
11905        let sort_mark = self.sort_mark_for(name);
11906        // Both header marks widen the column, or a sorted or drifting column's last
11907        // character would be pushed out of its cell.
11908        let header_width = crate::glyphs::cell_width(name)
11909            + crate::glyphs::cell_width(drift_mark)
11910            + crate::glyphs::cell_width(sort_mark);
11911        // The type row is part of the header, so a column is at least as wide as its
11912        // type name; "datetime" under a column called "ts" would otherwise clip.
11913        // A binary column's buffer holds the stub text; the type is the source's.
11914        // A unit from a delimited spec's unit row sits beside the type: `f64 · deg F`.
11915        let type_label = self.dtype_row.then(|| {
11916            let label = if is_binary {
11917                dtype_label(&DataType::Binary)
11918            } else {
11919                dtype_label(dtype)
11920            };
11921            match self.units.iter().find(|(column, _)| column == name) {
11922                Some((_, unit)) => format!("{label} {} {unit}", self.glyphs.middot),
11923                None => label,
11924            }
11925        });
11926        let type_width = type_label
11927            .as_deref()
11928            .map(crate::glyphs::cell_width)
11929            .unwrap_or(0);
11930        let cells_u16 = |w: usize| u16::try_from(w).unwrap_or(u16::MAX);
11931        let has_values = cells.iter().any(|c| matches!(c, SliceCell::Value(_)));
11932        ColumnSlice {
11933            name: name.to_string(),
11934            drift_mark,
11935            sort_mark,
11936            header_width: cells_u16(header_width),
11937            type_label,
11938            type_width: cells_u16(type_width),
11939            cells,
11940            value_width: cells_u16(value_width),
11941            has_values,
11942            width: cells_u16(header_width.max(type_width).max(value_width)),
11943            right_align,
11944            clips: is_binary || is_truncatable_dtype(dtype),
11945            cell_style,
11946            colour: if is_binary {
11947                self.binary_col
11948            } else {
11949                self.column_type_color(dtype)
11950            },
11951        }
11952    }
11953
11954    /// Draw fitted columns into `area` as a table. Every heading, type and value is
11955    /// fitted to its column here, at a grapheme boundary and marked where cut, so
11956    /// ratatui never truncates one itself: it cuts a right-aligned value from the left,
11957    /// which turns `1234567` into `34567`.
11958    fn draw_columns(
11959        &self,
11960        fitted: &FittedColumns,
11961        area: Rect,
11962        buf: &mut Buffer,
11963        state: &mut TableState,
11964        leading_gap: bool,
11965    ) -> (Vec<(u16, u16, String)>, usize) {
11966        let g = self.glyphs;
11967        let fit = |text: &str, width: u16| -> String {
11968            crate::glyphs::fit_cells(text, usize::from(width), g.ellipsis).into_owned()
11969        };
11970        // A null is drawn as a glyph in the dim colour, so it can never be mistaken
11971        // for an empty string or a zero that happens to be blank.
11972        let null_style = Style::default()
11973            .fg(self.dimmed)
11974            .add_modifier(Modifier::ITALIC);
11975        let columns = || fitted.cols.iter().zip(fitted.widths.iter().copied());
11976
11977        let rows: Vec<Row> = (0..fitted.rows)
11978            .map(|row_index| {
11979                let cells: Vec<Cell> = columns()
11980                    .map(|(col, w)| {
11981                        let span = match col.cells.get(row_index) {
11982                            Some(SliceCell::Null(glyph)) => Span::styled(fit(glyph, w), null_style),
11983                            Some(SliceCell::Value(text)) => {
11984                                let mut style = col.cell_style.unwrap_or_default();
11985                                if self.match_cells.as_ref().is_some_and(|cells| {
11986                                    cells.get(col.name.as_str()).is_some_and(|rows| {
11987                                        rows.contains(&(self.drawn_from + row_index))
11988                                    })
11989                                }) {
11990                                    style = style.patch(self.find_style);
11991                                }
11992                                Span::styled(fit(text, w), style)
11993                            }
11994                            None => return Cell::default(),
11995                        };
11996                        Cell::from(cell_line(vec![span], w, col.right_align))
11997                    })
11998                    .collect();
11999                let row_style = if row_index % 2 == 1 {
12000                    self.alternate_row_bg
12001                        .map(|c| Style::default().bg(c))
12002                        .unwrap_or_default()
12003                } else {
12004                    Style::default()
12005                };
12006                Row::new(cells).style(row_style)
12007            })
12008            .collect();
12009
12010        let header_row_style = if self.header_bg == Color::Reset {
12011            Style::default().fg(self.header_fg)
12012        } else {
12013            Style::default().bg(self.header_bg).fg(self.header_fg)
12014        };
12015        // The name takes the column's own colour, bold, so the header says what the
12016        // cells say without a mark in front of it; the type row beneath repeats the
12017        // colour in plain weight and spells the type out. Headings follow their
12018        // column's alignment; a left-aligned heading over right-aligned digits reads
12019        // as a rendering bug.
12020        let last = fitted.cols.len().saturating_sub(1);
12021        // The column cursor, when its column is among these.
12022        let cursor = self
12023            .current_column
12024            .as_deref()
12025            .and_then(|name| fitted.cols.iter().position(|c| c.name == name));
12026        // The current cell is drawn as found while a find's cell is the cursor's: a find
12027        // moves the cursor to the cell it lands on.
12028        let cell_style = if cursor.is_some() && self.find_column == self.current_column {
12029            self.find_style
12030        } else {
12031            self.cell_cursor_style
12032        };
12033        // Under a reversed row (`table_selected = "reversed"`), a reversed cell would
12034        // read as the rest of the row: the current cell is the one drawn upright.
12035        let cell_style = if self
12036            .highlight_style()
12037            .add_modifier
12038            .contains(Modifier::REVERSED)
12039        {
12040            cell_style.remove_modifier(Modifier::REVERSED)
12041        } else {
12042            cell_style
12043        };
12044        let headers: Vec<Cell> = columns()
12045            .enumerate()
12046            .map(|(i, (col, w))| {
12047                // The off-screen hint goes on the type row when there is one, else on
12048                // the name row: that line of the last column stops a cell short.
12049                let hint = u16::from(fitted.hint_cell && i == last && w > 1);
12050                let (name_w, type_w) = if col.type_label.is_some() {
12051                    (w, w - hint)
12052                } else {
12053                    (w - hint, w)
12054                };
12055                let name_style = match col.colour {
12056                    Some(c) => Style::default().fg(c).add_modifier(Modifier::BOLD),
12057                    None => Style::default().add_modifier(Modifier::BOLD),
12058                };
12059                // The marks are state, so a long name gives way to them: the name is
12060                // what gets clipped, never the sort direction or the drift footnote.
12061                let marks = crate::glyphs::cell_width(col.drift_mark)
12062                    + crate::glyphs::cell_width(col.sort_mark);
12063                let mut heading = Vec::with_capacity(3);
12064                match u16::try_from(marks).ok().filter(|&m| m < name_w) {
12065                    Some(marks) => {
12066                        heading.push(Span::styled(fit(&col.name, name_w - marks), name_style));
12067                        if !col.drift_mark.is_empty() {
12068                            heading.push(Span::styled(
12069                                col.drift_mark,
12070                                Style::default().fg(self.dimmed),
12071                            ));
12072                        }
12073                        // In the name's own style: the mark says how this column's
12074                        // values run, so it reads as part of the heading.
12075                        if !col.sort_mark.is_empty() {
12076                            heading.push(Span::styled(col.sort_mark, name_style));
12077                        }
12078                    }
12079                    None => heading.push(Span::styled(fit(&col.name, name_w), name_style)),
12080                }
12081                let mut lines = vec![cell_line(heading, name_w, col.right_align)];
12082                if let Some(label) = &col.type_label {
12083                    let type_style = match col.colour {
12084                        // A type the view gave, not the read: it shows.
12085                        _ if self.retyped.contains(&col.name) => Style::default().fg(self.accent),
12086                        Some(c) => Style::default().fg(c),
12087                        None => Style::default().fg(self.dimmed),
12088                    };
12089                    let label = Span::styled(fit(label, type_w), type_style);
12090                    lines.push(cell_line(vec![label], type_w, col.right_align));
12091                }
12092                let cell = Cell::from(Text::from(lines));
12093                if cursor == Some(i) {
12094                    cell.style(self.cell_cursor_style)
12095                } else {
12096                    cell
12097                }
12098            })
12099            .collect();
12100
12101        let mut table = Table::new(rows, fitted.widths.clone())
12102            .column_spacing(self.table_cell_padding)
12103            .header(
12104                Row::new(headers)
12105                    .style(header_row_style)
12106                    .height(self.header_height()),
12107            )
12108            .row_highlight_style(self.highlight_style())
12109            .column_highlight_style(self.column_cursor_style)
12110            .cell_highlight_style(cell_style);
12111        if leading_gap {
12112            // A blank selection column on every row: the Table offsets the header and
12113            // the cells past it and paints each row's tint across it, so the gap
12114            // stripes and highlights like the rest of the row.
12115            table = table
12116                .highlight_symbol(" ")
12117                .highlight_spacing(HighlightSpacing::Always);
12118        }
12119        // The frozen and scrolling sides share the row selection; the column is each
12120        // side's own, so it is set for this draw only.
12121        state.select_column(cursor);
12122        StatefulWidget::render(table, area, buf, state);
12123        state.select_column(None);
12124        // Where the Table put each column: past the gap's selection column, laid out
12125        // as it lays them out, so a click finds the column it drew.
12126        let lead = u16::from(leading_gap).min(area.width);
12127        let columns_area = Rect {
12128            x: area.x + lead,
12129            width: area.width - lead,
12130            ..area
12131        };
12132        let spans = ratatui::layout::Layout::horizontal(
12133            fitted
12134                .widths
12135                .iter()
12136                .map(|&w| ratatui::layout::Constraint::Length(w)),
12137        )
12138        .flex(ratatui::layout::Flex::Start)
12139        .spacing(self.table_cell_padding)
12140        .split(columns_area);
12141        let columns = spans
12142            .iter()
12143            .zip(&fitted.cols)
12144            .map(|(span, col)| (span.x, span.right(), col.name.clone()))
12145            .collect();
12146        (
12147            columns,
12148            fitted.rows.min(usize::from(
12149                area.height.saturating_sub(self.header_height()),
12150            )),
12151        )
12152    }
12153
12154    /// The width a scrolling column is drawn at: the width it was last drawn at in
12155    /// this view, or, for one not drawn since, measured from the rows on screen in the
12156    /// buffer held and learned as drawing it would learn it. What a sideways page is
12157    /// planned with.
12158    fn measure_column(
12159        &self,
12160        state: &mut DataTableState,
12161        name: &str,
12162        offset: usize,
12163        rows: usize,
12164        cap: u16,
12165    ) -> u16 {
12166        if let Some(width) = state.drawn_width(name) {
12167            return width;
12168        }
12169        let Some(page) = state.page_column(name, offset, rows) else {
12170            return crate::widgets::column_widths::UNSEEN_WIDTH;
12171        };
12172        let col = self.slice_column(
12173            &page,
12174            0,
12175            page.height(),
12176            &self.drifting_columns(),
12177            &mut String::new(),
12178        );
12179        let dtype = state.width_dtype(name);
12180        state.widths.width(name, &dtype, col.measure(), cap)
12181    }
12182
12183    /// Fit each column waiting for it to the rows on screen, from the buffer already
12184    /// held, so a column scrolled out of view is fitted to this page too.
12185    fn fit_pending(&self, state: &mut DataTableState, offset: usize, rows: usize, cap: u16) {
12186        let pending = state.widths.fits_pending();
12187        if pending.is_empty() || !state.buffer_on_hand() {
12188            return;
12189        }
12190        let drifting = self.drifting_columns();
12191        let mut scratch = String::new();
12192        for (name, dtype) in pending {
12193            let Some(page) = state.page_column(&name, offset, rows) else {
12194                continue;
12195            };
12196            let col = self.slice_column(&page, 0, page.height(), &drifting, &mut scratch);
12197            state.widths.fit(&name, &dtype, col.measure(), cap);
12198        }
12199    }
12200
12201    fn render_row_numbers(&self, area: Rect, buf: &mut Buffer, params: RowNumbersParams) {
12202        // Header row: same style as the rest of the column headers (fill full width so color matches)
12203        let header_style = if self.header_bg == Color::Reset {
12204            Style::default().fg(self.header_fg)
12205        } else {
12206            Style::default().bg(self.header_bg).fg(self.header_fg)
12207        };
12208        let header_h = self.header_height().min(area.height);
12209        let header_fill = " ".repeat(area.width as usize);
12210        for dy in 0..header_h {
12211            Paragraph::new(header_fill.clone())
12212                .style(header_style)
12213                .render(
12214                    Rect {
12215                        x: area.x,
12216                        y: area.y + dy,
12217                        width: area.width,
12218                        height: 1,
12219                    },
12220                    buf,
12221                );
12222        }
12223
12224        // Only render up to the actual number of rows in the data
12225        let rows_to_render = params
12226            .visible_rows
12227            .min(params.num_rows.saturating_sub(params.start_row));
12228
12229        if rows_to_render == 0 {
12230            return;
12231        }
12232
12233        let number = |row_idx: usize| params.numbers.get(row_idx).copied().unwrap_or_default();
12234        // Calculate width needed for largest row number
12235        let max_row_num = (0..rows_to_render).map(number).max().unwrap_or_default();
12236        let max_width = max_row_num.to_string().len();
12237
12238        // Render row numbers
12239        for row_idx in 0..rows_to_render.min(area.height.saturating_sub(header_h) as usize) {
12240            let row_num_text = number(row_idx).to_string();
12241
12242            // Right-align row numbers within the available width
12243            let padding = max_width.saturating_sub(row_num_text.len());
12244            let padded_text = format!("{}{}", " ".repeat(padding), row_num_text);
12245
12246            // Match main table background: default when row is even (or no alternate);
12247            // when alternate_row_bg is set, odd rows use that background. The selected
12248            // row carries the same tint as the table's own highlight.
12249            let is_selected = params.selected_row == Some(row_idx);
12250            let (fg, bg) = if is_selected {
12251                (
12252                    Color::Reset,
12253                    self.selected_bg
12254                        .or(self.alternate_row_bg.filter(|_| row_idx % 2 == 1)),
12255                )
12256            } else {
12257                (
12258                    self.row_numbers_fg,
12259                    self.alternate_row_bg.filter(|_| row_idx % 2 == 1),
12260                )
12261            };
12262            let row_num_style = match bg {
12263                Some(bg_color) => Style::default().fg(fg).bg(bg_color),
12264                None => Style::default().fg(fg),
12265            };
12266
12267            let y = area.y + row_idx as u16 + header_h;
12268            if y < area.y + area.height {
12269                Paragraph::new(padded_text).style(row_num_style).render(
12270                    Rect {
12271                        x: area.x,
12272                        y,
12273                        width: area.width,
12274                        height: 1,
12275                    },
12276                    buf,
12277                );
12278            }
12279        }
12280    }
12281}
12282
12283/// The table as the last frame drew it: what a click on it lands on.
12284#[derive(Debug, Clone, PartialEq, Eq)]
12285struct DrawnTable {
12286    /// The whole table, rail and header included.
12287    area: Rect,
12288    header: u16,
12289    /// The first row drawn and how many under the header.
12290    start_row: usize,
12291    rows: usize,
12292    columns: DrawnColumns,
12293}
12294
12295/// Each column drawn: its cells across, `[from, to)`, and its name.
12296pub type DrawnColumns = Vec<(u16, u16, String)>;
12297
12298/// What a click on the table lands on: the row on screen, counted from the top (none
12299/// on the header), and the column (none on the rail or the row numbers).
12300#[derive(Debug, Clone, PartialEq, Eq)]
12301pub struct CellHit {
12302    pub row: Option<usize>,
12303    pub column: Option<String>,
12304}
12305
12306impl DataTableState {
12307    /// Forget where the table was drawn: a frame that does not draw it leaves nothing
12308    /// there to click.
12309    pub fn forget_drawn(&mut self) {
12310        self.drawn = None;
12311    }
12312
12313    /// What the cell at `(x, y)` showed in the last frame. `None` off the table, or
12314    /// below its last row.
12315    pub fn drawn_cell(&self, x: u16, y: u16) -> Option<CellHit> {
12316        let drawn = self.drawn.as_ref()?;
12317        if !drawn.area.contains(ratatui::layout::Position { x, y }) {
12318            return None;
12319        }
12320        let below_header = usize::from(y - drawn.area.y).checked_sub(usize::from(drawn.header));
12321        let row = match below_header {
12322            Some(row) if row >= drawn.rows => return None,
12323            row => row,
12324        };
12325        let column = drawn
12326            .columns
12327            .iter()
12328            .find(|(from, to, _)| (*from..*to).contains(&x))
12329            .map(|(_, _, name)| name.clone());
12330        Some(CellHit { row, column })
12331    }
12332
12333    /// The column whose right edge the header cell at `(x, y)` is: the first cell
12334    /// of the gap after a column's last, where a drag resizes it.
12335    pub fn drawn_edge(&self, x: u16, y: u16) -> Option<String> {
12336        let drawn = self.drawn.as_ref()?;
12337        let header = drawn.area.y..drawn.area.y + drawn.header;
12338        if !header.contains(&y) || !drawn.area.contains(ratatui::layout::Position { x, y }) {
12339            return None;
12340        }
12341        if let Some(name) = Self::right_edge(drawn, x) {
12342            return Some(name);
12343        }
12344        if drawn
12345            .columns
12346            .iter()
12347            .any(|(from, to, _)| (*from..*to).contains(&x))
12348        {
12349            return None;
12350        }
12351        drawn
12352            .columns
12353            .iter()
12354            .find(|(_, to, _)| *to == x)
12355            .map(|(_, _, name)| name.clone())
12356    }
12357
12358    /// The edge of a column that reaches the table's right side, filled or cut
12359    /// there: no gap follows it, so its last header cell is its edge.
12360    fn right_edge(drawn: &DrawnTable, x: u16) -> Option<String> {
12361        let (_, to, name) = drawn.columns.iter().max_by_key(|(_, to, _)| *to)?;
12362        (*to >= drawn.area.right() && x + 1 == *to).then(|| name.clone())
12363    }
12364
12365    /// The column drawn across `x`, whatever the row: where a header dragged sideways
12366    /// is over.
12367    pub fn drawn_column_across(&self, x: u16) -> Option<String> {
12368        let drawn = self.drawn.as_ref()?;
12369        drawn
12370            .columns
12371            .iter()
12372            .find(|(from, to, _)| (*from..*to).contains(&x))
12373            .map(|(_, _, name)| name.clone())
12374    }
12375
12376    /// The header rows as drawn and each column's cells across, `[from, to)`: where a
12377    /// header drag draws its drop mark.
12378    pub fn drawn_header(&self) -> Option<(Rect, DrawnColumns)> {
12379        let drawn = self.drawn.as_ref()?;
12380        Some((
12381            Rect {
12382                height: drawn.header.min(drawn.area.height),
12383                ..drawn.area
12384            },
12385            drawn.columns.clone(),
12386        ))
12387    }
12388
12389    /// Put the cursor on what a click landed on: the row, when the rows drawn are
12390    /// still the view's, and the column. Returns whether it is there now, row and
12391    /// column both.
12392    pub fn point_at(&mut self, hit: &CellHit) -> bool {
12393        let mut landed = true;
12394        if let Some(row) = hit.row {
12395            if let Some(drawn) = self.drawn.as_ref()
12396                && drawn.start_row == self.start_row
12397                && row < drawn.rows
12398            {
12399                self.table_state.select(Some(row));
12400            } else {
12401                landed = false;
12402            }
12403        }
12404        if let Some(name) = &hit.column {
12405            self.set_current_column(name);
12406            landed &= self.current_column() == Some(name.as_str());
12407        }
12408        landed
12409    }
12410}
12411
12412impl StatefulWidget for DataTable {
12413    type State = DataTableState;
12414
12415    fn render(mut self, area: Rect, buf: &mut Buffer, state: &mut Self::State) {
12416        // The view's own sort, not the grouped original's: it is what ordered the
12417        // rows being drawn, so the header marks can never disagree with them.
12418        (self.sort_columns, self.sort_descending) = state.header_sort();
12419        self.current_column = state.current_column().map(str::to_string);
12420        self.units = state.units();
12421        self.retyped = state.retyped_columns();
12422        // One column on the left is the rail: blank on every row but the one the
12423        // cursor is on, where it carries the accent. It also holds the "columns off to
12424        // the left" hint in the header, so no header name ever gets a character
12425        // overwritten.
12426        let cap = self.text_cap(area.width);
12427        let whole = area;
12428        state.drawn = None;
12429        let rail_area = Rect {
12430            x: area.x,
12431            y: area.y,
12432            width: 1.min(area.width),
12433            height: area.height,
12434        };
12435        let area = Rect {
12436            x: area.x.saturating_add(1),
12437            y: area.y,
12438            width: area.width.saturating_sub(1),
12439            height: area.height,
12440        };
12441        let header_h = self.header_height();
12442        state.visible_termcols = area.width as usize;
12443        let new_visible_rows = (area.height as usize).saturating_sub(header_h as usize);
12444        let visible_rows_changed = new_visible_rows != state.visible_rows;
12445        state.visible_rows = new_visible_rows;
12446
12447        // Fewer rows (the footer grew a line): the page starts that much later, so the
12448        // row the cursor is on stays the row it is on.
12449        if let Some(selected) = state.table_state.selected()
12450            && selected >= state.visible_rows
12451            && state.visible_rows > 0
12452        {
12453            let overflow = selected - (state.visible_rows - 1);
12454            state.start_row += overflow;
12455            state.table_state.select(Some(state.visible_rows - 1));
12456        }
12457
12458        // Only a page the rows on hand do not cover needs a read: the footer growing
12459        // and shrinking a line must not re-read the buffer each time.
12460        if visible_rows_changed && !state.page_on_hand(state.start_row) {
12461            // The App event loop checks this flag after each render and triggers an
12462            // async collect.
12463            state.needs_recollect = true;
12464        }
12465
12466        // Only show errors in main view if not suppressed (e.g., when query input is active)
12467        // Query errors should only be shown in the query input frame
12468        if let Some(error) = state.error.as_ref()
12469            && !state.suppress_error_display
12470        {
12471            Paragraph::new(format!("Error: {}", user_message_from_polars(error)))
12472                .centered()
12473                .block(
12474                    Block::default()
12475                        .borders(Borders::NONE)
12476                        .padding(Padding::top(area.height / 2)),
12477                )
12478                .wrap(ratatui::widgets::Wrap { trim: true })
12479                .render(area, buf);
12480            return;
12481        }
12482        // If suppress_error_display is true, continue rendering the table normally
12483
12484        let start_row = state.start_to_draw();
12485        self.drawn_from = start_row;
12486        state.on_screen = None;
12487        // Only on the cursor's row (and, when drawn, its column): the cursor is what a
12488        // find moves, and a mark left behind would read as a second match.
12489        let selected = state.table_state.selected();
12490        self.find_column = self.find_cell.take().and_then(|(row, name)| {
12491            (row.checked_sub(start_row) == selected && selected.is_some()).then_some(name)
12492        });
12493
12494        // Where the scrolling columns are, for the cue drawn over the header after them.
12495        let mut scroll_indicator: Option<ScrollCue> = None;
12496
12497        // The numbers `#` shows, and the column wide enough for the widest.
12498        let numbers = if state.row_numbers {
12499            state.row_numbers_from(start_row, state.visible_rows)
12500        } else {
12501            Vec::new()
12502        };
12503        let row_num_width = if state.row_numbers {
12504            let widest = numbers.iter().max().copied().unwrap_or(1);
12505            widest.to_string().len().max(1) as u16 + 1 // +1 for spacing
12506        } else {
12507            0
12508        };
12509        let row_num_width = row_num_width.min(area.width);
12510        let data_area = Rect {
12511            x: area.x + row_num_width,
12512            width: area.width - row_num_width,
12513            ..area
12514        };
12515        let row_num_area = Rect {
12516            width: row_num_width,
12517            ..area
12518        };
12519        let visible_rows = state.visible_rows;
12520        let row_numbers = |start_row, num_rows, numbers, selected_row| RowNumbersParams {
12521            start_row,
12522            visible_rows,
12523            num_rows,
12524            numbers,
12525            selected_row,
12526        };
12527        let row_number_params = row_numbers(
12528            start_row,
12529            state.num_rows,
12530            numbers,
12531            state.table_state.selected(),
12532        );
12533
12534        // Both sides are cut to the same rows on screen, so the frozen columns are
12535        // measured on what they show, not on the head of the buffer.
12536        let offset = start_row.saturating_sub(state.buffered_start_row);
12537        let rows_room = (area.height as usize).saturating_sub(header_h as usize);
12538        let locked_slice = state
12539            .locked_df
12540            .as_ref()
12541            .and_then(|df| visible_or_header(df, offset, state.visible_rows));
12542        self.fit_pending(state, offset, state.visible_rows.min(rows_room), cap);
12543
12544        if state.df.is_some() || state.locked_df.is_some() {
12545            if state.row_numbers {
12546                self.render_row_numbers(row_num_area, buf, row_number_params);
12547            }
12548            let mut drawn_columns = Vec::new();
12549            let mut drawn_rows = 0;
12550            let mut scroll_area = data_area;
12551            let mut leading_gap = false;
12552            if let Some(locked) = locked_slice {
12553                let asked = state.locked_columns_count();
12554                // The rule runs down the header and the rows on screen, and stops
12555                // under the last: below it is no table to divide.
12556                let rule_bottom = (area.y + header_h)
12557                    .saturating_add(locked.height().min(rows_room) as u16)
12558                    .min(area.bottom());
12559                let mut fitted = self.fit_frozen_columns(
12560                    &locked,
12561                    locked.height().min(rows_room),
12562                    data_area.width,
12563                    state.column_order.len() <= asked,
12564                    area.width,
12565                    &mut Sizing {
12566                        widths: &mut state.widths,
12567                        schema: &state.schema,
12568                        cap,
12569                    },
12570                );
12571                state.fit_frozen(fitted.cols.len());
12572                // The state may not take the fit while no buffer is on hand; it then
12573                // keeps its count, and only what fits of it is drawn.
12574                let shown = state.frozen_shown().min(fitted.cols.len());
12575                fitted.cols.truncate(shown);
12576                fitted.widths.truncate(shown);
12577                let mut separator_x = data_area.x;
12578                if shown > 0 {
12579                    let gaps = self.table_cell_padding.saturating_mul(shown as u16 - 1);
12580                    let columns_width = fitted.widths.iter().sum::<u16>().saturating_add(gaps);
12581                    // One cell more than the columns: the space before the separator,
12582                    // which takes the header fill and the row tints.
12583                    let frozen_area = Rect {
12584                        width: columns_width.saturating_add(1).min(data_area.width),
12585                        ..data_area
12586                    };
12587                    let (columns, rows) =
12588                        self.draw_columns(&fitted, frozen_area, buf, &mut state.table_state, false);
12589                    drawn_columns.extend(columns);
12590                    drawn_rows = drawn_rows.max(rows);
12591                    separator_x = frozen_area.right();
12592                }
12593                if separator_x < data_area.right() {
12594                    // A broken rule while some frozen columns had to scroll: the window
12595                    // holds fewer than were asked for, and they come back with room.
12596                    let rule = if shown < asked {
12597                        self.glyphs.rule_broken
12598                    } else {
12599                        self.glyphs.rule
12600                    };
12601                    for y in area.y..rule_bottom {
12602                        let cell = &mut buf[(separator_x, y)];
12603                        cell.set_symbol(rule);
12604                        cell.set_style(Style::default().fg(self.separator_fg));
12605                    }
12606                }
12607                let scroll_x = separator_x.saturating_add(1).min(data_area.right());
12608                scroll_area = Rect {
12609                    x: scroll_x,
12610                    width: data_area.right() - scroll_x,
12611                    ..data_area
12612                };
12613                leading_gap = true;
12614            }
12615            // A page asked for lands here, where the room it is planned in is known.
12616            let room = Room {
12617                width: scroll_area.width,
12618                lead: u16::from(leading_gap),
12619                padding: self.table_cell_padding,
12620            };
12621            let rows = state.visible_rows.min(rows_room);
12622            state.land_column_moves(room, |state, name| {
12623                self.measure_column(state, name, offset, rows, cap)
12624            });
12625            if let Some(sliced_df) = state
12626                .df
12627                .as_ref()
12628                .and_then(|df| visible_or_header(df, offset, state.visible_rows))
12629            {
12630                let total_cols = sliced_df.width();
12631                let (shown, columns, rows) = self.render_scrolling(
12632                    &sliced_df,
12633                    scroll_area,
12634                    buf,
12635                    &mut state.table_state,
12636                    leading_gap,
12637                    Sizing {
12638                        widths: &mut state.widths,
12639                        schema: &state.schema,
12640                        cap,
12641                    },
12642                );
12643                drawn_columns.extend(columns);
12644                drawn_rows = drawn_rows.max(rows);
12645                let more_left = state.termcol_index > 0;
12646                let more_right = total_cols.saturating_sub(shown);
12647                let first = state.frozen_shown() + state.termcol_index + 1;
12648                let total = state.column_order.len();
12649                state.on_screen =
12650                    state
12651                        .cursor_index()
12652                        .filter(|_| total > 1)
12653                        .map(|cursor| OnScreen {
12654                            first,
12655                            last: first + shown.saturating_sub(1),
12656                            cursor: cursor + 1,
12657                            total,
12658                        });
12659                scroll_indicator = Some(ScrollCue {
12660                    area: scroll_area,
12661                    more_left,
12662                    more_right,
12663                });
12664            } else {
12665                // Every column frozen: all on screen, and the cursor walks them.
12666                let total = state.column_order.len();
12667                state.on_screen =
12668                    state
12669                        .cursor_index()
12670                        .filter(|_| total > 1)
12671                        .map(|cursor| OnScreen {
12672                            first: 1,
12673                            last: total,
12674                            cursor: cursor + 1,
12675                            total,
12676                        });
12677            }
12678            state.drawn = Some(DrawnTable {
12679                area: whole,
12680                header: header_h,
12681                start_row,
12682                rows: drawn_rows,
12683                columns: drawn_columns,
12684            });
12685        } else if !state.column_order.is_empty() {
12686            // No rows on hand, but a schema: the header alone, each column its own type.
12687            let empty_columns: Vec<_> = state
12688                .column_order
12689                .iter()
12690                .map(|name| {
12691                    let dtype = state
12692                        .schema
12693                        .get(name.as_str())
12694                        .cloned()
12695                        .unwrap_or(DataType::String);
12696                    Series::new_empty(name.as_str().into(), &dtype).into()
12697                })
12698                .collect();
12699            match DataFrame::new_infer_height(empty_columns) {
12700                Ok(empty_df) => {
12701                    if state.row_numbers {
12702                        self.render_row_numbers(
12703                            row_num_area,
12704                            buf,
12705                            row_numbers(0, 0, Vec::new(), None),
12706                        );
12707                    }
12708                    self.render_scrolling(
12709                        &empty_df,
12710                        data_area,
12711                        buf,
12712                        &mut state.table_state,
12713                        false,
12714                        Sizing {
12715                            widths: &mut state.widths,
12716                            schema: &state.schema,
12717                            cap,
12718                        },
12719                    );
12720                }
12721                _ => {
12722                    Paragraph::new("No data").render(area, buf);
12723                }
12724            }
12725        } else {
12726            // Truly empty: no schema, not loaded, or blank file
12727            Paragraph::new("No data").render(area, buf);
12728        }
12729
12730        // A table known to hold no rows says so under its header.
12731        let empty = state.num_rows_valid && state.num_rows == 0 && !state.column_order.is_empty();
12732        if empty && area.height > header_h && data_area.width > 0 {
12733            let line = Rect {
12734                y: area.y + header_h,
12735                height: 1,
12736                ..data_area
12737            };
12738            Paragraph::new("No rows")
12739                .style(Style::default().fg(self.dimmed))
12740                .render(line, buf);
12741        }
12742
12743        // The rail: the header rows take the header fill so the bar runs edge to edge,
12744        // and the selected row gets the accent mark.
12745        if rail_area.width > 0 && rail_area.height > 0 {
12746            let g = self.glyphs;
12747            let header_style = if self.header_bg == Color::Reset {
12748                Style::default().fg(self.header_fg)
12749            } else {
12750                Style::default().bg(self.header_bg).fg(self.header_fg)
12751            };
12752            for dy in 0..header_h.min(rail_area.height) {
12753                let cell = &mut buf[(rail_area.x, rail_area.y + dy)];
12754                cell.set_char(' ');
12755                cell.set_style(header_style);
12756            }
12757            if state.df.is_some()
12758                && !empty
12759                && let Some(sel) = state.table_state.selected()
12760            {
12761                let y = rail_area.y + header_h + sel as u16;
12762                if y < rail_area.y + rail_area.height {
12763                    let cell = &mut buf[(rail_area.x, y)];
12764                    cell.set_symbol(g.rail.trim_end());
12765                    let mut style = Style::default()
12766                        .fg(self.accent)
12767                        .add_modifier(Modifier::BOLD);
12768                    if let Some(bg) = self.selected_bg {
12769                        style = style.bg(bg);
12770                    }
12771                    cell.set_style(style);
12772                }
12773            }
12774        }
12775
12776        // Hints that more columns exist off-screen. The left one sits in the rail,
12777        // where nothing else lives. The right one says how many are hidden, and goes
12778        // on the type row when that row is on (its short labels leave room), else on
12779        // the name row, right-aligned into the slack after the last column.
12780        if let Some(cue) = scroll_indicator
12781            && cue.area.width > 0
12782            && cue.area.height > 0
12783        {
12784            let g = self.glyphs;
12785            let scroll_area = cue.area;
12786            let hidden = cue.more_right;
12787            let hint_style = if self.header_bg == Color::Reset {
12788                Style::default()
12789                    .fg(self.accent)
12790                    .add_modifier(Modifier::BOLD)
12791            } else {
12792                Style::default()
12793                    .bg(self.header_bg)
12794                    .fg(self.accent)
12795                    .add_modifier(Modifier::BOLD)
12796            };
12797            if cue.more_left && rail_area.width > 0 {
12798                let cell = &mut buf[(rail_area.x, rail_area.y)];
12799                cell.set_symbol(g.arrow_left);
12800                cell.set_style(hint_style);
12801            }
12802            if hidden > 0 {
12803                let y = if header_h > 1 {
12804                    scroll_area.y + 1
12805                } else {
12806                    scroll_area.y
12807                };
12808                // The count when the blank run at the end of the row holds it, the arrow
12809                // alone when not, so the count never covers a heading or a type.
12810                let right = scroll_area.x + scroll_area.width;
12811                let free = (scroll_area.x..right)
12812                    .rev()
12813                    .take_while(|&x| buf[(x, y)].symbol() == " ")
12814                    .count();
12815                let mut text = format!(" +{hidden} {}", g.arrow_right);
12816                if text.chars().count() > free {
12817                    text = g.arrow_right.to_string();
12818                }
12819                let w = text.chars().count() as u16;
12820                if scroll_area.width >= w {
12821                    let x0 = right - w;
12822                    for (i, ch) in text.chars().enumerate() {
12823                        let cell = &mut buf[(x0 + i as u16, y)];
12824                        cell.set_char(ch);
12825                        cell.set_style(hint_style);
12826                    }
12827                }
12828            }
12829        }
12830    }
12831}
12832
12833/// A partition column's type: the file's own if stored there, else inferred from its
12834/// values the way Polars does for a full scan.
12835pub(crate) fn partition_dtype(
12836    name: &str,
12837    file_schema: &Schema,
12838    values: &[(String, String)],
12839) -> DataType {
12840    use polars::io::csv::read::schema_inference::{finish_infer_field_schema, infer_field_schema};
12841    if let Some(dtype) = file_schema.get(name) {
12842        return dtype.clone();
12843    }
12844    let seen: PlIndexSet<DataType> = values
12845        .iter()
12846        .filter(|(k, v)| k == name && !v.is_empty() && v != "__HIVE_DEFAULT_PARTITION__")
12847        .map(|(_, v)| infer_field_schema(v, true, false))
12848        .collect();
12849    if seen.is_empty() {
12850        DataType::String
12851    } else {
12852        finish_infer_field_schema(&seen)
12853    }
12854}
12855
12856/// Everything a checkpoint promises to put back, as a test can compare it: the rows
12857/// each frame of the pipeline reads, the unsorted one analyses read among them
12858/// (collected here, on the test's thread), the schema,
12859/// the query, filters, sort and layout, the reshape, the drill, what the notes say, the
12860/// selection, and the count and buffer the view holds.
12861#[cfg(test)]
12862#[derive(Debug, PartialEq)]
12863pub(crate) struct ViewSnapshot {
12864    rows: std::result::Result<DataFrame, String>,
12865    analysis_rows: std::result::Result<DataFrame, String>,
12866    base_rows: std::result::Result<DataFrame, String>,
12867    reshaped_rows: Option<std::result::Result<DataFrame, String>>,
12868    schema: Arc<Schema>,
12869    queries: [String; 3],
12870    filters: String,
12871    sort: (Vec<String>, Vec<bool>, bool),
12872    layout: (Vec<String>, usize),
12873    reshape: String,
12874    grouped: (bool, bool),
12875    drill: (Option<usize>, Option<Vec<String>>, Option<Vec<String>>),
12876    drift: (bool, Arc<Vec<crate::schema_union::DriftGroup>>),
12877    notes: (Vec<crate::notes::Note>, bool, Vec<crate::notes::Note>),
12878    selection: (Option<usize>, usize, usize),
12879    count: (usize, bool, u64),
12880    buffer: (usize, usize, Option<DataFrame>),
12881    shown: Option<DataFrame>,
12882    error: Option<String>,
12883}
12884
12885#[cfg(test)]
12886impl ViewSnapshot {
12887    /// Whether the view's rows and count had been read.
12888    pub(crate) fn has_rows(&self) -> bool {
12889        self.buffer.2.is_some() && self.count.1
12890    }
12891}
12892
12893#[cfg(test)]
12894impl DataTableState {
12895    pub(crate) fn snapshot(&self) -> ViewSnapshot {
12896        let rows = |lf: &LazyFrame| lf.clone().collect().map_err(|e| e.to_string());
12897        ViewSnapshot {
12898            rows: rows(&self.lf),
12899            analysis_rows: rows(&self.analysis_lf()),
12900            base_rows: rows(&self.base_lf),
12901            reshaped_rows: self.reshaped_lf.as_ref().map(rows),
12902            schema: self.schema.clone(),
12903            queries: [
12904                self.active_query.clone(),
12905                self.active_sql_query.clone(),
12906                self.active_fuzzy_query.clone(),
12907            ],
12908            filters: format!("{:?}", self.filters),
12909            sort: (
12910                self.sort_columns.clone(),
12911                self.sort_descending.clone(),
12912                self.sort_ascending,
12913            ),
12914            layout: (self.column_order.clone(), self.locked_columns_count),
12915            reshape: format!(
12916                "{:?} {:?} {:?}",
12917                self.last_pivot_spec, self.last_melt_spec, self.reshape_source
12918            ),
12919            grouped: (self.grouped.is_some(), self.group_source.is_some()),
12920            drill: (
12921                self.drilled_down_group_index,
12922                self.drilled_down_group_key.clone(),
12923                self.drilled_down_group_key_columns.clone(),
12924            ),
12925            drift: (self.drift_column_present, self.drift_groups.clone()),
12926            notes: (self.notes.clone(), self.notes_seen, self.view_notes.clone()),
12927            selection: (
12928                self.table_state.selected(),
12929                self.start_row,
12930                self.termcol_index,
12931            ),
12932            count: (self.num_rows, self.num_rows_valid, self.len_generation),
12933            buffer: (
12934                self.buffered_start_row,
12935                self.buffered_end_row,
12936                self.buffered_df.clone(),
12937            ),
12938            shown: self.df.clone(),
12939            error: self.error.as_ref().map(|e| e.to_string()),
12940        }
12941    }
12942}
12943
12944#[cfg(test)]
12945mod checkpoint_tests {
12946    use super::*;
12947    use crate::filter_modal::{FilterOperator, LogicalOperator};
12948
12949    fn state() -> DataTableState {
12950        let lf = df!(
12951            "id" => (0..20i64).collect::<Vec<_>>(),
12952            "key" => (0..20).map(|i| if i % 2 == 0 { "a" } else { "b" }).collect::<Vec<_>>(),
12953            "val" => (0..20i64).map(|i| i * 10).collect::<Vec<_>>(),
12954        )
12955        .unwrap()
12956        .lazy();
12957        let mut state = DataTableState::from_lazyframe(lf, &crate::OpenOptions::default()).unwrap();
12958        state.visible_rows = 4;
12959        state.collect();
12960        state
12961    }
12962
12963    fn filter(column: &str, op: FilterOperator, value: &str) -> FilterStatement {
12964        FilterStatement {
12965            columns: Vec::new(),
12966            column: column.to_string(),
12967            operator: op,
12968            value: value.to_string(),
12969            logical_op: LogicalOperator::And,
12970        }
12971    }
12972
12973    /// A view with something at every stage: a query, a filter, a sort, a layout, a
12974    /// selection off the top, and its rows and count read.
12975    fn busy_view() -> DataTableState {
12976        let mut state = state();
12977        state.query("select id, val, key where val >= 20".to_string());
12978        state.filter(vec![filter("val", FilterOperator::Lt, "170")]);
12979        state.sort_by(vec!["val".to_string()], vec![true]);
12980        state.set_column_order(vec!["val".to_string(), "id".to_string(), "key".to_string()]);
12981        state.set_locked_columns(1);
12982        state.scroll_to(3);
12983        state.collect();
12984        state.table_state.select(Some(2));
12985        assert!(state.error().is_none(), "{:?}", state.error());
12986        assert!(state.is_num_rows_valid());
12987        assert!(state.display_df().is_some());
12988        state
12989    }
12990
12991    /// The same view over a melt, so a reshape is what a failure must leave in place.
12992    fn melted_view() -> DataTableState {
12993        let mut state = state();
12994        state
12995            .melt(&MeltSpec {
12996                index: vec!["id".to_string()],
12997                value_columns: vec!["val".to_string()],
12998                variable_name: "variable".to_string(),
12999                value_name: "value".to_string(),
13000            })
13001            .unwrap();
13002        state.filter(vec![filter("id", FilterOperator::Gt, "3")]);
13003        state.collect();
13004        state.table_state.select(Some(1));
13005        assert!(state.error().is_none(), "{:?}", state.error());
13006        state
13007    }
13008
13009    /// Fails as a view's step does: by the error the step leaves, or at the plan.
13010    fn planned(state: &mut DataTableState) -> std::result::Result<(), String> {
13011        if let Some(e) = state.error() {
13012            return Err(e.to_string());
13013        }
13014        state.check_plan().map_err(|e| e.to_string())
13015    }
13016
13017    /// Each step a view replays, ending in one that fails: every one puts back the
13018    /// whole view, and none of them reads a row on the way.
13019    #[test]
13020    fn a_transition_failing_after_each_step_puts_the_view_back() {
13021        type Steps = fn(&mut DataTableState) -> std::result::Result<(), String>;
13022        let cases: [(&str, Steps); 8] = [
13023            ("the query", |s| {
13024                s.query("select nope".to_string());
13025                planned(s)
13026            }),
13027            ("a filter after the query", |s| {
13028                s.query("select id, val".to_string());
13029                s.filter(vec![filter("key", FilterOperator::Eq, "a")]);
13030                planned(s)
13031            }),
13032            ("a sort after the filter", |s| {
13033                s.query("select id, val".to_string());
13034                s.filter(vec![filter("val", FilterOperator::Gt, "0")]);
13035                s.sort_by(vec!["key".to_string()], vec![false]);
13036                planned(s)
13037            }),
13038            ("a melt after the query", |s| {
13039                s.query("select id, val".to_string());
13040                s.melt(&MeltSpec {
13041                    index: vec!["id".to_string()],
13042                    value_columns: vec!["nope".to_string()],
13043                    variable_name: "variable".to_string(),
13044                    value_name: "value".to_string(),
13045                })
13046                .map_err(|e| e.to_string())?;
13047                planned(s)
13048            }),
13049            ("a filter after the melt", |s| {
13050                s.melt(&MeltSpec {
13051                    index: vec!["id".to_string()],
13052                    value_columns: vec!["val".to_string()],
13053                    variable_name: "variable".to_string(),
13054                    value_name: "value".to_string(),
13055                })
13056                .map_err(|e| e.to_string())?;
13057                s.filter(vec![filter("val", FilterOperator::Gt, "0")]);
13058                planned(s)
13059            }),
13060            ("a sort after the melt", |s| {
13061                s.melt(&MeltSpec {
13062                    index: vec!["id".to_string()],
13063                    value_columns: vec!["val".to_string()],
13064                    variable_name: "variable".to_string(),
13065                    value_name: "value".to_string(),
13066                })
13067                .map_err(|e| e.to_string())?;
13068                s.sort_by(vec!["key".to_string()], vec![false]);
13069                planned(s)
13070            }),
13071            ("the layout after the sort", |s| {
13072                s.query("select id, val".to_string());
13073                s.sort_by(vec!["val".to_string()], vec![false]);
13074                s.set_column_order(vec!["key".to_string()]);
13075                planned(s)
13076            }),
13077            ("a reset, then a query", |s| {
13078                s.reset();
13079                s.query("select id where nope > 1".to_string());
13080                planned(s)
13081            }),
13082        ];
13083        for prior in [busy_view as fn() -> DataTableState, melted_view] {
13084            for (step, steps) in cases {
13085                let mut state = prior();
13086                let before = state.snapshot();
13087                let failed = state.try_transition(steps);
13088                assert!(failed.is_err(), "{step}: the steps fail");
13089                assert_eq!(state.snapshot(), before, "{step}: the view is put back");
13090                assert!(
13091                    state.prepare_async_collect(None).is_none(),
13092                    "{step}: the rows on hand serve it, so nothing is read again"
13093                );
13094            }
13095        }
13096    }
13097
13098    /// Steps that all plan read nothing; when the new view's rows then fail, the
13099    /// checkpoint they returned puts back the view before them, rows and all.
13100    #[test]
13101    fn a_planned_view_whose_rows_fail_rolls_back_without_reading() {
13102        for prior in [busy_view as fn() -> DataTableState, melted_view] {
13103            let mut state = prior();
13104            let before = state.snapshot();
13105            let ((), saved) = state
13106                .try_transition(|s| {
13107                    s.query("select id, val where id > 4".to_string());
13108                    s.filter(vec![filter("val", FilterOperator::Lt, "150")]);
13109                    s.sort_by(vec!["id".to_string()], vec![false]);
13110                    planned(s)
13111                })
13112                .unwrap();
13113            assert!(!state.is_num_rows_valid(), "the view's count is not read");
13114            assert!(
13115                state.prepare_async_collect(None).is_some(),
13116                "nor its rows: both are left to the background read"
13117            );
13118
13119            state.roll_back(saved);
13120            assert_eq!(state.snapshot(), before);
13121        }
13122    }
13123
13124    /// The view before a query had no count yet; the count lands while the query's
13125    /// rows are read and they fail. The count comes back with the view, as its own.
13126    #[test]
13127    fn a_count_that_lands_meanwhile_comes_back_with_its_view() {
13128        let mut state = busy_view();
13129        state.invalidate_num_rows();
13130        let counting = state.len_generation();
13131        let ((), mut saved) = state
13132            .try_transition(|s| {
13133                s.query("select id".to_string());
13134                planned(s)
13135            })
13136            .unwrap();
13137
13138        assert!(
13139            !state.count_landed(counting, 7, None),
13140            "not a count of the frame on screen"
13141        );
13142        assert!(
13143            !saved.count_landed(state.len_generation(), 99, None),
13144            "nor is the query's count the view's"
13145        );
13146        assert!(saved.count_landed(counting, 7, None));
13147
13148        state.roll_back(saved);
13149        assert_eq!(state.len_generation(), counting);
13150        assert_eq!(state.num_rows_if_valid(), Some(7));
13151    }
13152
13153    /// A checkpoint over data that has since been replaced does not put its frames
13154    /// over the new data; the view returns to the data as loaded instead, with no
13155    /// rows on hand, for the next background read.
13156    #[test]
13157    fn a_checkpoint_over_replaced_data_returns_to_the_data() {
13158        let mut state = busy_view();
13159        let saved = state.rollback_point();
13160        let wider = df!(
13161            "id" => &[1i64, 2],
13162            "key" => &["a", "b"],
13163            "val" => &[10i64, 20],
13164            "more" => &[true, false],
13165        )
13166        .unwrap()
13167        .lazy();
13168        let schema = wider.clone().collect_schema().unwrap();
13169        state.replace_root(wider.clone(), schema.clone());
13170
13171        state.roll_back(saved);
13172        assert_eq!(state.schema(), &schema);
13173        assert!(state.get_active_query().is_empty());
13174        assert!(state.get_filters().is_empty());
13175        assert!(state.get_sort_columns().is_empty());
13176        assert_eq!(
13177            state.lf().clone().collect().unwrap(),
13178            wider.collect().unwrap()
13179        );
13180        assert!(
13181            state.prepare_async_collect(None).is_some(),
13182            "its rows are read in the background"
13183        );
13184    }
13185}
13186
13187#[cfg(test)]
13188mod tests {
13189    use super::*;
13190
13191    thread_local! {
13192        /// Calls of `compact_rows` on this thread: a test's own thread is the UI thread.
13193        pub(super) static COMPACTIONS: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
13194    }
13195
13196    fn compactions() -> usize {
13197        COMPACTIONS.with(std::cell::Cell::get)
13198    }
13199
13200    /// Rows a worker read for `[buffer_start, buffer_end)`, before they are fit.
13201    struct Fill {
13202        df: DataFrame,
13203        buffer_start: usize,
13204        buffer_end: usize,
13205        num_rows: usize,
13206        count_known: bool,
13207    }
13208
13209    impl DataTableState {
13210        /// Hand `fill` over as the collect worker does: fit as planned from what is
13211        /// held and shown now, then installed.
13212        fn land(&mut self, fill: Fill) {
13213            let plan = self.fill_plan(
13214                fill.buffer_start,
13215                fill.buffer_end,
13216                fill.num_rows,
13217                fill.count_known,
13218            );
13219            self.apply_async_collect(plan.fit(fill.df));
13220        }
13221    }
13222
13223    fn names(df: &DataFrame) -> Vec<String> {
13224        df.get_column_names()
13225            .iter()
13226            .map(|name| name.to_string())
13227            .collect()
13228    }
13229
13230    fn mixed_strings() -> LazyFrame {
13231        df!(
13232            "id" => &[1i64, 2, 3, 4],
13233            "amount" => &[" 10 ", "20", "", " 40"],
13234            "day" => &["2024-01-01", "2024-01-02", " ", "2024-01-04"],
13235            "at" => &["2024-01-01T10:00:00Z", "2024-01-01T11:00:00Z", "", "2024-01-01T12:00:00Z"],
13236            "score" => &[0.5f64, 1.5, 2.5, 3.5],
13237            "word" => &["a", " b", "c ", ""],
13238            "empty" => &[None::<&str>, None, None, None],
13239        )
13240        .unwrap()
13241        .lazy()
13242    }
13243
13244    /// String inference reads only the columns it types, and reads them exactly as
13245    /// the whole-frame sample it replaced did: trimmed, blanks null, same rows.
13246    #[test]
13247    fn the_inference_sample_holds_only_its_targets() {
13248        let targets: Vec<String> = ["amount", "day", "at", "word", "empty"]
13249            .map(String::from)
13250            .to_vec();
13251        let sample = DataTableState::string_inference_sample(mixed_strings(), &targets, 3).unwrap();
13252        assert_eq!(names(&sample), ["amount", "day", "at", "word", "empty"]);
13253        assert_eq!(sample.height(), 3);
13254
13255        // The sample as it was taken before: every column, the targets normalized.
13256        let blank = lit(PlSmallStr::from_static(""));
13257        let wide = mixed_strings()
13258            .limit(3)
13259            .with_columns(
13260                targets
13261                    .iter()
13262                    .map(|c| {
13263                        col(c.as_str())
13264                            .str()
13265                            .strip_chars(lit(PlSmallStr::from_static(" \t\n\r")))
13266                    })
13267                    .collect::<Vec<_>>(),
13268            )
13269            .with_columns(
13270                targets
13271                    .iter()
13272                    .map(|c| {
13273                        when(col(c.as_str()).eq(blank.clone()))
13274                            .then(Null {}.lit())
13275                            .otherwise(col(c.as_str()))
13276                            .alias(c.as_str())
13277                    })
13278                    .collect::<Vec<_>>(),
13279            )
13280            .collect()
13281            .unwrap();
13282        assert_eq!(wide.width(), 7);
13283        assert!(sample.equals_missing(&wide.select(targets.iter().map(String::as_str)).unwrap()));
13284
13285        let one =
13286            DataTableState::string_inference_sample(mixed_strings(), &["day".to_string()], 1_000)
13287                .unwrap();
13288        assert_eq!(names(&one), ["day"]);
13289        assert_eq!(one.height(), 4);
13290    }
13291
13292    /// The frame string inference returns keeps every column in its place, and types
13293    /// the targets as before: numbers, dates, timestamps, text left as text, an
13294    /// all-null column left alone.
13295    #[test]
13296    fn string_inference_keeps_the_whole_frame() {
13297        let utc = DataType::Datetime(TimeUnit::Microseconds, Some(TimeZone::UTC));
13298        let typed = |target: &ParseStringsTarget, types: StringTypes| {
13299            DataTableState::type_string_columns(
13300                mixed_strings(),
13301                target,
13302                1_000,
13303                types,
13304                &mut Vec::new(),
13305                &[],
13306                &mut Vec::new(),
13307            )
13308            .unwrap()
13309            .collect()
13310            .unwrap()
13311        };
13312        let all = StringTypes {
13313            dates: true,
13314            numbers: true,
13315        };
13316
13317        let df = typed(&ParseStringsTarget::All, all);
13318        let schema = df.schema();
13319        let order = names(&df);
13320        assert_eq!(
13321            order,
13322            ["id", "amount", "day", "at", "score", "word", "empty"]
13323        );
13324        assert_eq!(schema.get("id"), Some(&DataType::Int64));
13325        assert_eq!(schema.get("amount"), Some(&DataType::Int64));
13326        assert_eq!(schema.get("day"), Some(&DataType::Date));
13327        assert_eq!(schema.get("at"), Some(&utc));
13328        assert_eq!(schema.get("score"), Some(&DataType::Float64));
13329        assert_eq!(schema.get("word"), Some(&DataType::String));
13330        assert_eq!(schema.get("empty"), Some(&DataType::String));
13331        let amount: Vec<Option<i64>> = df.column("amount").unwrap().i64().unwrap().iter().collect();
13332        assert_eq!(amount, [Some(10), Some(20), None, Some(40)]);
13333        assert_eq!(df.column("day").unwrap().null_count(), 1);
13334        assert_eq!(df.column("at").unwrap().null_count(), 1);
13335        let word: Vec<Option<&str>> = df.column("word").unwrap().str().unwrap().iter().collect();
13336        assert_eq!(word, [Some("a"), Some("b"), Some("c"), Some("")]);
13337        let score: Vec<Option<f64>> = df.column("score").unwrap().f64().unwrap().iter().collect();
13338        assert_eq!(score, [Some(0.5), Some(1.5), Some(2.5), Some(3.5)]);
13339
13340        // Named columns: only those that are text are typed; the rest are as read.
13341        let some = typed(
13342            &ParseStringsTarget::Columns(vec!["day".into(), "id".into(), "missing".into()]),
13343            all,
13344        );
13345        assert_eq!(names(&some), order);
13346        assert_eq!(some.schema().get("day"), Some(&DataType::Date));
13347        assert_eq!(some.schema().get("amount"), Some(&DataType::String));
13348        assert_eq!(
13349            some.column("amount").unwrap().str().unwrap().get(0),
13350            Some(" 10 ")
13351        );
13352
13353        // Nothing to type: the frame comes back as it was.
13354        let none = typed(&ParseStringsTarget::Columns(vec!["id".into()]), all);
13355        assert!(none.equals_missing(&mixed_strings().collect().unwrap()));
13356
13357        // JSON: dates only. Numbers in strings stay text, untrimmed.
13358        let json = DataTableState::apply_parse_dates_to_json_lazyframe(
13359            mixed_strings(),
13360            &crate::OpenOptions::default(),
13361            &mut Vec::new(),
13362        )
13363        .unwrap()
13364        .collect()
13365        .unwrap();
13366        assert_eq!(names(&json), order);
13367        assert_eq!(json.schema().get("day"), Some(&DataType::Date));
13368        assert_eq!(json.schema().get("at"), Some(&utc));
13369        assert_eq!(json.schema().get("amount"), Some(&DataType::String));
13370        assert_eq!(
13371            json.column("word").unwrap().str().unwrap().get(1),
13372            Some(" b")
13373        );
13374    }
13375
13376    /// Data Quality's evidence rows read the downloaded file the dataset scans, so
13377    /// they hold it too: a view captured there is refused like the dataset's own.
13378    #[test]
13379    fn evidence_rows_hold_the_download_they_scan() {
13380        let dir = tempfile::tempdir().unwrap();
13381        let mut file = crate::download::TempDownload::create(Some(dir.path()), Some("csv"))
13382            .expect("a temp file");
13383        std::io::Write::write_all(&mut file, b"id\n1\n2\n").unwrap();
13384        let download = crate::download::TempDownload::keep(file);
13385        let state = DataTableState::from_csv(download.path(), &Default::default())
13386            .unwrap()
13387            .with_open(OpenFacts {
13388                download: Some(download.clone()),
13389                ..Default::default()
13390            });
13391        let path = download.path().to_path_buf();
13392        drop(download);
13393
13394        let view = state
13395            .quality_evidence_view(&crate::data_quality::QualityScope::WholeSource, lit(true))
13396            .unwrap();
13397        assert!(view.scans_a_download());
13398        drop(state);
13399        assert!(path.exists(), "the view still scans it");
13400        assert_eq!(collect_lazy(view.lf.clone(), false).unwrap().height(), 2);
13401        drop(view);
13402        assert!(!path.exists());
13403    }
13404
13405    /// The same for a compressed CSV: the evidence rows scan the decompressed copy,
13406    /// so they hold it, and it goes with the last state that scans it.
13407    #[test]
13408    fn evidence_rows_hold_the_decompressed_file_they_scan() {
13409        let source = tempfile::tempdir().unwrap();
13410        let scratch = tempfile::tempdir().unwrap();
13411        let gz = source.path().join("rows.csv.gz");
13412        let mut encoder = flate2::write::GzEncoder::new(
13413            std::fs::File::create(&gz).unwrap(),
13414            flate2::Compression::default(),
13415        );
13416        std::io::Write::write_all(&mut encoder, b"id\n1\n2\n").unwrap();
13417        encoder.finish().unwrap();
13418        let options = crate::OpenOptions {
13419            temp_dir: Some(scratch.path().to_path_buf()),
13420            ..Default::default()
13421        };
13422        let state = DataTableState::from_csv(&gz, &options).unwrap();
13423        let path = state
13424            .decompress_temp_file
13425            .as_ref()
13426            .expect("decompressed to a temp file")
13427            .path()
13428            .to_path_buf();
13429        assert!(path.starts_with(scratch.path()));
13430
13431        let view = state
13432            .quality_evidence_view(&crate::data_quality::QualityScope::WholeSource, lit(true))
13433            .unwrap();
13434        assert!(view.scans_a_temp_file());
13435        drop(state);
13436        assert!(path.exists(), "the view still scans it");
13437        assert_eq!(collect_lazy(view.lf.clone(), false).unwrap().height(), 2);
13438        drop(view);
13439        assert!(!path.exists());
13440    }
13441
13442    /// Scrolling sideways must not count the rows.
13443    ///
13444    /// A staged open shows a screen before the count is known, and the count on a
13445    /// cloud hive is a metadata read per object. `scroll_right` runs on the thread
13446    /// that draws and reads keys — `App::handle`'s `RIGHT_KEYS` arm calls it
13447    /// directly — so a count taken there is a freeze no keystroke can interrupt.
13448    /// The busy-key classifier admits Left/Right on the stated premise that column
13449    /// scroll never collects, which is what this holds.
13450    #[test]
13451    fn scrolling_sideways_does_not_count_the_rows() {
13452        use polars::prelude::*;
13453
13454        let frame = || {
13455            df!(
13456                "a" => &[1i64, 2, 3],
13457                "b" => &[4i64, 5, 6],
13458                "c" => &[7i64, 8, 9],
13459            )
13460            .unwrap()
13461            .lazy()
13462        };
13463        let mut lf = frame();
13464        let schema = std::sync::Arc::new((*lf.collect_schema().unwrap()).clone());
13465        let mut state = DataTableState::from_schema_and_lazyframe(
13466            schema,
13467            frame(),
13468            &crate::OpenOptions::default(),
13469            None,
13470        )
13471        .unwrap();
13472        assert_eq!(
13473            state.num_rows_if_valid(),
13474            None,
13475            "a staged open starts without a count"
13476        );
13477
13478        state.scroll_right();
13479        assert_eq!(
13480            state.num_rows_if_valid(),
13481            None,
13482            "scrolling right must leave the count to the background pass"
13483        );
13484        state.scroll_left();
13485        assert_eq!(
13486            state.num_rows_if_valid(),
13487            None,
13488            "and so must scrolling back"
13489        );
13490    }
13491
13492    /// And it must still scroll. Not counting is only the right answer if the new
13493    /// columns actually reach the screen, so this holds the other half: with a
13494    /// buffer in hand, Right moves the window the display is sliced from.
13495    #[test]
13496    fn scrolling_sideways_still_moves_the_columns() {
13497        use polars::prelude::*;
13498
13499        let frame = || {
13500            df!(
13501                "a" => &[1i64, 2, 3],
13502                "b" => &[4i64, 5, 6],
13503                "c" => &[7i64, 8, 9],
13504            )
13505            .unwrap()
13506            .lazy()
13507        };
13508        let mut lf = frame();
13509        let schema = std::sync::Arc::new((*lf.collect_schema().unwrap()).clone());
13510        let mut state = DataTableState::from_schema_and_lazyframe(
13511            schema,
13512            frame(),
13513            &crate::OpenOptions::default(),
13514            None,
13515        )
13516        .unwrap();
13517        // As the app has it once a page has landed: a count and a buffer.
13518        state.set_num_rows(3);
13519        state.visible_rows = 3;
13520        state.visible_termcols = 1;
13521        state.collect();
13522        let first = state
13523            .display_df()
13524            .map(|df| {
13525                df.get_column_names()
13526                    .iter()
13527                    .map(|n| n.to_string())
13528                    .collect::<Vec<_>>()
13529                    .join(",")
13530            })
13531            .unwrap_or_default();
13532
13533        state.scroll_right();
13534        let second = state
13535            .display_df()
13536            .map(|df| {
13537                df.get_column_names()
13538                    .iter()
13539                    .map(|n| n.to_string())
13540                    .collect::<Vec<_>>()
13541                    .join(",")
13542            })
13543            .unwrap_or_default();
13544        assert_ne!(first, second, "Right should show a different column window");
13545        assert!(!second.is_empty(), "and should show something");
13546
13547        state.scroll_left();
13548        let back = state
13549            .display_df()
13550            .map(|df| {
13551                df.get_column_names()
13552                    .iter()
13553                    .map(|n| n.to_string())
13554                    .collect::<Vec<_>>()
13555                    .join(",")
13556            })
13557            .unwrap_or_default();
13558        assert_eq!(back, first, "Left should come back to where it started");
13559    }
13560
13561    use crate::filter_modal::{FilterOperator, FilterStatement, LogicalOperator};
13562    use crate::pivot_melt_modal::{MeltSpec, PivotAggregation, PivotSpec};
13563
13564    /// The join drops the rows it read through the frame it replaced.
13565    ///
13566    /// The buffer holds rows read at the old schema, through the old scan. Kept, the
13567    /// next paint draws them under the new column order — and `display_slice_df` offsets
13568    /// into them from `buffered_start_row`, so where the user is deep in a dataset it is
13569    /// the wrong rows under the right numbers.
13570    #[test]
13571    fn the_join_drops_the_rows_read_through_the_frame_it_replaced() {
13572        let narrow = || df!("id" => &[1i64, 2]).unwrap().lazy();
13573        let wider = || {
13574            df!("id" => &[1i64, 2], "oops" => &["a", "b"])
13575                .unwrap()
13576                .lazy()
13577        };
13578        let dataset_of = |lf: LazyFrame| {
13579            let mut lf = lf;
13580            let schema = Arc::new((*lf.collect_schema().unwrap()).clone());
13581            let footer = crate::schema_union::FileFooter {
13582                schema,
13583                row_group_rows: vec![2],
13584                file_bytes: 0,
13585                row_group_bytes: Vec::new(),
13586                column_bytes: Vec::new(),
13587            };
13588            crate::schema_union::union_sampled(1, &[0], &[Some(footer)])
13589        };
13590
13591        let mut state = DataTableState::from_schema_and_lazyframe(
13592            dataset_of(narrow()).schema.clone(),
13593            narrow(),
13594            &crate::OpenOptions::default(),
13595            None,
13596        )
13597        .unwrap();
13598        state.visible_rows = 2;
13599        state.collect();
13600        assert!(
13601            state.buffered_df.is_some(),
13602            "there are rows on hand, read at the old schema"
13603        );
13604
13605        assert!(
13606            state
13607                .join_dataset_schema(FootersFound {
13608                    estimate: None,
13609                    dataset: dataset_of(wider()),
13610                    lf: wider(),
13611                    file_rows: Vec::new(),
13612                    files: Vec::new(),
13613                    row_groups: Vec::new(),
13614                    remote: None,
13615                })
13616                .is_ok()
13617        );
13618
13619        assert!(
13620            state.buffered_df.is_none(),
13621            "and they are let go, rather than drawn under the columns that replaced them"
13622        );
13623        assert_eq!(
13624            (state.buffered_start_row, state.buffered_end_row),
13625            (0, 0),
13626            "with nothing left saying which rows they were"
13627        );
13628    }
13629
13630    /// The join throws away a width measured on the frame it replaced.
13631    ///
13632    /// `bytes_per_row` prefers what was observed over what the schema estimates, and the
13633    /// observation belongs to the frame that just went. A dataset that opened two
13634    /// columns wide and gained thirty would plan its first page after the join from the
13635    /// two-column width — against a bucket, a read many times the budget the user set.
13636    /// `install_base` clears it for the same reason.
13637    #[test]
13638    fn the_join_does_not_keep_a_width_measured_on_the_frame_it_replaced() {
13639        let narrow = || df!("id" => &[1i64, 2]).unwrap().lazy();
13640        let wide = || {
13641            df!("id" => &[1i64, 2], "a" => &["x", "y"], "b" => &["x", "y"])
13642                .unwrap()
13643                .lazy()
13644        };
13645        let dataset_of = |lf: LazyFrame| {
13646            let mut lf = lf;
13647            let schema = Arc::new((*lf.collect_schema().unwrap()).clone());
13648            let footer = crate::schema_union::FileFooter {
13649                schema,
13650                row_group_rows: vec![2],
13651                file_bytes: 0,
13652                row_group_bytes: Vec::new(),
13653                column_bytes: Vec::new(),
13654            };
13655            crate::schema_union::union_sampled(1, &[0], &[Some(footer)])
13656        };
13657
13658        let mut state = DataTableState::from_schema_and_lazyframe(
13659            dataset_of(narrow()).schema.clone(),
13660            narrow(),
13661            &crate::OpenOptions::default(),
13662            None,
13663        )
13664        .unwrap();
13665        state.observed_bytes_per_row = Some(8);
13666        let measured_narrow = state.bytes_per_row();
13667
13668        assert!(
13669            state
13670                .join_dataset_schema(FootersFound {
13671                    estimate: None,
13672                    dataset: dataset_of(wide()),
13673                    lf: wide(),
13674                    file_rows: Vec::new(),
13675                    files: Vec::new(),
13676                    row_groups: Vec::new(),
13677                    remote: None,
13678                })
13679                .is_ok(),
13680            "nothing is built on the scan here, so the columns go straight in"
13681        );
13682
13683        assert!(
13684            state.bytes_per_row() > measured_narrow,
13685            "a row of the widened dataset is not planned at the width of the old one: \
13686             {} vs {measured_narrow}",
13687            state.bytes_per_row()
13688        );
13689    }
13690
13691    /// A dataset stops saying a count is coming once one has arrived.
13692    ///
13693    /// While a pass is reading its footers the dataset declines to count itself, because
13694    /// that pass is bringing the count — and the control bar shows a spinner in place of
13695    /// a number it would otherwise print as fact. When the pass lands on a view built on
13696    /// the scan the columns have to wait, but the row groups describe the same files the
13697    /// frame is already reading, so the number need not: without this the user watches
13698    /// that spinner over a count the dataset is holding.
13699    #[test]
13700    fn a_count_that_has_arrived_is_not_held_back_with_the_columns() {
13701        let rows = || df!("id" => (0..100i64).collect::<Vec<_>>()).unwrap().lazy();
13702        let wider = || {
13703            df!("id" => (0..100i64).collect::<Vec<_>>(), "oops" => vec!["a"; 100])
13704                .unwrap()
13705                .lazy()
13706        };
13707        let dataset_of = |lf: LazyFrame| {
13708            let mut lf = lf;
13709            let schema = Arc::new((*lf.collect_schema().unwrap()).clone());
13710            let footer = crate::schema_union::FileFooter {
13711                schema,
13712                row_group_rows: vec![100],
13713                file_bytes: 0,
13714                row_group_bytes: Vec::new(),
13715                column_bytes: Vec::new(),
13716            };
13717            crate::schema_union::union_sampled(1, &[0], &[Some(footer)])
13718        };
13719
13720        let mut state = DataTableState::from_schema_and_lazyframe(
13721            dataset_of(rows()).schema.clone(),
13722            rows(),
13723            &crate::OpenOptions::default(),
13724            None,
13725        )
13726        .unwrap()
13727        .with_open(OpenFacts {
13728            remote_source: true,
13729            remote_files: Some(RemoteFiles {
13730                urls: Arc::new(vec!["one".to_string()]),
13731                scan: Arc::new(move |_u: &[String], _t: &[PlSmallStr]| Ok(rows())),
13732                count: Arc::new(|_| Ok(vec![vec![100]])),
13733                offsets: None,
13734            }),
13735            footers_pending: Some(Arc::new(|_| None)),
13736            ..Default::default()
13737        });
13738        assert!(
13739            state.counts_itself_later(),
13740            "the pass is bringing a count, so the dataset is not going to fetch one"
13741        );
13742
13743        // The user is in a query when it lands, so the columns cannot go in.
13744        state.active_query = "select doubled: id * 2".to_string();
13745        let held = state.join_dataset_schema(FootersFound {
13746            estimate: None,
13747            dataset: dataset_of(wider()),
13748            lf: wider(),
13749            file_rows: vec![100],
13750            files: vec!["one".to_string()],
13751            row_groups: vec![vec![100]],
13752            remote: None,
13753        });
13754        assert!(held.is_err(), "the columns wait for the query to be let go");
13755
13756        // What the footers said about the files stays, so letting the query go gets the
13757        // total back rather than sending anyone to fetch it again.
13758        state.active_query.clear();
13759        state.restore_footer_count();
13760        assert_eq!(
13761            state.num_rows_if_valid(),
13762            Some(100),
13763            "the count is there, without a read to find it"
13764        );
13765        assert!(
13766            !state.counts_itself_later(),
13767            "so the dataset stops saying one is coming, and the bar prints a number \
13768             rather than a spinner over one it is holding"
13769        );
13770    }
13771
13772    /// A column the second pass could not see goes, rather than breaking every read.
13773    ///
13774    /// Columns only join — but the pass reads footers again, and a footer that parsed
13775    /// at the open can fail the second time: a transient error, or the object replaced
13776    /// between the two reads. If that was the only file with a column, the name is left
13777    /// naming nothing. Every page projects `column_order`, so a name the new schema
13778    /// does not have is not a blank column on screen, it is a scan that will not plan.
13779    #[test]
13780    fn a_column_the_second_pass_could_not_see_leaves_the_order() {
13781        let at_open = || {
13782            df!("id" => &[1i64], "only_the_first_file_had_this" => &["x"])
13783                .unwrap()
13784                .lazy()
13785        };
13786        let second_time = || df!("id" => &[1i64], "oops" => &["a"]).unwrap().lazy();
13787        let dataset_of = |lf: LazyFrame| {
13788            let mut lf = lf;
13789            let schema = Arc::new((*lf.collect_schema().unwrap()).clone());
13790            let footer = crate::schema_union::FileFooter {
13791                schema,
13792                row_group_rows: vec![1],
13793                file_bytes: 0,
13794                row_group_bytes: Vec::new(),
13795                column_bytes: Vec::new(),
13796            };
13797            crate::schema_union::union_sampled(1, &[0], &[Some(footer)])
13798        };
13799
13800        let mut state = DataTableState::from_schema_and_lazyframe(
13801            dataset_of(at_open()).schema.clone(),
13802            at_open(),
13803            &crate::OpenOptions::default(),
13804            None,
13805        )
13806        .unwrap();
13807        assert!(
13808            state
13809                .get_column_order()
13810                .iter()
13811                .any(|c| c == "only_the_first_file_had_this"),
13812            "the dataset opened with it"
13813        );
13814
13815        assert!(
13816            state
13817                .join_dataset_schema(FootersFound {
13818                    estimate: None,
13819                    dataset: dataset_of(second_time()),
13820                    lf: second_time(),
13821                    file_rows: Vec::new(),
13822                    files: Vec::new(),
13823                    row_groups: Vec::new(),
13824                    remote: None,
13825                })
13826                .is_ok()
13827        );
13828
13829        assert_eq!(
13830            state.get_column_order(),
13831            ["id", "oops"],
13832            "the name the pass can no longer account for is not left naming nothing"
13833        );
13834    }
13835
13836    /// Every way of building on the scan holds the arriving columns off, not just one.
13837    ///
13838    /// Each of these replaces the frame's root with a result of its own, whose columns
13839    /// are not the dataset's. The doc on `scan_is_the_root` says so of all of them; the
13840    /// query is the one the app-level test exercises, so this is the rest.
13841    #[test]
13842    fn anything_built_on_the_scan_holds_the_arriving_columns_off() {
13843        // A string column because a fuzzy search needs one to search.
13844        let narrow = || {
13845            df!("id" => &[1i64, 2], "v" => &[10i64, 20], "name" => &["one", "two"])
13846                .unwrap()
13847                .lazy()
13848        };
13849        let wider = || {
13850            df!(
13851                "id" => &[1i64, 2],
13852                "v" => &[10i64, 20],
13853                "name" => &["one", "two"],
13854                "oops" => &["a", "b"],
13855            )
13856            .unwrap()
13857            .lazy()
13858        };
13859        let found = || {
13860            let mut lf = wider();
13861            let schema = Arc::new((*lf.collect_schema().unwrap()).clone());
13862            let footer = crate::schema_union::FileFooter {
13863                schema,
13864                row_group_rows: vec![2],
13865                file_bytes: 0,
13866                row_group_bytes: Vec::new(),
13867                column_bytes: Vec::new(),
13868            };
13869            FootersFound {
13870                estimate: None,
13871                dataset: crate::schema_union::union_sampled(1, &[0], &[Some(footer)]),
13872                lf: wider(),
13873                file_rows: Vec::new(),
13874                files: Vec::new(),
13875                row_groups: Vec::new(),
13876                remote: None,
13877            }
13878        };
13879        let fresh = || {
13880            let mut lf = narrow();
13881            let schema = Arc::new((*lf.collect_schema().unwrap()).clone());
13882            DataTableState::from_schema_and_lazyframe(
13883                schema,
13884                narrow(),
13885                &crate::OpenOptions::default(),
13886                None,
13887            )
13888            .unwrap()
13889        };
13890
13891        // A filter and a sort are not built on the scan, they are the scan with
13892        // something done to it, and `apply_transformations` puts them back over
13893        // whatever the root becomes. Those the columns may join under.
13894        let mut sorted = fresh();
13895        sorted.sort(vec!["id".to_string()], true);
13896        assert!(
13897            sorted.join_dataset_schema(found()).is_ok(),
13898            "a sort is rebuilt over the wider scan, so the columns go in under it"
13899        );
13900
13901        let mut queried = fresh();
13902        queried.query("select doubled: v * 2".to_string());
13903        assert!(
13904            queried.join_dataset_schema(found()).is_err(),
13905            "a query's columns are its own"
13906        );
13907
13908        #[cfg(feature = "sql")]
13909        {
13910            let mut sql = fresh();
13911            sql.sql_query("SELECT id FROM df".to_string());
13912            assert!(sql.error.is_none(), "the statement runs: {:?}", sql.error);
13913            assert!(
13914                sql.join_dataset_schema(found()).is_err(),
13915                "and a SQL statement's are too"
13916            );
13917        }
13918
13919        let mut fuzzy = fresh();
13920        fuzzy.fuzzy_search("10".to_string());
13921        assert!(
13922            fuzzy.join_dataset_schema(found()).is_err(),
13923            "and what a fuzzy search matched is a result, not the dataset"
13924        );
13925
13926        let mut melted = fresh();
13927        melted
13928            .melt(&MeltSpec {
13929                index: vec!["id".to_string()],
13930                value_columns: vec!["v".to_string()],
13931                variable_name: "variable".to_string(),
13932                value_name: "value".to_string(),
13933            })
13934            .expect("the melt runs");
13935        assert!(
13936            melted.join_dataset_schema(found()).is_err(),
13937            "a melt's rows are not the dataset's rows"
13938        );
13939
13940        let mut pivoted = fresh();
13941        pivoted
13942            .pivot(&PivotSpec {
13943                index: vec!["id".to_string()],
13944                pivot_column: "name".to_string(),
13945                value_column: "v".to_string(),
13946                aggregation: PivotAggregation::First,
13947                sort_columns: None,
13948            })
13949            .expect("the pivot runs");
13950        assert!(
13951            pivoted.join_dataset_schema(found()).is_err(),
13952            "and a pivot's columns are made from the data, not read from it"
13953        );
13954
13955        let mut drilled = fresh();
13956        // The field rather than the drill itself, which needs a grouped frame to drill
13957        // into: what is being asked here is whether the clause is consulted.
13958        drilled.drilled_down_group_index = Some(0);
13959        assert!(
13960            drilled.join_dataset_schema(found()).is_err(),
13961            "and a drill-down is showing one group of it, not it"
13962        );
13963    }
13964
13965    /// Three files of two rows each, and every way they can conflict.
13966    ///
13967    /// Each case names the files that hold the column in a type it is not read in, and
13968    /// the rows those files own. The last file is the one that has no start after it,
13969    /// so a run ending there is the case an implementation is most likely to get wrong
13970    /// — and the one that two files of the same length hide, since dropping to the end
13971    /// of the dataset and dropping to the next file's start agree there.
13972    #[test]
13973    fn conflicting_files_become_the_runs_of_rows_they_own() {
13974        let starts = [0, 2, 4];
13975        let runs = |conflicts: [bool; 3]| conflicting_row_runs(&starts, 6, &conflicts);
13976
13977        assert_eq!(runs([false, false, false]), vec![], "nothing conflicts");
13978        assert_eq!(runs([true, false, false]), vec![(0, 2)], "the first file");
13979        assert_eq!(
13980            runs([false, true, false]),
13981            vec![(2, 4)],
13982            "a file in the middle ends where the next one begins, not at the end of \
13983             the dataset"
13984        );
13985        assert_eq!(
13986            runs([false, false, true]),
13987            vec![(4, 6)],
13988            "and the last one ends at the end of the dataset"
13989        );
13990        assert_eq!(
13991            runs([true, true, false]),
13992            vec![(0, 4)],
13993            "files that touch are one run"
13994        );
13995        assert_eq!(runs([true, true, true]), vec![(0, 6)], "as are all of them");
13996        assert_eq!(
13997            runs([true, false, true]),
13998            vec![(0, 2), (4, 6)],
13999            "files that do not touch are not"
14000        );
14001    }
14002
14003    /// A file of no rows owns no rows, so it neither makes a run nor breaks one.
14004    ///
14005    /// Zero-row Parquet files are written by any pipeline that partitions on a key
14006    /// with no data for some value, so this is a shape datui meets rather than one it
14007    /// has to imagine.
14008    #[test]
14009    fn a_file_of_no_rows_neither_makes_a_run_nor_splits_one() {
14010        // Files of 2, 0 and 2 rows: the middle one begins and ends at row 2.
14011        let starts = [0, 2, 2];
14012        assert_eq!(
14013            conflicting_row_runs(&starts, 4, &[false, true, false]),
14014            vec![],
14015            "a conflicting file with no rows keeps no row out"
14016        );
14017        assert_eq!(
14018            conflicting_row_runs(&starts, 4, &[true, false, true]),
14019            vec![(0, 4)],
14020            "and an empty file between two that conflict does not part them"
14021        );
14022        assert_eq!(
14023            conflicting_row_runs(&starts, 4, &[true, true, true]),
14024            vec![(0, 4)],
14025            "however it is flagged itself"
14026        );
14027    }
14028
14029    fn file_schema(
14030        columns: &[(&str, polars::prelude::DataType)],
14031        rows: usize,
14032    ) -> Option<crate::schema_union::FileFooter> {
14033        let mut schema = polars::prelude::Schema::with_capacity(columns.len());
14034        for (name, dtype) in columns {
14035            schema.with_column((*name).into(), dtype.clone());
14036        }
14037        Some(crate::schema_union::FileFooter {
14038            schema: Arc::new(schema),
14039            row_group_rows: vec![rows],
14040            file_bytes: 0,
14041            row_group_bytes: Vec::new(),
14042            column_bytes: Vec::new(),
14043        })
14044    }
14045
14046    fn create_test_lf() -> LazyFrame {
14047        df! (
14048            "a" => &[1, 2, 3],
14049            "b" => &["x", "y", "z"]
14050        )
14051        .unwrap()
14052        .lazy()
14053    }
14054
14055    fn create_large_test_lf() -> LazyFrame {
14056        df! (
14057            "a" => (0..100).collect::<Vec<i32>>(),
14058            "b" => (0..100).map(|i| format!("text_{}", i)).collect::<Vec<String>>(),
14059            "c" => (0..100).map(|i| i % 3).collect::<Vec<i32>>(),
14060            "d" => (0..100).map(|i| i % 5).collect::<Vec<i32>>()
14061        )
14062        .unwrap()
14063        .lazy()
14064    }
14065
14066    /// Pump Open(path, opts) and subsequent events until no event is returned.
14067    /// Returns true if a Crash event was seen.
14068    fn pump_open_until_done(
14069        app: &mut crate::App,
14070        rx: &std::sync::mpsc::Receiver<crate::AppEvent>,
14071        path: std::path::PathBuf,
14072        opts: crate::OpenOptions,
14073    ) -> bool {
14074        use crate::AppEvent;
14075        let mut next: Option<AppEvent> = Some(AppEvent::Open(vec![path], opts));
14076        let mut saw_crash = false;
14077        // Done once nothing is chained, queued or still owed; the deadline is only a
14078        // hang guard.
14079        let deadline = std::time::Instant::now() + std::time::Duration::from_secs(300);
14080        loop {
14081            match next.take() {
14082                Some(ev) => {
14083                    if matches!(ev, AppEvent::Crash(_)) {
14084                        saw_crash = true;
14085                        break;
14086                    }
14087                    next = app.event(&ev);
14088                }
14089                _ => match rx.try_recv() {
14090                    Ok(ev) => next = Some(ev),
14091                    Err(_) if app.count_waits_for_a_frame() => next = Some(AppEvent::FramePainted),
14092                    Err(_) if !crate::tests::work_pending(app) => break,
14093                    Err(_) => {
14094                        assert!(
14095                            std::time::Instant::now() < deadline,
14096                            "background work never reported back"
14097                        );
14098                        next = rx.recv_timeout(std::time::Duration::from_millis(50)).ok();
14099                    }
14100                },
14101            }
14102        }
14103        saw_crash
14104    }
14105
14106    /// CSV with 100 int-like rows then "N/A" then more ints. With infer_schema_length=100, Polars
14107    /// infers Int from the first 100 rows; the parse error surfaces from the async collect as a
14108    /// collect failure, which the app shows in the error modal rather than crashing.
14109    #[test]
14110    fn test_infer_schema_length_csv_short_inference_shows_error_modal() {
14111        use std::sync::mpsc;
14112
14113        let path = crate::tests::sample_data_dir().join("infer_schema_length_data.csv");
14114        let opts = crate::OpenOptions {
14115            infer_schema_length: Some(100),
14116            ..Default::default()
14117        };
14118
14119        let (tx, rx) = mpsc::channel();
14120        let mut app = crate::App::new(tx, crate::tests::test_runtime());
14121
14122        assert!(
14123            !pump_open_until_done(&mut app, &rx, path, opts),
14124            "load should not crash; parse failure should be surfaced via error modal"
14125        );
14126        assert!(
14127            app.error_modal.active,
14128            "error modal should be shown after parse failure"
14129        );
14130        assert!(!app.busy, "busy flag should be cleared after error");
14131    }
14132
14133    #[test]
14134    fn test_infer_schema_length_csv_succeeds_with_longer_inference() {
14135        use std::sync::mpsc;
14136
14137        let path = crate::tests::sample_data_dir().join("infer_schema_length_data.csv");
14138        let opts = crate::OpenOptions {
14139            infer_schema_length: Some(101),
14140            ..Default::default()
14141        };
14142
14143        let (tx, rx) = mpsc::channel();
14144        let mut app = crate::App::new(tx, crate::tests::test_runtime());
14145
14146        assert!(
14147            !pump_open_until_done(&mut app, &rx, path, opts),
14148            "load with infer_schema_length=101 should not crash"
14149        );
14150        let state = app.data_table_state.as_ref().unwrap();
14151        assert_eq!(state.schema.len(), 1);
14152        assert!(state.schema.contains("column"));
14153        assert_eq!(state.num_rows, 201);
14154    }
14155
14156    #[test]
14157    fn test_infer_schema_length_csv_succeeds_with_default() {
14158        use std::sync::mpsc;
14159
14160        let path = crate::tests::sample_data_dir().join("infer_schema_length_data.csv");
14161        let opts = crate::OpenOptions {
14162            infer_schema_length: Some(1000),
14163            ..Default::default()
14164        };
14165
14166        let (tx, rx) = mpsc::channel();
14167        let mut app = crate::App::new(tx, crate::tests::test_runtime());
14168
14169        assert!(
14170            !pump_open_until_done(&mut app, &rx, path, opts),
14171            "load with infer_schema_length=1000 (default) should not crash"
14172        );
14173        let state = app.data_table_state.as_ref().unwrap();
14174        assert_eq!(state.schema.len(), 1);
14175        assert_eq!(state.num_rows, 201);
14176    }
14177
14178    #[test]
14179    fn test_from_csv() {
14180        // Ensure sample data is generated before running test
14181        // Test uncompressed CSV loading
14182        let path = crate::tests::sample_data_dir().join("3-sfd-header.csv");
14183        let state = DataTableState::from_csv(&path, &Default::default()).unwrap(); // Uses default buffer params from options
14184        assert_eq!(state.schema.len(), 6); // id, integer_col, float_col, string_col, boolean_col, date_col
14185    }
14186
14187    #[test]
14188    fn test_from_csv_gzipped() {
14189        // Ensure sample data is generated before running test
14190        // Test gzipped CSV loading
14191        let path = crate::tests::sample_data_dir().join("mixed_types.csv.gz");
14192        let state = DataTableState::from_csv(&path, &Default::default()).unwrap(); // Uses default buffer params from options
14193        assert_eq!(state.schema.len(), 6); // id, integer_col, float_col, string_col, boolean_col, date_col
14194    }
14195
14196    #[test]
14197    fn test_from_parquet() {
14198        // Ensure sample data is generated before running test
14199        let path = crate::tests::sample_data_dir().join("people.parquet");
14200        let state = DataTableState::from_parquet(&path, &OpenOptions::default()).unwrap();
14201        assert!(!state.schema.is_empty());
14202    }
14203
14204    #[test]
14205    fn test_from_ipc() {
14206        use polars::prelude::IpcWriter;
14207        use std::io::BufWriter;
14208        let mut df = df!(
14209            "x" => &[1_i32, 2, 3],
14210            "y" => &["a", "b", "c"]
14211        )
14212        .unwrap();
14213        let dir = tempfile::tempdir().unwrap();
14214        let path = dir.path().join("datui_test_ipc.arrow");
14215        let file = std::fs::File::create(&path).unwrap();
14216        let mut writer = BufWriter::new(file);
14217        IpcWriter::new(&mut writer).finish(&mut df).unwrap();
14218        drop(writer);
14219        let state = DataTableState::from_ipc(&path, &OpenOptions::default()).unwrap();
14220        assert_eq!(state.schema.len(), 2);
14221        assert!(state.schema.contains("x"));
14222        assert!(state.schema.contains("y"));
14223    }
14224
14225    #[test]
14226    fn test_from_avro() {
14227        use polars::io::avro::AvroWriter;
14228        use std::io::BufWriter;
14229        let mut df = df!(
14230            "id" => &[1_i32, 2, 3],
14231            "name" => &["alice", "bob", "carol"]
14232        )
14233        .unwrap();
14234        let dir = tempfile::tempdir().unwrap();
14235        let path = dir.path().join("datui_test_avro.avro");
14236        let file = std::fs::File::create(&path).unwrap();
14237        let mut writer = BufWriter::new(file);
14238        AvroWriter::new(&mut writer).finish(&mut df).unwrap();
14239        drop(writer);
14240        let state = DataTableState::from_avro(&path, &OpenOptions::default()).unwrap();
14241        assert_eq!(state.schema.len(), 2);
14242        assert!(state.schema.contains("id"));
14243        assert!(state.schema.contains("name"));
14244    }
14245
14246    #[test]
14247    fn test_from_orc() {
14248        use arrow::array::{Int64Array, StringArray};
14249        use arrow::datatypes::{DataType, Field, Schema};
14250        use arrow::record_batch::RecordBatch;
14251        use orc_rust::ArrowWriterBuilder;
14252        use std::io::BufWriter;
14253        use std::sync::Arc;
14254
14255        let schema = Arc::new(Schema::new(vec![
14256            Field::new("id", DataType::Int64, false),
14257            Field::new("name", DataType::Utf8, false),
14258        ]));
14259        let id_array = Arc::new(Int64Array::from(vec![1_i64, 2, 3]));
14260        let name_array = Arc::new(StringArray::from(vec!["a", "b", "c"]));
14261        let batch = RecordBatch::try_new(schema.clone(), vec![id_array, name_array]).unwrap();
14262
14263        let dir = tempfile::tempdir().unwrap();
14264        let path = dir.path().join("datui_test_orc.orc");
14265        let file = std::fs::File::create(&path).unwrap();
14266        let writer = BufWriter::new(file);
14267        let mut orc_writer = ArrowWriterBuilder::new(writer, schema).try_build().unwrap();
14268        orc_writer.write(&batch).unwrap();
14269        orc_writer.close().unwrap();
14270
14271        let state = DataTableState::from_orc(&path, &OpenOptions::default()).unwrap();
14272        assert_eq!(state.schema.len(), 2);
14273        assert!(state.schema.contains("id"));
14274        assert!(state.schema.contains("name"));
14275    }
14276
14277    /// `--delimiter` reaches the in-memory readers of every compression, and the
14278    /// one-row read that per-column null values are built from (#290), including
14279    /// for a file that is only readable once decompressed.
14280    #[test]
14281    fn test_delimiter_reaches_every_csv_reader() {
14282        use std::io::Write;
14283        let dir = tempfile::tempdir().unwrap();
14284        let body = b"id|name\n1|NA\n2|x\n";
14285        let bz = dir.path().join("t.csv.bz2");
14286        let mut enc =
14287            bzip2::write::BzEncoder::new(File::create(&bz).unwrap(), bzip2::Compression::best());
14288        enc.write_all(body).unwrap();
14289        enc.finish().unwrap();
14290        let xz = dir.path().join("t.csv.xz");
14291        let mut enc = xz2::write::XzEncoder::new(File::create(&xz).unwrap(), 6);
14292        enc.write_all(body).unwrap();
14293        enc.finish().unwrap();
14294        let plain = dir.path().join("t.csv");
14295        std::fs::write(&plain, body).unwrap();
14296
14297        let in_memory = OpenOptions {
14298            delimiter: Some(b'|'),
14299            decompress_in_memory: true,
14300            ..Default::default()
14301        };
14302        // A global and a per-column value together make the read ask for the schema;
14303        // the column's own value replaces the global one, so only `x` is null.
14304        let null_values = OpenOptions {
14305            delimiter: Some(b'|'),
14306            null_values: Some(vec!["NA".into(), "name=x".into()]),
14307            ..Default::default()
14308        };
14309        // The same in memory, where the column names have to come from the
14310        // decompressed bytes rather than the file on disk.
14311        let null_values_in_memory = OpenOptions {
14312            decompress_in_memory: true,
14313            ..null_values.clone()
14314        };
14315        for (what, path, opts) in [
14316            ("bzip2", &bz, &in_memory),
14317            ("xz", &xz, &in_memory),
14318            ("null values", &plain, &null_values),
14319            ("null values, bzip2", &bz, &null_values_in_memory),
14320            ("null values, xz", &xz, &null_values_in_memory),
14321        ] {
14322            let state = DataTableState::from_csv(path, opts).unwrap();
14323            let df = state.lf.clone().collect().unwrap();
14324            let names: Vec<_> = df
14325                .get_column_names()
14326                .iter()
14327                .map(|n| n.to_string())
14328                .collect();
14329            assert_eq!(names, ["id", "name"], "{what}");
14330            if opts.null_values.is_some() {
14331                assert_eq!(df.column("name").unwrap().null_count(), 1, "{what}");
14332            }
14333        }
14334    }
14335
14336    #[test]
14337    fn test_from_delimited_tsv_has_header() {
14338        let dir = tempfile::tempdir().unwrap();
14339        let path = dir.path().join("datui_test_tsv_header.tsv");
14340        let content = "a\tb\tc\td\n1\t2\t3\t4\n5\t6\t7\t8\n";
14341        std::fs::write(&path, content).unwrap();
14342        let opts = OpenOptions {
14343            has_header: Some(true),
14344            ..Default::default()
14345        };
14346        let mut state = DataTableState::from_delimited(&path, b'\t', &opts).unwrap();
14347        state.collect();
14348        assert_eq!(state.schema.len(), 4);
14349        assert!(state.schema.contains("a"));
14350        assert!(state.schema.contains("b"));
14351        assert!(state.schema.contains("c"));
14352        assert!(state.schema.contains("d"));
14353        assert_eq!(state.num_rows, 2);
14354    }
14355
14356    #[test]
14357    fn test_from_delimited_tsv_no_header() {
14358        let dir = tempfile::tempdir().unwrap();
14359        let path = dir.path().join("datui_test_tsv_no_header.tsv");
14360        let content = "a\tb\tc\td\n1\t2\t3\t4\n5\t6\t7\t8\n";
14361        std::fs::write(&path, content).unwrap();
14362        let opts = OpenOptions {
14363            has_header: Some(false),
14364            ..Default::default()
14365        };
14366        let mut state = DataTableState::from_delimited(&path, b'\t', &opts).unwrap();
14367        state.collect();
14368        assert_eq!(state.schema.len(), 4);
14369        assert!(state.schema.contains("column_1"));
14370        assert!(state.schema.contains("column_2"));
14371        assert!(state.schema.contains("column_3"));
14372        assert!(state.schema.contains("column_4"));
14373        assert_eq!(state.num_rows, 3);
14374    }
14375
14376    #[test]
14377    fn test_from_delimited_psv_no_header() {
14378        let dir = tempfile::tempdir().unwrap();
14379        let path = dir.path().join("datui_test_psv_no_header.psv");
14380        let content = "x|y|z\n10|20|30\n40|50|60\n";
14381        std::fs::write(&path, content).unwrap();
14382        let opts = OpenOptions {
14383            has_header: Some(false),
14384            ..Default::default()
14385        };
14386        let mut state = DataTableState::from_delimited(&path, b'|', &opts).unwrap();
14387        state.collect();
14388        assert_eq!(state.schema.len(), 3);
14389        assert!(state.schema.contains("column_1"));
14390        assert!(state.schema.contains("column_2"));
14391        assert!(state.schema.contains("column_3"));
14392        assert_eq!(state.num_rows, 3);
14393    }
14394
14395    #[test]
14396    fn test_filter() {
14397        let lf = create_test_lf();
14398        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
14399        let filters = vec![FilterStatement {
14400            columns: Vec::new(),
14401            column: "a".to_string(),
14402            operator: FilterOperator::Gt,
14403            value: "2".to_string(),
14404            logical_op: LogicalOperator::And,
14405        }];
14406        state.filter(filters);
14407        let df = state.lf.clone().collect().unwrap();
14408        assert_eq!(df.shape().0, 1);
14409        assert_eq!(df.column("a").unwrap().get(0).unwrap(), AnyValue::Int32(3));
14410    }
14411
14412    #[test]
14413    fn test_sort() {
14414        let lf = create_test_lf();
14415        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
14416        state.sort(vec!["a".to_string()], false);
14417        let df = state.lf.clone().collect().unwrap();
14418        assert_eq!(df.column("a").unwrap().get(0).unwrap(), AnyValue::Int32(3));
14419    }
14420
14421    #[test]
14422    fn test_query() {
14423        let lf = create_test_lf();
14424        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
14425        state.query("select b where a = 2".to_string());
14426        let df = state.lf.clone().collect().unwrap();
14427        assert_eq!(df.shape(), (1, 1));
14428        assert_eq!(
14429            df.column("b").unwrap().get(0).unwrap(),
14430            AnyValue::String("y")
14431        );
14432    }
14433
14434    #[test]
14435    fn test_query_date_accessors() {
14436        use chrono::NaiveDate;
14437        let df = df!(
14438            "event_date" => [
14439                NaiveDate::from_ymd_opt(2024, 1, 15).unwrap(),
14440                NaiveDate::from_ymd_opt(2024, 6, 20).unwrap(),
14441                NaiveDate::from_ymd_opt(2024, 12, 31).unwrap(),
14442            ],
14443            "name" => &["a", "b", "c"],
14444        )
14445        .unwrap();
14446        let lf = df.lazy();
14447        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
14448
14449        // Select with date accessors
14450        state.query("select name, year: event_date.year, month: event_date.month".to_string());
14451        assert!(
14452            state.error.is_none(),
14453            "query should succeed: {:?}",
14454            state.error
14455        );
14456        let df = state.lf.clone().collect().unwrap();
14457        assert_eq!(df.shape(), (3, 3));
14458        assert_eq!(
14459            df.column("year").unwrap().get(0).unwrap(),
14460            AnyValue::Int32(2024)
14461        );
14462        assert_eq!(
14463            df.column("month").unwrap().get(0).unwrap(),
14464            AnyValue::Int8(1)
14465        );
14466        assert_eq!(
14467            df.column("month").unwrap().get(1).unwrap(),
14468            AnyValue::Int8(6)
14469        );
14470
14471        // Filter with date accessor
14472        state.query("select name, event_date where event_date.month = 12".to_string());
14473        assert!(
14474            state.error.is_none(),
14475            "filter should succeed: {:?}",
14476            state.error
14477        );
14478        let df = state.lf.clone().collect().unwrap();
14479        assert_eq!(df.height(), 1);
14480        assert_eq!(
14481            df.column("name").unwrap().get(0).unwrap(),
14482            AnyValue::String("c")
14483        );
14484
14485        // Filter with YYYY.MM.DD date literal
14486        state.query("select name, event_date where event_date.date > 2024.06.15".to_string());
14487        assert!(
14488            state.error.is_none(),
14489            "date literal filter should succeed: {:?}",
14490            state.error
14491        );
14492        let df = state.lf.clone().collect().unwrap();
14493        assert_eq!(
14494            df.height(),
14495            2,
14496            "2024-06-20 and 2024-12-31 are after 2024-06-15"
14497        );
14498
14499        // String accessors: upper, lower, len, ends_with
14500        state.query(
14501            "select name, upper_name: name.upper, name_len: name.len where name.ends_with[\"c\"]"
14502                .to_string(),
14503        );
14504        assert!(
14505            state.error.is_none(),
14506            "string accessors should succeed: {:?}",
14507            state.error
14508        );
14509        let df = state.lf.clone().collect().unwrap();
14510        assert_eq!(df.height(), 1, "only 'c' ends with 'c'");
14511        assert_eq!(
14512            df.column("upper_name").unwrap().get(0).unwrap(),
14513            AnyValue::String("C")
14514        );
14515
14516        // Query that returns 0 rows: df and locked_df must be cleared for correct empty-table render
14517        state.query("select where event_date.date = 2020.01.01".to_string());
14518        assert!(state.error.is_none());
14519        assert_eq!(state.num_rows, 0);
14520        state.visible_rows = 10;
14521        state.collect();
14522        assert!(state.df.is_none(), "df must be cleared when num_rows is 0");
14523        assert!(
14524            state.locked_df.is_none(),
14525            "locked_df must be cleared when num_rows is 0"
14526        );
14527    }
14528
14529    #[test]
14530    fn test_select_next_previous() {
14531        let lf = create_large_test_lf();
14532        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
14533        state.visible_rows = 10;
14534        state.table_state.select(Some(5));
14535
14536        state.select_next();
14537        assert_eq!(state.table_state.selected(), Some(6));
14538
14539        state.select_previous();
14540        assert_eq!(state.table_state.selected(), Some(5));
14541    }
14542
14543    #[test]
14544    fn test_page_up_down() {
14545        let lf = create_large_test_lf();
14546        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
14547        state.visible_rows = 20;
14548        state.collect();
14549
14550        assert_eq!(state.start_row, 0);
14551        state.page_down();
14552        assert_eq!(state.start_row, 20);
14553        state.page_down();
14554        assert_eq!(state.start_row, 40);
14555        state.page_up();
14556        assert_eq!(state.start_row, 20);
14557        state.page_up();
14558        assert_eq!(state.start_row, 0);
14559    }
14560
14561    #[test]
14562    fn test_scroll_left_right() {
14563        let lf = create_large_test_lf();
14564        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
14565        assert_eq!(state.termcol_index, 0);
14566        state.scroll_right();
14567        assert_eq!(state.termcol_index, 1);
14568        state.scroll_right();
14569        assert_eq!(state.termcol_index, 2);
14570        state.scroll_left();
14571        assert_eq!(state.termcol_index, 1);
14572        state.scroll_left();
14573        assert_eq!(state.termcol_index, 0);
14574    }
14575
14576    #[test]
14577    fn test_reverse() {
14578        let lf = create_test_lf();
14579        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
14580        state.sort(vec!["a".to_string()], true);
14581        assert_eq!(
14582            state
14583                .lf
14584                .clone()
14585                .collect()
14586                .unwrap()
14587                .column("a")
14588                .unwrap()
14589                .get(0)
14590                .unwrap(),
14591            AnyValue::Int32(1)
14592        );
14593        state.reverse();
14594        assert_eq!(
14595            state
14596                .lf
14597                .clone()
14598                .collect()
14599                .unwrap()
14600                .column("a")
14601                .unwrap()
14602                .get(0)
14603                .unwrap(),
14604            AnyValue::Int32(3)
14605        );
14606    }
14607
14608    fn column_values(state: &DataTableState, name: &str) -> Vec<Option<i64>> {
14609        let df = state.lf.clone().collect().unwrap();
14610        df.column(name)
14611            .unwrap()
14612            .cast(&DataType::Int64)
14613            .unwrap()
14614            .i64()
14615            .unwrap()
14616            .iter()
14617            .collect()
14618    }
14619
14620    #[test]
14621    fn test_sort_puts_nulls_last_in_both_directions() {
14622        let lf = df!("a" => &[Some(2i64), None, Some(3), None, Some(1)])
14623            .unwrap()
14624            .lazy();
14625        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
14626
14627        state.sort(vec!["a".to_string()], true);
14628        assert_eq!(
14629            column_values(&state, "a"),
14630            [Some(1), Some(2), Some(3), None, None]
14631        );
14632
14633        state.sort(vec!["a".to_string()], false);
14634        assert_eq!(
14635            column_values(&state, "a"),
14636            [Some(3), Some(2), Some(1), None, None]
14637        );
14638
14639        state.reverse();
14640        assert_eq!(
14641            column_values(&state, "a"),
14642            [Some(1), Some(2), Some(3), None, None]
14643        );
14644    }
14645
14646    /// Ties keep their order, so the page read at the top (a top-k to Polars) and
14647    /// the page read below it agree on the rows they share.
14648    #[test]
14649    fn a_sort_with_ties_reads_the_same_rows_page_by_page() {
14650        let df = df!(
14651            "k" => (0..3000i64).map(|i| i % 3).collect::<Vec<_>>(),
14652            "v" => (0..3000i64).collect::<Vec<_>>(),
14653        )
14654        .unwrap();
14655        let sorted = df
14656            .lazy()
14657            .sort_by_exprs([col("k")], sort_options(vec![false]));
14658        let top = sorted.clone().slice(0, 200).collect().unwrap();
14659        let below = sorted.clone().slice(100, 100).collect().unwrap();
14660        assert!(top.slice(100, 100).equals(&below));
14661        let one = sorted.slice(5, 1).collect().unwrap();
14662        assert!(top.slice(5, 1).equals(&one));
14663    }
14664
14665    /// `IN (SELECT …)` reads the subquery's values once rather than once per row,
14666    /// which on the streaming engine ran out of memory for a page of 100k rows
14667    /// (#509), and still returns what polars-sql's own plan does, NULLs included.
14668    /// The data is small, so that without the fix the test fails on the plan rather
14669    /// than on memory. A user's own list question of the same shape is left alone.
14670    #[cfg(feature = "sql")]
14671    #[test]
14672    fn a_sql_in_subquery_reads_its_values_once() {
14673        let mut df = df!(
14674            "k" => (0..3000i64).map(|i| i % 3).collect::<Vec<_>>(),
14675            "v" => (0..3000i64).map(|i| (i % 7 != 0).then_some(i)).collect::<Vec<_>>(),
14676        )
14677        .unwrap();
14678        // A first value that is a NULL list: its length is NULL, not 1 as exploded.
14679        let lists: ListChunked = (0..3000i64)
14680            .map(|i| (i != 0).then(|| Series::new(PlSmallStr::EMPTY, [i])))
14681            .collect();
14682        df.with_column(lists.into_column().with_name("l".into()))
14683            .unwrap();
14684        let n = df.height();
14685        for (sql, rows) in [
14686            (
14687                "SELECT * FROM df WHERE v IN (SELECT v FROM df WHERE k = 1)",
14688                None,
14689            ),
14690            // The values hold a NULL, so no row is NOT IN them.
14691            (
14692                "SELECT * FROM df WHERE v NOT IN (SELECT v FROM df WHERE k = 1)",
14693                Some(0),
14694            ),
14695            (
14696                "SELECT * FROM df WHERE v NOT IN (SELECT v FROM df WHERE k = 1 AND v IS NOT NULL)",
14697                None,
14698            ),
14699            // No values: every row, a NULL `v` too.
14700            (
14701                "SELECT * FROM df WHERE v NOT IN (SELECT v FROM df WHERE k = 5)",
14702                Some(n),
14703            ),
14704            (
14705                "SELECT * FROM df WHERE v IN (SELECT v FROM df WHERE k = 5)",
14706                Some(0),
14707            ),
14708            (
14709                "SELECT * FROM df WHERE k = 2 OR v IN (SELECT v FROM df WHERE k = 1 LIMIT 100)",
14710                None,
14711            ),
14712            (
14713                "SELECT * FROM df WHERE v IN (SELECT MIN(v) FROM df GROUP BY v % 100)",
14714                None,
14715            ),
14716            (
14717                "SELECT * FROM df WHERE v IN (SELECT v FROM df WHERE v NOT IN (SELECT v FROM df WHERE k = 0 AND v IS NOT NULL))",
14718                None,
14719            ),
14720            // Unknown, not false, for a value outside a set holding a NULL, and for a
14721            // NULL value.
14722            (
14723                "SELECT * FROM df WHERE (v NOT IN (SELECT v FROM df WHERE k = 1)) IS NULL",
14724                None,
14725            ),
14726            // Only NULLs: every row unknown.
14727            (
14728                "SELECT * FROM df WHERE (v IN (SELECT v FROM df WHERE v IS NULL)) IS NULL",
14729                Some(n),
14730            ),
14731            // No values: false, a NULL `v` too.
14732            (
14733                "SELECT * FROM df WHERE (v IN (SELECT v FROM df WHERE k = 5)) IS NULL",
14734                Some(0),
14735            ),
14736            (
14737                "SELECT * FROM df WHERE ARRAY_LENGTH(FIRST(l)) IS NULL AND v NOT IN (SELECT v FROM df WHERE k = 5)",
14738                Some(n),
14739            ),
14740        ] {
14741            let mut ctx = polars_sql::SQLContext::new();
14742            ctx.register("df", df.clone().lazy());
14743            let raw = ctx.execute(sql).unwrap();
14744            // Else the check on the view's plan below would pass for nothing.
14745            assert!(
14746                (&raw.logical_plan).into_iter().any(asks_per_row),
14747                "{sql}: polars-sql's plan"
14748            );
14749            let expected = raw.collect().unwrap();
14750            if let Some(rows) = rows {
14751                assert_eq!(expected.height(), rows, "{sql}");
14752            }
14753            let mut state =
14754                DataTableState::from_lazyframe(df.clone().lazy(), &OpenOptions::default()).unwrap();
14755            state.sql_query(sql.to_string());
14756            assert!(state.error.is_none(), "{sql}: {:?}", state.error);
14757            assert!(
14758                !(&state.lf.logical_plan).into_iter().any(asks_per_row),
14759                "{sql}"
14760            );
14761            for streaming in [false, cfg!(feature = "streaming")] {
14762                let got = collect_lazy(state.lf.clone(), streaming).unwrap();
14763                assert!(
14764                    got.equals_missing(&expected),
14765                    "{sql}, streaming {streaming}"
14766                );
14767            }
14768        }
14769    }
14770
14771    /// An `IN (SELECT …)` subquery returns exactly the statement's columns, after a
14772    /// QUALIFY as after a WHERE: polars-sql leaves the column holding the values in a
14773    /// QUALIFY's result (#519). The rows are polars-sql's own, and a user's column
14774    /// named like polars-sql's is kept.
14775    #[cfg(feature = "sql")]
14776    #[test]
14777    fn a_sql_in_subquery_returns_only_the_statements_columns() {
14778        const LOOKALIKE: &str = "_POLARS_TMP_999999999";
14779        let df = df!(
14780            "k" => (0..300i64).map(|i| i % 3).collect::<Vec<_>>(),
14781            "i" => (0..300i64).collect::<Vec<_>>(),
14782            "w" => (0..300i64).map(|i| (i % 7 != 0).then_some(i % 11)).collect::<Vec<_>>(),
14783            LOOKALIKE => (0..300i64).collect::<Vec<_>>(),
14784        )
14785        .unwrap();
14786        let all = ["k", "i", "w", LOOKALIKE];
14787        for (sql, columns) in [
14788            (
14789                "SELECT k, i, ROW_NUMBER() OVER (PARTITION BY k ORDER BY i) AS r FROM df \
14790                 QUALIFY r IN (SELECT w FROM df WHERE w < 3)",
14791                &["k", "i", "r"][..],
14792            ),
14793            (
14794                "SELECT k, i, ROW_NUMBER() OVER (PARTITION BY k ORDER BY i) AS r FROM df \
14795                 QUALIFY r NOT IN (SELECT w FROM df WHERE w > 3 AND w IS NOT NULL) \
14796                 ORDER BY i DESC LIMIT 50",
14797                &["k", "i", "r"][..],
14798            ),
14799            (
14800                "SELECT * FROM df \
14801                 QUALIFY ROW_NUMBER() OVER (PARTITION BY k ORDER BY i) IN (SELECT w FROM df)",
14802                &all[..],
14803            ),
14804            (
14805                "SELECT DISTINCT k, ROW_NUMBER() OVER (PARTITION BY k ORDER BY i) AS r, _POLARS_TMP_999999999 FROM df \
14806                 QUALIFY r IN (SELECT w FROM df WHERE w < 3)",
14807                &["k", "r", LOOKALIKE][..],
14808            ),
14809            (
14810                "SELECT * FROM (SELECT k, i, ROW_NUMBER() OVER (PARTITION BY k ORDER BY i) AS r FROM df \
14811                 QUALIFY r IN (SELECT w FROM df WHERE w < 3)) WHERE i > 1",
14812                &["k", "i", "r"][..],
14813            ),
14814            (
14815                "SELECT k, i, ROW_NUMBER() OVER (PARTITION BY k ORDER BY i) AS r FROM df \
14816                 QUALIFY r IN (SELECT w FROM df WHERE w < 3) AND i IN (SELECT w FROM df)",
14817                &["k", "i", "r"][..],
14818            ),
14819            (
14820                "WITH q AS (SELECT k, i, ROW_NUMBER() OVER (PARTITION BY k ORDER BY i) AS r FROM df \
14821                 QUALIFY r IN (SELECT w FROM df WHERE w < 3)) \
14822                 SELECT * FROM q UNION ALL SELECT * FROM q",
14823                &["k", "i", "r"][..],
14824            ),
14825            // A join suffixes the right side's copy of both polars-sql's column and
14826            // the user's.
14827            (
14828                "WITH q AS (SELECT *, ROW_NUMBER() OVER (PARTITION BY k ORDER BY i) AS r FROM df \
14829                 QUALIFY r IN (SELECT w FROM df WHERE w < 3)) \
14830                 SELECT * FROM q a JOIN q b ON a.i = b.i JOIN q c ON a.i = c.i ORDER BY a.i",
14831                &[
14832                    "k",
14833                    "i",
14834                    "w",
14835                    LOOKALIKE,
14836                    "r",
14837                    "k:b",
14838                    "i:b",
14839                    "w:b",
14840                    "_POLARS_TMP_999999999:b",
14841                    "r:b",
14842                    "k:c",
14843                    "i:c",
14844                    "w:c",
14845                    "_POLARS_TMP_999999999:c",
14846                    "r:c",
14847                ][..],
14848            ),
14849            ("SELECT * FROM df WHERE i IN (SELECT w FROM df)", &all[..]),
14850            (
14851                "SELECT k, i FROM df WHERE i NOT IN (SELECT w FROM df WHERE w IS NOT NULL)",
14852                &["k", "i"][..],
14853            ),
14854        ] {
14855            let mut ctx = polars_sql::SQLContext::new();
14856            ctx.register("df", df.clone().lazy());
14857            let raw = ctx.execute(sql).unwrap().collect().unwrap();
14858            assert!(raw.height() > 0, "{sql}");
14859            let expected = raw.select(columns.iter().copied()).unwrap();
14860            let mut state =
14861                DataTableState::from_lazyframe(df.clone().lazy(), &OpenOptions::default()).unwrap();
14862            state.sql_query(sql.to_string());
14863            assert!(state.error.is_none(), "{sql}: {:?}", state.error);
14864            let names: Vec<&str> = state.schema.iter_names().map(|n| n.as_str()).collect();
14865            assert_eq!(names, columns, "{sql}");
14866            for streaming in [false, cfg!(feature = "streaming")] {
14867                let got = collect_lazy(state.lf.clone(), streaming).unwrap();
14868                assert!(
14869                    got.equals_missing(&expected),
14870                    "{sql}, streaming {streaming}: {got:?}"
14871                );
14872            }
14873        }
14874    }
14875
14876    /// The ORDER BY of the SQL in effect is state, so it leaves its mark: the sort
14877    /// mark on each column it orders by, as named in the result, until the sidebar
14878    /// sorts or another query runs (#688, item 13).
14879    #[cfg(feature = "sql")]
14880    #[test]
14881    fn a_sql_order_by_marks_the_header_until_the_sidebar_sorts() {
14882        let df = df!("k" => [1i64, 2, 3], "v" => [3i64, 2, 1]).unwrap();
14883        let marks = |sql: &str| {
14884            let mut state =
14885                DataTableState::from_lazyframe(df.clone().lazy(), &OpenOptions::default()).unwrap();
14886            state.sql_query(sql.to_string());
14887            assert!(state.error.is_none(), "{sql}: {:?}", state.error);
14888            state.header_sort()
14889        };
14890        let owned = |names: &[&str]| names.iter().map(|n| n.to_string()).collect::<Vec<_>>();
14891        assert_eq!(
14892            marks("SELECT v, k FROM df ORDER BY k DESC"),
14893            (owned(&["k"]), vec![true])
14894        );
14895        assert_eq!(
14896            marks("SELECT * FROM df ORDER BY k DESC, v LIMIT 2"),
14897            (owned(&["k", "v"]), vec![true, false])
14898        );
14899        assert_eq!(
14900            marks("SELECT v AS w, k FROM df ORDER BY w"),
14901            (owned(&["w"]), vec![false]),
14902            "named as in the result"
14903        );
14904        assert_eq!(
14905            marks("SELECT k, SUM(v) AS s FROM df GROUP BY k ORDER BY s DESC"),
14906            (owned(&["s"]), vec![true])
14907        );
14908        // An expression, or a column the result leaves out, leaves no mark.
14909        assert_eq!(marks("SELECT * FROM df ORDER BY k + 1"), (vec![], vec![]));
14910        assert_eq!(marks("SELECT v FROM df ORDER BY k"), (vec![], vec![]));
14911        assert_eq!(marks("SELECT * FROM df"), (vec![], vec![]));
14912
14913        // On the header, and the sidebar's sort replaces it.
14914        let mut state =
14915            DataTableState::from_lazyframe(df.clone().lazy(), &OpenOptions::default()).unwrap();
14916        state.sql_query("SELECT * FROM df ORDER BY k DESC".to_string());
14917        state.collect();
14918        let area = Rect::new(0, 0, 30, 6);
14919        let mut buf = Buffer::empty(area);
14920        DataTable::default().render(area, &mut buf, &mut state);
14921        let header = row_string(&buf, area, 0);
14922        let g = crate::glyphs::get();
14923        assert!(header.contains(&format!("k{}", g.sort_desc)), "{header:?}");
14924        state.sort_by(vec!["v".to_string()], vec![false]);
14925        assert_eq!(state.header_sort(), (owned(&["v"]), vec![false]));
14926        // A new query names its own order, or none.
14927        state.sql_query("SELECT * FROM df".to_string());
14928        assert_eq!(state.header_sort(), (vec![], vec![]));
14929    }
14930
14931    /// A SQL ORDER BY keeps tied rows in order, as the sidebar's sort does: the page
14932    /// read at the top (a top-k to Polars) and the next page agree on the rows they
14933    /// share, through a LIMIT and a subquery, and over a grouping, a join, a union or
14934    /// a DISTINCT, whose rows would otherwise come in any order (#495). With either
14935    /// engine.
14936    #[cfg(feature = "sql")]
14937    #[test]
14938    fn a_sql_order_by_with_ties_reads_the_same_rows_page_by_page() {
14939        let df = df!(
14940            "k" => (0..5000i64).map(|i| i % 3).collect::<Vec<_>>(),
14941            "v" => (0..5000i64).collect::<Vec<_>>(),
14942        )
14943        .unwrap();
14944        for sql in [
14945            "SELECT * FROM df ORDER BY k",
14946            "SELECT v, k FROM df ORDER BY k DESC",
14947            "SELECT * FROM df ORDER BY k LIMIT 4000",
14948            "SELECT * FROM (SELECT * FROM df ORDER BY k) WHERE v >= 0",
14949            "SELECT v % 1000 AS g, COUNT(*) AS n, MIN(v) AS v FROM df GROUP BY g ORDER BY n",
14950            "SELECT a.k, a.v FROM df a JOIN df b ON a.v = b.v ORDER BY a.k",
14951            "SELECT a.k, a.v FROM df a LEFT JOIN df b ON a.v = b.v + 1 ORDER BY a.k",
14952            "SELECT k, v FROM df UNION SELECT k, v FROM df ORDER BY k",
14953            "SELECT DISTINCT v % 1000 AS g, v % 1000 AS v FROM df ORDER BY g % 3",
14954        ] {
14955            let mut state =
14956                DataTableState::from_lazyframe(df.clone().lazy(), &OpenOptions::default()).unwrap();
14957            state.sql_query(sql.to_string());
14958            assert!(state.error.is_none(), "{sql}: {:?}", state.error);
14959            let unstable = (&state.lf.logical_plan).into_iter().any(|node| {
14960                matches!(
14961                    node,
14962                    polars::lazy::dsl::DslPlan::Sort { sort_options, .. }
14963                        if !sort_options.maintain_order
14964                )
14965            });
14966            assert!(!unstable, "{sql}");
14967            // The app pages with the streaming engine by default, which runs the top
14968            // page's top-k its own way.
14969            for streaming in [false, cfg!(feature = "streaming")] {
14970                let page = |offset, len| {
14971                    collect_lazy(state.lf.clone().slice(offset, len), streaming).unwrap()
14972                };
14973                let top = page(0, 200);
14974                let next = page(100, 200);
14975                assert!(top.slice(100, 100).equals(&next.slice(0, 100)), "{sql}");
14976                let one = page(1500, 1);
14977                let around = page(1400, 200);
14978                assert!(around.slice(100, 1).equals(&one), "{sql}");
14979                // Ties keep the order they were read in.
14980                let v = top.column("v").unwrap().i64().unwrap();
14981                assert!(
14982                    v.into_no_null_iter().is_sorted(),
14983                    "{sql}, streaming {streaming}: {:?}",
14984                    v.head(Some(10))
14985                );
14986            }
14987        }
14988    }
14989
14990    /// A SQL result with no ORDER BY reads the same rows page by page, and a Sort &
14991    /// Filter sort over it keeps its ties in that order: groupings, distincts, unions
14992    /// and joins give their rows in one order, so every page agrees with a read of
14993    /// the whole result, and a LIMIT keeps the same rows (#508). Both engines give the
14994    /// same order. A grouping sorted by its keys leaves its groups' order to the sort.
14995    #[cfg(feature = "sql")]
14996    #[test]
14997    fn a_sql_result_without_order_by_reads_the_same_rows_page_by_page() {
14998        let df = df!(
14999            "k" => (0..5000i64).map(|i| i % 3).collect::<Vec<_>>(),
15000            "v" => (0..5000i64).collect::<Vec<_>>(),
15001        )
15002        .unwrap();
15003        for sql in [
15004            "SELECT a.k, a.v FROM df a JOIN df b ON a.v = b.v",
15005            "SELECT a.k, a.v, b.v AS w FROM df a LEFT JOIN df b ON a.v = b.v + 1",
15006            "SELECT k, v FROM df UNION ALL SELECT k, v FROM df",
15007            "SELECT k, v FROM df UNION SELECT k, v FROM df",
15008            "SELECT v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY g LIMIT 300",
15009            "SELECT v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY g",
15010            "SELECT g, COUNT(*) AS n FROM (SELECT v % 1000 AS g FROM df) GROUP BY g",
15011            "SELECT DISTINCT v % 1000 AS g FROM df",
15012            "SELECT * FROM df WHERE v IN (SELECT v FROM df WHERE k = 1) LIMIT 1000",
15013        ] {
15014            for sort in [false, true] {
15015                let mut state =
15016                    DataTableState::from_lazyframe(df.clone().lazy(), &OpenOptions::default())
15017                        .unwrap();
15018                state.sql_query(sql.to_string());
15019                assert!(state.error.is_none(), "{sql}: {:?}", state.error);
15020                if sort {
15021                    // By the first column, which most of these results repeat.
15022                    let first = state.schema.get_at_index(0).unwrap().0.to_string();
15023                    state.sort(vec![first], true);
15024                    assert!(state.error.is_none(), "{sql}: {:?}", state.error);
15025                }
15026                let mut fulls = Vec::new();
15027                for streaming in [false, cfg!(feature = "streaming")] {
15028                    let read = |lf: LazyFrame| collect_lazy(lf, streaming).unwrap();
15029                    let full = read(state.lf.clone());
15030                    let middle = full.height() as i64 / 2;
15031                    for offset in [0, 100, middle] {
15032                        let page = read(state.lf.clone().slice(offset, 200));
15033                        assert!(
15034                            page.equals_missing(&full.slice(offset, 200)),
15035                            "{sql}, sort {sort}, streaming {streaming}, offset {offset}"
15036                        );
15037                    }
15038                    let one = read(state.lf.clone().slice(middle + 7, 1));
15039                    assert!(
15040                        one.equals_missing(&full.slice(middle + 7, 1)),
15041                        "{sql}, sort {sort}, streaming {streaming}"
15042                    );
15043                    fulls.push(full);
15044                }
15045                assert!(
15046                    fulls[0].equals_missing(&fulls[1]),
15047                    "{sql}, sort {sort}: the engines disagree"
15048                );
15049            }
15050        }
15051        // Keeping the groups' order as well would double a large grouping's time.
15052        let mut state = DataTableState::from_lazyframe(df.lazy(), &OpenOptions::default()).unwrap();
15053        state.sql_query("SELECT v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY g".to_string());
15054        assert!((&state.lf.logical_plan).into_iter().any(|node| matches!(
15055            node,
15056            polars::lazy::dsl::DslPlan::GroupBy {
15057                maintain_order: false,
15058                ..
15059            }
15060        )));
15061    }
15062
15063    /// A SQL grouping whose ORDER BY covers every key, by alias, ordinal or name,
15064    /// leaves its groups' order to the sort (#523), and still reads the same rows page
15065    /// by page, in the order polars-sql's own plan gives, with either engine, NULL and
15066    /// NaN keys too. A sort that leaves any key out, sorts by an expression of one, or
15067    /// reads the groups through a LIMIT, a filter or a computed column keeps the
15068    /// groups' order.
15069    #[cfg(feature = "sql")]
15070    #[test]
15071    fn a_sql_grouping_sorted_by_its_keys_leaves_the_order_to_the_sort() {
15072        use polars::lazy::dsl::DslPlan;
15073        // NaNs group as one and sort as one, as do 0.0 and -0.0.
15074        let floats = [0.0, -0.0, f64::NAN, -f64::NAN, 1.0, f64::INFINITY, -1.5];
15075        let df = df!(
15076            "k" => (0..5000i64).map(|i| i % 3).collect::<Vec<_>>(),
15077            "v" => (0..5000i64).collect::<Vec<_>>(),
15078            "w" => (0..5000i64).map(|i| (i % 11 != 0).then_some(i % 700)).collect::<Vec<_>>(),
15079            "f" => (0..5000usize)
15080                .map(|i| (i % 9 != 0).then_some(floats[i % floats.len()]))
15081                .collect::<Vec<_>>(),
15082        )
15083        .unwrap();
15084        let groups_ordered = |plan: &DslPlan| -> Vec<bool> {
15085            plan.into_iter()
15086                .filter_map(|node| match node {
15087                    DslPlan::GroupBy { maintain_order, .. } => Some(*maintain_order),
15088                    _ => None,
15089                })
15090                .collect()
15091        };
15092        for (sql, ordered) in [
15093            (
15094                "SELECT v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY g ORDER BY g",
15095                false,
15096            ),
15097            (
15098                "SELECT v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY v % 1000 ORDER BY 1 DESC LIMIT 300",
15099                false,
15100            ),
15101            (
15102                "SELECT k AS kk, v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY k, g ORDER BY n, g, kk",
15103                false,
15104            ),
15105            (
15106                "SELECT k AS d, COUNT(*) AS n FROM df GROUP BY k ORDER BY k DESC",
15107                false,
15108            ),
15109            (
15110                "SELECT w, MIN(v) AS v FROM df GROUP BY w HAVING COUNT(*) > 1 ORDER BY ALL",
15111                false,
15112            ),
15113            (
15114                "SELECT w, COUNT(*) AS n FROM df GROUP BY w ORDER BY w DESC NULLS FIRST",
15115                false,
15116            ),
15117            (
15118                "SELECT k, f, COUNT(*) AS n FROM df GROUP BY k, f ORDER BY f DESC NULLS FIRST, k",
15119                false,
15120            ),
15121            (
15122                "SELECT w % 7 AS w, COUNT(*) AS n FROM df GROUP BY w % 7 ORDER BY w",
15123                false,
15124            ),
15125            (
15126                "SELECT v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY g ORDER BY n",
15127                true,
15128            ),
15129            (
15130                "SELECT k, w, COUNT(*) AS n FROM df GROUP BY k, w ORDER BY k, w + 0",
15131                true,
15132            ),
15133            (
15134                "SELECT * FROM (SELECT w, COUNT(*) AS n FROM df GROUP BY w) WHERE n > 7 ORDER BY w",
15135                true,
15136            ),
15137            (
15138                "SELECT k, v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY k, g ORDER BY g",
15139                true,
15140            ),
15141            (
15142                "SELECT v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY v % 1000 ORDER BY v % 1000",
15143                true,
15144            ),
15145            (
15146                "SELECT * FROM (SELECT v % 1000 AS g, COUNT(*) AS n FROM df GROUP BY g LIMIT 300) ORDER BY g",
15147                true,
15148            ),
15149            (
15150                "SELECT g, ROW_NUMBER() OVER () AS r FROM (SELECT v % 1000 AS g FROM df GROUP BY g) ORDER BY g",
15151                true,
15152            ),
15153        ] {
15154            let mut state =
15155                DataTableState::from_lazyframe(df.clone().lazy(), &OpenOptions::default()).unwrap();
15156            state.sql_query(sql.to_string());
15157            assert!(state.error.is_none(), "{sql}: {:?}", state.error);
15158            assert_eq!(groups_ordered(&state.lf.logical_plan), [ordered], "{sql}");
15159            let mut ctx = polars_sql::SQLContext::new();
15160            ctx.register("df", df.clone().lazy());
15161            let raw = ctx.execute(sql).unwrap();
15162            for streaming in [false, cfg!(feature = "streaming")] {
15163                let read = |lf: LazyFrame| collect_lazy(lf, streaming).unwrap();
15164                let full = read(state.lf.clone());
15165                if !ordered {
15166                    // No ties, so polars-sql's unstable sort gives the one order too.
15167                    assert!(
15168                        full.equals_missing(&read(raw.clone())),
15169                        "{sql}, streaming {streaming}"
15170                    );
15171                }
15172                let middle = full.height() as i64 / 2;
15173                for offset in [0, 100, middle] {
15174                    let page = read(state.lf.clone().slice(offset, 200));
15175                    assert!(
15176                        page.equals_missing(&full.slice(offset, 200)),
15177                        "{sql}, streaming {streaming}, offset {offset}"
15178                    );
15179                }
15180            }
15181        }
15182    }
15183
15184    #[test]
15185    fn test_multi_column_sort_puts_nulls_last_in_every_column() {
15186        let lf = df!(
15187            "a" => &[Some(1i64), None, Some(1), Some(2), Some(1)],
15188            "b" => &[Some(5i64), Some(9), None, Some(7), Some(6)],
15189        )
15190        .unwrap()
15191        .lazy();
15192        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15193
15194        state.sort(vec!["a".to_string(), "b".to_string()], false);
15195        assert_eq!(
15196            column_values(&state, "a"),
15197            [Some(2), Some(1), Some(1), Some(1), None]
15198        );
15199        assert_eq!(
15200            column_values(&state, "b"),
15201            [Some(7), Some(6), Some(5), None, Some(9)]
15202        );
15203    }
15204
15205    #[test]
15206    fn test_by_query_puts_null_group_last() {
15207        let lf = df!(
15208            "g" => &[Some(2i64), None, Some(1), Some(2)],
15209            "v" => &[1i64, 2, 3, 4],
15210        )
15211        .unwrap()
15212        .lazy();
15213        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15214        state.query("select sum v by g".to_string());
15215        assert!(state.error.is_none(), "{:?}", state.error);
15216        assert_eq!(column_values(&state, "g"), [Some(1), Some(2), None]);
15217    }
15218
15219    #[test]
15220    fn test_filter_multiple() {
15221        let lf = create_large_test_lf();
15222        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15223        let filters = vec![
15224            FilterStatement {
15225                columns: Vec::new(),
15226                column: "c".to_string(),
15227                operator: FilterOperator::Eq,
15228                value: "1".to_string(),
15229                logical_op: LogicalOperator::And,
15230            },
15231            FilterStatement {
15232                columns: Vec::new(),
15233                column: "d".to_string(),
15234                operator: FilterOperator::Eq,
15235                value: "2".to_string(),
15236                logical_op: LogicalOperator::And,
15237            },
15238        ];
15239        state.filter(filters);
15240        let df = state.lf.clone().collect().unwrap();
15241        assert_eq!(df.shape().0, 7);
15242    }
15243
15244    #[test]
15245    fn test_filter_and_sort() {
15246        let lf = create_large_test_lf();
15247        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15248        let filters = vec![FilterStatement {
15249            columns: Vec::new(),
15250            column: "c".to_string(),
15251            operator: FilterOperator::Eq,
15252            value: "1".to_string(),
15253            logical_op: LogicalOperator::And,
15254        }];
15255        state.filter(filters);
15256        state.sort(vec!["a".to_string()], false);
15257        let df = state.lf.clone().collect().unwrap();
15258        assert_eq!(df.column("a").unwrap().get(0).unwrap(), AnyValue::Int32(97));
15259    }
15260
15261    /// Minimal long-format data for pivot tests: id, date, key, value.
15262    /// Includes duplicates for aggregation (e.g. (1,d1,A) appears twice).
15263    fn create_pivot_long_lf() -> LazyFrame {
15264        let df = df!(
15265            "id" => &[1_i32, 1, 1, 2, 2, 2, 1, 2],
15266            "date" => &["d1", "d1", "d1", "d1", "d1", "d1", "d1", "d1"],
15267            "key" => &["A", "B", "C", "A", "B", "C", "A", "B"],
15268            "value" => &[10.0_f64, 20.0, 30.0, 40.0, 50.0, 60.0, 11.0, 51.0],
15269        )
15270        .unwrap();
15271        df.lazy()
15272    }
15273
15274    /// Wide-format data for melt tests: id, date, c1, c2, c3.
15275    fn create_melt_wide_lf() -> LazyFrame {
15276        let df = df!(
15277            "id" => &[1_i32, 2, 3],
15278            "date" => &["d1", "d2", "d3"],
15279            "c1" => &[10.0_f64, 20.0, 30.0],
15280            "c2" => &[11.0, 21.0, 31.0],
15281            "c3" => &[12.0, 22.0, 32.0],
15282        )
15283        .unwrap();
15284        df.lazy()
15285    }
15286
15287    #[test]
15288    fn test_pivot_basic() {
15289        let lf = create_pivot_long_lf();
15290        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15291        let spec = PivotSpec {
15292            index: vec!["id".to_string(), "date".to_string()],
15293            pivot_column: "key".to_string(),
15294            value_column: "value".to_string(),
15295            aggregation: PivotAggregation::Last,
15296            sort_columns: None,
15297        };
15298        state.pivot(&spec).unwrap();
15299        let df = state.lf.clone().collect().unwrap();
15300        let names: Vec<&str> = df.get_column_names().iter().map(|s| s.as_str()).collect();
15301        assert!(names.contains(&"id"));
15302        assert!(names.contains(&"date"));
15303        assert!(names.contains(&"A"));
15304        assert!(names.contains(&"B"));
15305        assert!(names.contains(&"C"));
15306        assert_eq!(df.height(), 2);
15307    }
15308
15309    #[test]
15310    fn test_pivot_aggregation_last() {
15311        let lf = create_pivot_long_lf();
15312        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15313        let spec = PivotSpec {
15314            index: vec!["id".to_string(), "date".to_string()],
15315            pivot_column: "key".to_string(),
15316            value_column: "value".to_string(),
15317            aggregation: PivotAggregation::Last,
15318            sort_columns: None,
15319        };
15320        state.pivot(&spec).unwrap();
15321        let df = state.lf.clone().collect().unwrap();
15322        let a_col = df.column("A").unwrap();
15323        let row0 = a_col.get(0).unwrap();
15324        let row1 = a_col.get(1).unwrap();
15325        assert_eq!(row0, AnyValue::Float64(11.0));
15326        assert_eq!(row1, AnyValue::Float64(40.0));
15327    }
15328
15329    #[test]
15330    fn test_pivot_aggregation_first() {
15331        let lf = create_pivot_long_lf();
15332        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15333        let spec = PivotSpec {
15334            index: vec!["id".to_string(), "date".to_string()],
15335            pivot_column: "key".to_string(),
15336            value_column: "value".to_string(),
15337            aggregation: PivotAggregation::First,
15338            sort_columns: None,
15339        };
15340        state.pivot(&spec).unwrap();
15341        let df = state.lf.clone().collect().unwrap();
15342        let a_col = df.column("A").unwrap();
15343        assert_eq!(a_col.get(0).unwrap(), AnyValue::Float64(10.0));
15344        assert_eq!(a_col.get(1).unwrap(), AnyValue::Float64(40.0));
15345    }
15346
15347    #[test]
15348    fn test_pivot_aggregation_min_max() {
15349        let lf = create_pivot_long_lf();
15350        let mut state_min = DataTableState::new(lf.clone(), None, None, None, None, true).unwrap();
15351        state_min
15352            .pivot(&PivotSpec {
15353                index: vec!["id".to_string(), "date".to_string()],
15354                pivot_column: "key".to_string(),
15355                value_column: "value".to_string(),
15356                aggregation: PivotAggregation::Min,
15357                sort_columns: None,
15358            })
15359            .unwrap();
15360        let df_min = state_min.lf.clone().collect().unwrap();
15361        assert_eq!(
15362            df_min.column("A").unwrap().get(0).unwrap(),
15363            AnyValue::Float64(10.0)
15364        );
15365
15366        let mut state_max = DataTableState::new(lf, None, None, None, None, true).unwrap();
15367        state_max
15368            .pivot(&PivotSpec {
15369                index: vec!["id".to_string(), "date".to_string()],
15370                pivot_column: "key".to_string(),
15371                value_column: "value".to_string(),
15372                aggregation: PivotAggregation::Max,
15373                sort_columns: None,
15374            })
15375            .unwrap();
15376        let df_max = state_max.lf.clone().collect().unwrap();
15377        assert_eq!(
15378            df_max.column("A").unwrap().get(0).unwrap(),
15379            AnyValue::Float64(11.0)
15380        );
15381    }
15382
15383    #[test]
15384    fn test_pivot_aggregation_avg_count() {
15385        let lf = create_pivot_long_lf();
15386        let mut state_avg = DataTableState::new(lf.clone(), None, None, None, None, true).unwrap();
15387        state_avg
15388            .pivot(&PivotSpec {
15389                index: vec!["id".to_string(), "date".to_string()],
15390                pivot_column: "key".to_string(),
15391                value_column: "value".to_string(),
15392                aggregation: PivotAggregation::Avg,
15393                sort_columns: None,
15394            })
15395            .unwrap();
15396        let df_avg = state_avg.lf.clone().collect().unwrap();
15397        let a = df_avg.column("A").unwrap().get(0).unwrap();
15398        if let AnyValue::Float64(x) = a {
15399            assert!((x - 10.5).abs() < 1e-6);
15400        } else {
15401            panic!("expected float");
15402        }
15403
15404        let mut state_count = DataTableState::new(lf, None, None, None, None, true).unwrap();
15405        state_count
15406            .pivot(&PivotSpec {
15407                index: vec!["id".to_string(), "date".to_string()],
15408                pivot_column: "key".to_string(),
15409                value_column: "value".to_string(),
15410                aggregation: PivotAggregation::Count,
15411                sort_columns: None,
15412            })
15413            .unwrap();
15414        let df_count = state_count.lf.clone().collect().unwrap();
15415        let a = df_count.column("A").unwrap().get(0).unwrap();
15416        assert_eq!(a, AnyValue::UInt32(2));
15417    }
15418
15419    #[test]
15420    fn test_pivot_string_first_last() {
15421        let df = df!(
15422            "id" => &[1_i32, 1, 2, 2],
15423            "key" => &["X", "Y", "X", "Y"],
15424            "value" => &["low", "mid", "high", "mid"],
15425        )
15426        .unwrap();
15427        let lf = df.lazy();
15428        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15429        let spec = PivotSpec {
15430            index: vec!["id".to_string()],
15431            pivot_column: "key".to_string(),
15432            value_column: "value".to_string(),
15433            aggregation: PivotAggregation::Last,
15434            sort_columns: None,
15435        };
15436        state.pivot(&spec).unwrap();
15437        let out = state.lf.clone().collect().unwrap();
15438        assert_eq!(
15439            out.column("X").unwrap().get(0).unwrap(),
15440            AnyValue::String("low")
15441        );
15442        assert_eq!(
15443            out.column("Y").unwrap().get(0).unwrap(),
15444            AnyValue::String("mid")
15445        );
15446    }
15447
15448    #[test]
15449    fn test_melt_basic() {
15450        let lf = create_melt_wide_lf();
15451        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15452        let spec = MeltSpec {
15453            index: vec!["id".to_string(), "date".to_string()],
15454            value_columns: vec!["c1".to_string(), "c2".to_string(), "c3".to_string()],
15455            variable_name: "variable".to_string(),
15456            value_name: "value".to_string(),
15457        };
15458        state.melt(&spec).unwrap();
15459        let df = state.lf.clone().collect().unwrap();
15460        assert_eq!(df.height(), 9);
15461        let names: Vec<&str> = df.get_column_names().iter().map(|s| s.as_str()).collect();
15462        assert!(names.contains(&"variable"));
15463        assert!(names.contains(&"value"));
15464        assert!(names.contains(&"id"));
15465        assert!(names.contains(&"date"));
15466    }
15467
15468    /// Dates past the calendar in the second row: a date, and datetimes in ms and
15469    /// us with and without a zone. The first row is 1970-01-01.
15470    fn past_calendar_lf() -> LazyFrame {
15471        let paris = TimeZone::opt_try_new(Some("Europe/Paris")).unwrap();
15472        let datetime = |name: &str, unit, zone: Option<TimeZone>| {
15473            Series::new(name.into(), [0, i64::MIN + 1])
15474                .cast(&DataType::Datetime(unit, zone))
15475                .unwrap()
15476                .into_column()
15477        };
15478        DataFrame::new_infer_height(vec![
15479            Column::new("id".into(), [1i32, 2]),
15480            Column::new("s".into(), ["a", "b"]),
15481            Series::new("d".into(), [0, i32::MAX])
15482                .cast(&DataType::Date)
15483                .unwrap()
15484                .into_column(),
15485            datetime("t_ms", TimeUnit::Milliseconds, None),
15486            datetime("t_us", TimeUnit::Microseconds, None),
15487            datetime("t_ms_tz", TimeUnit::Milliseconds, paris.clone()),
15488            datetime("t_us_tz", TimeUnit::Microseconds, paris),
15489        ])
15490        .unwrap()
15491        .lazy()
15492    }
15493
15494    const PAST_CALENDAR: [&str; 5] = ["d", "t_ms", "t_us", "t_ms_tz", "t_us_tz"];
15495
15496    /// `column`'s two values as text: Polars' own for the first, and the stored
15497    /// number the table shows for the one past the calendar.
15498    fn past_calendar_text(column: &str) -> [String; 2] {
15499        let df = past_calendar_lf().collect().unwrap();
15500        let values = df.column(column).unwrap();
15501        let first = values
15502            .slice(0, 1)
15503            .cast(&DataType::String)
15504            .unwrap()
15505            .str()
15506            .unwrap()
15507            .get(0)
15508            .unwrap()
15509            .to_string();
15510        let past = crate::exact::past_calendar_text(&values.get(1).unwrap()).unwrap();
15511        [first, past]
15512    }
15513
15514    /// Pivoted on a date past the calendar, its new column is named by the stored
15515    /// number, the columns still in date order; Polars' own naming panicked (#506).
15516    /// Without one, the columns are named as they always were.
15517    #[test]
15518    fn a_pivot_on_a_date_past_the_calendar_names_it_by_its_stored_number() {
15519        for on in PAST_CALENDAR {
15520            let [first, past] = past_calendar_text(on);
15521            for with_past in [true, false] {
15522                let lf = past_calendar_lf()
15523                    .filter(col("id").eq(lit(1)).or(lit(with_past)))
15524                    .select([col("id"), col(on), col("s")]);
15525                let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15526                state
15527                    .pivot(&PivotSpec {
15528                        index: vec!["id".to_string()],
15529                        pivot_column: on.to_string(),
15530                        value_column: "s".to_string(),
15531                        aggregation: PivotAggregation::First,
15532                        sort_columns: None,
15533                    })
15534                    .unwrap();
15535                let df = state.lf.clone().collect().unwrap();
15536                let names: Vec<&str> = df.get_column_names().iter().map(|n| n.as_str()).collect();
15537                let dates = match (with_past, on) {
15538                    (false, _) => vec![first.as_str()],
15539                    // i32::MAX days is after 1970, i64::MIN + 1 before it.
15540                    (true, "d") => vec![first.as_str(), past.as_str()],
15541                    (true, _) => vec![past.as_str(), first.as_str()],
15542                };
15543                assert_eq!(names[1..], dates, "{on}");
15544                assert_eq!(
15545                    df.column(&first).unwrap().str().unwrap().get(0),
15546                    Some("a"),
15547                    "{on}"
15548                );
15549                if with_past {
15550                    assert_eq!(
15551                        df.column(&past).unwrap().str().unwrap().get(1),
15552                        Some("b"),
15553                        "{on}"
15554                    );
15555                }
15556            }
15557        }
15558    }
15559
15560    /// Melted with text, a date past the calendar is its stored number, where the
15561    /// cast to text panicked (#506), and one in range Polars' own text. Melted with
15562    /// dates only, the values stay dates.
15563    #[test]
15564    fn a_melt_of_dates_with_text_writes_a_date_past_the_calendar_as_its_number() {
15565        let melt = |columns: [&str; 2]| {
15566            let mut state =
15567                DataTableState::new(past_calendar_lf(), None, None, None, None, true).unwrap();
15568            state
15569                .melt(&MeltSpec {
15570                    index: vec!["id".to_string()],
15571                    value_columns: columns.map(String::from).to_vec(),
15572                    variable_name: "variable".to_string(),
15573                    value_name: "value".to_string(),
15574                })
15575                .unwrap();
15576            state.lf.clone().collect().unwrap()
15577        };
15578        for column in PAST_CALENDAR {
15579            let [first, past] = past_calendar_text(column);
15580            let df = melt([column, "s"]);
15581            let values: Vec<Option<&str>> =
15582                df.column("value").unwrap().str().unwrap().iter().collect();
15583            assert_eq!(
15584                values,
15585                [
15586                    Some(first.as_str()),
15587                    Some(past.as_str()),
15588                    Some("a"),
15589                    Some("b")
15590                ],
15591                "{column}"
15592            );
15593        }
15594        let df = melt(["t_ms", "t_us"]);
15595        assert!(matches!(
15596            df.column("value").unwrap().dtype(),
15597            DataType::Datetime(..)
15598        ));
15599    }
15600
15601    /// SQL's three plan rewrites compose: in a join filtered by an `IN` subquery, a
15602    /// date past the calendar met with text is its stored number (#506), the join
15603    /// keeps one row order (#508), and the subquery's values are read once (#509).
15604    #[cfg(feature = "sql")]
15605    #[test]
15606    fn a_sql_join_with_an_in_subquery_and_a_date_past_the_calendar() {
15607        use polars::lazy::dsl::DslPlan;
15608        for c in PAST_CALENDAR {
15609            let [first, past] = past_calendar_text(c);
15610            let sql = format!(
15611                "SELECT a.id, COALESCE(a.{c}, b.s) AS x FROM df a JOIN df b ON a.id = b.id \
15612                 WHERE a.id IN (SELECT t.id FROM df t WHERE t.s <> 'z')"
15613            );
15614            let mut state =
15615                DataTableState::new(past_calendar_lf(), None, None, None, None, true).unwrap();
15616            state.sql_query(sql.clone());
15617            assert!(state.error.is_none(), "{sql}: {:?}", state.error);
15618            let plan = &state.lf.logical_plan;
15619            assert!(!plan.into_iter().any(asks_per_row), "{sql}");
15620            let joins: Vec<_> = plan
15621                .into_iter()
15622                .filter_map(|node| match node {
15623                    DslPlan::Join { options, .. } => Some(options.args.maintain_order),
15624                    _ => None,
15625                })
15626                .collect();
15627            assert!(!joins.is_empty(), "{sql}");
15628            assert!(
15629                joins.iter().all(|order| *order != MaintainOrderJoin::None),
15630                "{sql}"
15631            );
15632            for streaming in [false, cfg!(feature = "streaming")] {
15633                let df = collect_lazy(state.lf.clone(), streaming).unwrap();
15634                let ids: Vec<Option<i32>> =
15635                    df.column("id").unwrap().i32().unwrap().iter().collect();
15636                assert_eq!(ids, [Some(1), Some(2)], "{sql}, streaming {streaming}");
15637                let x: Vec<Option<&str>> = df.column("x").unwrap().str().unwrap().iter().collect();
15638                assert_eq!(
15639                    x,
15640                    [Some(first.as_str()), Some(past.as_str())],
15641                    "{sql}, streaming {streaming}"
15642                );
15643                let page = collect_lazy(state.lf.clone().slice(1, 1), streaming).unwrap();
15644                assert!(page.equals_missing(&df.slice(1, 1)), "{sql}");
15645            }
15646        }
15647    }
15648
15649    /// A q query forgets the melt it replaces; rolled back, the melt is
15650    /// what SQL runs against again, not only what the table shows.
15651    #[test]
15652    fn a_rollback_brings_back_the_melt_a_query_forgot() {
15653        let lf = create_melt_wide_lf();
15654        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15655        let spec = MeltSpec {
15656            index: vec!["id".to_string(), "date".to_string()],
15657            value_columns: vec!["c1".to_string(), "c2".to_string(), "c3".to_string()],
15658            variable_name: "variable".to_string(),
15659            value_name: "value".to_string(),
15660        };
15661        state.melt(&spec).unwrap();
15662        let saved = state.rollback_point();
15663        state.query("select id".to_string());
15664        assert!(state.last_melt_spec().is_none());
15665        state.roll_back(saved);
15666        assert!(state.last_melt_spec().is_some());
15667        assert_eq!(state.query_root().collect().unwrap().height(), 9);
15668    }
15669
15670    #[test]
15671    fn test_melt_all_except_index() {
15672        let lf = create_melt_wide_lf();
15673        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15674        let spec = MeltSpec {
15675            index: vec!["id".to_string(), "date".to_string()],
15676            value_columns: vec!["c1".to_string(), "c2".to_string(), "c3".to_string()],
15677            variable_name: "var".to_string(),
15678            value_name: "val".to_string(),
15679        };
15680        state.melt(&spec).unwrap();
15681        let df = state.lf.clone().collect().unwrap();
15682        assert!(df.column("var").is_ok());
15683        assert!(df.column("val").is_ok());
15684    }
15685
15686    #[test]
15687    fn test_pivot_on_current_view_after_filter() {
15688        let lf = create_pivot_long_lf();
15689        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15690        state.filter(vec![FilterStatement {
15691            columns: Vec::new(),
15692            column: "id".to_string(),
15693            operator: FilterOperator::Eq,
15694            value: "1".to_string(),
15695            logical_op: LogicalOperator::And,
15696        }]);
15697        let spec = PivotSpec {
15698            index: vec!["id".to_string(), "date".to_string()],
15699            pivot_column: "key".to_string(),
15700            value_column: "value".to_string(),
15701            aggregation: PivotAggregation::Last,
15702            sort_columns: None,
15703        };
15704        state.pivot(&spec).unwrap();
15705        let df = state.lf.clone().collect().unwrap();
15706        assert_eq!(df.height(), 1);
15707        let id_col = df.column("id").unwrap();
15708        assert_eq!(id_col.get(0).unwrap(), AnyValue::Int32(1));
15709    }
15710
15711    /// The one-pass pivot gives what the lazy pivot over the whole view gave, for every
15712    /// aggregation, with nulls in the index, the pivot column and the values, pairs with
15713    /// no rows, and more rows than one streaming morsel, so `first` and `last` are
15714    /// checked for order across batches.
15715    #[test]
15716    fn a_pivot_in_one_pass_matches_the_lazy_pivot() {
15717        let n = 250_000usize;
15718        let view = df!(
15719            "g" => (0..n)
15720                .map(|i| (i % 13 != 0).then_some((i % 97) as i64))
15721                .collect::<Vec<_>>(),
15722            "key" => (0..n)
15723                .map(|i| (i % 17 != 0).then(|| format!("k{}", (i * 7) % 11)))
15724                .collect::<Vec<_>>(),
15725            "v" => (0..n)
15726                .map(|i| (i % 7 != 0).then_some((i % 1_000) as f64))
15727                .collect::<Vec<_>>(),
15728        )
15729        .unwrap()
15730        .lazy()
15731        // Pairs with no rows at all: `k3` never meets a `g` divisible by five.
15732        .filter(
15733            (col("g") % lit(5i64))
15734                .neq(lit(0i64))
15735                .or(col("key").neq(lit("k3")))
15736                .fill_null(lit(true)),
15737        );
15738
15739        // The pivot as it ran before: the new columns read first, then the lazy pivot.
15740        let lazy_pivot = |spec: &PivotSpec| {
15741            let on = spec.pivot_column.as_str();
15742            let value = spec.value_column.as_str();
15743            let on_columns = view
15744                .clone()
15745                .select([col(on)])
15746                .unique(None, UniqueKeepStrategy::Any)
15747                .sort([on], SortMultipleOptions::default().with_nulls_last(true))
15748                .collect()
15749                .unwrap();
15750            let index = if spec.index.is_empty() {
15751                all() - by_name([on, value], true, false)
15752            } else {
15753                by_name(spec.index.iter().map(String::as_str), true, false)
15754            };
15755            view.clone()
15756                .pivot(
15757                    by_name([on], true, false),
15758                    Arc::new(on_columns),
15759                    index,
15760                    by_name([value], true, false),
15761                    pivot_agg_expr(spec.aggregation, element()),
15762                    true,
15763                    PlSmallStr::from_static("_"),
15764                    PivotColumnNaming::Auto,
15765                )
15766                .collect()
15767                .unwrap()
15768        };
15769        let close = |a: &DataFrame, b: &DataFrame| {
15770            a.get_column_names() == b.get_column_names()
15771                && a.height() == b.height()
15772                && a.columns().iter().zip(b.columns()).all(|(x, y)| {
15773                    if x.dtype().is_float() {
15774                        let (x, y) = (x.f64().unwrap(), y.f64().unwrap());
15775                        x.iter().zip(y.iter()).all(|pair| match pair {
15776                            (Some(x), Some(y)) => (x - y).abs() <= 1e-9 * x.abs().max(1.0),
15777                            (x, y) => x.is_none() && y.is_none(),
15778                        })
15779                    } else {
15780                        x.as_materialized_series()
15781                            .equals_missing(y.as_materialized_series())
15782                    }
15783                })
15784        };
15785
15786        for index in [vec!["g".to_string()], Vec::new()] {
15787            for aggregation in PivotAggregation::ALL {
15788                let spec = PivotSpec {
15789                    index: index.clone(),
15790                    pivot_column: "key".to_string(),
15791                    value_column: "v".to_string(),
15792                    aggregation,
15793                    sort_columns: None,
15794                };
15795                let expected = lazy_pivot(&spec);
15796                for streaming in [false, true] {
15797                    let pivoted = PivotJob {
15798                        view: view.clone(),
15799                        spec: spec.clone(),
15800                        streaming,
15801                    }
15802                    .run()
15803                    .unwrap();
15804                    assert!(
15805                        close(&pivoted, &expected),
15806                        "{aggregation:?}, index {index:?}, streaming {streaming}:\n\
15807                         {pivoted:?}\n{expected:?}"
15808                    );
15809                }
15810            }
15811        }
15812    }
15813
15814    #[test]
15815    fn test_fuzzy_token_regex() {
15816        assert_eq!(fuzzy_token_regex("foo"), "(?i).*f.*o.*o.*");
15817        assert_eq!(fuzzy_token_regex("a"), "(?i).*a.*");
15818        // Regex-special characters are escaped
15819        let pat = fuzzy_token_regex("[");
15820        assert!(pat.contains("\\["));
15821    }
15822
15823    #[test]
15824    fn test_fuzzy_search() {
15825        // Filter logic is covered by test_fuzzy_search_regex_direct. This test runs the full
15826        // path through DataTableState; it requires sample data (CSV with string column).
15827        crate::tests::ensure_sample_data();
15828        let path = crate::tests::sample_data_dir().join("3-sfd-header.csv");
15829        let mut state = DataTableState::from_csv(&path, &Default::default()).unwrap();
15830        state.visible_rows = 10;
15831        state.collect();
15832        let before = state.num_rows;
15833        state.fuzzy_search("string".to_string());
15834        assert!(state.error.is_none(), "{:?}", state.error);
15835        assert!(state.num_rows <= before, "fuzzy search should filter rows");
15836        state.fuzzy_search("".to_string());
15837        state.collect();
15838        assert_eq!(state.num_rows, before, "empty fuzzy search should reset");
15839        assert!(state.get_active_fuzzy_query().is_empty());
15840    }
15841
15842    #[test]
15843    fn test_fuzzy_search_regex_direct() {
15844        // Sanity check: Polars str().contains with our regex matches "alice" for pattern ".*a.*l.*i.*"
15845        let lf = df!("name" => &["alice", "bob", "carol"]).unwrap().lazy();
15846        let pattern = fuzzy_token_regex("alice");
15847        let out = lf
15848            .filter(col("name").str().contains(lit(pattern.clone()), false))
15849            .collect()
15850            .unwrap();
15851        assert_eq!(out.height(), 1, "regex {:?} should match alice", pattern);
15852
15853        // Two columns OR (as in fuzzy_search)
15854        let lf2 = df!(
15855            "id" => &[1i32, 2, 3],
15856            "name" => &["alice", "bob", "carol"],
15857            "city" => &["NYC", "LA", "Boston"]
15858        )
15859        .unwrap()
15860        .lazy();
15861        let pat = fuzzy_token_regex("alice");
15862        let expr = col("name")
15863            .str()
15864            .contains(lit(pat.clone()), false)
15865            .or(col("city").str().contains(lit(pat), false));
15866        let out2 = lf2.clone().filter(expr).collect().unwrap();
15867        assert_eq!(out2.height(), 1);
15868
15869        // Replicate exact fuzzy_search logic: schema from original_lf, string_cols, then filter
15870        let schema = lf2.clone().collect_schema().unwrap();
15871        let string_cols: Vec<String> = schema
15872            .iter()
15873            .filter(|(_, dtype)| dtype.is_string())
15874            .map(|(name, _)| name.to_string())
15875            .collect();
15876        assert!(
15877            !string_cols.is_empty(),
15878            "df! string cols should be detected"
15879        );
15880        let pattern = fuzzy_token_regex("alice");
15881        let token_expr = string_cols
15882            .iter()
15883            .map(|c| col(c.as_str()).str().contains(lit(pattern.clone()), false))
15884            .reduce(|a, b| a.or(b))
15885            .unwrap();
15886        let out3 = lf2.filter(token_expr).collect().unwrap();
15887        assert_eq!(
15888            out3.height(),
15889            1,
15890            "fuzzy_search-style filter should match 1 row"
15891        );
15892    }
15893
15894    /// A fuzzy search replaces the sort along with the rest of the pipeline: a descending
15895    /// sort before it must not leave the result reversed with no sort column to show it.
15896    #[test]
15897    fn fuzzy_search_after_a_descending_sort_is_not_reversed() {
15898        let lf = df!(
15899            "id" => &[1i32, 2, 3],
15900            "name" => &["alice", "bob", "carol"]
15901        )
15902        .unwrap()
15903        .lazy();
15904        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15905        state.sort(vec!["id".to_string()], false);
15906        state.fuzzy_search("a".to_string());
15907        assert!(state.error.is_none(), "{:?}", state.error);
15908        assert!(state.view_sort_columns().is_empty());
15909        assert!(state.view_sort_ascending());
15910
15911        // The sidebar re-applies the (empty) filters and sort over the result.
15912        state.filter(Vec::new());
15913        let df = state.lf.clone().collect().unwrap();
15914        assert_eq!(df.height(), 2);
15915        assert_eq!(df.column("id").unwrap().get(0).unwrap(), AnyValue::Int32(1));
15916    }
15917
15918    /// A reshape likewise replaces the sort. It runs over the sorted view, so its rows
15919    /// come out descending, and a sidebar action afterwards must leave them that way
15920    /// rather than reverse a frame that shows no sort column.
15921    #[test]
15922    fn melt_after_a_descending_sort_is_not_reversed() {
15923        let lf = df!("id" => &[1i32, 2, 3], "c1" => &[10i32, 20, 30])
15924            .unwrap()
15925            .lazy();
15926        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15927        state.sort(vec!["id".to_string()], false);
15928        state
15929            .melt(&MeltSpec {
15930                index: vec!["id".to_string()],
15931                value_columns: vec!["c1".to_string()],
15932                variable_name: "var".to_string(),
15933                value_name: "val".to_string(),
15934            })
15935            .unwrap();
15936        assert!(state.view_sort_columns().is_empty());
15937        assert!(state.view_sort_ascending());
15938        let melted = state.lf.clone().collect().unwrap();
15939        assert_eq!(
15940            melted.column("id").unwrap().get(0).unwrap(),
15941            AnyValue::Int32(3)
15942        );
15943
15944        state.filter(Vec::new());
15945        assert!(state.lf.clone().collect().unwrap().equals(&melted));
15946    }
15947
15948    #[test]
15949    fn test_fuzzy_search_no_string_columns() {
15950        let lf = df!("a" => &[1i32, 2, 3], "b" => &[10i64, 20, 30])
15951            .unwrap()
15952            .lazy();
15953        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15954        state.fuzzy_search("x".to_string());
15955        assert!(state.error.is_some());
15956    }
15957
15958    /// What the Search hint promises: every word's letters in order, in any text
15959    /// column. Each word may match a different column; letters out of order do not.
15960    #[test]
15961    fn search_matches_every_words_letters_in_order_in_any_text_column() {
15962        let rows = |query: &str| {
15963            let lf = df!(
15964                "name" => &["Smith", "Marion", "Smith"],
15965                "city" => &["London", "London", "Paris"]
15966            )
15967            .unwrap()
15968            .lazy();
15969            let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
15970            state.fuzzy_search(query.to_string());
15971            assert!(state.error.is_none(), "{:?}", state.error);
15972            state.lf.clone().collect().unwrap().height()
15973        };
15974        assert_eq!(rows("smth"), 2, "letters in order, not adjacent, any case");
15975        assert_eq!(rows("smth ldn"), 1, "every word, each in its own column");
15976        assert_eq!(rows("htims"), 0, "letters out of order");
15977    }
15978
15979    /// By-queries must produce results sorted by the group columns (age_group, then team)
15980    /// so that output order is deterministic and practical. Raw data is deliberately out of order.
15981    #[test]
15982    fn test_by_query_result_sorted_by_group_columns() {
15983        // Build a small table: age_group (1-5, out of order), team (Red/Blue/Green), score (0-100)
15984        let df = df!(
15985            "age_group" => &[3i64, 1, 5, 2, 4, 1, 2, 3, 4, 5, 1, 2, 3, 4, 5],
15986            "team" => &[
15987                "Red", "Blue", "Green", "Red", "Blue", "Green", "Green", "Red", "Blue",
15988                "Green", "Red", "Blue", "Red", "Blue", "Green",
15989            ],
15990            "score" => &[50.0f64, 10.0, 90.0, 20.0, 30.0, 40.0, 60.0, 70.0, 80.0, 15.0, 25.0, 35.0, 45.0, 55.0, 65.0],
15991        )
15992        .unwrap();
15993        let lf = df.lazy();
15994        let options = crate::OpenOptions::default();
15995        let mut state = DataTableState::from_lazyframe(lf, &options).unwrap();
15996        state.query("select avg score by age_group, team".to_string());
15997        assert!(
15998            state.error.is_none(),
15999            "query should succeed: {:?}",
16000            state.error
16001        );
16002        let result = state.lf.collect().unwrap();
16003        // Result must be sorted by group columns (age_group, then team)
16004        let sorted = result
16005            .sort(
16006                ["age_group", "team"],
16007                SortMultipleOptions::default().with_order_descending(false),
16008            )
16009            .unwrap();
16010        assert_eq!(
16011            result, sorted,
16012            "by-query result must be sorted by (age_group, team)"
16013        );
16014    }
16015
16016    /// Computed group keys (e.g. Fare: 1+floor Fare % 25) must be sorted by their result column
16017    /// values, not by re-evaluating the expression on the result.
16018    #[test]
16019    fn test_by_query_computed_group_key_sorted_by_result_column() {
16020        let df = df!(
16021            "x" => &[7.0f64, 12.0, 3.0, 22.0, 17.0, 8.0],
16022            "v" => &[1.0f64, 2.0, 3.0, 4.0, 5.0, 6.0],
16023        )
16024        .unwrap();
16025        let lf = df.lazy();
16026        let options = crate::OpenOptions::default();
16027        let mut state = DataTableState::from_lazyframe(lf, &options).unwrap();
16028        // bucket: 1+floor(x)%3 -> values 1,2,3; raw x order 7,12,3,22,17,8 -> buckets 2,2,1,2,2,2
16029        state.query("select sum v by bucket: 1+floor x % 3".to_string());
16030        assert!(
16031            state.error.is_none(),
16032            "query should succeed: {:?}",
16033            state.error
16034        );
16035        let result = state.lf.collect().unwrap();
16036        let bucket = result.column("bucket").unwrap();
16037        // Must be sorted by bucket (1, 2, 3)
16038        for i in 1..result.height() {
16039            let prev: i64 = bucket.get(i - 1).unwrap().try_extract().unwrap_or(0);
16040            let curr: i64 = bucket.get(i).unwrap().try_extract().unwrap_or(0);
16041            assert!(
16042                curr >= prev,
16043                "bucket column must be sorted: {} then {}",
16044                prev,
16045                curr
16046            );
16047        }
16048    }
16049
16050    // Read the header (top) row of a rendered buffer as a string.
16051    fn header_row_string(buf: &Buffer, area: Rect) -> String {
16052        (area.x..area.x + area.width)
16053            .map(|x| buf[(x, area.y)].symbol().to_string())
16054            .collect()
16055    }
16056
16057    // Read an arbitrary row of a rendered buffer as a string (y = 0 is the header).
16058    fn row_string(buf: &Buffer, area: Rect, y: u16) -> String {
16059        (area.x..area.x + area.width)
16060            .map(|x| buf[(x, area.y + y)].symbol().to_string())
16061            .collect()
16062    }
16063
16064    fn table_with_format(preset: &str, align: bool) -> DataTable {
16065        DataTable::default().with_number_format(NumberFormatSettings {
16066            format: crate::numfmt::NumberFormat::preset(preset).unwrap(),
16067            enabled: true,
16068            exclude: Vec::new(),
16069            align_numeric_right: align,
16070        })
16071    }
16072
16073    #[test]
16074    fn grouping_is_off_by_default() {
16075        // Upgrading must not change how anything renders.
16076        let table = DataTable::default();
16077        let df = df!("pos" => &[1234567i64]).unwrap();
16078        let area = Rect::new(0, 0, 30, 3);
16079        let mut buf = Buffer::empty(area);
16080        let mut ts = TableState::default();
16081        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16082        let row = row_string(&buf, area, 1);
16083        assert!(row.contains("1234567"), "got: {row:?}");
16084        assert!(!row.contains("1,234,567"), "got: {row:?}");
16085    }
16086
16087    #[test]
16088    fn thousands_separators_are_applied_to_integer_columns() {
16089        let table = table_with_format("thousands", false);
16090        // Genomic coordinates, the case from issue #51.
16091        let df = df!("chromStart" => &[248956422i64, 3088269832]).unwrap();
16092        let area = Rect::new(0, 0, 30, 4);
16093        let mut buf = Buffer::empty(area);
16094        let mut ts = TableState::default();
16095        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16096        assert!(row_string(&buf, area, 1).contains("248,956,422"));
16097        assert!(row_string(&buf, area, 2).contains("3,088,269,832"));
16098    }
16099
16100    #[test]
16101    fn column_width_accounts_for_separators() {
16102        // The separators widen the column; the heading must not be clipped and
16103        // the value must render in full.
16104        let table = table_with_format("thousands", false);
16105        let df = df!("n" => &[1234567i64]).unwrap();
16106        let area = Rect::new(0, 0, 12, 3);
16107        let mut buf = Buffer::empty(area);
16108        let mut ts = TableState::default();
16109        let shown = table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16110        assert_eq!(shown, 1);
16111        assert!(row_string(&buf, area, 1).contains("1,234,567"));
16112    }
16113
16114    #[test]
16115    fn strings_are_untouched_and_integers_group_uniformly() {
16116        let table = table_with_format("thousands", false);
16117        let df = df!(
16118            "chrom" => &["chr1"],
16119            "n" => &[2024i32],
16120        )
16121        .unwrap();
16122        let area = Rect::new(0, 0, 30, 3);
16123        let mut buf = Buffer::empty(area);
16124        let mut ts = TableState::default();
16125        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16126        let row = row_string(&buf, area, 1);
16127        assert!(row.contains("chr1"), "got: {row:?}");
16128        // No magnitude threshold: a column must not mix grouped and ungrouped
16129        // values, so four-digit numbers group like everything else.
16130        assert!(row.contains("2,024"), "got: {row:?}");
16131    }
16132
16133    #[test]
16134    fn excluding_a_column_is_how_identifier_columns_stay_plain() {
16135        // The replacement for a digit threshold: name the columns that hold
16136        // identifiers rather than quantities.
16137        let table = DataTable::default().with_number_format(NumberFormatSettings {
16138            format: crate::numfmt::NumberFormat::preset("thousands").unwrap(),
16139            enabled: true,
16140            exclude: vec![crate::numfmt::Glob::new("year")],
16141            align_numeric_right: false,
16142        });
16143        let df = df!(
16144            "year" => &[2024i32],
16145            "count" => &[2024i32],
16146        )
16147        .unwrap();
16148        let area = Rect::new(0, 0, 40, 3);
16149        let mut buf = Buffer::empty(area);
16150        let mut ts = TableState::default();
16151        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16152        let row = row_string(&buf, area, 1);
16153        assert!(row.contains("2024"), "excluded column stays plain: {row:?}");
16154        assert!(row.contains("2,024"), "other column groups: {row:?}");
16155    }
16156
16157    #[test]
16158    fn numeric_columns_and_their_headers_render_flush_right() {
16159        let table = table_with_format("none", true);
16160        // Header "value" is 5 wide; the values are shorter, so they must be
16161        // padded on the left, not the right.
16162        let df = df!("value" => &[7i64, 42]).unwrap();
16163        let area = Rect::new(0, 0, 5, 4);
16164        let mut buf = Buffer::empty(area);
16165        let mut ts = TableState::default();
16166        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16167        assert_eq!(row_string(&buf, area, 1), "    7");
16168        assert_eq!(row_string(&buf, area, 2), "   42");
16169        assert_eq!(header_row_string(&buf, area), "value");
16170    }
16171
16172    #[test]
16173    fn header_follows_its_column_alignment() {
16174        // A wide numeric column: the heading must sit flush right over the
16175        // digits rather than floating left.
16176        let table = table_with_format("none", true);
16177        let df = df!("n" => &[1234567i64]).unwrap();
16178        let area = Rect::new(0, 0, 7, 3);
16179        let mut buf = Buffer::empty(area);
16180        let mut ts = TableState::default();
16181        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16182        assert_eq!(header_row_string(&buf, area), "      n");
16183        assert_eq!(row_string(&buf, area, 1), "1234567");
16184    }
16185
16186    #[test]
16187    fn non_numeric_columns_stay_left_aligned() {
16188        let table = table_with_format("none", true);
16189        let df = df!("name" => &["ab"]).unwrap();
16190        let area = Rect::new(0, 0, 4, 3);
16191        let mut buf = Buffer::empty(area);
16192        let mut ts = TableState::default();
16193        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16194        assert_eq!(row_string(&buf, area, 1), "ab  ");
16195        assert_eq!(header_row_string(&buf, area), "name");
16196    }
16197
16198    #[test]
16199    fn alignment_can_be_turned_off() {
16200        let table = table_with_format("none", false);
16201        let df = df!("value" => &[7i64]).unwrap();
16202        let area = Rect::new(0, 0, 5, 3);
16203        let mut buf = Buffer::empty(area);
16204        let mut ts = TableState::default();
16205        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16206        assert_eq!(row_string(&buf, area, 1), "7    ");
16207    }
16208
16209    #[test]
16210    fn excluded_columns_are_not_grouped() {
16211        let table = DataTable::default().with_number_format(NumberFormatSettings {
16212            format: crate::numfmt::NumberFormat::preset("thousands").unwrap(),
16213            enabled: true,
16214            exclude: vec![crate::numfmt::Glob::new("*_id")],
16215            align_numeric_right: false,
16216        });
16217        let df = df!(
16218            "sample_id" => &[1234567i64],
16219            "count" => &[1234567i64],
16220        )
16221        .unwrap();
16222        let area = Rect::new(0, 0, 40, 3);
16223        let mut buf = Buffer::empty(area);
16224        let mut ts = TableState::default();
16225        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16226        let row = row_string(&buf, area, 1);
16227        assert!(row.contains("1234567"), "excluded column raw: {row:?}");
16228        assert!(row.contains("1,234,567"), "other column grouped: {row:?}");
16229    }
16230
16231    #[test]
16232    fn disabled_formatting_renders_raw_digits() {
16233        // What the `,` toggle does: same settings, enabled = false.
16234        let mut settings = NumberFormatSettings {
16235            format: crate::numfmt::NumberFormat::preset("thousands").unwrap(),
16236            enabled: true,
16237            exclude: Vec::new(),
16238            align_numeric_right: false,
16239        };
16240        settings.enabled = false;
16241        let table = DataTable::default().with_number_format(settings);
16242        let df = df!("n" => &[1234567i64]).unwrap();
16243        let area = Rect::new(0, 0, 20, 3);
16244        let mut buf = Buffer::empty(area);
16245        let mut ts = TableState::default();
16246        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16247        assert!(row_string(&buf, area, 1).contains("1234567"));
16248    }
16249
16250    #[test]
16251    fn binary_stub_columns_are_never_formatted_or_aligned() {
16252        // The stub is a placeholder, not data.
16253        let table = DataTable {
16254            binary_cols: std::collections::HashSet::from(["blob".to_string()]),
16255            ..table_with_format("thousands", true)
16256        };
16257        let df = df!("blob" => &[binary_stub()]).unwrap();
16258        let area = Rect::new(0, 0, 10, 3);
16259        let mut buf = Buffer::empty(area);
16260        let mut ts = TableState::default();
16261        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16262        assert!(row_string(&buf, area, 1).starts_with(binary_stub()));
16263    }
16264
16265    /// The buffer holds a binary column as stub text; the type row still says binary.
16266    #[test]
16267    fn a_binary_column_s_type_row_says_binary() {
16268        let table = DataTable {
16269            binary_cols: std::collections::HashSet::from(["blob".to_string()]),
16270            dtype_row: true,
16271            ..table_with_format("thousands", true)
16272        };
16273        let df = df!("blob" => &[binary_stub()], "s" => &["x"]).unwrap();
16274        let area = Rect::new(0, 0, 30, 4);
16275        let mut buf = Buffer::empty(area);
16276        let mut ts = TableState::default();
16277        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16278        let types = row_string(&buf, area, 1);
16279        assert!(types.contains("binary"), "{types:?}");
16280        assert_eq!(types.matches("str").count(), 1, "{types:?}");
16281    }
16282
16283    /// A line break or a tab in a value is marked in the one-line cell; drawn as
16284    /// is, ratatui drops it and `line1\nline2` reads `line1line2`.
16285    #[test]
16286    fn breaks_tabs_and_controls_are_marked_in_a_cell() {
16287        let table = DataTable::default();
16288        let df = df!("s" => ["line1\nline2", "tab\tseparated", "esc\u{1b}[0m"]).unwrap();
16289        let area = Rect::new(0, 0, 30, 4);
16290        let mut buf = Buffer::empty(area);
16291        let mut ts = TableState::default();
16292        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16293        let g = table.glyphs;
16294        assert!(
16295            row_string(&buf, area, 1).starts_with(&format!("line1{}line2", g.newline_mark)),
16296            "{:?}",
16297            row_string(&buf, area, 1)
16298        );
16299        assert!(row_string(&buf, area, 2).starts_with(&format!("tab{}separated", g.tab_mark)));
16300        assert!(row_string(&buf, area, 3).starts_with(&format!("esc{}[0m", g.control_mark)));
16301    }
16302
16303    /// A direction control in a value is marked too: drawn, a terminal that lays
16304    /// out bidirectional text would reverse the rest of the row.
16305    #[test]
16306    fn direction_controls_are_marked_in_a_cell() {
16307        let table = DataTable::default();
16308        let df = df!("s" => ["a\u{202e}evil\u{202c}z"]).unwrap();
16309        let area = Rect::new(0, 0, 30, 3);
16310        let mut buf = Buffer::empty(area);
16311        table.render_dataframe(&df, area, &mut buf, &mut TableState::default(), false, 0);
16312        let m = table.glyphs.control_mark;
16313        let row = row_string(&buf, area, 1);
16314        assert!(row.starts_with(&format!("a{m}evil{m}z")), "{row:?}");
16315        assert!(
16316            !buf.content()
16317                .iter()
16318                .any(|c| c.symbol().contains('\u{202e}'))
16319        );
16320    }
16321
16322    /// A cell measures only the start of a huge value, and says it goes on.
16323    #[test]
16324    fn a_huge_value_is_measured_by_its_start() {
16325        let table = DataTable::default();
16326        let huge = "x".repeat(crate::exact::CELL_PREVIEW_BYTES * 4);
16327        let df = df!("s" => [huge.as_str()]).unwrap();
16328        let mut scratch = String::new();
16329        let slice = table.slice_column(&df, 0, 1, &HashSet::new(), &mut scratch);
16330        let ellipsis = crate::glyphs::cell_width(table.glyphs.ellipsis);
16331        assert_eq!(
16332            usize::from(slice.value_width),
16333            crate::exact::CELL_PREVIEW_BYTES + ellipsis
16334        );
16335    }
16336
16337    #[test]
16338    fn trailing_overflow_string_column_is_truncated_not_dropped() {
16339        // A string trailing column that doesn't fully fit should still be shown truncated
16340        // (filling the remaining width) rather than dropped entirely (leaving blank space).
16341        let table = DataTable::default();
16342        let df = df!(
16343            "a" => &[1i32, 2, 3],
16344            "wide_text" => &["aaaaaaaaaa", "bbbbbbbbbb", "cccccccccc"],
16345        )
16346        .unwrap();
16347        // "a" needs width 1; with padding that's used_width 2, leaving 6 for "wide_text",
16348        // whose full width (10) overflows -> it must be shown truncated to the remaining 6.
16349        let area = Rect::new(0, 0, 8, 4);
16350        let mut buf = Buffer::empty(area);
16351        let mut ts = TableState::default();
16352        let shown = table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16353        assert_eq!(
16354            shown, 2,
16355            "the overflowing trailing string column should be kept (truncated)"
16356        );
16357        // Part of the heading should be visible so the user knows what the column is,
16358        // behind the clip marker (three cells in the ASCII set).
16359        assert!(
16360            header_row_string(&buf, area).contains("wid"),
16361            "truncated column heading should be visible: {:?}",
16362            header_row_string(&buf, area)
16363        );
16364    }
16365
16366    #[test]
16367    fn binary_stub_cells_are_styled_with_binary_color_and_italic() {
16368        // Binary columns render the `‹binary›` stub; those cells should be colored with
16369        // binary_col and italicized so they read as a placeholder, while ordinary columns
16370        // keep their normal (non-italic) styling.
16371        let table = DataTable {
16372            binary_col: Some(Color::DarkGray),
16373            binary_cols: std::collections::HashSet::from(["blob".to_string()]),
16374            ..DataTable::default()
16375        };
16376        let df = df!(
16377            "a" => &[1i32, 2],
16378            "blob" => &[binary_stub(), binary_stub()],
16379        )
16380        .unwrap();
16381        let area = Rect::new(0, 0, 20, 4);
16382        let mut buf = Buffer::empty(area);
16383        let mut ts = TableState::default();
16384        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16385
16386        // A data row (y = 1; y = 0 is the header). The stub cells should be dark gray + italic.
16387        let stub_styled = (area.x..area.x + area.width).any(|x| {
16388            let cell = &buf[(x, 1)];
16389            cell.fg == Color::DarkGray && cell.modifier.contains(Modifier::ITALIC)
16390        });
16391        assert!(
16392            stub_styled,
16393            "binary stub cells should be colored with binary_col and italicized"
16394        );
16395        // The non-binary "a" column must not be italicized.
16396        let any_italic_non_darkgray = (area.x..area.x + area.width).any(|x| {
16397            let cell = &buf[(x, 1)];
16398            cell.modifier.contains(Modifier::ITALIC) && cell.fg != Color::DarkGray
16399        });
16400        assert!(
16401            !any_italic_non_darkgray,
16402            "only binary columns should be italicized"
16403        );
16404    }
16405
16406    #[test]
16407    fn trailing_overflow_numeric_column_is_dropped_not_truncated() {
16408        // A numeric column must NOT be shown truncated: a partial number reads as a wrong value.
16409        let table = DataTable::default();
16410        let df = df!(
16411            "a" => &[1i32, 2, 3],
16412            "wide_number" => &[111_111_111i64, 222_222_222, 333_333_333],
16413        )
16414        .unwrap();
16415        let area = Rect::new(0, 0, 8, 4);
16416        let mut buf = Buffer::empty(area);
16417        let mut ts = TableState::default();
16418        let shown = table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
16419        assert_eq!(
16420            shown, 1,
16421            "an overflowing numeric column should be dropped, not truncated"
16422        );
16423    }
16424
16425    #[test]
16426    fn byte_clamp_keeps_view_near_end_of_buffer() {
16427        // Regression: jumping to the END of a large dataset used to go blank because the
16428        // max_buffered_mb byte-clamp trimmed the buffer's HEAD (slice(0, max_rows)), discarding
16429        // exactly the tail rows the view needed. The clamp must keep the view in range.
16430        let lf = df!("a" => &["seed"]).unwrap().lazy();
16431        // 1 MB byte budget.
16432        let mut state = DataTableState::new(lf, None, None, None, Some(1), true).unwrap();
16433        state.num_rows = 1000;
16434        state.num_rows_valid = true;
16435        state.visible_rows = 40;
16436        state.start_row = 960; // jump-to-end position (num_rows - visible_rows)
16437
16438        // Buffer spans [900, 1000); the view [960, 1000) sits at its tail. ~2 MB of data forces
16439        // a trim to roughly half the rows.
16440        let buffer_start = 900;
16441        let big: Vec<String> = (0..100).map(|_| "z".repeat(20_000)).collect();
16442        let df = df!("a" => big).unwrap();
16443
16444        let fitted = state.fill_plan(buffer_start, 1000, 1000, true).fit(df);
16445        let (sliced, eff_start) = (fitted.df, fitted.start);
16446        let eff_end = eff_start + sliced.height();
16447        assert!(
16448            sliced.height() < 100,
16449            "expected a trim below the byte budget"
16450        );
16451        assert!(
16452            eff_start <= state.start_row,
16453            "view start {} fell before kept buffer start {}",
16454            state.start_row,
16455            eff_start
16456        );
16457        assert!(
16458            eff_end >= state.start_row + state.visible_rows,
16459            "view end {} fell after kept buffer end {}",
16460            state.start_row + state.visible_rows,
16461            eff_end
16462        );
16463        assert_eq!(
16464            sliced.height(),
16465            eff_end - eff_start,
16466            "df height must match range"
16467        );
16468    }
16469
16470    /// The address range of every buffer behind `arr`: values, validity, offsets, string
16471    /// views and data, and nested children. Panics on a type it cannot walk, so a test
16472    /// cannot pass by skipping a column.
16473    fn array_storage(arr: &dyn polars_arrow::array::Array, out: &mut Vec<(usize, usize)>) {
16474        use polars_arrow::array::{
16475            BinaryViewArray, BooleanArray, FixedSizeListArray, ListArray, NullArray,
16476            PrimitiveArray, StructArray, Utf8ViewArray,
16477        };
16478        fn push<T>(out: &mut Vec<(usize, usize)>, s: &[T]) {
16479            if !s.is_empty() {
16480                let start = s.as_ptr() as usize;
16481                out.push((start, start + std::mem::size_of_val(s)));
16482            }
16483        }
16484        let any = arr.as_any();
16485        // A null array's validity is Polars' shared zeroed bitmap, owned by no frame.
16486        if any.downcast_ref::<NullArray>().is_some() {
16487            return;
16488        }
16489        if let Some(validity) = arr.validity() {
16490            push(out, validity.as_slice().0);
16491        }
16492        macro_rules! primitive {
16493            ($($t:ty),*) => {$(
16494                if let Some(a) = any.downcast_ref::<PrimitiveArray<$t>>() {
16495                    return push(out, a.values().as_slice());
16496                }
16497            )*};
16498        }
16499        primitive!(i8, i16, i32, i64, i128, u8, u16, u32, u64, f32, f64);
16500        if let Some(a) = any.downcast_ref::<BooleanArray>() {
16501            push(out, a.values().as_slice().0);
16502        } else if let Some(a) = any.downcast_ref::<Utf8ViewArray>() {
16503            push(out, a.views().as_slice());
16504            a.data_buffers()
16505                .iter()
16506                .for_each(|b| push(out, b.as_slice()));
16507        } else if let Some(a) = any.downcast_ref::<BinaryViewArray>() {
16508            push(out, a.views().as_slice());
16509            a.data_buffers()
16510                .iter()
16511                .for_each(|b| push(out, b.as_slice()));
16512        } else if let Some(a) = any.downcast_ref::<ListArray<i64>>() {
16513            push(out, a.offsets().as_slice());
16514            array_storage(a.values().as_ref(), out);
16515        } else if let Some(a) = any.downcast_ref::<FixedSizeListArray>() {
16516            array_storage(a.values().as_ref(), out);
16517        } else if let Some(a) = any.downcast_ref::<StructArray>() {
16518            a.values()
16519                .iter()
16520                .for_each(|v| array_storage(v.as_ref(), out));
16521        } else {
16522            panic!("no storage walk for {:?}", arr.dtype());
16523        }
16524    }
16525
16526    fn frame_storage(df: &DataFrame) -> Vec<(usize, usize)> {
16527        let mut out = Vec::new();
16528        for column in df.columns() {
16529            for chunk in column.as_materialized_series().chunks() {
16530                array_storage(chunk.as_ref(), &mut out);
16531            }
16532        }
16533        out
16534    }
16535
16536    /// Whether any buffer behind `held` lies inside one behind `source`.
16537    fn shares_storage(held: &DataFrame, source: &DataFrame) -> bool {
16538        let source = frame_storage(source);
16539        frame_storage(held)
16540            .iter()
16541            .any(|&(s, e)| source.iter().any(|&(ss, se)| s < se && ss < e))
16542    }
16543
16544    /// Every kind of column a buffer holds: fixed width, booleans, short and long
16545    /// strings, binary, nulls in each, a categorical, an enum, a datetime, a decimal, a
16546    /// list of strings, a fixed-size array, a struct and an all-null column.
16547    fn mixed_frame(start: usize, end: usize) -> DataFrame {
16548        let rows = start..end;
16549        let long = |i: usize| format!("{i:>8}-{}", "x".repeat(40));
16550        let mut df = df!(
16551            "id" => rows.clone().map(|i| i as i64).collect::<Vec<_>>(),
16552            "maybe" => rows.clone().map(|i| (i % 3 != 0).then_some(i as f64)).collect::<Vec<_>>(),
16553            "flag" => rows.clone().map(|i| (i % 5 != 0).then_some(i % 2 == 0)).collect::<Vec<_>>(),
16554            "short" => rows.clone().map(|i| format!("s{}", i % 100)).collect::<Vec<_>>(),
16555            "long" => rows.clone().map(|i| (i % 7 != 0).then(|| long(i))).collect::<Vec<_>>(),
16556        )
16557        .unwrap();
16558        let n = end - start;
16559        let cat = df
16560            .column("short")
16561            .unwrap()
16562            .cast(&DataType::from_categories(Categories::global()))
16563            .unwrap()
16564            .with_name("cat".into());
16565        let when = df
16566            .column("id")
16567            .unwrap()
16568            .cast(&DataType::Datetime(TimeUnit::Milliseconds, None))
16569            .unwrap()
16570            .with_name("when".into());
16571        let list: ListChunked = rows
16572            .clone()
16573            .map(|i| Some(Series::new("".into(), [long(i), format!("t{i}")])))
16574            .collect();
16575        let nested = StructChunked::from_columns(
16576            "nested".into(),
16577            n,
16578            &[
16579                df.column("id").unwrap().clone(),
16580                df.column("long").unwrap().clone(),
16581            ],
16582        )
16583        .unwrap();
16584        let labels: Vec<String> = (0..100).map(|i| format!("s{i}")).collect();
16585        let labels = FrozenCategories::new(labels.iter().map(String::as_str)).unwrap();
16586        let label = df
16587            .column("short")
16588            .unwrap()
16589            .cast(&DataType::from_frozen_categories(labels))
16590            .unwrap()
16591            .with_name("label".into());
16592        let bytes = df
16593            .column("long")
16594            .unwrap()
16595            .cast(&DataType::Binary)
16596            .unwrap()
16597            .with_name("bytes".into());
16598        let price = df
16599            .column("maybe")
16600            .unwrap()
16601            .cast(&DataType::Decimal(18, 2))
16602            .unwrap()
16603            .with_name("price".into());
16604        let tags = list.with_name("tags".into()).into_column();
16605        let pair = tags
16606            .cast(&DataType::Array(Box::new(DataType::String), 2))
16607            .unwrap()
16608            .with_name("pair".into());
16609        for column in [
16610            cat,
16611            label,
16612            bytes,
16613            when,
16614            price,
16615            tags,
16616            pair,
16617            nested.into_column(),
16618        ] {
16619            df.with_column(column).unwrap();
16620        }
16621        df.with_column(Series::new_null("nothing".into(), n).into_column())
16622            .unwrap();
16623        df
16624    }
16625
16626    #[test]
16627    fn compacted_rows_own_their_storage() {
16628        // A slice of a fill keeps the fill's buffers, rechunked or not; the compacted
16629        // rows share none of them and read the same.
16630        let source = mixed_frame(0, 20_000);
16631        let slice = source.slice(1_000, 1_000);
16632        assert!(shares_storage(&slice, &source), "the probe sees a slice");
16633        let mut rechunked = slice.clone();
16634        rechunked.rechunk_mut();
16635        assert!(
16636            shares_storage(&rechunked, &source),
16637            "a lone chunk rechunked is still the slice"
16638        );
16639
16640        let kept = compact_rows(source.clone(), 1_000, 1_000, None);
16641        assert!(!shares_storage(&kept, &source));
16642        assert!(kept.equals_missing(&slice));
16643        assert_eq!(kept.schema(), source.schema());
16644
16645        // Chunks a slice keeps whole are sliced; past a quarter more, the rows are copied.
16646        let mut chunked = mixed_frame(0, 1_000);
16647        for i in 1..4 {
16648            chunked
16649                .vstack_mut(&mixed_frame(i * 1_000, (i + 1) * 1_000))
16650                .unwrap();
16651        }
16652        let chunk = |i: usize| chunked.slice(i as i64 * 1_000, 1_000);
16653        let sliced = trim_rows(chunked.clone(), 1_000, 2_000, None);
16654        assert!(shares_storage(&sliced, &chunk(1)) && shares_storage(&sliced, &chunk(2)));
16655        assert!(!shares_storage(&sliced, &chunk(0)) && !shares_storage(&sliced, &chunk(3)));
16656        let copied = trim_rows(chunked.clone(), 1_500, 1_000, None);
16657        assert!((0..4).all(|i| !shares_storage(&copied, &chunk(i))));
16658        assert!(copied.equals_missing(&chunked.slice(1_500, 1_000)));
16659
16660        // Two chunks: rows from one of them, and rows across both. Across a seam, the
16661        // rows either side of it are copied apart.
16662        let mut stitched = mixed_frame(0, 3_000);
16663        stitched.vstack_mut(&mixed_frame(3_000, 6_000)).unwrap();
16664        for (offset, len) in [(3_500, 1_000), (2_500, 1_000), (0, 6_000)] {
16665            for seam in [None, Some(3_000)] {
16666                let kept = compact_rows(stitched.clone(), offset, len, seam);
16667                assert!(!shares_storage(&kept, &stitched), "{offset}+{len}");
16668                assert!(kept.equals_missing(&stitched.slice(offset as i64, len)));
16669                assert_eq!(kept.schema(), stitched.schema());
16670                let across = seam.is_some() && offset < 3_000 && 3_000 < offset + len;
16671                let chunks = if across { 2 } else { 1 };
16672                assert!(
16673                    kept.columns()
16674                        .iter()
16675                        .filter_map(Column::as_series)
16676                        .all(|s| s.n_chunks() == chunks),
16677                    "{offset}+{len} {seam:?}"
16678                );
16679            }
16680        }
16681
16682        // A constant column is cut to length, not written out a row at a time.
16683        let constant = "k".repeat(200);
16684        let with_constant = mixed_frame(0, 2_000)
16685            .lazy()
16686            .with_column(lit(constant.as_str()).alias("constant"))
16687            .collect()
16688            .unwrap();
16689        assert!(matches!(
16690            with_constant.column("constant").unwrap(),
16691            Column::Scalar(_)
16692        ));
16693        let kept = compact_rows(with_constant.clone(), 500, 1_000, None);
16694        let Column::Scalar(cut) = kept.column("constant").unwrap() else {
16695            panic!("the constant column was built out");
16696        };
16697        assert_eq!(cut.len(), 1_000);
16698        assert!(kept.equals_missing(&with_constant.slice(500, 1_000)));
16699    }
16700
16701    /// A state with a 1 MB byte budget and the first column locked.
16702    fn trimming_state(lf: LazyFrame) -> DataTableState {
16703        let mut state = DataTableState::new(lf, None, None, None, Some(1), true).unwrap();
16704        state.locked_columns_count = 1;
16705        state.visible_rows = 40;
16706        state
16707    }
16708
16709    fn assert_view_rows(state: &DataTableState, source: &DataFrame) {
16710        let (start, end) = (state.buffered_start(), state.buffered_end());
16711        assert!(start <= state.start_row && state.start_row + 40 <= end);
16712        let held = state.buffered_df.as_ref().unwrap();
16713        assert_eq!(held.height(), end - start);
16714        assert!(held.equals_missing(&source.slice(start as i64, end - start)));
16715        let id = |df: &DataFrame, row: usize| df.column("id").unwrap().i64().unwrap().get(row);
16716        let locked = state.locked_df.as_ref().unwrap();
16717        assert_eq!(locked.get_column_names(), ["id"]);
16718        assert_eq!(
16719            id(locked, state.start_row - start),
16720            Some(state.start_row as i64)
16721        );
16722        let shown = state.df.as_ref().unwrap();
16723        assert_eq!(shown.height(), held.height());
16724        assert!(
16725            shown.column("id").is_err(),
16726            "the locked column is not repeated"
16727        );
16728        for frame in [held, locked, shown] {
16729            assert!(!shares_storage(frame, source), "trimmed rows are let go");
16730        }
16731    }
16732
16733    #[test]
16734    fn a_trimmed_fill_lets_go_of_the_rows_it_drops() {
16735        const N: usize = 10_000;
16736        let source = mixed_frame(0, N);
16737        let budget_rows = 1024 * 1024 / (source.estimated_size() / N);
16738        assert!(budget_rows < N * 3 / 4, "the budget trims the fill");
16739
16740        // Asynchronous: a plain fill over the budget.
16741        let mut state = trimming_state(source.clone().lazy());
16742        state.num_rows = N;
16743        state.num_rows_valid = true;
16744        state.start_row = 7_500;
16745        // The worker makes the copy; the UI thread installs it as it is.
16746        let plan = state.fill_plan(0, N, N, true);
16747        let fill = source.clone();
16748        let result = std::thread::spawn(move || {
16749            let result = plan.fit(fill);
16750            assert_eq!(compactions(), 1, "the worker copies the rows it keeps");
16751            result
16752        })
16753        .join()
16754        .unwrap();
16755        let before = compactions();
16756        state.apply_async_collect(result);
16757        assert_eq!(compactions(), before, "the install copies nothing");
16758        assert!(state.buffered_end() - state.buffered_start() <= budget_rows);
16759        assert_view_rows(&state, &source);
16760
16761        // Synchronous: the same fill collected on the spot from an eager source, which
16762        // stays held by the frame itself; the buffer's copy does not. A binary column
16763        // is drawn as a stub there, so it is left out.
16764        let source = source.drop("bytes").unwrap();
16765        let mut state = trimming_state(source.clone().lazy());
16766        state.num_rows = N;
16767        state.num_rows_valid = true;
16768        state.start_row = 300;
16769        state.load_buffer(0, N);
16770        assert!(state.error.is_none());
16771        assert_view_rows(&state, &source);
16772    }
16773
16774    #[test]
16775    fn a_fill_cut_around_a_view_since_left_is_not_installed() {
16776        // The worker cuts around the view the fill was planned for. A view that jumped
16777        // past the rows kept, while it was out, keeps what is held and asks again. So
16778        // does a read that came back short, whose end the cut dropped.
16779        const N: usize = 10_000;
16780        let source = mixed_frame(0, N);
16781        for short in [false, true] {
16782            let mut state = trimming_state(source.clone().lazy());
16783            state.num_rows = N;
16784            state.num_rows_valid = !short;
16785            state.start_row = 300;
16786            let asked = if short { N + 5_000 } else { N };
16787            let result = state.fill_plan(0, asked, asked, !short).fit(source.clone());
16788            assert!(
16789                result.start + result.df.height() < 9_000,
16790                "the fill was cut"
16791            );
16792            state.start_row = 9_000;
16793            state.needs_recollect = false;
16794            state.apply_async_collect(result);
16795            assert!(
16796                state.buffered_df.is_none(),
16797                "nothing drawn under wrong numbers (short {short})"
16798            );
16799            assert!(state.needs_recollect);
16800            assert_eq!((state.num_rows, state.num_rows_valid), (N, true));
16801        }
16802    }
16803
16804    #[test]
16805    fn an_untrimmed_fill_is_kept_as_collected() {
16806        // Ordinary paging copies nothing: the buffer is the fill itself.
16807        let source = mixed_frame(0, 200);
16808        let mut state = trimming_state(source.clone().lazy());
16809        state.land(Fill {
16810            df: source.clone(),
16811            buffer_start: 0,
16812            buffer_end: 200,
16813            num_rows: 200,
16814            count_known: true,
16815        });
16816        let held = state.buffered_df.as_ref().unwrap();
16817        assert_eq!(frame_storage(held), frame_storage(&source));
16818    }
16819
16820    #[test]
16821    fn a_stitched_trim_lets_go_of_both_groups() {
16822        // Forward and back across a row group boundary: the union is trimmed to the
16823        // cap, and neither fetched group is held behind the kept rows. The copy is the
16824        // worker's; installing it copies nothing. Leaving the stitched rows cuts the
16825        // buffer down to one group and lets the other go, again without a copy.
16826        const G: usize = 1_000_000;
16827        const CAP: usize = DEFAULT_MAX_BUFFERED_ROWS;
16828        let rows = |start: usize, end: usize| {
16829            Fill {
16830            df: df!(
16831                "id" => (start as i64..end as i64).collect::<Vec<i64>>(),
16832                "name" => (start..end).map(|i| format!("{i:>8}-{}", "y".repeat(24))).collect::<Vec<_>>(),
16833            )
16834            .unwrap(),
16835            buffer_start: start,
16836            buffer_end: end,
16837            num_rows: 10 * G,
16838            count_known: true,
16839        }
16840        };
16841        let check = |state: &DataTableState, fetched: &[&DataFrame]| {
16842            let (start, end) = (state.buffered_start(), state.buffered_end());
16843            assert!(start <= state.start_row && state.start_row + 40 <= end);
16844            assert_eq!(end - start, CAP, "trimmed back to the cap");
16845            let held = state.buffered_df.as_ref().unwrap();
16846            let ids = held.column("id").unwrap().i64().unwrap();
16847            assert_eq!(ids.get(0), Some(start as i64));
16848            assert_eq!(ids.get(CAP - 1), Some(end as i64 - 1));
16849            for df in fetched {
16850                assert!(!shares_storage(held, df));
16851                assert!(!shares_storage(state.locked_df.as_ref().unwrap(), df));
16852                assert!(!shares_storage(state.df.as_ref().unwrap(), df));
16853            }
16854        };
16855        let lf = rows(0, 1).df.lazy();
16856        for forward in [true, false] {
16857            let mut state = DataTableState::new(lf.clone(), None, None, None, None, true)
16858                .unwrap()
16859                .with_open(OpenFacts {
16860                    remote_source: true,
16861                    row_groups: vec![vec![G; 10]],
16862                    ..Default::default()
16863                });
16864            state.locked_columns_count = 1;
16865            state.visible_rows = 40;
16866            let (first, view) = if forward {
16867                (G - 60, G - 20)
16868            } else {
16869                (G + 20, G - 20)
16870            };
16871            assert!(state.scroll_to(first));
16872            let request = state.prepare_async_collect(None).expect("one group");
16873            let first = rows(request.buffer_start, request.buffer_end);
16874            let first_df = first.df.clone();
16875            state.land(first);
16876
16877            assert!(state.scroll_to(view));
16878            let request = state.prepare_async_collect(None).expect("the other group");
16879            assert_eq!(
16880                request.buffer_start == G,
16881                forward,
16882                "fetched alone, to stitch on"
16883            );
16884            let second = rows(request.buffer_start, request.buffer_end);
16885            let second_df = second.df.clone();
16886            // Fit on a worker, as the app does, against the rows held when planned.
16887            let plan = request.plan;
16888            let result = std::thread::spawn(move || {
16889                let result = plan.fit(second.df);
16890                assert!(compactions() > 0, "the worker copies the rows it keeps");
16891                result
16892            })
16893            .join()
16894            .unwrap();
16895            let before = compactions();
16896            state.apply_async_collect(result);
16897            assert_eq!(compactions(), before, "the install copies nothing");
16898            check(&state, &[&first_df, &second_df]);
16899
16900            // A window inside one group: the rows of the other are let go.
16901            let stitched = state.buffered_df.clone().unwrap();
16902            let (start, end) = (state.buffered_start(), state.buffered_end());
16903            let (start, end, other) = if forward {
16904                (G, end, stitched.slice(0, G - start))
16905            } else {
16906                (start, G, stitched.slice((G - start) as i64, end - G))
16907            };
16908            assert!(state.scroll_to(if forward { G } else { G - 40 }));
16909            assert!(state.holds_buffer(start, end));
16910            assert_eq!(compactions(), before, "the cut falls on the seam: a slice");
16911            state.slice_buffer_into_display();
16912            assert_eq!((state.buffered_start(), state.buffered_end()), (start, end));
16913            let held = state.buffered_df.as_ref().unwrap();
16914            assert_eq!(held.height(), end - start);
16915            assert!(!shares_storage(held, &other));
16916            assert!(!shares_storage(state.df.as_ref().unwrap(), &other));
16917            let ids = held.column("id").unwrap().i64().unwrap();
16918            assert_eq!(ids.get(0), Some(state.buffered_start() as i64));
16919        }
16920    }
16921
16922    #[test]
16923    fn the_files_holding_a_range_of_rows() {
16924        let offsets = [0, 100, 100, 250, 400];
16925        assert_eq!(files_holding(&offsets, 0, 10), Some((0, 0)));
16926        assert_eq!(
16927            files_holding(&offsets, 95, 10),
16928            Some((0, 2)),
16929            "the empty file is skipped"
16930        );
16931        assert_eq!(files_holding(&offsets, 100, 10), Some((2, 2)));
16932        assert_eq!(files_holding(&offsets, 390, 50), Some((3, 3)));
16933        assert_eq!(files_holding(&offsets, 400, 10), None);
16934        assert_eq!(files_holding(&[0], 0, 10), None);
16935    }
16936
16937    #[test]
16938    fn a_buffer_over_many_small_files_opens_a_few() {
16939        // A thousand files of ten rows each.
16940        let offsets: Vec<usize> = (0..=1000).map(|i| i * 10).collect();
16941        // A window of 2,000 rows around row 5,000 spans 200 files; it keeps 16.
16942        let (start, end) = limit_files(&offsets, 5_000, 5_040, 4_000, 6_000, 16);
16943        assert_eq!(
16944            files_holding(&offsets, start, end - start).map(|(a, b)| b - a + 1),
16945            Some(16)
16946        );
16947        assert!(
16948            start <= 5_000 && 5_040 <= end,
16949            "the view stays: {start}..{end}"
16950        );
16951        // A view spanning more files than the limit keeps all of them.
16952        let (start, end) = limit_files(&offsets, 0, 400, 0, 400, 16);
16953        assert_eq!((start, end), (0, 400));
16954        // Few files: unchanged.
16955        assert_eq!(limit_files(&offsets, 0, 40, 0, 100, 16), (0, 100));
16956    }
16957
16958    /// A dataset written a file a day is mostly empty files in its quiet years; a window
16959    /// opens only the files with rows, and the limit counts only those (#659).
16960    #[test]
16961    fn a_window_passes_over_empty_files() {
16962        // Every other file empty: file 2i holds rows 10i..10i+10.
16963        let offsets: Vec<usize> = (0..=1000_usize).map(|i| i.div_ceil(2) * 10).collect();
16964        let (first, last) = files_holding(&offsets, 0, 40).unwrap();
16965        assert_eq!((first, last), (0, 6));
16966        assert_eq!(files_with_rows(&offsets, first, last), vec![0, 2, 4, 6]);
16967        let (start, end) = limit_files(&offsets, 0, 40, 0, 2_000, 16);
16968        let (first, last) = files_holding(&offsets, start, end - start).unwrap();
16969        assert_eq!(
16970            files_with_rows(&offsets, first, last).len(),
16971            16,
16972            "sixteen files with rows, not eight and the empty ones between"
16973        );
16974    }
16975
16976    /// The two checks that read footers rather than values: a column a file never had,
16977    /// and a column a file holds in a type the scan cannot read.
16978    ///
16979    /// Both are invisible to every measurement over values — an absent cell arrives as
16980    /// a null and a conflicting one is not read at all — so the only way to test them
16981    /// is through a dataset whose files genuinely disagree.
16982    #[test]
16983    fn absent_columns_and_type_conflicts_are_measured_from_the_footers() {
16984        use crate::data_quality::{DataQualityPlan, ObservationKind, QualityCompute, QualityScope};
16985        use crate::schema_union::{DatasetSchema, SchemaOrigin, union_file_schemas};
16986        use polars::prelude::{DataType, IntoLazy, df};
16987
16988        let urls: Vec<String> = vec!["a".to_string(), "b".to_string(), "c".to_string()];
16989        // `a` agrees with the schema; `b` holds `n` as text, which the scan cannot read
16990        // as the Int64 the majority wrote; `c` has no `fee` column at all.
16991        let scan: FileScan = Arc::new(move |urls: &[String], as_text: &[PlSmallStr]| {
16992            let reading_text = as_text.contains(&PlSmallStr::from("n"));
16993            let frames: Vec<LazyFrame> = urls
16994                .iter()
16995                .map(|url| match url.as_str() {
16996                    "a" => df!(
16997                        "id" => &[0i64, 1, 2],
16998                        "n" => &[10i64, 20, 30],
16999                        "fee" => &[1.5f64, 2.5, 3.5],
17000                        crate::schema_union::DRIFT_COLUMN => &[0u32, 1, 2],
17001                    )
17002                    .unwrap()
17003                    .lazy()
17004                    .with_column(col("n").cast(if reading_text {
17005                        DataType::String
17006                    } else {
17007                        DataType::Int64
17008                    })),
17009                    "b" => {
17010                        let frame = df!(
17011                            "id" => &[3i64, 4],
17012                            "n" => &["sixty", "seventy"],
17013                            "fee" => &[4.5f64, 5.5],
17014                            crate::schema_union::DRIFT_COLUMN => &[3u32, 4],
17015                        )
17016                        .unwrap()
17017                        .lazy();
17018                        if reading_text {
17019                            frame
17020                        } else {
17021                            // Not read from this file at all, as the real scan leaves it.
17022                            frame.with_column(lit(NULL).cast(DataType::Int64).alias("n"))
17023                        }
17024                    }
17025                    _ => df!(
17026                        "id" => &[5i64, 6],
17027                        "n" => &[50i64, 60],
17028                        crate::schema_union::DRIFT_COLUMN => &[5u32, 6],
17029                    )
17030                    .unwrap()
17031                    .lazy()
17032                    // A column the file never had reads as null, which is exactly why
17033                    // no measurement over values can tell it from one.
17034                    .with_column(lit(NULL).cast(DataType::Float64).alias("fee"))
17035                    .select([
17036                        col("id"),
17037                        col("n").cast(if reading_text {
17038                            DataType::String
17039                        } else {
17040                            DataType::Int64
17041                        }),
17042                        col("fee"),
17043                        col(crate::schema_union::DRIFT_COLUMN),
17044                    ]),
17045                })
17046                .collect();
17047            polars::prelude::concat(frames, Default::default())
17048        });
17049
17050        let dataset: DatasetSchema = union_file_schemas(
17051            &[
17052                file_schema(
17053                    &[
17054                        ("id", DataType::Int64),
17055                        ("n", DataType::Int64),
17056                        ("fee", DataType::Float64),
17057                    ],
17058                    3,
17059                ),
17060                file_schema(
17061                    &[
17062                        ("id", DataType::Int64),
17063                        ("n", DataType::String),
17064                        ("fee", DataType::Float64),
17065                    ],
17066                    2,
17067                ),
17068                file_schema(&[("id", DataType::Int64), ("n", DataType::Int64)], 2),
17069            ],
17070            SchemaOrigin::AllFooters(3),
17071        );
17072        let state = DataTableState::from_schema_and_lazyframe(
17073            dataset.schema.clone(),
17074            scan(&urls, &[]).unwrap(),
17075            &crate::OpenOptions::default(),
17076            None,
17077        )
17078        .unwrap()
17079        .with_open(OpenFacts {
17080            remote_source: true,
17081            remote_files: Some(RemoteFiles {
17082                urls: Arc::new(urls.clone()),
17083                scan,
17084                count: Arc::new(|_| Ok(vec![vec![3], vec![2], vec![2]])),
17085                offsets: None,
17086            }),
17087            dataset: Some(DatasetAtOpen {
17088                schema: dataset,
17089                file_rows: vec![3, 2, 2],
17090                files: urls.clone(),
17091            }),
17092            ..Default::default()
17093        });
17094        assert!(state.drifts(), "the three files do not agree");
17095
17096        let (lf, source) = state.data_quality_source_scan();
17097        let mut source = source.expect("every file is counted, so rows map to files");
17098        source.conflict_scan = state.quality_conflict_scan();
17099        let lf = crate::data_quality::prepare_source_quality_scan(lf, Some(&source)).unwrap();
17100        let plan = DataQualityPlan {
17101            scope: QualityScope::WholeSource,
17102            compute: QualityCompute::Full,
17103            ..DataQualityPlan::default()
17104        };
17105        let results =
17106            crate::data_quality::compute_data_quality(&lf, Some(7), &plan, Some(&source), false)
17107                .unwrap();
17108
17109        let absent = results
17110            .observations
17111            .iter()
17112            .find(|observation| observation.kind == ObservationKind::Absent)
17113            .expect("`fee` is absent from the third file");
17114        assert_eq!(absent.column, "fee");
17115        assert_eq!(
17116            (absent.affected_rows, absent.evaluated_rows),
17117            (2, 7),
17118            "the third file's two rows, out of the source's seven"
17119        );
17120        assert_eq!(
17121            absent
17122                .files
17123                .iter()
17124                .map(|file| file.number)
17125                .collect::<Vec<_>>(),
17126            vec![3],
17127            "named by the number the Scope page gives it"
17128        );
17129        assert_eq!(
17130            absent.fact, "1 of 3 files has no such column",
17131            "every footer was read, so the count is a total rather than a floor"
17132        );
17133
17134        let conflict = results
17135            .observations
17136            .iter()
17137            .find(|observation| observation.kind == ObservationKind::TypeConflict)
17138            .expect("`n` is text in the second file");
17139        assert_eq!(conflict.column, "n");
17140        assert_eq!((conflict.affected_rows, conflict.evaluated_rows), (2, 7));
17141        let file = conflict.files.first().expect("the file that disagrees");
17142        assert_eq!(file.number, 2);
17143        assert_eq!(file.stored_type.as_deref(), Some("str"));
17144        assert_eq!(
17145            file.examples,
17146            vec!["sixty".to_string(), "seventy".to_string()],
17147            "the values the conflict hides, read at the type that file wrote"
17148        );
17149
17150        // The same two checks at the budget that reads no values at all: the footers
17151        // were read when the dataset opened, so there is nothing left to pay for.
17152        let metadata = crate::data_quality::compute_data_quality(
17153            &lf,
17154            Some(7),
17155            &DataQualityPlan {
17156                scope: QualityScope::WholeSource,
17157                compute: QualityCompute::Metadata,
17158                ..DataQualityPlan::default()
17159            },
17160            Some(&source),
17161            false,
17162        )
17163        .unwrap();
17164        assert_eq!(metadata.evaluated_rows, 0, "no value was read");
17165        assert_eq!(
17166            metadata
17167                .observations
17168                .iter()
17169                .map(|observation| (observation.kind, observation.affected_rows))
17170                .collect::<Vec<_>>(),
17171            vec![
17172                (ObservationKind::Absent, 2),
17173                (ObservationKind::TypeConflict, 2),
17174            ],
17175            "both are reported without reading a value"
17176        );
17177
17178        // The drill-in is the files themselves: an absent cell has no value to filter.
17179        let scope = absent.evidence_scope().expect("a scope, not a predicate");
17180        assert_eq!(scope, QualityScope::SourceFiles(vec![3]));
17181        assert!(absent.evidence_predicate().is_none());
17182        let evidence = state
17183            .quality_evidence_view(&scope, lit(true))
17184            .expect("the rows the third file contributed");
17185        let rows = collect_lazy(evidence.lf.clone(), false).unwrap();
17186        assert_eq!(
17187            rows.column("id")
17188                .unwrap()
17189                .i64()
17190                .unwrap()
17191                .into_no_null_iter()
17192                .collect::<Vec<_>>(),
17193            vec![5, 6],
17194            "the file that has no `fee`, and only that file"
17195        );
17196        assert!(
17197            rows.column(crate::schema_union::DRIFT_COLUMN).is_err(),
17198            "the hidden scan index is never handed back as user data"
17199        );
17200    }
17201
17202    /// The local route to the values a type conflict hides: real Parquet files that
17203    /// disagree, read through `lenient_scan` rather than through a dataset's own scan
17204    /// closure. The remote route above shares none of this code.
17205    #[test]
17206    fn a_local_dataset_reads_the_values_a_type_conflict_hides() {
17207        use crate::data_quality::{DataQualityPlan, ObservationKind, QualityCompute, QualityScope};
17208        use crate::schema_union::{DatasetSchema, SchemaOrigin, union_file_schemas};
17209        use polars::prelude::{DataType, ParquetWriter, df};
17210
17211        let dir = tempfile::tempdir().unwrap();
17212        let write = |name: &str, mut frame: polars::prelude::DataFrame| -> String {
17213            let path = dir.path().join(name);
17214            let file = std::fs::File::create(&path).unwrap();
17215            ParquetWriter::new(file).finish(&mut frame).unwrap();
17216            path.to_string_lossy().to_string()
17217        };
17218        // `n` is an integer in the first file and text in the second, so the scan reads
17219        // it as Int64 and leaves the second file's values behind entirely.
17220        let files = vec![
17221            write(
17222                "a.parquet",
17223                df!("id" => &[0i64, 1, 2], "n" => &[10i64, 20, 30]).unwrap(),
17224            ),
17225            write(
17226                "b.parquet",
17227                df!("id" => &[3i64, 4], "n" => &["sixty", "seventy"]).unwrap(),
17228            ),
17229        ];
17230        let dataset: DatasetSchema = union_file_schemas(
17231            &[
17232                file_schema(&[("id", DataType::Int64), ("n", DataType::Int64)], 3),
17233                file_schema(&[("id", DataType::Int64), ("n", DataType::String)], 2),
17234            ],
17235            SchemaOrigin::AllFooters(2),
17236        );
17237        let file_rows = vec![3usize, 2];
17238        let drift = crate::schema_union::ScanDrift::new(&files, &dataset, &file_rows);
17239        let lf = crate::schema_union::lenient_scan(
17240            &files,
17241            dataset.schema.clone(),
17242            None,
17243            drift.as_ref(),
17244            &[],
17245        )
17246        .unwrap();
17247        let state = DataTableState::from_schema_and_lazyframe(
17248            dataset.schema.clone(),
17249            lf,
17250            &crate::OpenOptions::default(),
17251            None,
17252        )
17253        .unwrap()
17254        .with_open(OpenFacts {
17255            dataset: Some(DatasetAtOpen {
17256                schema: dataset,
17257                file_rows,
17258                files,
17259            }),
17260            ..Default::default()
17261        });
17262        assert!(state.drifts(), "the two files disagree on `n`");
17263        assert_eq!(
17264            state.quality_conflict_reads(),
17265            1,
17266            "one column, in one file, before anything runs"
17267        );
17268
17269        let (lf, source) = state.data_quality_source_scan();
17270        let mut source = source.expect("every file is counted");
17271        source.conflict_scan = state.quality_conflict_scan();
17272        let lf = crate::data_quality::prepare_source_quality_scan(lf, Some(&source)).unwrap();
17273        let plan = DataQualityPlan {
17274            scope: QualityScope::WholeSource,
17275            compute: QualityCompute::Full,
17276            ..DataQualityPlan::default()
17277        };
17278        let results =
17279            crate::data_quality::compute_data_quality(&lf, Some(5), &plan, Some(&source), false)
17280                .unwrap();
17281        let conflict = results
17282            .observations
17283            .iter()
17284            .find(|observation| observation.kind == ObservationKind::TypeConflict)
17285            .expect("`n` is text in the second file");
17286        let file = conflict.files.first().expect("the file that disagrees");
17287        assert_eq!(file.number, 2);
17288        assert_eq!(
17289            file.examples,
17290            vec!["sixty".to_string(), "seventy".to_string()],
17291            "read at the type that file wrote, not as the null the scan hands back"
17292        );
17293    }
17294
17295    /// A remote dataset reads its column as text too, and its windowed reads with it.
17296    ///
17297    /// The remote branch takes a different route: the scan closure it was opened with
17298    /// is asked again with the column named, and `buffer_lf` passes the same names on
17299    /// every later window. Miss that second half and a cloud dataset would read the
17300    /// first screen as text and the next one as it was, disagreeing with its own
17301    /// schema.
17302    #[test]
17303    fn a_remote_dataset_reads_a_conflicting_column_as_text_on_every_window() {
17304        use crate::schema_union::{DatasetSchema, SchemaOrigin, union_file_schemas};
17305        use polars::prelude::{DataType, IntoLazy, df};
17306
17307        let urls: Vec<String> = vec!["a".to_string(), "b".to_string()];
17308        // The scan the dataset was opened with, standing in for the cloud one: the
17309        // first file holds `n` as an integer and the second as text.
17310        let scan: FileScan = Arc::new(move |urls: &[String], as_text: &[PlSmallStr]| {
17311            let frames: Vec<LazyFrame> = urls
17312                .iter()
17313                .map(|url| {
17314                    // The hidden row index the real scan stamps, numbered from where
17315                    // the file's rows begin in the dataset so a window keeps its place.
17316                    let frame = if url == "a" {
17317                        df!(
17318                            "id" => &[0i64, 1, 2],
17319                            "n" => &[10i64, 20, 30],
17320                            crate::schema_union::DRIFT_COLUMN => &[0u32, 1, 2],
17321                        )
17322                        .unwrap()
17323                    } else {
17324                        df!(
17325                            "id" => &[3i64, 4],
17326                            "n" => &["sixty", "seventy"],
17327                            crate::schema_union::DRIFT_COLUMN => &[3u32, 4],
17328                        )
17329                        .unwrap()
17330                    };
17331                    let lf = frame.lazy();
17332                    if as_text.contains(&PlSmallStr::from("n")) {
17333                        lf.with_column(col("n").cast(DataType::String))
17334                    } else if url == "a" {
17335                        lf
17336                    } else {
17337                        // Not read from this file at all, as the real scan leaves it.
17338                        lf.with_column(lit(NULL).cast(DataType::Int64).alias("n"))
17339                    }
17340                })
17341                .collect();
17342            polars::prelude::concat(frames, Default::default())
17343        });
17344
17345        let dataset: DatasetSchema = union_file_schemas(
17346            &[
17347                file_schema(&[("id", DataType::Int64), ("n", DataType::Int64)], 3),
17348                file_schema(&[("id", DataType::Int64), ("n", DataType::String)], 2),
17349            ],
17350            SchemaOrigin::AllFooters(2),
17351        );
17352        // The dataset's schema, which does not name the hidden row index the frame
17353        // carries — as the real open does it.
17354        let mut state = DataTableState::from_schema_and_lazyframe(
17355            dataset.schema.clone(),
17356            scan(&urls, &[]).unwrap(),
17357            &crate::OpenOptions::default(),
17358            None,
17359        )
17360        .unwrap()
17361        .with_open(OpenFacts {
17362            remote_source: true,
17363            remote_files: Some(RemoteFiles {
17364                urls: Arc::new(urls.clone()),
17365                scan,
17366                count: Arc::new(|_| Ok(vec![vec![3], vec![2]])),
17367                offsets: None,
17368            }),
17369            dataset: Some(DatasetAtOpen {
17370                schema: dataset,
17371                file_rows: vec![3, 2],
17372                files: urls.clone(),
17373            }),
17374            ..Default::default()
17375        });
17376        let groups = (state.remote_files_counter().unwrap())(&Default::default()).unwrap();
17377        assert!(state.count_landed(state.len_generation(), 5, Some(&groups)));
17378        assert!(state.drifts(), "the two files disagree on `n`");
17379
17380        assert!(
17381            state.read_column_as_text("n").unwrap(),
17382            "the offer is taken"
17383        );
17384        assert_eq!(
17385            state.schema.get("n"),
17386            Some(&DataType::String),
17387            "the column is text now"
17388        );
17389
17390        // Every window, not only the first: the second file's rows are on their own
17391        // page, and they are the ones the conflict was hiding.
17392        let text = |state: &DataTableState, start: usize, rows: usize| -> Vec<String> {
17393            collect_lazy(state.buffer_lf(start, rows).unwrap(), false)
17394                .unwrap()
17395                .column("n")
17396                .unwrap()
17397                .str()
17398                .unwrap()
17399                .iter()
17400                .map(|value| value.unwrap_or("null").to_string())
17401                .collect()
17402        };
17403        assert_eq!(text(&state, 0, 3), ["10", "20", "30"]);
17404        assert_eq!(
17405            text(&state, 3, 2),
17406            ["sixty", "seventy"],
17407            "the page that needed the text read most"
17408        );
17409    }
17410
17411    #[test]
17412    fn a_counted_remote_dataset_reads_only_the_files_a_buffer_needs() {
17413        use polars::prelude::IntoLazy;
17414        let part = |from: i32| {
17415            polars::df!("n" => (from..from + 100).collect::<Vec<i32>>())
17416                .unwrap()
17417                .lazy()
17418        };
17419        let urls: Vec<String> = (0..5).map(|i| format!("file{i}")).collect();
17420        let asked = Arc::new(std::sync::Mutex::new(Vec::<Vec<String>>::new()));
17421        let scan: FileScan = {
17422            let asked = asked.clone();
17423            Arc::new(move |urls: &[String], _as_text: &[PlSmallStr]| {
17424                asked.lock().unwrap().push(urls.to_vec());
17425                let frames: Vec<LazyFrame> = urls
17426                    .iter()
17427                    .map(|u| part(u.trim_start_matches("file").parse::<i32>().unwrap() * 100))
17428                    .collect();
17429                polars::prelude::concat(frames, Default::default())
17430            })
17431        };
17432        let full = scan(&urls, &[]).unwrap();
17433        asked.lock().unwrap().clear();
17434        let mut state = DataTableState::from_lazyframe(full, &crate::OpenOptions::default())
17435            .unwrap()
17436            .with_open(OpenFacts {
17437                remote_source: true,
17438                remote_files: Some(RemoteFiles {
17439                    urls: Arc::new(urls),
17440                    scan,
17441                    count: Arc::new(|_| Ok(vec![vec![50, 50]; 5])),
17442                    offsets: None,
17443                }),
17444                ..Default::default()
17445            });
17446        let groups = (state.remote_files_counter().unwrap())(&Default::default()).unwrap();
17447        assert!(state.count_landed(state.len_generation(), 500, Some(&groups)));
17448        assert_eq!(state.num_rows_if_valid(), Some(500));
17449        assert!(state.remote_files_counter().is_none(), "counted once");
17450
17451        let df = collect_lazy(state.buffer_lf(350, 20).unwrap(), false).unwrap();
17452        let values: Vec<i32> = df
17453            .column("n")
17454            .unwrap()
17455            .i32()
17456            .unwrap()
17457            .into_no_null_iter()
17458            .collect();
17459        assert_eq!(values, (350..370).collect::<Vec<i32>>());
17460        assert_eq!(*asked.lock().unwrap(), vec![vec!["file3".to_string()]]);
17461
17462        asked.lock().unwrap().clear();
17463        let df = collect_lazy(state.buffer_lf(190, 20).unwrap(), false).unwrap();
17464        assert_eq!(df.height(), 20);
17465        assert_eq!(
17466            *asked.lock().unwrap(),
17467            vec![vec!["file1".to_string(), "file2".to_string()]],
17468            "a range across a boundary reads both files"
17469        );
17470    }
17471
17472    /// What an open finds arrives in one step, in the order it depends on: a many-file
17473    /// dataset's row groups land after its files, so they count it and place each page
17474    /// in its files; a single object's footer counts it.
17475    #[test]
17476    fn an_open_s_findings_arrive_together() {
17477        let lf = || df!("a" => (0..100i32).collect::<Vec<_>>()).unwrap().lazy();
17478        let many = DataTableState::from_lazyframe(lf(), &crate::OpenOptions::default())
17479            .unwrap()
17480            .with_open(OpenFacts {
17481                remote_source: true,
17482                remote_files: Some(RemoteFiles {
17483                    urls: Arc::new(vec!["one".to_string(), "two".to_string()]),
17484                    scan: Arc::new(move |_: &[String], _: &[PlSmallStr]| Ok(lf())),
17485                    count: Arc::new(|_| Err("counted at the open".to_string())),
17486                    offsets: None,
17487                }),
17488                row_groups: vec![vec![30, 30], vec![40]],
17489                ..Default::default()
17490            });
17491        assert_eq!(many.num_rows_if_valid(), Some(100));
17492        assert_eq!(
17493            many.files_a_page_reads(50, 20),
17494            Some(2),
17495            "rows 50..70 span both"
17496        );
17497        assert!(
17498            many.remote_files_counter().is_none(),
17499            "nothing left to count"
17500        );
17501
17502        let one = DataTableState::from_lazyframe(lf(), &crate::OpenOptions::default())
17503            .unwrap()
17504            .with_open(OpenFacts {
17505                remote_source: true,
17506                row_groups: vec![vec![60, 40]],
17507                ..Default::default()
17508            });
17509        assert_eq!(one.num_rows_if_valid(), Some(100));
17510        assert!(one.is_remote_source());
17511    }
17512
17513    #[test]
17514    fn a_remote_source_buffers_one_window_and_pages_inside_it_for_free() {
17515        // Every buffer fill of an object-store scan downloads whole row groups, so the
17516        // buffer is one window of `max_buffered_rows` rather than a few pages: paging
17517        // inside it asks for nothing, and a jump asks once.
17518        let lf = df!("a" => &[0i32]).unwrap().lazy();
17519        let mut state = DataTableState::new(lf, None, None, Some(10_000), None, true)
17520            .unwrap()
17521            .with_open(OpenFacts {
17522                remote_source: true,
17523                ..Default::default()
17524            });
17525        state.num_rows = 1_000_000;
17526        state.num_rows_valid = true;
17527        state.visible_rows = 40;
17528        let window = |start: usize| Fill {
17529            df: df!("a" => (0..10_000).collect::<Vec<i32>>()).unwrap(),
17530            buffer_start: start,
17531            buffer_end: start + 10_000,
17532            num_rows: 1_000_000,
17533            count_known: true,
17534        };
17535
17536        let request = state.prepare_async_collect(None).expect("first fill");
17537        assert_eq!((request.buffer_start, request.buffer_end), (0, 10_000));
17538        state.land(window(0));
17539        for _ in 0..20 {
17540            assert!(!state.page_down(), "a page inside the window needs no fill");
17541        }
17542
17543        assert!(state.scroll_to_end());
17544        let request = state
17545            .prepare_async_collect(None)
17546            .expect("the jump fills once");
17547        assert_eq!(
17548            (request.buffer_start, request.buffer_end),
17549            (990_000, 1_000_000)
17550        );
17551        state.land(window(990_000));
17552
17553        assert!(state.scroll_to_start(), "Home after End must fill again");
17554        let request = state
17555            .prepare_async_collect(None)
17556            .expect("one fill at the top");
17557        assert_eq!((request.buffer_start, request.buffer_end), (0, 10_000));
17558    }
17559
17560    #[test]
17561    fn align_to_row_groups_takes_the_view_groups_whole_and_lookahead_within_the_cap() {
17562        let offsets = [0, 1_000_000, 2_000_000, 3_000_000, 3_500_000];
17563        // A window inside one group is that group, whatever the row cap.
17564        assert_eq!(
17565            align_to_row_groups(&offsets, 50, 97, 0, 100_000, 100_000),
17566            (0, 1_000_000)
17567        );
17568        // A view straddling a boundary takes both groups, over the cap.
17569        assert_eq!(
17570            align_to_row_groups(&offsets, 999_980, 1_000_020, 950_000, 1_050_000, 100_000),
17571            (0, 2_000_000)
17572        );
17573        // Groups the window reaches into come along while they fit, the one ahead first.
17574        assert_eq!(
17575            align_to_row_groups(
17576                &offsets, 1_500_000, 1_500_047, 950_000, 2_050_000, 2_000_000
17577            ),
17578            (1_000_000, 3_000_000)
17579        );
17580        assert_eq!(
17581            align_to_row_groups(&offsets, 1_500_000, 1_500_047, 950_000, 2_050_000, 0),
17582            (0, 3_000_000)
17583        );
17584        // The last, short group; and a window past the data is clamped to it.
17585        assert_eq!(
17586            align_to_row_groups(
17587                &offsets, 3_400_000, 3_400_047, 3_350_000, 3_450_000, 100_000
17588            ),
17589            (3_000_000, 3_500_000)
17590        );
17591        // No groups known: the window is left alone.
17592        assert_eq!(align_to_row_groups(&[0], 5, 10, 0, 100, 50), (0, 100));
17593    }
17594
17595    #[test]
17596    fn a_remote_object_is_read_inside_its_row_group() {
17597        // Ten 1M-row groups with the default 100k-row cap: the window is planned inside
17598        // the group the view is in, so it never pulls the next group before the view
17599        // reaches it; crossing fetches rows of the next group alone, stitched on to the
17600        // ones on hand and trimmed back to the cap.
17601        const G: usize = 1_000_000;
17602        const CAP: usize = DEFAULT_MAX_BUFFERED_ROWS;
17603        let lf = df!("a" => &[0i32]).unwrap().lazy();
17604        let mut state = DataTableState::new(lf, None, None, None, None, true)
17605            .unwrap()
17606            .with_open(OpenFacts {
17607                remote_source: true,
17608                row_groups: vec![vec![G; 10]],
17609                ..Default::default()
17610            });
17611        assert_eq!(state.num_rows, 10 * G);
17612        state.visible_rows = 40;
17613        let rows = |start: usize, end: usize| Fill {
17614            df: df!("a" => (start as i32..end as i32).collect::<Vec<i32>>()).unwrap(),
17615            buffer_start: start,
17616            buffer_end: end,
17617            num_rows: 10 * G,
17618            count_known: true,
17619        };
17620
17621        let request = state.prepare_async_collect(None).expect("first fill");
17622        assert_eq!((request.buffer_start, request.buffer_end), (0, CAP));
17623        state.land(rows(0, CAP));
17624        for _ in 0..20 {
17625            assert!(!state.page_down(), "a page inside the window needs no fill");
17626        }
17627
17628        // A jump to the end of group 0 is clipped to it: group 1 is not touched yet.
17629        assert!(state.scroll_to(G - 60));
17630        let request = state
17631            .prepare_async_collect(None)
17632            .expect("the end of group 0");
17633        assert_eq!((request.buffer_start, request.buffer_end), (G - CAP, G));
17634        state.land(rows(G - CAP, G));
17635
17636        // A view straddling the boundary fetches rows of group 1 alone.
17637        assert!(state.scroll_to(G - 20));
17638        let request = state.prepare_async_collect(None).expect("into group 1");
17639        assert_eq!((request.buffer_start, request.buffer_end), (G, G + CAP / 2));
17640        state.land(rows(G, G + CAP / 2));
17641        let (held_start, held_end) = (state.buffered_start(), state.buffered_end());
17642        assert!(
17643            held_start <= G - 20 && G + 20 <= held_end,
17644            "the view is on hand"
17645        );
17646        assert_eq!(held_end - held_start, CAP, "trimmed back to the cap");
17647        let held = state.buffered_df.as_ref().unwrap();
17648        assert_eq!(held.height(), CAP);
17649        assert_eq!(
17650            held.column("a").unwrap().i32().unwrap().get(G - held_start),
17651            Some(G as i32),
17652            "stitched in order"
17653        );
17654        assert!(!state.page_down());
17655
17656        // End and Home are one window each.
17657        assert!(state.scroll_to_end());
17658        let request = state.prepare_async_collect(None).expect("the last window");
17659        assert_eq!(
17660            (request.buffer_start, request.buffer_end),
17661            (10 * G - CAP, 10 * G)
17662        );
17663        state.land(rows(10 * G - CAP, 10 * G));
17664        assert!(state.scroll_to_start());
17665        let request = state.prepare_async_collect(None).expect("the first window");
17666        assert_eq!((request.buffer_start, request.buffer_end), (0, CAP));
17667    }
17668
17669    #[test]
17670    fn small_row_groups_are_fetched_whole() {
17671        // 40k-row groups under a 100k cap: a fill is whole groups, as many as fit,
17672        // and crossing into the next fetches exactly that group.
17673        const G: usize = 40_000;
17674        let lf = df!("a" => &[0i32]).unwrap().lazy();
17675        let mut state = DataTableState::new(lf, None, None, None, None, true)
17676            .unwrap()
17677            .with_open(OpenFacts {
17678                remote_source: true,
17679                row_groups: vec![vec![G; 25]],
17680                ..Default::default()
17681            });
17682        state.visible_rows = 40;
17683        let rows = |start: usize, end: usize| Fill {
17684            df: df!("a" => (start as i32..end as i32).collect::<Vec<i32>>()).unwrap(),
17685            buffer_start: start,
17686            buffer_end: end,
17687            num_rows: 25 * G,
17688            count_known: true,
17689        };
17690
17691        let request = state.prepare_async_collect(None).expect("first fill");
17692        assert_eq!((request.buffer_start, request.buffer_end), (0, 2 * G));
17693        state.land(rows(0, 2 * G));
17694
17695        assert!(state.scroll_to(2 * G - 20));
17696        let request = state.prepare_async_collect(None).expect("the next group");
17697        assert_eq!((request.buffer_start, request.buffer_end), (2 * G, 3 * G));
17698        state.land(rows(2 * G, 3 * G));
17699        let (held_start, held_end) = (state.buffered_start(), state.buffered_end());
17700        assert!(held_start <= 2 * G - 20 && 2 * G + 20 <= held_end);
17701        assert!(held_end - held_start <= DEFAULT_MAX_BUFFERED_ROWS);
17702    }
17703
17704    #[test]
17705    fn a_wide_schema_is_budgeted_before_the_collect() {
17706        // 1,000 Float64 columns are 8,000 bytes a row: a 512 MB budget allows 67,108
17707        // rows, so the planned window is that and not the 100k row cap. The rows are
17708        // never materialized only to be trimmed after the collect.
17709        let columns: Vec<Column> = (0..1000)
17710            .map(|i| Series::new(format!("f{i}").into(), &[0.0f64]).into())
17711            .collect();
17712        let lf = DataFrame::new(1, columns).unwrap().lazy();
17713        let mut state = DataTableState::new(lf, None, None, None, None, true)
17714            .unwrap()
17715            .with_open(OpenFacts {
17716                remote_source: true,
17717                row_groups: vec![vec![1_000_000; 3]],
17718                ..Default::default()
17719            });
17720        assert_eq!(
17721            estimate_bytes_per_row(&state.schema, &state.column_order, &[]),
17722            8_000
17723        );
17724        state.visible_rows = 40;
17725
17726        let request = state.prepare_async_collect(None).expect("first fill");
17727        let planned = request.buffer_end - request.buffer_start;
17728        assert!(
17729            planned <= 512 * 1024 * 1024 / 8_000,
17730            "planned {planned} rows over the byte budget"
17731        );
17732        assert!(planned >= 40, "never below a screen");
17733        assert_eq!(
17734            request.buffer_start, 0,
17735            "a window at the top starts at the top"
17736        );
17737
17738        // The screen is the floor, whatever the budget.
17739        let tiny = |bytes: usize| {
17740            let mut tiny = DataTableState::new(
17741                df!("a" => &["x".repeat(2_000)]).unwrap().lazy(),
17742                None,
17743                None,
17744                None,
17745                Some(1),
17746                true,
17747            )
17748            .unwrap()
17749            .with_open(OpenFacts {
17750                column_bytes: vec![("a".to_string(), bytes)],
17751                ..Default::default()
17752            });
17753            tiny.visible_rows = 40;
17754            tiny.byte_cap_rows()
17755        };
17756        assert_eq!(tiny(2_000), 1024 * 1024 / 2_016);
17757        assert_eq!(tiny(1 << 20), 40);
17758    }
17759
17760    #[test]
17761    fn a_collected_buffer_measures_the_next_plan() {
17762        // The schema guesses 40 bytes for a string; the first buffer shows the strings
17763        // are 2 KB, and the next window is planned on that.
17764        let big: Vec<String> = (0..100).map(|_| "z".repeat(2_000)).collect();
17765        let lf = df!("a" => &big).unwrap().lazy();
17766        let mut state = DataTableState::new(lf, None, None, None, Some(1), true).unwrap();
17767        state.num_rows = 1_000_000;
17768        state.num_rows_valid = true;
17769        state.visible_rows = 40;
17770        let guessed = state.byte_cap_rows();
17771        assert_eq!(guessed, 1024 * 1024 / STRING_BYTES_GUESS);
17772        state.land(Fill {
17773            df: df!("a" => &big).unwrap(),
17774            buffer_start: 0,
17775            buffer_end: 100,
17776            num_rows: 1_000_000,
17777            count_known: true,
17778        });
17779        let measured = state.byte_cap_rows();
17780        assert!(
17781            (400..=600).contains(&measured),
17782            "about 1 MB / 2 KB rows, got {measured}"
17783        );
17784    }
17785
17786    #[test]
17787    fn a_filtered_remote_scan_falls_back_to_the_page_window() {
17788        // `filter(..).slice(0, N)` stops at the first N matches, so a window of a few
17789        // pages stops at the first row group with any; the 100k window read forty.
17790        use crate::filter_modal::{FilterOperator, FilterStatement, LogicalOperator};
17791        let lf = df!("a" => (0..1_000i32).collect::<Vec<i32>>())
17792            .unwrap()
17793            .lazy();
17794        let mut state = DataTableState::new(lf, None, None, Some(10_000), None, true)
17795            .unwrap()
17796            .with_open(OpenFacts {
17797                remote_source: true,
17798                row_groups: vec![vec![500, 500]],
17799                parquet_count_dir: Some(PathBuf::from("/hive")),
17800                ..Default::default()
17801            });
17802        state.visible_rows = 40;
17803        state.defer_collect = true;
17804
17805        let request = state.prepare_async_collect(None).expect("first fill");
17806        assert_eq!(
17807            (request.buffer_start, request.buffer_end),
17808            (0, 1_000),
17809            "both groups fit the remote window"
17810        );
17811
17812        state.filter(vec![FilterStatement {
17813            columns: Vec::new(),
17814            column: "a".to_string(),
17815            operator: FilterOperator::Gt,
17816            value: "990".to_string(),
17817            logical_op: LogicalOperator::And,
17818        }]);
17819        let request = state.prepare_async_collect(None).expect("filtered fill");
17820        assert_eq!(
17821            (request.buffer_start, request.buffer_end),
17822            (0, 7 * 40),
17823            "a page plus three either side, not the remote window"
17824        );
17825
17826        state.filter(Vec::new());
17827        assert!(
17828            state.remote_window(),
17829            "with the filters cleared the frame is the scan as loaded again"
17830        );
17831        assert_eq!(
17832            state.num_rows_if_valid(),
17833            Some(1_000),
17834            "and its footer answers the count"
17835        );
17836
17837        // The local hive footer count is gated the same way.
17838        assert_eq!(state.parquet_count_dir(), Some(PathBuf::from("/hive")));
17839        state.sort(vec!["a".to_string()], true);
17840        assert!(state.parquet_count_dir().is_none());
17841        state.sort(Vec::new(), true);
17842        assert_eq!(state.parquet_count_dir(), Some(PathBuf::from("/hive")));
17843        state.reverse();
17844        assert!(!state.remote_window(), "reversed is not as loaded");
17845    }
17846
17847    #[test]
17848    fn a_count_below_the_view_brings_the_view_back() {
17849        // A filter applied deep in the data: the frame turns out to have 100 rows and
17850        // the view was at 9,990. It comes back to the data and asks for a fill.
17851        let lf = df!("a" => (0..10_000i32).collect::<Vec<i32>>())
17852            .unwrap()
17853            .lazy();
17854        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
17855        state.visible_rows = 10;
17856        state.num_rows = 10_000;
17857        state.num_rows_valid = true;
17858        assert!(state.scroll_to_end());
17859        assert_eq!(state.start_row, 9_990);
17860        state.set_num_rows(100);
17861        assert_eq!(state.start_row, 90);
17862        assert!(state.needs_recollect);
17863
17864        // A slice deep in the frame that found nothing is not the count.
17865        state.needs_recollect = false;
17866        state.num_rows_valid = false;
17867        state.start_row = 9_990;
17868        state.land(Fill {
17869            df: df!("a" => Vec::<i32>::new()).unwrap(),
17870            buffer_start: 9_990,
17871            buffer_end: 10_060,
17872            num_rows: 10_060,
17873            count_known: false,
17874        });
17875        assert!(
17876            !state.num_rows_valid,
17877            "only the count can say where it ends"
17878        );
17879    }
17880
17881    #[test]
17882    fn a_stale_stitch_keeps_the_rows_on_hand() {
17883        // A fill planned to run on from rows since replaced neither abuts what is held
17884        // nor shows the view: installing it would draw rows under the wrong numbers.
17885        const G: usize = 1_000_000;
17886        let lf = df!("a" => &[0i32]).unwrap().lazy();
17887        let mut state = DataTableState::new(lf, None, None, None, None, true)
17888            .unwrap()
17889            .with_open(OpenFacts {
17890                remote_source: true,
17891                row_groups: vec![vec![G; 10]],
17892                ..Default::default()
17893            });
17894        state.visible_rows = 40;
17895        let rows = |start: usize, end: usize| Fill {
17896            df: df!("a" => (start as i32..end as i32).collect::<Vec<i32>>()).unwrap(),
17897            buffer_start: start,
17898            buffer_end: end,
17899            num_rows: 10 * G,
17900            count_known: true,
17901        };
17902        assert!(state.scroll_to(G - 60));
17903        let request = state
17904            .prepare_async_collect(None)
17905            .expect("the end of group 0");
17906        state.land(rows(request.buffer_start, request.buffer_end));
17907        assert!(state.scroll_to(G - 20));
17908        let stitch = state.prepare_async_collect(None).expect("into group 1");
17909        assert_eq!(stitch.buffer_start, G);
17910
17911        // Meanwhile a synchronous collect moved the view and replaced the buffer.
17912        assert!(state.scroll_to(500_000));
17913        state.land(rows(450_000, 550_000));
17914        state.needs_recollect = false;
17915
17916        // Fit as the worker did, against the rows on hand when it was planned.
17917        let fill = rows(stitch.buffer_start, stitch.buffer_end);
17918        state.apply_async_collect(stitch.plan.fit(fill.df));
17919        assert_eq!(
17920            (state.buffered_start(), state.buffered_end()),
17921            (450_000, 550_000),
17922            "the rows on hand stay"
17923        );
17924        assert!(state.needs_recollect, "and a fill is asked for");
17925    }
17926
17927    #[test]
17928    fn a_fill_that_holds_the_first_row_is_kept_when_the_view_grew() {
17929        // The terminal grew while a row group was downloading: the fill holds the view's
17930        // first row but not its last. It is installed, and the rest asked for, rather than
17931        // thrown away.
17932        const G: usize = 1_000_000;
17933        let lf = df!("a" => &[0i32]).unwrap().lazy();
17934        let mut state = DataTableState::new(lf, None, None, None, None, true)
17935            .unwrap()
17936            .with_open(OpenFacts {
17937                remote_source: true,
17938                row_groups: vec![vec![G; 10]],
17939                ..Default::default()
17940            });
17941        state.visible_rows = 40;
17942        assert!(state.scroll_to(G - 60));
17943        let request = state
17944            .prepare_async_collect(None)
17945            .expect("the end of group 0");
17946        assert!(request.buffer_end <= G);
17947
17948        state.visible_rows = 120; // resized while the fetch was out
17949        state.needs_recollect = false;
17950        state.land(Fill {
17951            df: df!("a" => (request.buffer_start as i32..request.buffer_end as i32)
17952                .collect::<Vec<i32>>())
17953            .unwrap(),
17954            buffer_start: request.buffer_start,
17955            buffer_end: request.buffer_end,
17956            num_rows: 10 * G,
17957            count_known: true,
17958        });
17959        assert_eq!(
17960            (state.buffered_start(), state.buffered_end()),
17961            (request.buffer_start, request.buffer_end),
17962            "the downloaded rows are kept"
17963        );
17964        assert!(state.needs_recollect, "and the rest of the view is fetched");
17965    }
17966
17967    #[test]
17968    fn the_byte_cap_governs_the_alignment() {
17969        // Rows of a kilobyte under a 64 MB budget allow 66k rows, fewer than the 100k row
17970        // cap; with 40k-row groups the first fill must be group 0 exactly, not a window
17971        // cut through group 1 that a later refill downloads again.
17972        const G: usize = 40_000;
17973        let lf = df!("a" => &["x"]).unwrap().lazy();
17974        let mut state = DataTableState::new(lf, None, None, None, Some(64), true)
17975            .unwrap()
17976            .with_open(OpenFacts {
17977                remote_source: true,
17978                row_groups: vec![vec![G; 5]],
17979                column_bytes: vec![("a".to_string(), 1_000)],
17980                ..Default::default()
17981            });
17982        state.visible_rows = 40;
17983        let cap = state.byte_cap_rows();
17984        assert!(
17985            (G..2 * G).contains(&cap),
17986            "cap {cap} between one and two groups"
17987        );
17988        let rows = |start: usize, end: usize| Fill {
17989            df: df!("a" => (start..end).map(|i| "x".repeat(8 + i % 3)).collect::<Vec<_>>())
17990                .unwrap(),
17991            buffer_start: start,
17992            buffer_end: end,
17993            num_rows: 5 * G,
17994            count_known: true,
17995        };
17996
17997        let request = state.prepare_async_collect(None).expect("first fill");
17998        assert_eq!((request.buffer_start, request.buffer_end), (0, G));
17999        state.land(rows(0, G));
18000
18001        assert!(state.scroll_to(G - 20));
18002        let request = state.prepare_async_collect(None).expect("into group 1");
18003        assert_eq!(request.buffer_start, G, "group 1 alone");
18004        assert!(request.buffer_end <= 2 * G);
18005    }
18006
18007    #[test]
18008    fn a_new_base_is_measured_afresh() {
18009        // The width measured on a buffer of the old frame does not plan the new one.
18010        let big: Vec<String> = (0..100).map(|_| "z".repeat(2_000)).collect();
18011        let lf = df!("a" => &big, "b" => (0..100i32).collect::<Vec<i32>>())
18012            .unwrap()
18013            .lazy();
18014        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18015        state.visible_rows = 10;
18016        state.defer_collect = true;
18017        state.land(Fill {
18018            df: df!("a" => &big, "b" => (0..100i32).collect::<Vec<i32>>()).unwrap(),
18019            buffer_start: 0,
18020            buffer_end: 100,
18021            num_rows: 100,
18022            count_known: true,
18023        });
18024        assert!(state.observed_bytes_per_row.is_some());
18025        state.query("select b".to_string());
18026        assert!(state.observed_bytes_per_row.is_none());
18027        assert_eq!(
18028            state.bytes_per_row(),
18029            4,
18030            "the narrow frame, from its schema"
18031        );
18032    }
18033
18034    #[test]
18035    fn a_nested_column_takes_its_width_from_the_footer() {
18036        let schema = Schema::from_iter([Field::new(
18037            "l".into(),
18038            DataType::List(Box::new(DataType::Float64)),
18039        )]);
18040        let columns = vec!["l".to_string()];
18041        assert_eq!(estimate_bytes_per_row(&schema, &columns, &[]), 64);
18042        assert_eq!(
18043            estimate_bytes_per_row(&schema, &columns, &[("l".to_string(), 800)]),
18044            800
18045        );
18046    }
18047
18048    #[test]
18049    fn a_reset_remote_scan_is_pristine_again() {
18050        // A query makes the frame a predicate over the object, so the row-group window
18051        // and footer count stand down; clearing it brings both back without a len().
18052        let lf = df!("a" => (0..100).collect::<Vec<i32>>()).unwrap().lazy();
18053        let mut state = DataTableState::new(lf, None, None, None, None, true)
18054            .unwrap()
18055            .with_open(OpenFacts {
18056                remote_source: true,
18057                row_groups: vec![vec![60, 40]],
18058                ..Default::default()
18059            });
18060        assert!(state.remote_window());
18061        state.query("select a where a > 50".to_string());
18062        assert!(!state.remote_window());
18063        state.query(String::new());
18064        assert!(state.remote_window());
18065        assert_eq!(state.num_rows_if_valid(), Some(100));
18066    }
18067
18068    #[test]
18069    fn quality_source_scope_ignores_current_query_and_evidence_matches_scope() {
18070        let lf = df!("a" => &[1i32, 2, 3, 4]).unwrap().lazy();
18071        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18072        state.drift_files = vec!["first.parquet".into(), "second.parquet".into()];
18073        state.drift_file_starts = vec![0, 2];
18074        state.query("select a where a > 2".to_string());
18075        let (current, _) = state.data_quality_scan(false);
18076        let (source, context) = state.data_quality_source_scan();
18077        let source =
18078            crate::data_quality::prepare_source_quality_scan(source, context.as_ref()).unwrap();
18079        assert_eq!(current.collect().unwrap().height(), 2);
18080        assert_eq!(source.collect().unwrap().height(), 4);
18081        assert_eq!(state.quality_source_file_count(), 2);
18082        let (raw, mapping) = state.data_quality_source_scan();
18083        let indexed =
18084            crate::data_quality::prepare_source_quality_scan(raw, mapping.as_ref()).unwrap();
18085        let first_file = crate::data_quality::apply_quality_scope(
18086            indexed,
18087            &crate::data_quality::QualityScope::SourceFiles(vec![1]),
18088            mapping.as_ref(),
18089        )
18090        .unwrap()
18091        .collect()
18092        .unwrap();
18093        assert_eq!(first_file.height(), 2);
18094        assert_eq!(
18095            first_file.column("a").unwrap().i32().unwrap().get(0),
18096            Some(1)
18097        );
18098
18099        let evidence = state
18100            .quality_evidence_view(
18101                &crate::data_quality::QualityScope::WholeSource,
18102                col("a").eq(lit(1)),
18103            )
18104            .unwrap();
18105        assert_eq!(evidence.visible_lf().collect().unwrap().height(), 1);
18106        let bounded = state
18107            .quality_evidence_view(
18108                &crate::data_quality::QualityScope::FirstRows(1),
18109                col("a").eq(lit(4)),
18110            )
18111            .unwrap();
18112        assert_eq!(bounded.visible_lf().collect().unwrap().height(), 0);
18113        let file_evidence = state
18114            .quality_evidence_view(
18115                &crate::data_quality::QualityScope::SourceFiles(vec![1]),
18116                col("a").eq(lit(1)),
18117            )
18118            .unwrap();
18119        assert_eq!(file_evidence.visible_lf().collect().unwrap().height(), 1);
18120    }
18121
18122    #[test]
18123    fn source_time_roles_can_use_columns_hidden_by_current_query() {
18124        let lf = df!("a" => &[1i32, 2], "event" => &[20_000i32, 20_001])
18125            .unwrap()
18126            .lazy()
18127            .with_columns([col("event").cast(DataType::Date)]);
18128        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18129        state.query("select a".to_string());
18130        assert!(
18131            state
18132                .quality_temporal_columns(&crate::data_quality::QualityScope::CurrentView)
18133                .is_empty()
18134        );
18135        assert_eq!(
18136            state.quality_temporal_columns(&crate::data_quality::QualityScope::WholeSource),
18137            vec!["event"]
18138        );
18139    }
18140
18141    #[test]
18142    fn binary_columns_are_stubbed_in_display_buffer() {
18143        // Binary columns must not be materialized into the display buffer (their blobs can be huge
18144        // and reading them is what stalls jump-to-end); the buffer holds a stub instead.
18145        let a = Series::new("a".into(), &[1i32, 2, 3]);
18146        let blob = Series::new("blob".into(), &["aaaa", "bbbb", "cccc"])
18147            .cast(&DataType::Binary)
18148            .unwrap();
18149        let lf = DataFrame::new_infer_height(vec![a.into(), blob.into()])
18150            .unwrap()
18151            .lazy();
18152        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18153        state.visible_rows = 10;
18154        state.collect();
18155
18156        let df = state.df.as_ref().expect("display df present");
18157        let col = df.column("blob").expect("blob column present in buffer");
18158        assert_eq!(
18159            col.dtype(),
18160            &DataType::String,
18161            "binary column should be stubbed (not read as binary)"
18162        );
18163        assert_eq!(col.str().unwrap().get(0).unwrap(), binary_stub());
18164        // A non-binary column is untouched.
18165        assert_eq!(df.column("a").unwrap().dtype(), &DataType::Int32);
18166    }
18167
18168    #[test]
18169    fn analysis_describe_stubs_binary_columns_without_reading_blobs() {
18170        // Describe (and the other analysis tools) route the frame through `binary_stub_exprs`
18171        // so binary blobs are never materialized — reading multi-GB blobs across partitions is
18172        // what froze the process. The binary column still appears, as a stub.
18173        let a = Series::new("a".into(), &[1i32, 2, 3]);
18174        let blob = Series::new("blob".into(), &["aaaa", "bbbb", "cccc"])
18175            .cast(&DataType::Binary)
18176            .unwrap();
18177        let lf = DataFrame::new_infer_height(vec![a.into(), blob.into()])
18178            .unwrap()
18179            .lazy();
18180        let state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18181
18182        let analysis_lf = state.lf.clone().select(state.binary_stub_exprs());
18183        let results = crate::statistics::compute_describe_from_lazy(
18184            &analysis_lf,
18185            Some(3),
18186            &crate::sampling::Sample {
18187                method: crate::sampling::SampleMethod::EveryRow,
18188                ..crate::sampling::Sample::default()
18189            },
18190            false,
18191        )
18192        .expect("describe should not fail on binary columns");
18193
18194        let blob_stat = results
18195            .column_statistics
18196            .iter()
18197            .find(|c| c.name == "blob")
18198            .expect("binary column present in describe");
18199        // Stubbed to a constant string, so describe treats it as categorical and its min/max is
18200        // the stub — never the raw bytes.
18201        assert_eq!(blob_stat.dtype, DataType::String);
18202        let cat = blob_stat
18203            .categorical_stats
18204            .as_ref()
18205            .expect("stubbed binary column has categorical stats");
18206        assert_eq!(cat.min.as_deref(), Some(binary_stub()));
18207        assert_eq!(cat.max.as_deref(), Some(binary_stub()));
18208        // The numeric column is still described normally.
18209        let a_stat = results
18210            .column_statistics
18211            .iter()
18212            .find(|c| c.name == "a")
18213            .expect("numeric column present in describe");
18214        assert!(a_stat.numeric_stats.is_some());
18215    }
18216
18217    #[test]
18218    fn trailing_overflow_binary_column_is_truncated() {
18219        // Binary columns render as text (e.g. b"...") and should be truncated like strings —
18220        // this is the EDGAR `txt_bytes` case where a wide Binary column was wrongly dropped.
18221        let table = DataTable::default();
18222        let a = Series::new("a".into(), &[1i32, 2, 3]);
18223        let bin = Series::new(
18224            "wide_bytes".into(),
18225            &["aaaaaaaaaa", "bbbbbbbbbb", "cccccccccc"],
18226        )
18227        .cast(&DataType::Binary)
18228        .unwrap();
18229        let df = DataFrame::new_infer_height(vec![a.into(), bin.into()]).unwrap();
18230        let area = Rect::new(0, 0, 8, 4);
18231        let mut buf = Buffer::empty(area);
18232        let mut ts = TableState::default();
18233        let shown = table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
18234        assert_eq!(
18235            shown, 2,
18236            "an overflowing binary column should be shown truncated"
18237        );
18238    }
18239
18240    #[test]
18241    fn tiny_remaining_width_drops_overflow_string_column() {
18242        // Even a string column should not render a useless 1-2 char sliver.
18243        let table = DataTable::default();
18244        let df = df!(
18245            "abcd" => &[1i32, 2, 3],
18246            "next" => &["yyyy", "yyyy", "yyyy"],
18247        )
18248        .unwrap();
18249        // "abcd" is 4 wide; used_width becomes 5, leaving only 1 (< MIN) for "next".
18250        let area = Rect::new(0, 0, 5, 4);
18251        let mut buf = Buffer::empty(area);
18252        let mut ts = TableState::default();
18253        let shown = table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
18254        assert_eq!(shown, 1, "a sub-minimal sliver should not be shown");
18255    }
18256
18257    #[test]
18258    fn more_columns_indicator_appears_and_tracks_scroll() {
18259        // 4 columns that can't all fit: a right-edge marker should signal off-screen columns,
18260        // and after scrolling right a left-edge marker should appear too.
18261        let mut state =
18262            DataTableState::new(create_large_test_lf(), None, None, None, None, true).unwrap();
18263        state.visible_rows = 3;
18264        state.collect();
18265
18266        let area = Rect::new(0, 0, 6, 4); // narrow: not all 4 columns fit
18267        let mut buf = Buffer::empty(area);
18268        DataTable::default().render(area, &mut buf, &mut state);
18269        let header = header_row_string(&buf, area);
18270        // The arrows come from the glyph set for the locale, which is ASCII on Windows CI.
18271        let g = crate::glyphs::get();
18272        assert!(
18273            header.contains(g.arrow_right),
18274            "expected right indicator, header: {header:?}"
18275        );
18276        assert!(
18277            !header.contains(g.arrow_left),
18278            "should not show left indicator at offset 0: {header:?}"
18279        );
18280
18281        // Scroll right: now columns exist both left and right of the viewport.
18282        state.scroll_right();
18283        let mut buf2 = Buffer::empty(area);
18284        DataTable::default().render(area, &mut buf2, &mut state);
18285        let header2 = header_row_string(&buf2, area);
18286        assert!(
18287            header2.contains(g.arrow_left),
18288            "expected left indicator after scroll: {header2:?}"
18289        );
18290    }
18291
18292    #[test]
18293    fn sorted_column_header_carries_the_direction_mark() {
18294        // A sorted view must not look identical to an unsorted one: the sorted
18295        // column's header says so, and only that column's.
18296        let g = crate::glyphs::get();
18297        let table = DataTable::default().with_sort(vec!["age".to_string()], vec![false]);
18298        let df = df!("name" => &["ann"], "age" => &[41i32]).unwrap();
18299        let area = Rect::new(0, 0, 20, 3);
18300        let mut buf = Buffer::empty(area);
18301        let mut ts = TableState::default();
18302        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
18303        let header = header_row_string(&buf, area);
18304        assert!(
18305            header.contains(&format!("age{}", g.sort_asc)),
18306            "the sorted column is marked: {header:?}"
18307        );
18308        assert!(
18309            !header.contains(&format!("name{}", g.sort_asc)),
18310            "the unsorted column is not: {header:?}"
18311        );
18312        assert!(
18313            !header.contains(g.sort_desc),
18314            "an ascending sort never shows the descending mark: {header:?}"
18315        );
18316    }
18317
18318    #[test]
18319    fn the_direction_mark_flips_with_the_sort() {
18320        let g = crate::glyphs::get();
18321        let table = DataTable::default().with_sort(vec!["age".to_string()], vec![true]);
18322        let df = df!("name" => &["ann"], "age" => &[41i32]).unwrap();
18323        let area = Rect::new(0, 0, 20, 3);
18324        let mut buf = Buffer::empty(area);
18325        let mut ts = TableState::default();
18326        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
18327        let header = header_row_string(&buf, area);
18328        assert!(
18329            header.contains(&format!("age{}", g.sort_desc)),
18330            "a descending sort points down: {header:?}"
18331        );
18332        assert!(!header.contains(g.sort_asc), "and never up: {header:?}");
18333    }
18334
18335    #[test]
18336    fn every_column_of_a_multi_sort_is_marked() {
18337        // Marks only, no position numbers: the columns all run the same way.
18338        let g = crate::glyphs::get();
18339        let table = DataTable::default().with_sort(
18340            vec!["name".to_string(), "age".to_string()],
18341            vec![false, false],
18342        );
18343        let df = df!("name" => &["ann"], "age" => &[41i32]).unwrap();
18344        let area = Rect::new(0, 0, 20, 3);
18345        let mut buf = Buffer::empty(area);
18346        let mut ts = TableState::default();
18347        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
18348        let header = header_row_string(&buf, area);
18349        for name in ["name", "age"] {
18350            assert!(
18351                header.contains(&format!("{name}{}", g.sort_asc)),
18352                "{name} carries the mark: {header:?}"
18353            );
18354        }
18355    }
18356
18357    #[test]
18358    fn the_sort_mark_composes_with_the_drift_mark() {
18359        // A column can be sorted and drifting at once; the header carries both
18360        // marks and the width arithmetic counts both, so nothing is clipped.
18361        let g = crate::glyphs::get();
18362        let table = DataTable::default()
18363            .with_sort(vec!["age".to_string()], vec![false])
18364            .with_drift(
18365                Vec::new(),
18366                Arc::new(vec![crate::schema_union::DriftGroup {
18367                    absent: vec!["age".into()],
18368                    unread: Vec::new(),
18369                }]),
18370            );
18371        let df = df!("age" => &[41i32]).unwrap();
18372        let area = Rect::new(0, 0, 20, 3);
18373        let mut buf = Buffer::empty(area);
18374        let mut ts = TableState::default();
18375        table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
18376        let header = header_row_string(&buf, area);
18377        assert!(
18378            header.contains(&format!("age{}{}", g.drift_mark, g.sort_asc)),
18379            "footnote first, direction after: {header:?}"
18380        );
18381        assert!(
18382            row_string(&buf, area, 1).contains("41"),
18383            "the widened header does not clip the value"
18384        );
18385    }
18386
18387    #[test]
18388    fn the_header_mark_follows_the_state_sort_and_its_reverse() {
18389        // Through the stateful render: the marks come from the state being drawn,
18390        // so applying a sort shows them and `reverse` flips them, with no caller
18391        // wiring in between.
18392        let g = crate::glyphs::get();
18393        let mut state =
18394            DataTableState::new(create_test_lf(), None, None, None, None, true).unwrap();
18395        state.visible_rows = 3;
18396        state.sort(vec!["a".to_string()], true);
18397
18398        let area = Rect::new(0, 0, 20, 5);
18399        let mut buf = Buffer::empty(area);
18400        DataTable::default().render(area, &mut buf, &mut state);
18401        let header = header_row_string(&buf, area);
18402        assert!(
18403            header.contains(&format!("a{}", g.sort_asc)),
18404            "sorted ascending: {header:?}"
18405        );
18406
18407        state.reverse();
18408        let mut buf2 = Buffer::empty(area);
18409        DataTable::default().render(area, &mut buf2, &mut state);
18410        let header2 = header_row_string(&buf2, area);
18411        assert!(
18412            header2.contains(&format!("a{}", g.sort_desc)),
18413            "reversed: {header2:?}"
18414        );
18415        assert!(
18416            !header2.contains(g.sort_asc),
18417            "the old direction is gone: {header2:?}"
18418        );
18419    }
18420
18421    /// A state over `id` (frozen), `name`, `city`, `amount`, thirty rows, drawn once at
18422    /// 80×24 so the room and widths are known.
18423    fn cursor_fixture() -> (DataTableState, Rect) {
18424        let n = 30;
18425        let lf = df!(
18426            "id" => (0..n).collect::<Vec<i64>>(),
18427            "name" => (0..n).map(|i| format!("name {i}")).collect::<Vec<_>>(),
18428            "city" => (0..n).map(|i| format!("city {i}")).collect::<Vec<_>>(),
18429            "amount" => (0..n).map(|i| i as f64 * 1.5).collect::<Vec<_>>(),
18430        )
18431        .unwrap()
18432        .lazy();
18433        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18434        state.visible_rows = 22;
18435        state.set_locked_columns(1);
18436        state.table_state.select(Some(0));
18437        let area = Rect::new(0, 0, 80, 24);
18438        DataTable::default().render(area, &mut Buffer::empty(area), &mut state);
18439        (state, area)
18440    }
18441
18442    /// The x range of the column headed `name` on the header row.
18443    fn column_span(buf: &Buffer, area: Rect, name: &str) -> std::ops::Range<u16> {
18444        let header = row_string(buf, area, 0);
18445        let at = header.find(name).expect("the column is drawn") as u16;
18446        at..at + name.len() as u16
18447    }
18448
18449    /// A click finds the cell drawn under it, at 80×24 and on a wide screen: each
18450    /// column where its heading is, frozen or scrolling, and each row where its cells
18451    /// are, with row numbers on and the view scrolled down.
18452    #[test]
18453    fn a_click_finds_the_cell_drawn_under_it() {
18454        for (width, height) in [(80, 24), (200, 50)] {
18455            let n = 400;
18456            let mut columns = vec![Column::new("id".into(), (0..n).collect::<Vec<i64>>())];
18457            for c in 0..30 {
18458                columns.push(Column::new(
18459                    format!("col_{c:02}").as_str().into(),
18460                    (0..n).map(|i| format!("v{i}_{c}")).collect::<Vec<_>>(),
18461                ));
18462            }
18463            let lf = DataFrame::new_infer_height(columns).unwrap().lazy();
18464            let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18465            state.set_locked_columns(1);
18466            state.toggle_row_numbers();
18467            let area = Rect::new(0, 0, width, height);
18468            let render = |state: &mut DataTableState| {
18469                let mut buf = Buffer::empty(area);
18470                DataTable::default().render(area, &mut buf, state);
18471                buf
18472            };
18473            // The first frame sets how many rows fit; the rows are read for it.
18474            render(&mut state);
18475            state.collect();
18476            for scrolled in [false, true] {
18477                if scrolled {
18478                    state.page_down();
18479                    state.collect();
18480                }
18481                let buf = render(&mut state);
18482                let drawn = state.drawn.clone().expect("the table was drawn");
18483                let header = row_string(&buf, area, 0);
18484                assert!(drawn.columns.len() > 3, "{width}x{height}: {header:?}");
18485                assert_eq!(drawn.columns[0].2, "id", "the frozen column first");
18486                for (_, _, name) in &drawn.columns {
18487                    // In cells, not bytes: the frozen separator is a wide glyph.
18488                    let at = header.find(name.as_str()).expect("heading drawn");
18489                    let from = header[..at].chars().count() as u16;
18490                    for x in [from, from + name.len() as u16 - 1] {
18491                        let hit = state.drawn_cell(x, 0).expect("on the table");
18492                        assert_eq!(hit.row, None, "the header is no row");
18493                        assert_eq!(hit.column.as_deref(), Some(name.as_str()), "at {x}");
18494                    }
18495                }
18496                // The rail and the row numbers are no column, but are the row.
18497                let y = drawn.header + 5;
18498                assert_eq!(
18499                    state.drawn_cell(0, y),
18500                    Some(CellHit {
18501                        row: Some(5),
18502                        column: None
18503                    })
18504                );
18505                // A cell: the row and column whose value is drawn there.
18506                let (from, to, name) = drawn.columns[2].clone();
18507                let c: usize = name["col_".len()..].parse().unwrap();
18508                let hit = state.drawn_cell(from, y).expect("a cell");
18509                let row = drawn.start_row + 5;
18510                let text: String = (from..to).map(|x| buf[(x, y)].symbol()).collect();
18511                assert_eq!(text.trim(), format!("v{row}_{c}"), "{width}x{height}");
18512                state.point_at(&hit);
18513                assert_eq!(state.table_state.selected(), Some(5));
18514                assert_eq!(state.current_column(), Some(name.as_str()));
18515                // Off the table's right or bottom edge is nothing.
18516                assert_eq!(state.drawn_cell(width, y), None);
18517                assert_eq!(state.drawn_cell(0, height), None);
18518            }
18519        }
18520    }
18521
18522    /// The column cursor at 80×24: its header and cells take the column tint, the
18523    /// current cell (the cursor's row and column) the cell tint, and the rest of the
18524    /// current row keeps the row tint. Frozen columns take it the same way.
18525    #[test]
18526    fn the_column_cursor_tints_its_header_and_cells() {
18527        let (mut state, area) = cursor_fixture();
18528        let row_tint = Color::Rgb(0x28, 0x34, 0x57);
18529        let column_tint = Color::Rgb(0x29, 0x2e, 0x42);
18530        let cell_tint = Color::Rgb(0x3b, 0x42, 0x61);
18531        let table = || {
18532            DataTable {
18533                selection_style: Style::default().bg(row_tint),
18534                ..DataTable::default()
18535            }
18536            .with_cursor_styles(
18537                crate::config::column_cursor_style(Some(column_tint)),
18538                crate::config::cell_cursor_style(Some(cell_tint)),
18539            )
18540        };
18541        state.move_cursor(CursorMove::Right);
18542        state.move_cursor(CursorMove::Right);
18543        assert_eq!(state.current_column(), Some("city"));
18544        let mut buf = Buffer::empty(area);
18545        table().render(area, &mut buf, &mut state);
18546        let city = column_span(&buf, area, "city");
18547        let name = column_span(&buf, area, "name");
18548        for x in city.clone() {
18549            assert_eq!(buf[(x, 0)].bg, cell_tint, "the header, at {x}");
18550            assert!(buf[(x, 0)].modifier.contains(Modifier::BOLD));
18551            assert_eq!(buf[(x, 1)].bg, cell_tint, "the current cell, at {x}");
18552            for y in 2..area.height {
18553                assert_eq!(buf[(x, y)].bg, column_tint, "the column, at {x},{y}");
18554            }
18555        }
18556        for x in name.clone() {
18557            assert_eq!(buf[(x, 1)].bg, row_tint, "the rest of the row");
18558            assert_ne!(buf[(x, 0)].bg, cell_tint, "another header");
18559            assert_ne!(buf[(x, 2)].bg, column_tint, "another column");
18560        }
18561
18562        // Frozen: the same marks, left of the separator.
18563        state.move_cursor(CursorMove::First);
18564        assert_eq!(state.current_column(), Some("id"));
18565        let mut buf = Buffer::empty(area);
18566        table().render(area, &mut buf, &mut state);
18567        let id = column_span(&buf, area, "id");
18568        for x in id {
18569            assert_eq!(buf[(x, 0)].bg, cell_tint);
18570            assert_eq!(buf[(x, 1)].bg, cell_tint);
18571            assert_eq!(buf[(x, 5)].bg, column_tint);
18572        }
18573        for x in city {
18574            assert_ne!(buf[(x, 0)].bg, cell_tint, "city lets go of it");
18575        }
18576    }
18577
18578    /// Where the tints would not show (16 colors, `NO_COLOR`), the header and the
18579    /// current cell are reversed, so the cursor is still on screen, and nothing else is.
18580    #[test]
18581    fn the_column_cursor_shows_without_its_tints() {
18582        let (mut state, area) = cursor_fixture();
18583        state.move_cursor(CursorMove::Right);
18584        for tint in [Color::Black, Color::White, Color::Reset] {
18585            let mut buf = Buffer::empty(area);
18586            DataTable::default()
18587                .with_cursor_styles(
18588                    crate::config::column_cursor_style(Some(tint)),
18589                    crate::config::cell_cursor_style(Some(tint)),
18590                )
18591                .render(area, &mut buf, &mut state);
18592            let name = column_span(&buf, area, "name");
18593            let reversed = |x, y| buf[(x, y)].modifier.contains(Modifier::REVERSED);
18594            for x in name {
18595                assert!(reversed(x, 0), "{tint:?}: the header");
18596                assert!(reversed(x, 1), "{tint:?}: the current cell");
18597                assert!(!reversed(x, 2), "{tint:?}: not the rest of the column");
18598            }
18599            let city = column_span(&buf, area, "city");
18600            assert!(!reversed(city.start, 0) && !reversed(city.start, 1));
18601        }
18602    }
18603
18604    /// Under a reversed row, the current cell is drawn upright, tinted or not, so it
18605    /// stands out from the row; the header keeps its own mark.
18606    #[test]
18607    fn the_current_cell_stands_out_of_a_reversed_row() {
18608        let (mut state, area) = cursor_fixture();
18609        state.move_cursor(CursorMove::Right);
18610        for tint in [Color::Rgb(0x3b, 0x42, 0x61), Color::Black, Color::Reset] {
18611            let mut buf = Buffer::empty(area);
18612            DataTable {
18613                selection_style: Style::default().add_modifier(Modifier::REVERSED),
18614                ..DataTable::default()
18615            }
18616            .with_cursor_styles(
18617                crate::config::column_cursor_style(Some(tint)),
18618                crate::config::cell_cursor_style(Some(tint)),
18619            )
18620            .render(area, &mut buf, &mut state);
18621            let reversed = |x, y| buf[(x, y)].modifier.contains(Modifier::REVERSED);
18622            for x in column_span(&buf, area, "name") {
18623                assert!(!reversed(x, 1), "{tint:?}: the current cell is upright");
18624                assert!(buf[(x, 1)].modifier.contains(Modifier::BOLD));
18625            }
18626            let city = column_span(&buf, area, "city");
18627            assert!(reversed(city.start, 1), "{tint:?}: the rest of the row");
18628        }
18629    }
18630
18631    /// The cursor follows its column by name when the columns are reordered or
18632    /// frozen, and a column hidden from under it hands it to the one in its place.
18633    #[test]
18634    fn the_column_cursor_follows_its_column_by_name() {
18635        let (mut state, area) = cursor_fixture();
18636        state.move_cursor(CursorMove::Right);
18637        state.move_cursor(CursorMove::Right);
18638        assert_eq!(state.current_column(), Some("city"));
18639        let order = |names: &[&str]| names.iter().map(|n| n.to_string()).collect::<Vec<_>>();
18640        state.set_column_order(order(&["city", "id", "name", "amount"]));
18641        assert_eq!(state.current_column(), Some("city"));
18642        assert_eq!(state.current_column_index(), Some(0));
18643        state.set_locked_columns(0);
18644        state.set_column_order(order(&["id", "name", "city", "amount"]));
18645        state.set_locked_columns(3);
18646        assert_eq!(state.current_column(), Some("city"), "frozen now");
18647        // Hidden: the column now in its place takes it.
18648        state.set_locked_columns(0);
18649        state.set_column_order(order(&["id", "name", "amount"]));
18650        assert_eq!(state.current_column(), Some("amount"));
18651        // Hidden at the end: the last column.
18652        state.set_column_order(order(&["id", "name"]));
18653        assert_eq!(state.current_column(), Some("name"));
18654        state.set_column_order(Vec::new());
18655        assert_eq!(state.current_column(), None);
18656        DataTable::default().render(area, &mut Buffer::empty(area), &mut state);
18657    }
18658
18659    /// `h` `l` cross from the frozen columns to the scrolling ones and back in the
18660    /// shown order; the view moves only when the cursor would leave it, and a page
18661    /// puts the cursor on the new page's first column.
18662    #[test]
18663    fn the_column_cursor_scrolls_only_at_the_edges() {
18664        let n = 5;
18665        let names: Vec<String> = (0..40).map(|i| format!("column_{i:02}")).collect();
18666        let columns: Vec<Column> = names
18667            .iter()
18668            .map(|name| Column::new(name.as_str().into(), (0..n).collect::<Vec<i64>>()))
18669            .collect();
18670        let lf = DataFrame::new_infer_height(columns).unwrap().lazy();
18671        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18672        state.visible_rows = 5;
18673        state.set_locked_columns(2);
18674        let area = Rect::new(0, 0, 80, 8);
18675        let draw = |state: &mut DataTableState| {
18676            DataTable::default().render(area, &mut Buffer::empty(area), state);
18677            state.columns_on_screen().unwrap()
18678        };
18679        let start = draw(&mut state);
18680        assert_eq!((start.first, start.cursor), (3, 1));
18681        state.move_cursor(CursorMove::Right);
18682        state.move_cursor(CursorMove::Right);
18683        let on = draw(&mut state);
18684        assert_eq!((on.first, on.cursor), (3, 3), "into the scrolling side");
18685        // Walk to the right edge: nothing scrolls until the cursor would leave (a
18686        // column cut at the edge counts as leaving), then just enough.
18687        let mut before = on;
18688        let past = loop {
18689            state.move_cursor(CursorMove::Right);
18690            let now = draw(&mut state);
18691            assert_eq!(now.cursor, before.cursor + 1);
18692            if now.first != before.first {
18693                break now;
18694            }
18695            before = now;
18696        };
18697        assert!(past.first > 3 && past.cursor <= past.last, "{past:?}");
18698        assert!(past.cursor >= start.last, "not before the edge: {past:?}");
18699        assert!(past.first <= before.last, "no column skipped: {past:?}");
18700        // Back to the left edge, and one more scrolls back a column.
18701        for _ in past.first..past.cursor {
18702            state.move_cursor(CursorMove::Left);
18703        }
18704        assert_eq!(draw(&mut state).first, past.first);
18705        state.move_cursor(CursorMove::Left);
18706        assert_eq!(draw(&mut state).first, past.first - 1);
18707        // Pages: the cursor starts the new page; on the last, it takes the last column.
18708        state.move_cursor(CursorMove::PageRight);
18709        let page = draw(&mut state);
18710        assert_eq!(page.cursor, page.first);
18711        state.move_cursor(CursorMove::Last);
18712        let last = draw(&mut state);
18713        assert_eq!((last.cursor, last.last), (40, 40));
18714        state.move_cursor(CursorMove::PageRight);
18715        assert_eq!(draw(&mut state).cursor, 40);
18716        // From a frozen column, the next is the first scrolling column: the view
18717        // goes back to it.
18718        state.go_to_column("column_01");
18719        assert_eq!(draw(&mut state).first, last.first, "frozen: on screen");
18720        state.move_cursor(CursorMove::Right);
18721        let back = draw(&mut state);
18722        assert_eq!((back.first, back.cursor), (3, 3));
18723        // `[` on the first page: its first column, then the first of all.
18724        state.move_cursor(CursorMove::Right);
18725        state.move_cursor(CursorMove::PageLeft);
18726        assert_eq!(draw(&mut state).cursor, 3);
18727        state.move_cursor(CursorMove::PageLeft);
18728        assert_eq!(draw(&mut state).cursor, 1);
18729        state.move_cursor(CursorMove::Left);
18730        assert_eq!(draw(&mut state).cursor, 1, "nothing left of the first");
18731    }
18732
18733    /// Every cell right of the frozen separator sits one cell off it, as the cells
18734    /// left of it do, and the gap takes its row's tint. A right-aligned number as
18735    /// wide as its column, a negative one most often, used to touch the line:
18736    /// `│-9.930889` (#386).
18737    #[test]
18738    fn the_frozen_separator_has_a_gap_on_both_sides() {
18739        let lf = df!(
18740            "carrier" => &["AA", "UA", "9E"],
18741            "delay" => &[-9.930889f64, 3.5, 12.25],
18742            "name" => &["American", "United", "Endeavor"],
18743        )
18744        .unwrap()
18745        .lazy();
18746        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18747        state.visible_rows = 3;
18748        state.set_locked_columns(1);
18749        state.table_state.select(Some(0));
18750
18751        let area = Rect::new(0, 0, 40, 6);
18752        let mut buf = Buffer::empty(area);
18753        let table = DataTable {
18754            header_bg: Color::Indexed(238),
18755            alternate_row_bg: Some(Color::Indexed(236)),
18756            selection_style: Style::default().bg(Color::Indexed(24)),
18757            ..DataTable::default()
18758        };
18759        table.render(area, &mut buf, &mut state);
18760
18761        let rows: Vec<String> = (0..area.height)
18762            .map(|y| row_string(&buf, area, y))
18763            .collect();
18764        assert!(
18765            rows[1].contains(&format!("{} -9.930889", crate::glyphs::get().rule)),
18766            "{rows:#?}"
18767        );
18768        let rule = crate::glyphs::get().rule;
18769        let sep = (0..area.width)
18770            .find(|&x| buf[(x, 0)].symbol() == rule)
18771            .expect("a separator");
18772        for y in 0..area.height {
18773            let row = &rows[y as usize];
18774            assert_eq!(buf[(sep - 1, y)].symbol(), " ", "row {y}: {row:?}");
18775            assert_eq!(buf[(sep + 1, y)].symbol(), " ", "row {y}: {row:?}");
18776            assert_eq!(
18777                buf[(sep + 1, y)].bg,
18778                buf[(sep + 2, y)].bg,
18779                "row {y}'s gap takes the row's tint: {row:?}"
18780            );
18781        }
18782        // The header, the highlight and the stripe, not only unstyled rows.
18783        assert_eq!(buf[(sep + 1, 0)].bg, Color::Indexed(238));
18784        assert_eq!(buf[(sep + 1, 1)].bg, Color::Indexed(24));
18785        assert_eq!(buf[(sep + 1, 2)].bg, Color::Indexed(236));
18786    }
18787
18788    /// The frozen separator runs down the header and the rows, and stops under the
18789    /// last: a grouped view of seven rows had it running down the empty screen.
18790    #[test]
18791    fn the_frozen_separator_stops_at_the_last_row() {
18792        let lf = df!(
18793            "carrier" => &["AA", "UA", "9E"],
18794            "delay" => &[-9.9f64, 3.5, 12.25],
18795        )
18796        .unwrap()
18797        .lazy();
18798        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18799        state.visible_rows = 3;
18800        state.set_locked_columns(1);
18801        state.table_state.select(Some(0));
18802        let area = Rect::new(0, 0, 30, 10);
18803        let mut buf = Buffer::empty(area);
18804        DataTable::default().render(area, &mut buf, &mut state);
18805        let rule = crate::glyphs::get().rule;
18806        let sep = (0..area.width)
18807            .find(|&x| buf[(x, 0)].symbol() == rule)
18808            .expect("a separator");
18809        let ruled: Vec<u16> = (0..area.height)
18810            .filter(|&y| buf[(sep, y)].symbol() == rule)
18811            .collect();
18812        let last = ruled.last().copied().unwrap();
18813        assert_eq!(
18814            ruled,
18815            (0..=last).collect::<Vec<_>>(),
18816            "unbroken to the last row"
18817        );
18818        let header = DataTable::default().header_height();
18819        assert_eq!(last, header + 2, "under the third row, no further");
18820    }
18821
18822    /// A frozen column whose type is wider than its name and values still gets its
18823    /// whole width and the gap before the separator. The width pass left the type row
18824    /// out, so `id` over `i64` ran onto the line (`i64│`) and a one-letter string
18825    /// column did not fit at all.
18826    #[test]
18827    fn a_frozen_column_is_as_wide_as_its_type() {
18828        let lf = df!(
18829            "id" => &[1i64, 2],
18830            "k" => &["x", "y"],
18831            "v" => &[-3.5f64, 4.25],
18832        )
18833        .unwrap()
18834        .lazy();
18835        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18836        state.visible_rows = 2;
18837        state.set_locked_columns(2);
18838        let area = Rect::new(0, 0, 30, 4);
18839        let mut buf = Buffer::empty(area);
18840        DataTable {
18841            dtype_row: true,
18842            ..DataTable::default()
18843        }
18844        .render(area, &mut buf, &mut state);
18845
18846        let rows: Vec<String> = (0..area.height)
18847            .map(|y| row_string(&buf, area, y))
18848            .collect();
18849        let rule = crate::glyphs::get().rule;
18850        assert!(rows[0].contains(&format!(" id k   {rule}")), "{rows:#?}");
18851        assert!(rows[1].contains(&format!("i64 str {rule}")), "{rows:#?}");
18852        assert!(rows[2].contains(&format!("  1 x   {rule}")), "{rows:#?}");
18853    }
18854
18855    fn list_state() -> DataTableState {
18856        let many: Vec<String> = (0..12).map(|i| format!("t{i}")).collect();
18857        let tags = Series::new(
18858            "tags".into(),
18859            &[
18860                Series::new("".into(), &["a", "b"]),
18861                Series::new("".into(), many),
18862            ],
18863        );
18864        let id = Series::new("id".into(), &[1i64, 2]);
18865        let more = Series::new(
18866            "more".into(),
18867            &[
18868                Series::new("".into(), &["x"]),
18869                Series::new("".into(), &["y"]),
18870            ],
18871        );
18872        let lf = DataFrame::new_infer_height(vec![id.into(), tags.into(), more.into()])
18873            .unwrap()
18874            .lazy();
18875        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18876        state.visible_rows = 2;
18877        state.collect();
18878        state
18879    }
18880
18881    /// A list cell reads as it did when the buffer held lists as text, and the type
18882    /// row names the list once the rows have landed, not `str`.
18883    #[test]
18884    fn a_list_column_draws_its_items_under_its_list_type() {
18885        let mut state = list_state();
18886        let area = Rect::new(0, 0, 80, 4);
18887        let mut buf = Buffer::empty(area);
18888        DataTable {
18889            dtype_row: true,
18890            ..DataTable::default()
18891        }
18892        .render(area, &mut buf, &mut state);
18893        let rows: Vec<String> = (0..area.height)
18894            .map(|y| row_string(&buf, area, y))
18895            .collect();
18896        assert!(rows[1].contains("list[str]"), "{rows:#?}");
18897        assert!(!rows[1].contains(" str "), "{rows:#?}");
18898        assert!(rows[2].contains("[a, b]"), "{rows:#?}");
18899        // The column's width cap cuts the rest; `exact` tests the whole preview.
18900        assert!(
18901            rows[3].contains("[t0, t1, t2, t3, t4, t5, t6, t7"),
18902            "{rows:#?}"
18903        );
18904    }
18905
18906    /// The display frames keep lists as lists, frozen or scrolling, through a
18907    /// sideways scroll: a step re-cuts the buffer and formats nothing; only the
18908    /// cells drawn are formatted.
18909    #[test]
18910    fn a_sideways_scroll_keeps_lists_in_the_display_frames() {
18911        let mut state = list_state();
18912        state.set_locked_columns(2);
18913        state.scroll_right();
18914        let is_list = |df: &DataFrame, name: &str| {
18915            matches!(df.column(name).unwrap().dtype(), DataType::List(_))
18916        };
18917        assert!(is_list(state.locked_df.as_ref().unwrap(), "tags"));
18918        assert!(is_list(state.df.as_ref().unwrap(), "more"));
18919    }
18920
18921    /// A page over columns not drawn yet waits for the draw, which measures them from
18922    /// the rows on hand and lands it: `Last` ends the page with the last column whole,
18923    /// and the one before that would not have fitted. Nothing moves before then.
18924    #[test]
18925    fn a_page_over_columns_not_drawn_lands_at_the_draw() {
18926        let names: Vec<String> = (0..30).map(|i| format!("col{i:02}")).collect();
18927        let columns: Vec<Column> = names
18928            .iter()
18929            .enumerate()
18930            .map(|(i, name)| {
18931                let value = "x".repeat(3 + i % 5);
18932                Series::new(name.as_str().into(), vec![value; 3]).into()
18933            })
18934            .collect();
18935        let lf = DataFrame::new_infer_height(columns).unwrap().lazy();
18936        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18937        state.visible_rows = 3;
18938        state.collect();
18939        let area = Rect::new(0, 0, 50, 4);
18940        let draw = |state: &mut DataTableState| {
18941            let mut buf = Buffer::empty(area);
18942            DataTable::default().render(area, &mut buf, state);
18943            row_string(&buf, area, 0)
18944        };
18945        draw(&mut state);
18946        assert!(state.drawn_width("col29").is_none(), "not drawn yet");
18947
18948        state.scroll_columns(ColumnMove::Last);
18949        assert_eq!(state.termcol_index, 0, "nothing moves before the draw");
18950        let header = draw(&mut state);
18951        assert!(header.trim_end().ends_with("col29"), "{header}");
18952        let start = state.termcol_index;
18953        assert!(start > 0);
18954        // The column before the page would not have fitted beside it.
18955        let room = state.scroll_room.unwrap();
18956        let used: u16 = names[start - 1..]
18957            .iter()
18958            .map(|n| state.shown_width(n).unwrap() + room.padding)
18959            .sum::<u16>()
18960            - room.padding;
18961        assert!(used > room.width, "{used} in {}", room.width);
18962
18963        // Back, which lands at the draw again; then forward over columns drawn now,
18964        // which needs none.
18965        state.scroll_columns(ColumnMove::PageLeft);
18966        draw(&mut state);
18967        let back = state.termcol_index;
18968        assert!(back < start);
18969        state.scroll_columns(ColumnMove::PageRight);
18970        assert_eq!(state.termcol_index, start);
18971        state.scroll_columns(ColumnMove::First);
18972        assert_eq!(state.termcol_index, 0);
18973    }
18974
18975    /// A state over thirty text columns of mixed widths, drawn once at 50 wide.
18976    fn paging_state() -> (DataTableState, impl Fn(&mut DataTableState)) {
18977        let columns: Vec<Column> = (0..30)
18978            .map(|i| {
18979                let value = "x".repeat(3 + (i * 7) % 11);
18980                Series::new(format!("col{i:02}").into(), vec![value; 3]).into()
18981            })
18982            .collect();
18983        let lf = DataFrame::new_infer_height(columns).unwrap().lazy();
18984        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
18985        state.visible_rows = 3;
18986        state.collect();
18987        let draw = |state: &mut DataTableState| {
18988            let area = Rect::new(0, 0, 50, 4);
18989            DataTable::default().render(area, &mut Buffer::empty(area), state);
18990        };
18991        draw(&mut state);
18992        (state, draw)
18993    }
18994
18995    /// Keys typed faster than frames, over columns not drawn yet, land at the draw in
18996    /// the order typed, exactly where they land with a frame after each, and keep
18997    /// waiting while no rows are on hand to measure with.
18998    #[test]
18999    fn moves_typed_before_a_draw_land_in_order() {
19000        use ColumnMove::*;
19001        let keys = [PageRight, PageRight, StepRight, PageLeft, PageRight];
19002        let (mut paced, draw) = paging_state();
19003        for mv in keys {
19004            paced.scroll_columns(mv);
19005            draw(&mut paced);
19006        }
19007        assert!(paced.termcol_index > 0);
19008
19009        let (mut typed, draw) = paging_state();
19010        typed.defer_collect = true;
19011        for mv in keys {
19012            typed.scroll_columns(mv);
19013        }
19014        draw(&mut typed);
19015        assert_eq!(typed.termcol_index, 0, "no rows to measure with: they wait");
19016        typed.defer_collect = false;
19017        draw(&mut typed);
19018        assert_eq!(typed.termcol_index, paced.termcol_index);
19019        assert!(typed.column_moves.is_empty());
19020
19021        // First drops what waits; nothing is left to land later.
19022        let (mut first, draw) = paging_state();
19023        first.scroll_columns(Last);
19024        first.scroll_columns(StepRight);
19025        first.scroll_columns(ColumnMove::First);
19026        draw(&mut first);
19027        assert_eq!(first.termcol_index, 0);
19028    }
19029
19030    /// `[` straight after `]` goes back to the page `]` left, even where the page
19031    /// before, packed from the right, would start further left.
19032    #[test]
19033    fn page_left_after_page_right_retraces() {
19034        let (mut state, draw) = paging_state();
19035        let room = state.scroll_room.unwrap();
19036        let p = room.padding;
19037        let usable = room.width - room.lead;
19038        // Three of `a` fill the first page, two of `b` the second with a column too
19039        // wide for the room after them; packed from the right, the page ending before
19040        // that wide column would take in two of the `a` columns as well.
19041        let a = (usable - 2 * p) / 3;
19042        let b = (usable.saturating_sub(2 * a + 3 * p) / 2)
19043            .max(crate::widgets::column_widths::MIN_WIDTH);
19044        let widths = [a, a, a, b, b, usable + 10];
19045        state.set_width_choices(
19046            widths
19047                .iter()
19048                .enumerate()
19049                .map(|(i, &w)| (format!("col{i:02}"), WidthChoice::Manual(w))),
19050        );
19051        draw(&mut state);
19052        state.scroll_columns(ColumnMove::PageRight);
19053        draw(&mut state);
19054        assert_eq!(state.termcol_index, 3);
19055        state.scroll_columns(ColumnMove::PageRight);
19056        draw(&mut state);
19057        assert_eq!(state.termcol_index, 5);
19058        let names = state.scrolling_names().to_vec();
19059        let packed =
19060            crate::widgets::column_paging::plan(ColumnMove::PageLeft, 5, names.len(), room, |i| {
19061                state.drawn_width(&names[i])
19062            });
19063        assert!(packed < Some(3), "the packed page differs: {packed:?}");
19064        state.scroll_columns(ColumnMove::PageLeft);
19065        assert_eq!(state.termcol_index, 3, "back to the page left");
19066        state.scroll_columns(ColumnMove::PageLeft);
19067        assert_eq!(state.termcol_index, 0);
19068        // Not straight after `]`: the page before, packed from the right.
19069        state.scroll_columns(ColumnMove::PageRight);
19070        state.scroll_columns(ColumnMove::PageRight);
19071        state.scroll_columns(ColumnMove::StepRight);
19072        state.scroll_columns(ColumnMove::StepLeft);
19073        state.scroll_columns(ColumnMove::PageLeft);
19074        assert_eq!(Some(state.termcol_index), packed);
19075    }
19076
19077    /// The hidden-columns count goes in the blank run after the last column, or not
19078    /// at all: at no width does it cover the type under a heading. The separator's gap
19079    /// took the one cell of slack that used to keep `+2 >` clear of `str` at 60
19080    /// columns, and the count then wrote over the `r`.
19081    #[test]
19082    fn the_hidden_count_never_covers_a_type() {
19083        let lf = df!(
19084            "k" => &["x"],
19085            "origin" => &["JFK"],
19086            "dest" => &["LAX"],
19087            "tail" => &["N1"],
19088            "name" => &["Endeavor"],
19089        )
19090        .unwrap()
19091        .lazy();
19092        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
19093        state.visible_rows = 1;
19094        state.set_locked_columns(1);
19095        let table = || DataTable {
19096            dtype_row: true,
19097            ..DataTable::default()
19098        };
19099        assert_eq!(table().header_height(), 2, "the count goes on the type row");
19100        for width in 8..=40 {
19101            let area = Rect::new(0, 0, width, 3);
19102            let mut buf = Buffer::empty(area);
19103            table().render(area, &mut buf, &mut state);
19104            let names = row_string(&buf, area, 0);
19105            let types = row_string(&buf, area, 1);
19106            // Every heading shown whole has its whole type under it; these are all
19107            // strings, so each starts where its name does.
19108            for name in ["origin", "dest", "tail", "name"] {
19109                if let Some(at) = names.find(&format!(" {name}")) {
19110                    let x = names[..at].chars().count() + 1;
19111                    let under: String = types.chars().skip(x).take(3).collect();
19112                    assert_eq!(under, "str", "width {width}:\n{names}\n{types}");
19113                }
19114            }
19115        }
19116    }
19117
19118    /// The gap is paid for in the width budget: at every width, the columns right of
19119    /// the separator are whole or absent. Left out, the last column that fits exactly
19120    /// comes out one cell short, and a number cut short reads as a different number.
19121    /// The one exception is a first column with no room for its value at all, which
19122    /// shows a preview behind the clip marker rather than nothing.
19123    #[test]
19124    fn the_separator_gap_never_cuts_a_number_short() {
19125        let values = ["-987", "654", "-32", "10"];
19126        let lf = df!(
19127            "k" => &["x"],
19128            "a" => &[-987i64],
19129            "b" => &[654i64],
19130            "c" => &[-32i64],
19131            "d" => &[10i64],
19132        )
19133        .unwrap()
19134        .lazy();
19135        let mut state = DataTableState::new(lf, None, None, None, None, true).unwrap();
19136        state.visible_rows = 1;
19137        state.set_locked_columns(1);
19138        let data_row = DataTable::default().header_height();
19139        for width in 8..=30 {
19140            let area = Rect::new(0, 0, width, data_row + 1);
19141            let mut buf = Buffer::empty(area);
19142            DataTable::default().render(area, &mut buf, &mut state);
19143            let row = row_string(&buf, area, data_row);
19144            let g = crate::glyphs::get();
19145            let (_, scrolled) = row.split_once(g.rule).expect("a separator");
19146            for (i, token) in scrolled.split_whitespace().enumerate() {
19147                let previewed = i == 0 && token.ends_with(g.ellipsis);
19148                assert!(
19149                    values.contains(&token) || previewed,
19150                    "width {width}: {token:?} is cut short in {row:?}"
19151                );
19152            }
19153        }
19154    }
19155
19156    fn glyph_sets() -> [&'static crate::glyphs::Glyphs; 2] {
19157        [crate::glyphs::unicode(), crate::glyphs::ascii()]
19158    }
19159
19160    fn set_name(g: &crate::glyphs::Glyphs) -> &'static str {
19161        if g.unicode { "unicode" } else { "ascii" }
19162    }
19163
19164    /// The cells of row `y` from `x0` on, as the terminal shows them: a wide
19165    /// character once, without the blank its second cell holds.
19166    fn drawn_from(buf: &Buffer, y: u16, x0: u16) -> String {
19167        use ratatui::buffer::CellWidth;
19168        let mut row = String::new();
19169        let mut x = x0;
19170        while x < buf.area.right() {
19171            let symbol = buf[(x, y)].symbol();
19172            row.push_str(symbol);
19173            x += symbol.cell_width().max(1);
19174        }
19175        row
19176    }
19177
19178    /// Draw `state` at `width` × `height` with `table`, row by row as shown.
19179    fn draw(table: DataTable, state: &mut DataTableState, width: u16, height: u16) -> Vec<String> {
19180        let area = Rect::new(0, 0, width, height);
19181        let mut buf = Buffer::empty(area);
19182        table.render(area, &mut buf, state);
19183        (0..height).map(|y| drawn_from(&buf, y, 0)).collect()
19184    }
19185
19186    /// A state over `df` with its first page buffered for `visible_rows` rows.
19187    fn state_of(df: &DataFrame, visible_rows: usize) -> DataTableState {
19188        let mut state =
19189            DataTableState::new(df.clone().lazy(), None, None, None, None, true).unwrap();
19190        state.visible_rows = visible_rows;
19191        state.collect();
19192        state
19193    }
19194
19195    /// A heading wider than the table is clipped, marked, over values that fit whole:
19196    /// it never leaves the table blank, at any width, in either glyph set, with or
19197    /// without the type row. It used to drop the column, and with it everything
19198    /// after, leaving the rail and an off-screen hint over nothing.
19199    #[test]
19200    fn a_long_header_never_blanks_the_table() {
19201        let name = format!("numeric_header_{}", "x".repeat(90));
19202        let df = DataFrame::new_infer_height(vec![
19203            Series::new(name.as_str().into(), &[1i64, 22, 333]).into(),
19204            Series::new("tail".into(), &["t1", "t2", "t3"]).into(),
19205        ])
19206        .unwrap();
19207        for g in glyph_sets() {
19208            for dtype_row in [false, true] {
19209                for width in [12u16, 20, 60, 80, 120] {
19210                    let table = || DataTable {
19211                        glyphs: g,
19212                        dtype_row,
19213                        ..DataTable::default()
19214                    };
19215                    let header_h = usize::from(table().header_height());
19216                    let mut state = state_of(&df, 3);
19217                    let rows = draw(table(), &mut state, width, header_h as u16 + 3);
19218                    let ctx = format!(
19219                        "{} glyphs, type row {dtype_row}, width {width}:\n{}",
19220                        set_name(g),
19221                        rows.join("\n")
19222                    );
19223                    assert!(rows[0].contains("num"), "{ctx}");
19224                    for (i, value) in ["1", "22", "333"].iter().enumerate() {
19225                        assert!(
19226                            rows[header_h + i].split_whitespace().any(|t| t == *value),
19227                            "{value} is whole on its row: {ctx}"
19228                        );
19229                    }
19230                    if dtype_row && width >= 20 {
19231                        assert!(rows[1].contains("i64"), "{ctx}");
19232                    }
19233                    // Wider than any cap on automatic widths: always clipped, and from
19234                    // 60 columns on, with the column after it beside it.
19235                    assert!(rows[0].contains(g.ellipsis), "{ctx}");
19236                    if width >= 60 {
19237                        assert!(rows[0].contains("tail"), "{ctx}");
19238                    }
19239                }
19240            }
19241        }
19242    }
19243
19244    /// A clipped heading gives way to its marks: the name is cut, never the sort
19245    /// direction, which is state.
19246    #[test]
19247    fn a_clipped_heading_keeps_its_sort_mark() {
19248        let name = format!("numeric_header_{}", "x".repeat(90));
19249        let df =
19250            DataFrame::new_infer_height(vec![Series::new(name.as_str().into(), &[1i64]).into()])
19251                .unwrap();
19252        for g in glyph_sets() {
19253            let table = DataTable {
19254                glyphs: g,
19255                ..DataTable::default()
19256            }
19257            .with_sort(vec![name.clone()], vec![true]);
19258            let area = Rect::new(0, 0, 30, 2);
19259            let mut buf = Buffer::empty(area);
19260            let mut ts = TableState::default();
19261            table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
19262            let header = header_row_string(&buf, area);
19263            assert!(
19264                header
19265                    .trim_end()
19266                    .ends_with(&format!("{}{}", g.ellipsis, g.sort_desc)),
19267                "{}: {header:?}",
19268                set_name(g)
19269            );
19270        }
19271    }
19272
19273    /// A heading whose drawn width is not its string's width (`لا` draws two cells,
19274    /// a halfwidth sound mark one) is drawn whole over right-aligned numbers, with its
19275    /// sort mark after it. ratatui's own alignment pushed the last letter off the cell,
19276    /// and the mark landed on it.
19277    #[test]
19278    fn a_heading_is_placed_by_the_cells_it_draws() {
19279        for name in ["الاسم", "ガギ"] {
19280            let df = DataFrame::new_infer_height(vec![
19281                Series::new(name.into(), &[1i64]).into(),
19282                Series::new("tail".into(), &["x"]).into(),
19283            ])
19284            .unwrap();
19285            let table = DataTable::default().with_sort(vec![name.to_string()], vec![false]);
19286            let mark = table.glyphs.sort_asc;
19287            let area = Rect::new(0, 0, 30, 3);
19288            let mut buf = Buffer::empty(area);
19289            let mut ts = TableState::default();
19290            table.render_dataframe(&df, area, &mut buf, &mut ts, false, 0);
19291            let rows: Vec<String> = (0..3).map(|y| drawn_from(&buf, y, 0)).collect();
19292            assert!(rows[0].starts_with(&format!("{name}{mark}")), "{rows:#?}");
19293            let width = crate::glyphs::cell_width(name) + 1;
19294            assert!(
19295                rows[1].starts_with(&format!("{:>width$} x", "1")),
19296                "{rows:#?}"
19297            );
19298        }
19299    }
19300
19301    /// A number too wide for the whole table shows its leading digits behind the clip
19302    /// marker: something rather than nothing, and never its trailing digits passing
19303    /// for a whole number, which is what ratatui's own cut of a right-aligned value
19304    /// would show.
19305    #[test]
19306    fn a_number_wider_than_the_table_is_a_marked_preview() {
19307        let full = "-1234567890123456789";
19308        let df = df!("n" => &[-1234567890123456789i64]).unwrap();
19309        for g in glyph_sets() {
19310            for width in 3u16..=24 {
19311                let mut state = state_of(&df, 1);
19312                let table = DataTable {
19313                    glyphs: g,
19314                    ..DataTable::default()
19315                };
19316                let rows = draw(table, &mut state, width, 2);
19317                let shown: String = rows[1].chars().skip(1).collect::<String>();
19318                let shown = shown.trim();
19319                let ctx = format!("{} glyphs, width {width}: {rows:?}", set_name(g));
19320                if usize::from(width) > full.len() {
19321                    assert_eq!(shown, full, "{ctx}");
19322                } else if g.ellipsis.starts_with(shown) {
19323                    // Room for no more than the marker.
19324                    assert!(!shown.is_empty(), "{ctx}");
19325                } else {
19326                    let kept = shown
19327                        .strip_suffix(g.ellipsis)
19328                        .unwrap_or_else(|| panic!("a clipped number carries the marker: {ctx}"));
19329                    assert!(full.starts_with(kept), "{ctx}");
19330                }
19331            }
19332        }
19333    }
19334
19335    /// Wide characters are measured in cells: a column of four CJK characters is
19336    /// eight cells wide, and shows whole beside the columns after it while there is
19337    /// room. Counted in characters, it was given four and clipped with space to spare.
19338    #[test]
19339    fn wide_characters_are_measured_in_cells() {
19340        let values = ["東京大阪", "京都横浜", "名古屋市"];
19341        let df = df!(
19342            "a" => &values,
19343            "b" => &[1i64, 2, 3],
19344            "tail" => &["x", "y", "z"],
19345        )
19346        .unwrap();
19347        for g in glyph_sets() {
19348            for width in [16u16, 30, 80] {
19349                let mut state = state_of(&df, 3);
19350                let table = DataTable {
19351                    glyphs: g,
19352                    ..DataTable::default()
19353                };
19354                let rows = draw(table, &mut state, width, 4);
19355                let ctx = format!("{} glyphs, width {width}: {rows:#?}", set_name(g));
19356                assert!(rows[0].contains("tail"), "{ctx}");
19357                for (i, value) in values.iter().enumerate() {
19358                    assert!(rows[1 + i].contains(value), "{ctx}");
19359                }
19360            }
19361        }
19362    }
19363
19364    /// A value cut where it meets the edge keeps whole graphemes and gains the clip
19365    /// marker: no half of a wide character, no accent split from its letter, no
19366    /// joined emoji broken apart, at any width, in either glyph set.
19367    #[test]
19368    fn a_clipped_cell_keeps_whole_graphemes() {
19369        let values = [
19370            "東京大阪名古屋横浜",
19371            "e\u{301}e\u{301}e\u{301}e\u{301}e\u{301}e\u{301}e\u{301}",
19372            "👩\u{200d}👩\u{200d}👧👍🏽🇯🇵 and more",
19373            "plain text that runs on",
19374        ];
19375        let df = df!("id" => &[1i64, 2, 3, 4], "text" => &values).unwrap();
19376        for g in glyph_sets() {
19377            for width in 4u16..=30 {
19378                let mut state = state_of(&df, 4);
19379                let area = Rect::new(0, 0, width, 5);
19380                let mut buf = Buffer::empty(area);
19381                DataTable {
19382                    glyphs: g,
19383                    ..DataTable::default()
19384                }
19385                .render(area, &mut buf, &mut state);
19386                // The rail, `id` two cells wide, and one cell of padding.
19387                let text_x = 4;
19388                if text_x >= width {
19389                    continue;
19390                }
19391                for (i, value) in values.iter().enumerate() {
19392                    let y = 1 + i as u16;
19393                    let shown = drawn_from(&buf, y, text_x);
19394                    let shown = shown.trim_end();
19395                    let ctx = format!("{} glyphs, width {width}, row {i}: {shown:?}", set_name(g));
19396                    if shown.is_empty() || shown == *value {
19397                        continue;
19398                    }
19399                    let kept = shown
19400                        .strip_suffix(g.ellipsis)
19401                        .unwrap_or_else(|| panic!("a clipped value is marked: {ctx}"));
19402                    let span = Span::raw(*value);
19403                    let mut whole = String::new();
19404                    for grapheme in span.styled_graphemes(Style::default()) {
19405                        if whole.len() >= kept.len() {
19406                            break;
19407                        }
19408                        whole.push_str(grapheme.symbol);
19409                    }
19410                    assert_eq!(whole, kept, "{ctx}");
19411                    // And each cell holds a whole grapheme of the value or the marker.
19412                    for x in text_x..width {
19413                        let symbol = buf[(x, y)].symbol();
19414                        assert!(
19415                            symbol == " "
19416                                || g.ellipsis.contains(symbol)
19417                                || span
19418                                    .styled_graphemes(Style::default())
19419                                    .any(|gr| gr.symbol == symbol),
19420                            "cell {x} holds {symbol:?}: {ctx}"
19421                        );
19422                    }
19423                }
19424            }
19425        }
19426    }
19427
19428    /// A frozen text column wider than the window is clipped, marked and still
19429    /// frozen, with the column after it beside it. It used to vanish, leaving only
19430    /// the column after it.
19431    #[test]
19432    fn a_frozen_long_text_is_clipped_not_dropped() {
19433        let url = format!("https://example.com/{}", "long-segment/".repeat(15));
19434        let df = df!("url" => &[url.as_str(), "short"], "tail" => &[7i64, 8]).unwrap();
19435        for g in glyph_sets() {
19436            for width in [40u16, 60, 80, 120] {
19437                let mut state = state_of(&df, 2);
19438                state.set_locked_columns(1);
19439                let table = DataTable {
19440                    glyphs: g,
19441                    ..DataTable::default()
19442                };
19443                let rows = draw(table, &mut state, width, 3);
19444                let ctx = format!("{} glyphs, width {width}: {rows:#?}", set_name(g));
19445                let (frozen, scrolled) = rows[0].split_once(g.rule).expect(&ctx);
19446                assert!(frozen.contains("url"), "{ctx}");
19447                assert!(scrolled.contains("tail"), "{ctx}");
19448                let (frozen, scrolled) = rows[1].split_once(g.rule).expect(&ctx);
19449                assert!(frozen.contains("https://exa"), "{ctx}");
19450                assert!(frozen.trim_end().ends_with(g.ellipsis), "{ctx}");
19451                assert!(scrolled.split_whitespace().any(|t| t == "7"), "{ctx}");
19452                assert_eq!(state.frozen_shown(), 1, "{ctx}");
19453            }
19454        }
19455    }
19456
19457    /// The frozen columns are measured on the rows on screen, like the rest. They
19458    /// were measured on the head of the buffer, which is not the page on screen once
19459    /// the view has moved into it, so a longer value on screen was cut short.
19460    #[test]
19461    fn frozen_columns_are_measured_on_the_rows_on_screen() {
19462        let long = "a much longer frozen value";
19463        let names: Vec<String> = (0..400)
19464            .map(|i| {
19465                if i == 302 {
19466                    long.to_string()
19467                } else {
19468                    format!("n{i}")
19469                }
19470            })
19471            .collect();
19472        let df = df!("name" => names, "v" => (0..400i64).collect::<Vec<_>>()).unwrap();
19473        let mut state = state_of(&df, 5);
19474        state.set_locked_columns(1);
19475        state.scroll_to(300);
19476        state.collect();
19477        assert!(
19478            state.buffered_start_row < state.start_row,
19479            "the page is not the head of the buffer: {} vs {}",
19480            state.buffered_start_row,
19481            state.start_row
19482        );
19483        let rows = draw(DataTable::default(), &mut state, 80, 6);
19484        assert!(
19485            rows.iter().any(|row| row.contains(long)),
19486            "the value on screen is whole: {rows:#?}"
19487        );
19488    }
19489
19490    /// Frozen columns are spaced like scrolling ones: the configured padding on both
19491    /// sides of the separator, at every setting. The frozen side used a hardcoded
19492    /// single space.
19493    #[test]
19494    fn frozen_and_scrolling_columns_share_the_padding() {
19495        let df = df!(
19496            "id" => &[1i64, 2],
19497            "k" => &["x", "y"],
19498            "v" => &[3i64, 4],
19499            "w" => &[5i64, 6],
19500        )
19501        .unwrap();
19502        for padding in [0u16, 1, 2, 3] {
19503            let mut state = state_of(&df, 2);
19504            state.set_locked_columns(2);
19505            let table = DataTable {
19506                table_cell_padding: padding,
19507                ..DataTable::default()
19508            };
19509            let rows = draw(table, &mut state, 40, 3);
19510            let gap = " ".repeat(usize::from(padding));
19511            let rule = crate::glyphs::get().rule;
19512            assert!(
19513                rows[0].starts_with(&format!(" id{gap}k {rule} v{gap}w")),
19514                "padding {padding}: {rows:#?}"
19515            );
19516            assert!(
19517                rows[1].contains(&format!(" 1{gap}x {rule} 3{gap}5")),
19518                "padding {padding}: {rows:#?}"
19519            );
19520        }
19521    }
19522
19523    /// The column names, in order, for the frozen-prefix tests.
19524    const PHONETIC: [&str; 6] = ["alpha", "bravo", "charlie", "delta", "echo", "foxtrot"];
19525
19526    fn phonetic_frame() -> DataFrame {
19527        let columns: Vec<Column> = PHONETIC
19528            .iter()
19529            .map(|n| {
19530                Series::new(
19531                    (*n).into(),
19532                    &[format!("{n}-value-one"), format!("{n}-value-two")],
19533                )
19534                .into()
19535            })
19536            .collect();
19537        DataFrame::new_infer_height(columns).unwrap()
19538    }
19539
19540    /// Every column name some frame shows while scrolling right from the start.
19541    fn names_reached(
19542        state: &mut DataTableState,
19543        g: &'static crate::glyphs::Glyphs,
19544        width: u16,
19545    ) -> Vec<&'static str> {
19546        let mut seen = Vec::new();
19547        for _ in 0..PHONETIC.len() + 2 {
19548            let table = DataTable {
19549                glyphs: g,
19550                ..DataTable::default()
19551            };
19552            let rows = draw(table, state, width, 3);
19553            for name in PHONETIC {
19554                if rows[0].contains(name) && !seen.contains(&name) {
19555                    seen.push(name);
19556                }
19557            }
19558            state.scroll_right();
19559        }
19560        seen
19561    }
19562
19563    /// Frozen columns that cannot all fit beside a usable scrolling column: as many
19564    /// as fit stay frozen, the broken rule says some had to scroll, those lead the
19565    /// scrolling side, every column stays reachable, and the request stands, so a
19566    /// wider window freezes them all again.
19567    #[test]
19568    fn a_frozen_prefix_too_wide_scrolls_until_there_is_room() {
19569        let df = phonetic_frame();
19570        for g in glyph_sets() {
19571            for row_numbers in [false, true] {
19572                let mut state = state_of(&df, 2);
19573                state.row_numbers = row_numbers;
19574                state.set_locked_columns(4);
19575                let table = || DataTable {
19576                    glyphs: g,
19577                    ..DataTable::default()
19578                };
19579                let rows = draw(table(), &mut state, 60, 3);
19580                let ctx = format!(
19581                    "{} glyphs, row numbers {row_numbers}: {rows:#?}",
19582                    set_name(g)
19583                );
19584                assert_eq!(state.locked_columns_count(), 4, "{ctx}");
19585                let shown = state.frozen_shown();
19586                assert!((1..4).contains(&shown), "{shown} frozen: {ctx}");
19587                let (frozen, scrolled) = rows[0].split_once(g.rule_broken).expect(&ctx);
19588                assert!(!rows[0].contains(g.rule), "{ctx}");
19589                assert!(frozen.contains(PHONETIC[0]), "{ctx}");
19590                assert!(
19591                    scrolled.trim_start().starts_with(PHONETIC[shown]),
19592                    "the first column left out leads the scrolling side: {ctx}"
19593                );
19594
19595                let reached = names_reached(&mut state, g, 60);
19596                assert_eq!(reached.len(), PHONETIC.len(), "{reached:?}: {ctx}");
19597
19598                let rows = draw(table(), &mut state, 160, 3);
19599                assert_eq!(state.frozen_shown(), 4, "{rows:#?}");
19600                let (frozen, _) = rows[0].split_once(g.rule).expect("the plain rule");
19601                for name in &PHONETIC[..4] {
19602                    assert!(frozen.contains(name), "{rows:#?}");
19603                }
19604            }
19605        }
19606    }
19607
19608    /// A rollback puts back the frozen fit its columns were sliced for: a wider
19609    /// layout since then must re-slice them, or the columns that had to scroll show
19610    /// twice, frozen and scrolling.
19611    #[test]
19612    fn a_rollback_keeps_the_frozen_fit_its_columns_were_sliced_for() {
19613        let df = phonetic_frame();
19614        let mut state = state_of(&df, 2);
19615        state.set_locked_columns(4);
19616        draw(DataTable::default(), &mut state, 60, 3);
19617        assert!(state.frozen_shown() < 4);
19618        let saved = state.rollback_point();
19619        draw(DataTable::default(), &mut state, 200, 3);
19620        assert_eq!(state.frozen_shown(), 4);
19621        state.roll_back(saved);
19622        let rows = draw(DataTable::default(), &mut state, 200, 3);
19623        for name in PHONETIC {
19624            assert_eq!(rows[0].matches(name).count(), 1, "{name}: {rows:#?}");
19625        }
19626    }
19627
19628    /// With every column frozen there is nothing to scroll, until the window is too
19629    /// narrow for them all: then the ones that do not fit scroll, and each is still
19630    /// reachable.
19631    #[test]
19632    fn with_every_column_frozen_each_is_still_reachable() {
19633        let df = phonetic_frame();
19634        for g in glyph_sets() {
19635            let mut state = state_of(&df, 2);
19636            state.set_locked_columns(PHONETIC.len());
19637            let table = || DataTable {
19638                glyphs: g,
19639                ..DataTable::default()
19640            };
19641            let rows = draw(table(), &mut state, 200, 3);
19642            assert_eq!(state.frozen_shown(), PHONETIC.len(), "{rows:#?}");
19643            assert!(rows[0].contains(g.rule), "{rows:#?}");
19644            assert!(!rows[0].contains(g.arrow_right), "{rows:#?}");
19645
19646            let rows = draw(table(), &mut state, 60, 3);
19647            assert!(state.frozen_shown() < PHONETIC.len(), "{rows:#?}");
19648            assert!(rows[0].contains(g.rule_broken), "{rows:#?}");
19649            let reached = names_reached(&mut state, g, 60);
19650            assert_eq!(reached.len(), PHONETIC.len(), "{reached:?}");
19651        }
19652    }
19653
19654    /// The page from the issue's reproduction: one description runs to a long URL.
19655    /// That page still shows its rows, with the description clipped and marked
19656    /// rather than any column vanishing for it.
19657    #[test]
19658    fn a_page_with_one_long_value_is_not_blank() {
19659        let url = format!("https://example.com/{}", "long-segment/".repeat(15));
19660        let n = 80usize;
19661        let df = df!(
19662            "id" => (0..n as i64).collect::<Vec<_>>(),
19663            "description" => (0..n)
19664                .map(|i| if i == 24 { url.clone() } else { format!("item {i}") })
19665                .collect::<Vec<_>>(),
19666            "amount" => (0..n).map(|i| i as f64 * 1.5).collect::<Vec<_>>(),
19667            "status" => (0..n).map(|i| if i % 2 == 0 { "open" } else { "closed" }).collect::<Vec<_>>(),
19668        )
19669        .unwrap();
19670        for g in glyph_sets() {
19671            for width in [60u16, 80, 120] {
19672                let mut state = state_of(&df, 20);
19673                state.page_down();
19674                state.collect();
19675                let table = DataTable {
19676                    glyphs: g,
19677                    ..DataTable::default()
19678                };
19679                let rows = draw(table, &mut state, width, 21);
19680                let ctx = format!("{} glyphs, width {width}: {rows:#?}", set_name(g));
19681                assert!(rows[0].contains("id"), "{ctx}");
19682                assert!(rows[0].contains("desc"), "{ctx}");
19683                let long = rows
19684                    .iter()
19685                    .find(|row| row.contains("https://example.com/"))
19686                    .expect(&ctx);
19687                assert!(long.contains(g.ellipsis), "{ctx}");
19688                for row in &rows[1..] {
19689                    assert!(!row.trim().is_empty(), "{ctx}");
19690                }
19691            }
19692        }
19693    }
19694
19695    /// The issue's long-URL frame: six columns over 80 rows, every description short
19696    /// but row 24's.
19697    fn long_url_frame() -> DataFrame {
19698        let url = format!("https://example.com/{}", "long-segment/".repeat(15));
19699        let n = 80usize;
19700        let start = NaiveDate::from_ymd_opt(2024, 1, 1)
19701            .unwrap()
19702            .and_hms_opt(0, 0, 0)
19703            .unwrap();
19704        df!(
19705            "id" => (0..n as i64).collect::<Vec<_>>(),
19706            "description" => (0..n)
19707                .map(|i| if i == 24 { url.clone() } else { format!("item {i}") })
19708                .collect::<Vec<_>>(),
19709            "amount" => (0..n).map(|i| i as f64 * 1.5).collect::<Vec<_>>(),
19710            "status" => (0..n).map(|i| if i % 2 == 0 { "open" } else { "closed" }).collect::<Vec<_>>(),
19711            "timestamp" => (0..n)
19712                .map(|i| start + chrono::Duration::hours(i as i64))
19713                .collect::<Vec<_>>(),
19714            "uuid" => (0..n)
19715                .map(|i| format!("00000000-0000-0000-0000-{:012x}", i * 7919 + 1))
19716                .collect::<Vec<_>>(),
19717        )
19718        .unwrap()
19719    }
19720
19721    /// Paging down past a page with one long value, and back, moves no column: the
19722    /// heading row is the same on every page, at every width, in both glyph sets.
19723    /// Measured per page, the long URL turned a six-column view into a two-column one.
19724    /// On its page the URL is clipped and marked.
19725    #[test]
19726    fn widths_hold_still_across_pages() {
19727        let df = long_url_frame();
19728        for g in glyph_sets() {
19729            for width in [60u16, 80, 120] {
19730                let mut state = state_of(&df, 20);
19731                let table = || DataTable {
19732                    glyphs: g,
19733                    ..DataTable::default()
19734                };
19735                let first = draw(table(), &mut state, width, 21);
19736                let ctx =
19737                    |rows: &[String]| format!("{} glyphs, width {width}: {rows:#?}", set_name(g));
19738                if width >= 80 {
19739                    assert!(first[0].contains("timestamp"), "{}", ctx(&first));
19740                }
19741                let mut saw_url = false;
19742                for step in 0..6 {
19743                    if step < 3 {
19744                        state.page_down();
19745                    } else {
19746                        state.page_up();
19747                    }
19748                    state.collect();
19749                    let rows = draw(table(), &mut state, width, 21);
19750                    assert_eq!(rows[0], first[0], "page {step}: {}", ctx(&rows));
19751                    if let Some(row) = rows.iter().find(|r| r.contains("https://")) {
19752                        saw_url = true;
19753                        let clipped = row.split_whitespace().nth(1).unwrap();
19754                        assert!(clipped.ends_with(g.ellipsis), "{}", ctx(&rows));
19755                    }
19756                }
19757                assert!(saw_url, "the long URL's page was drawn");
19758            }
19759        }
19760    }
19761
19762    /// A frozen column keeps its width across pages, so the number of columns that
19763    /// stay frozen does not change with the page either.
19764    #[test]
19765    fn frozen_columns_hold_still_across_pages() {
19766        let df = long_url_frame();
19767        for g in glyph_sets() {
19768            for width in [60u16, 80, 120] {
19769                let mut state = state_of(&df, 20);
19770                state.set_locked_columns(3);
19771                let table = || DataTable {
19772                    glyphs: g,
19773                    ..DataTable::default()
19774                };
19775                let first = draw(table(), &mut state, width, 21);
19776                let frozen = state.frozen_shown();
19777                for _ in 0..3 {
19778                    state.page_down();
19779                    state.collect();
19780                    let rows = draw(table(), &mut state, width, 21);
19781                    let ctx = format!("{} glyphs, width {width}: {rows:#?}", set_name(g));
19782                    assert_eq!(state.frozen_shown(), frozen, "{ctx}");
19783                    assert_eq!(rows[0], first[0], "{ctx}");
19784                }
19785            }
19786        }
19787    }
19788
19789    /// A number's column widens for a wider number on a later page, so it is never
19790    /// cut, and does not narrow again on a page of narrower ones.
19791    #[test]
19792    fn a_number_column_widens_and_stays_wide() {
19793        let values: Vec<i64> = (0..60)
19794            .map(|i| if i == 30 { 123_456_789 } else { i })
19795            .collect();
19796        let df = df!("n" => values, "t" => (0..60).map(|i| format!("t{i}")).collect::<Vec<_>>())
19797            .unwrap();
19798        let mut state = state_of(&df, 20);
19799        draw(DataTable::default(), &mut state, 60, 21);
19800        assert_eq!(state.shown_width("n"), Some(2));
19801        state.page_down();
19802        state.collect();
19803        let rows = draw(DataTable::default(), &mut state, 60, 21);
19804        assert!(rows.iter().any(|r| r.contains("123456789")), "{rows:#?}");
19805        state.page_down();
19806        state.collect();
19807        draw(DataTable::default(), &mut state, 60, 21);
19808        assert_eq!(state.shown_width("n"), Some(9));
19809    }
19810
19811    /// A struct is a preview like text: one long value on a later page is clipped
19812    /// at the cap, and the columns after it stay on screen on the pages after.
19813    #[test]
19814    fn a_long_struct_value_does_not_widen_its_column_for_good() {
19815        let n = 60usize;
19816        let y: Vec<String> = (0..n)
19817            .map(|i| {
19818                if i == 30 {
19819                    "a very long struct value ".repeat(6)
19820                } else {
19821                    "short".to_string()
19822                }
19823            })
19824            .collect();
19825        let s = StructChunked::from_series(
19826            "s".into(),
19827            n,
19828            [
19829                Series::new("x".into(), (0..n as i64).collect::<Vec<_>>()),
19830                Series::new("y".into(), y),
19831            ]
19832            .iter(),
19833        )
19834        .unwrap()
19835        .into_series();
19836        let df = DataFrame::new_infer_height(vec![
19837            Series::new("id".into(), (0..n as i64).collect::<Vec<_>>()).into(),
19838            s.into(),
19839            Series::new("tail".into(), (0..n as i64).collect::<Vec<_>>()).into(),
19840        ])
19841        .unwrap();
19842        let mut state = state_of(&df, 20);
19843        for page in 0..3 {
19844            let rows = draw(DataTable::default(), &mut state, 80, 21);
19845            assert!(rows[0].contains("tail"), "page {page}: {rows:#?}");
19846            assert!(
19847                state.shown_width("s").is_some_and(|w| w <= 32),
19848                "page {page}: {rows:#?}"
19849            );
19850            if page == 1 {
19851                let long = rows.iter().find(|r| r.contains("{30,")).unwrap();
19852                assert!(long.contains(crate::glyphs::get().ellipsis), "{rows:#?}");
19853            }
19854            state.page_down();
19855            state.collect();
19856        }
19857    }
19858
19859    /// Automatic text and headings stop at two fifths of the terminal, marked where
19860    /// cut: 32 cells at 80 columns, 48 at 120.
19861    #[test]
19862    fn automatic_text_stops_at_the_cap_and_is_marked() {
19863        let long = "x".repeat(200);
19864        let df = df!("text" => &[long.as_str()], "tail" => &[1i64]).unwrap();
19865        for g in glyph_sets() {
19866            for (width, cap) in [(80u16, 32u16), (120, 48)] {
19867                let mut state = state_of(&df, 1);
19868                let table = DataTable {
19869                    glyphs: g,
19870                    ..DataTable::default()
19871                };
19872                let rows = draw(table, &mut state, width, 2);
19873                let ctx = format!("{} glyphs, width {width}: {rows:#?}", set_name(g));
19874                assert_eq!(state.shown_width("text"), Some(cap), "{ctx}");
19875                let row: String = rows[1].chars().skip(1).collect();
19876                let value = row.split_whitespace().next().unwrap();
19877                assert!(value.ends_with(g.ellipsis), "{ctx}");
19878                assert_eq!(crate::glyphs::cell_width(value), usize::from(cap), "{ctx}");
19879                assert!(rows[0].contains("tail"), "{ctx}");
19880            }
19881        }
19882    }
19883
19884    /// The last column drawn takes the room to the table's right edge: a long text on
19885    /// the far right runs to the edge rather than stopping at the cap. A width set by
19886    /// hand is drawn as set, and a number stays under its heading.
19887    #[test]
19888    fn the_last_column_runs_to_the_right_edge() {
19889        let long = "x".repeat(200);
19890        let df = df!("id" => &[1i64], "text" => &[long.as_str()]).unwrap();
19891        let ellipsis = crate::glyphs::get().ellipsis;
19892        let mut state = state_of(&df, 1);
19893        let rows = draw(DataTable::default(), &mut state, 120, 2);
19894        let value = rows[1].trim_end();
19895        assert!(value.ends_with(ellipsis), "{rows:#?}");
19896        assert_eq!(crate::glyphs::cell_width(value), 120, "{rows:#?}");
19897        assert_eq!(state.shown_width("text"), Some(48), "learned at the cap");
19898        assert!(state.on_screen_width("text").unwrap() > 48, "drawn past it");
19899
19900        state.set_width_choices([("text".to_string(), WidthChoice::Manual(20))]);
19901        let rows = draw(DataTable::default(), &mut state, 120, 2);
19902        assert_eq!(state.shown_width("text"), Some(20), "{rows:#?}");
19903        assert!(
19904            crate::glyphs::cell_width(rows[1].trim_end()) < 40,
19905            "{rows:#?}"
19906        );
19907
19908        let df = df!("text" => &["ab"], "n" => &[5i64]).unwrap();
19909        let mut state = state_of(&df, 1);
19910        let rows = draw(DataTable::default(), &mut state, 120, 2);
19911        assert!(
19912            crate::glyphs::cell_width(rows[1].trim_end()) < 20,
19913            "{rows:#?}"
19914        );
19915    }
19916
19917    /// The cap follows the terminal, not the table: a sidebar narrowing the table
19918    /// moves no column.
19919    #[test]
19920    fn a_sidebar_moves_no_column() {
19921        let df = long_url_frame();
19922        let mut state = state_of(&df, 20);
19923        state.page_down();
19924        state.collect();
19925        let table = || DataTable {
19926            screen_width: 120,
19927            ..DataTable::default()
19928        };
19929        let whole = draw(table(), &mut state, 120, 21);
19930        let widths: Vec<_> = ["id", "description", "amount"]
19931            .iter()
19932            .map(|c| state.shown_width(c))
19933            .collect();
19934        let beside = draw(table(), &mut state, 70, 21);
19935        for (c, before) in ["id", "description", "amount"].iter().zip(&widths) {
19936            assert_eq!(state.shown_width(c), *before, "{c}: {beside:#?}");
19937        }
19938        assert!(
19939            whole[0].starts_with(&beside[0][..40]),
19940            "{whole:#?} {beside:#?}"
19941        );
19942    }
19943
19944    /// A width set by hand: exact for text, values clipped and marked; it survives
19945    /// paging, scrolling, reordering, hiding and resizing; fit and reset change it.
19946    #[test]
19947    fn a_manual_width_survives_everything_but_fit_and_reset() {
19948        let df = long_url_frame();
19949        let mut state = state_of(&df, 20);
19950        state.set_width_choices([("description".to_string(), WidthChoice::Manual(6))]);
19951        let rows = draw(DataTable::default(), &mut state, 80, 21);
19952        assert_eq!(state.shown_width("description"), Some(6), "{rows:#?}");
19953        // Six cells, the ellipsis among them: `item …`, or `ite...` in ASCII.
19954        let ellipsis = crate::glyphs::get().ellipsis;
19955        let kept = 6 - crate::glyphs::display_width(ellipsis);
19956        let clipped = format!("{}{ellipsis}", &"item 10"[..kept]);
19957        assert!(rows.iter().any(|r| r.contains(&clipped)), "{rows:#?}");
19958
19959        state.page_down();
19960        state.collect();
19961        state.scroll_right();
19962        draw(DataTable::default(), &mut state, 80, 21);
19963        state.scroll_left();
19964        // Moved to the end, then hidden and shown again.
19965        let mut order = state.headers();
19966        order.retain(|c| c != "description");
19967        order.push("description".to_string());
19968        state.set_column_order(order.clone());
19969        draw(DataTable::default(), &mut state, 200, 21);
19970        assert_eq!(state.shown_width("description"), Some(6));
19971        order.pop();
19972        state.set_column_order(order.clone());
19973        draw(DataTable::default(), &mut state, 40, 21);
19974        order.insert(1, "description".to_string());
19975        state.set_column_order(order);
19976        draw(DataTable::default(), &mut state, 120, 21);
19977        assert_eq!(state.width_choice("description"), WidthChoice::Manual(6));
19978        assert_eq!(state.shown_width("description"), Some(6));
19979
19980        // Fit takes the rows on screen: page two holds the URL.
19981        state.set_width_choices([("description".to_string(), WidthChoice::Fit)]);
19982        draw(DataTable::default(), &mut state, 120, 21);
19983        let url_width = 20 + 13 * 15;
19984        assert_eq!(
19985            state.width_choice("description"),
19986            WidthChoice::Manual(url_width)
19987        );
19988        state.reset();
19989        assert_eq!(state.width_choice("description"), WidthChoice::Auto);
19990    }
19991
19992    /// A column scrolled out of view is fitted to the rows on screen too.
19993    #[test]
19994    fn a_column_out_of_view_is_fitted_to_the_page() {
19995        let df = long_url_frame();
19996        let mut state = state_of(&df, 20);
19997        state.page_down();
19998        state.collect();
19999        for _ in 0..3 {
20000            state.scroll_right();
20001        }
20002        draw(DataTable::default(), &mut state, 80, 21);
20003        state.set_width_choices([("description".to_string(), WidthChoice::Fit)]);
20004        let rows = draw(DataTable::default(), &mut state, 80, 21);
20005        assert!(!rows[0].contains("description"), "{rows:#?}");
20006        assert_eq!(
20007            state.width_choice("description"),
20008            WidthChoice::Manual(20 + 13 * 15)
20009        );
20010    }
20011
20012    /// A number column set narrower than its numbers still shows them whole.
20013    #[test]
20014    fn a_number_column_set_narrow_still_shows_whole_numbers() {
20015        let df = df!("n" => &[1_234_567i64, 2], "t" => &["a", "b"]).unwrap();
20016        let mut state = state_of(&df, 2);
20017        state.set_width_choices([("n".to_string(), WidthChoice::Manual(4))]);
20018        let rows = draw(DataTable::default(), &mut state, 40, 3);
20019        assert!(rows[1].contains("1234567"), "{rows:#?}");
20020    }
20021
20022    /// The same name with another type, as a query can make it, is another column:
20023    /// it starts from an automatic width, and the first gets its own back.
20024    #[test]
20025    fn a_column_whose_type_changes_starts_afresh() {
20026        let df = df!("a" => &["x", "y"], "n" => &[1i64, 2]).unwrap();
20027        let mut state = state_of(&df, 2);
20028        state.set_width_choices([("a".to_string(), WidthChoice::Manual(9))]);
20029        state.query("select a: n".to_string());
20030        assert_eq!(state.width_choice("a"), WidthChoice::Auto);
20031        state.query("select a, n".to_string());
20032        assert_eq!(state.width_choice("a"), WidthChoice::Manual(9));
20033    }
20034
20035    /// A column ending at the table's edge with more after it leaves its heading's
20036    /// last cell to the off-screen hint, so the hint covers neither a letter nor the
20037    /// clip marker, and the values keep the full width: a number that fits is whole,
20038    /// never a preview.
20039    #[test]
20040    fn the_offscreen_hint_takes_a_heading_cell_not_a_value_cell() {
20041        let df = df!(
20042            "aaaaaaaaaa" => &["aaaaaaaaaa"],
20043            "bbbbbbb" => &[1_234_567i64],
20044            "c" => &["x"],
20045        )
20046        .unwrap();
20047        for g in glyph_sets() {
20048            for dtype_row in [false, true] {
20049                let table = DataTable {
20050                    glyphs: g,
20051                    dtype_row,
20052                    ..DataTable::default()
20053                };
20054                let header_h = usize::from(table.header_height());
20055                let mut state = state_of(&df, 1);
20056                // The rail, ten cells, a gap and seven: the second column ends at the edge.
20057                let rows = draw(table, &mut state, 19, header_h as u16 + 1);
20058                let ctx = format!("{} glyphs, type row {dtype_row}: {rows:#?}", set_name(g));
20059                assert!(rows[header_h].ends_with(" 1234567"), "{ctx}");
20060                let hint_row = &rows[header_h - 1];
20061                assert!(hint_row.ends_with(g.arrow_right), "{ctx}");
20062                let before = hint_row.strip_suffix(g.arrow_right).unwrap();
20063                if dtype_row {
20064                    assert!(before.trim_end().ends_with("i64"), "{ctx}");
20065                } else {
20066                    assert!(before.ends_with(g.ellipsis), "{ctx}");
20067                }
20068            }
20069        }
20070    }
20071
20072    /// A followed file's page near its end, and a filtered view's count after rows
20073    /// arrive, are read from the marks the watcher made, not from the file's start.
20074    #[test]
20075    fn a_followed_view_reads_and_counts_from_its_marks() {
20076        use std::io::Write as _;
20077        let dir = tempfile::tempdir().unwrap();
20078        let path = dir.path().join("grow.csv");
20079        let mut text = String::from("t,n\n");
20080        for i in 0..20_000 {
20081            text.push_str(&format!("{i},{}\n", i % 7));
20082        }
20083        std::fs::write(&path, &text).unwrap();
20084        let scan = LazyCsvReader::new(PlRefPath::try_from_path(&path).unwrap())
20085            .with_ignore_errors(true)
20086            .finish()
20087            .unwrap();
20088        let options = crate::OpenOptions::default();
20089        let (lf, tail) =
20090            crate::follow::bound_to_complete(scan, &path, crate::FileFormat::Csv, &options)
20091                .unwrap();
20092        let (tx, rx) = std::sync::mpsc::channel();
20093        let mut state = DataTableState::new(lf, None, None, None, None, false).unwrap();
20094        let rows = tail.rows();
20095        state.start_following(crate::follow::Follow::start(
20096            tail,
20097            std::time::Duration::from_secs(3_600),
20098            tx,
20099            None,
20100        ));
20101        state.follow_to(rows, false);
20102        let from_marks = |lf: &LazyFrame| format!("{:?}", lf.logical_plan).contains("FOLLOWED");
20103        let page = state.buffer_lf(19_990, 10).unwrap();
20104        assert!(from_marks(&page));
20105        let t = |df: DataFrame| df.column("t").unwrap().i64().unwrap().to_vec();
20106        assert_eq!(t(page.collect().unwrap()).first(), Some(&Some(19_990)));
20107
20108        state.defer_collect = true;
20109        state.filter(vec![FilterStatement {
20110            columns: Vec::new(),
20111            column: "n".to_string(),
20112            operator: crate::filter_modal::FilterOperator::Eq,
20113            value: "3".to_string(),
20114            logical_op: crate::filter_modal::LogicalOperator::And,
20115        }]);
20116        let matches = |n: usize| (0..n).filter(|i| i % 7 == 3).count();
20117        // The first count of the filter reads the whole file.
20118        assert!(state.source_counter().is_none());
20119        state.set_num_rows(matches(20_000));
20120
20121        let mut out = std::fs::OpenOptions::new()
20122            .append(true)
20123            .open(&path)
20124            .unwrap();
20125        let more: String = (20_000..20_050)
20126            .map(|i| format!("{i},{}\n", i % 7))
20127            .collect();
20128        out.write_all(more.as_bytes()).unwrap();
20129        let follow = state.follow_mut().unwrap();
20130        follow.check_now();
20131        // A watcher that hears changes may report before the check it was asked for.
20132        let (rows, restarted) = loop {
20133            let crate::AppEvent::Followed(news) =
20134                rx.recv_timeout(std::time::Duration::from_secs(30)).unwrap()
20135            else {
20136                panic!("the watcher said something else");
20137            };
20138            follow.take(&news.change);
20139            if follow.waiting() == 50 {
20140                break follow.catch_up();
20141            }
20142        };
20143        assert_eq!(rows, 20_050);
20144        state.follow_to(rows, restarted);
20145        assert!(!state.is_num_rows_valid());
20146        let counter = state.source_counter().expect("counts the new rows alone");
20147        assert_eq!(counter().unwrap(), matches(20_050));
20148        state.set_num_rows(matches(20_050));
20149        let last = state.buffer_lf(matches(20_050) - 3, 3).unwrap();
20150        assert!(from_marks(&last));
20151        let expected: Vec<_> = (0..20_050i64).filter(|i| i % 7 == 3).map(Some).collect();
20152        assert_eq!(t(last.collect().unwrap()), expected[expected.len() - 3..]);
20153    }
20154}