Skip to main content

datui_lib/table/
copy.rs

1//! Drilling into groups and values, the inspector's rows, and what a copy takes.
2
3use super::*;
4
5/// The grouped view and pipeline state a drill-down saved, so filters and sort inside
6/// the group apply to it and `drill_up` restores the grouped view.
7#[derive(Clone)]
8pub(super) struct GroupedView {
9    pub(super) lf: LazyFrame,
10    pub(super) base_lf: LazyFrame,
11    /// The schemas of `base_lf` and of the view, so coming back resolves neither.
12    base_schema: Arc<Schema>,
13    schema: Arc<Schema>,
14    pub(super) filters: Vec<FilterStatement>,
15    pub(super) sort_columns: Vec<String>,
16    pub(super) sort_descending: Vec<bool>,
17    pub(super) sort_ascending: bool,
18    /// Whether `lf` has the hidden drift column and its groups' meaning, so drilling up
19    /// restores those cells and their notes.
20    drift: bool,
21    drift_groups: Arc<Vec<crate::formats::schema_union::DriftGroup>>,
22    /// Whether `lf` numbers its rows itself (`#`).
23    view_numbered: bool,
24    notes: Vec<crate::notes::Note>,
25    pub(super) group_source: Option<GroupSource>,
26    /// Where the user was, so coming back puts the cursor on the group drilled into
27    /// with the key columns still frozen.
28    column_order: Vec<String>,
29    locked_columns_count: usize,
30    start_row: usize,
31    termcol_index: usize,
32    cursor_column: Option<String>,
33    selected: Option<usize>,
34    /// Drilled into from Value Counts rather than from a grouped row.
35    by_value: bool,
36    /// Each drill that led here, as its key columns and values in one row: the group
37    /// (or value) drilled into, then each value it was narrowed to. See [`DrillPlace`].
38    steps: Vec<DataFrame>,
39    /// How `base_lf` was built, for Copy as Python.
40    base_steps: Vec<Step>,
41    lineage: Lineage,
42}
43
44/// The rows a grouped result came from and how its keys were computed, so a drill-down
45/// finds a group's rows even from aggregates. Recorded by the grouping query, not
46/// inferred from (possibly renamed) columns.
47#[derive(Clone)]
48pub(super) struct GroupSource {
49    /// The rows before grouping, after any filter the query applied first.
50    pub(super) rows: LazyFrame,
51    /// Each key's column in the result, with the expression that computes it from `rows`.
52    pub(super) keys: Vec<(PlSmallStr, Expr)>,
53    /// Columns `rows` carries only to compute keys, left out of a drill.
54    pub(super) scratch: Vec<PlSmallStr>,
55    /// Whether the result's list columns are each group's rows, as a `by` query's are.
56    /// A SQL result's lists are values it computed, such as `ARRAY_AGG`.
57    pub(super) rows_in_lists: bool,
58    /// The same as Copy as Python steps: how `rows` was built, and each key as
59    /// Python code, aliases undone. None where the script cannot say.
60    pub(super) python_rows: Option<Vec<Step>>,
61    pub(super) python_keys: Vec<Option<String>>,
62    /// Which loaded column each column of `rows` is.
63    pub(super) lineage: Lineage,
64}
65
66/// One field the row inspector lists: a column of the frame on screen.
67#[derive(Debug, Clone, PartialEq)]
68pub struct InspectField {
69    pub name: String,
70    pub dtype: DataType,
71    /// Hidden from the table, so not among the rows read for it.
72    pub hidden: bool,
73}
74
75impl InspectField {
76    /// Whether the table's rows hold this field's value: a hidden column is not
77    /// read for them, and a binary one is read as a stub.
78    pub fn buffered(&self) -> bool {
79        !self.hidden && !matches!(self.dtype, DataType::Binary)
80    }
81}
82
83/// The selected row as the buffer holds it; see [`DataTableState::inspect_row`].
84#[derive(Clone)]
85pub struct InspectRow {
86    /// The row's index in the view.
87    pub row: usize,
88    /// The frame it is a row of (`len_generation`), which a sort, a filter or a
89    /// query replaces.
90    pub frame: u64,
91    /// The row number the table shows.
92    pub display_row: usize,
93    /// The row, one column per table column, raw.
94    pub values: DataFrame,
95    /// Which drift group its file is in, when the files differ.
96    pub drift_group: Option<u32>,
97}
98
99/// What a null means where the files of a dataset differ.
100#[derive(Debug, Clone, Copy, PartialEq, Eq)]
101pub enum NullKind {
102    Null,
103    Absent,
104    Conflict,
105}
106
107/// A drill-down as the drills that reached it and what was done inside, to take
108/// again over a new read of the same data (a reopen). A group is found again by its
109/// keys, since its row's position or lists may have changed.
110#[derive(Debug, Clone)]
111pub struct DrillPlace {
112    /// Whether the first step is a grouped row, found by reading the grouped view,
113    /// rather than a value from Value Counts.
114    by_group: bool,
115    /// Each drill's key columns and values, as one row.
116    steps: Vec<DataFrame>,
117    filters: Vec<FilterStatement>,
118    sort_columns: Vec<String>,
119    sort_descending: Vec<bool>,
120    column_order: Vec<String>,
121    locked_columns_count: usize,
122}
123
124impl DrillPlace {
125    /// Whether its first drill is into a grouped row, found again by reading.
126    pub fn by_group(&self) -> bool {
127        self.by_group
128    }
129
130    /// The group or value drilled into first, as `key = value` pairs.
131    pub fn describe(&self) -> String {
132        self.steps.first().map(describe).unwrap_or_default()
133    }
134}
135
136/// A drill's keys as `key = value` pairs.
137fn describe(keys: &DataFrame) -> String {
138    keys.columns()
139        .iter()
140        .map(|c| {
141            let value = c.get(0).map(|v| crate::exact::str_value(&v).to_string());
142            format!("{} = {}", c.name(), value.unwrap_or_default())
143        })
144        .collect::<Vec<_>>()
145        .join(", ")
146}
147
148/// A drill a reopen could not take again: the files no longer hold its group or
149/// value (`key = value`), or hold it as another type.
150#[derive(Debug)]
151pub struct DrillGone(pub String);
152
153impl std::fmt::Display for DrillGone {
154    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
155        write!(f, "No rows with {} in the current files", self.0)
156    }
157}
158
159impl std::error::Error for DrillGone {}
160
161/// The view row index column a search for a group to drill into again reads.
162const GROUP_ROW: &str = "__datui_group_row";
163
164/// The row a drill into a group reads, from [`DataTableState::drill_row`].
165pub enum DrillRow {
166    /// Taken from the rows on screen.
167    Buffered(DataFrame),
168    /// Not on hand: the one-row frame to collect, off the UI thread.
169    Read(Box<LazyFrame>),
170}
171
172/// A group's rows, drilled from its row of a grouped view.
173pub(super) struct GroupRows {
174    lf: LazyFrame,
175    /// The key columns of the grouped view and the row's values in them, as text.
176    key_columns: Vec<String>,
177    key_values: Vec<String>,
178    /// Columns of `lf` that hold the keys as they stand, to lead the view.
179    lead: Vec<String>,
180    /// How `lf` is built, as Copy as Python steps.
181    steps: Vec<Step>,
182    /// Which loaded column each column of `lf` is.
183    lineage: Lineage,
184}
185
186/// The rows an export writes. See [`DataTableState::export_frame`].
187pub struct ExportFrame {
188    pub(super) lf: LazyFrame,
189    pub(super) files: Option<SourceFiles>,
190}
191
192/// The dataset's files in scan order, and the row each starts at.
193pub(super) struct SourceFiles {
194    pub(super) names: Arc<Vec<String>>,
195    pub(super) starts: Arc<Vec<usize>>,
196}
197
198impl ExportFrame {
199    /// Rows that are not the view's, such as a column's value counts.
200    pub fn of(lf: LazyFrame) -> Self {
201        Self { lf, files: None }
202    }
203}
204
205impl ExportFrame {
206    /// The name of the column an export adds when asked to say where each row is from.
207    pub const SOURCE_FILE_COLUMN: &'static str = "source_file";
208
209    /// The plan. Naming files reads the schema (possibly resolving the scan), so off the UI
210    /// thread; names map from the row index per batch, so streamed exports never hold
211    /// every row.
212    pub fn into_lazy(self) -> PolarsResult<LazyFrame> {
213        let Some(SourceFiles { names, starts }) = self.files else {
214            return Ok(self.lf);
215        };
216        let mut lf = self.lf;
217        let schema = lf.collect_schema()?;
218        let name = Self::free_name(schema.iter_names().map(|n| n.as_str()));
219        let index = crate::formats::schema_union::DRIFT_COLUMN;
220        let file_of = move |rows: Column| -> PolarsResult<Column> {
221            let rows = rows.strict_cast(&DataType::UInt64)?;
222            let named: StringChunked = rows
223                .u64()?
224                .iter()
225                .map(|row| {
226                    let row = row? as usize;
227                    let file = starts
228                        .partition_point(|&start| start <= row)
229                        .saturating_sub(1);
230                    names.get(file).map(String::as_str)
231                })
232                .collect();
233            Ok(named.with_name(rows.name().clone()).into_column())
234        };
235        // Added last, after the dataset's own columns, as the index comes off.
236        Ok(lf
237            .with_column(
238                col(index)
239                    .map(file_of, |_, field| {
240                        Ok(Field::new(field.name().clone(), DataType::String))
241                    })
242                    .alias(name),
243            )
244            .drop(by_name([index], true, false)))
245    }
246
247    /// A name for the source-file column no column has: `source_file` may exist already,
248    /// and adding a same-named column would silently replace it.
249    fn free_name<'a>(taken: impl Iterator<Item = &'a str>) -> String {
250        let taken: HashSet<&str> = taken.collect();
251        std::iter::once(Self::SOURCE_FILE_COLUMN.to_string())
252            .chain((1..).map(|n| format!("{}_{n}", Self::SOURCE_FILE_COLUMN)))
253            .find(|candidate| !taken.contains(candidate.as_str()))
254            .expect("some suffix is free")
255    }
256}
257
258impl DataTableState {
259    /// The group drilled into, as its key columns and their values.
260    pub fn drilled_group_key(&self) -> Option<(&[String], &[String])> {
261        let values = self.view.drilled_down_group_key.as_deref()?;
262        let columns = self
263            .view
264            .drilled_down_group_key_columns
265            .as_deref()
266            .unwrap_or_default();
267        Some((columns, values))
268    }
269
270    /// The drill-down on screen, to take again over a new read; `None` when not drilled.
271    pub fn drill_place(&self) -> Option<DrillPlace> {
272        let grouped = self.view.grouped.as_ref()?;
273        Some(DrillPlace {
274            by_group: !grouped.by_value,
275            steps: grouped.steps.clone(),
276            filters: self.view.filters.clone(),
277            sort_columns: self.view.sort_columns.clone(),
278            sort_descending: self.view.sort_descending.clone(),
279            column_order: self.view.column_order.clone(),
280            locked_columns_count: self.view.locked_columns_count,
281        })
282    }
283
284    /// The column order and locked count of the view a drill-down left, while drilled.
285    pub fn grouped_column_order(&self) -> Option<(&[String], usize)> {
286        let grouped = self.view.grouped.as_ref()?;
287        Some((&grouped.column_order, grouped.locked_columns_count))
288    }
289
290    /// What finding `place`'s group again reads: its first row in this view with the
291    /// same keys, led by its position. `None` when its first drill was a value, which
292    /// needs no read; [`DrillGone`] when a key is no longer a value of its column.
293    /// Run off this thread; [`Self::found_group`] reads the answer.
294    pub fn find_group(&self, place: &DrillPlace) -> Option<Result<LazyFrame>> {
295        let keys = place.steps.first().filter(|_| place.by_group)?;
296        if !self.can_drill_down() {
297            return Some(Err(color_eyre::eyre::eyre!("the view is not grouped")));
298        }
299        let mut matches = lit(true);
300        for key in keys.columns() {
301            let value = key.get(0).ok().map(|v| v.into_static());
302            let Some((dtype, value)) =
303                value.and_then(|v| self.value_in_column(key.name().as_str(), v))
304            else {
305                return Some(Err(DrillGone(place.describe()).into()));
306            };
307            let scalar = Scalar::new(dtype, value);
308            matches = matches.and(col(key.name().clone()).eq_missing(lit(scalar)));
309        }
310        let columns = std::iter::once(GROUP_ROW.to_string()).chain(self.drill_columns());
311        Some(Ok(self
312            .visible_lf()
313            .with_row_index(GROUP_ROW, None)
314            .filter(matches)
315            .select(columns.map(|c| col(c.as_str())).collect::<Vec<_>>())
316            .slice(0, 1)))
317    }
318
319    /// `value` as a value of `column` in this view, in the column's type: cast strictly
320    /// when its own type differs (a file rewritten since it was read). `None` when
321    /// there is no such column or the value is not one of its type.
322    fn value_in_column(
323        &self,
324        column: &str,
325        value: AnyValue<'static>,
326    ) -> Option<(DataType, AnyValue<'static>)> {
327        let dtype = self.view.schema.get(column)?.clone();
328        if value.dtype() == dtype || value.is_null() {
329            return Some((dtype, value));
330        }
331        let cast = Series::from_any_values(PlSmallStr::EMPTY, &[value], true)
332            .and_then(|s| s.strict_cast(&dtype))
333            .ok()?;
334        let value = cast.get(0).ok()?.into_static();
335        Some((dtype, value))
336    }
337
338    /// The group [`Self::find_group`] read: its view row and its row, or `None` when no
339    /// group has those keys now.
340    pub fn found_group(read: DataFrame) -> Result<Option<(usize, DataFrame)>> {
341        if read.height() == 0 {
342            return Ok(None);
343        }
344        let index = read
345            .column(GROUP_ROW)?
346            .get(0)?
347            .extract::<usize>()
348            .ok_or_else(|| color_eyre::eyre::eyre!("no row position"))?;
349        Ok(Some((index, read.drop(GROUP_ROW)?)))
350    }
351
352    /// Take `place`'s drill-down again over this view (as one transition: see
353    /// [`Self::try_transition`]), returning [`DrillGone`] when a value it narrowed to
354    /// is no longer one of its column's type: into the group `found` (from
355    /// [`Self::find_group`]) or the first value, through each value narrowed to, then
356    /// the filters, sort and column order inside it.
357    pub fn redrill(&mut self, place: &DrillPlace, found: Option<(usize, DataFrame)>) -> Result<()> {
358        let mut steps = place.steps.iter();
359        if place.by_group {
360            steps.next();
361            let Some((index, row)) = found else {
362                return Ok(());
363            };
364            self.drill_down_with_row(index, &row)?;
365        }
366        for keys in steps {
367            let Some(key) = keys.columns().first() else {
368                continue;
369            };
370            let column = key.name().as_str();
371            let Some((_, value)) = self.value_in_column(column, key.get(0)?.into_static()) else {
372                return Err(DrillGone(describe(keys)).into());
373            };
374            self.drill_into_value(column, value)?;
375        }
376        if !place.filters.is_empty() {
377            self.filter(place.filters.clone());
378        }
379        if !place.sort_columns.is_empty() {
380            self.sort_by(place.sort_columns.clone(), place.sort_descending.clone());
381        }
382        if let Some(error) = self.error().cloned() {
383            return Err(color_eyre::eyre::eyre!("{error}"));
384        }
385        let shown: HashSet<&str> = self.view.schema.iter_names().map(|n| n.as_str()).collect();
386        if place
387            .column_order
388            .iter()
389            .all(|c| shown.contains(c.as_str()))
390        {
391            self.set_column_order(place.column_order.clone());
392            self.set_locked_columns(place.locked_columns_count);
393        }
394        Ok(())
395    }
396
397    /// The columns and types of `df`, the table SQL runs against, from the known schema; a
398    /// drilled group or reshape only resolves its plan, reading nothing.
399    pub fn sql_table_columns(&self) -> Vec<(String, DataType)> {
400        let schema = if self.view.grouped.is_none() && self.view.reshaped_lf.is_none() {
401            Some(self.original_schema.clone())
402        } else {
403            self.query_root().collect_schema().ok()
404        };
405        schema
406            .map(|schema| {
407                schema
408                    .iter()
409                    .filter(|(name, _)| name.as_str() != crate::formats::schema_union::DRIFT_COLUMN)
410                    .map(|(name, dtype)| (name.to_string(), dtype.clone()))
411                    .collect()
412            })
413            .unwrap_or_default()
414    }
415
416    /// Rows `df` holds, when that is known without counting: the data as loaded,
417    /// once its count has come back.
418    pub fn sql_table_rows(&self) -> Option<usize> {
419        if self.view.grouped.is_some() || self.view.reshaped_lf.is_some() {
420            return None;
421        }
422        self.pristine_rows
423    }
424
425    pub fn get_active_fuzzy_query(&self) -> &str {
426        &self.view.active_fuzzy_query
427    }
428
429    pub fn last_pivot_spec(&self) -> Option<&PivotSpec> {
430        self.view.last_pivot_spec.as_ref()
431    }
432
433    pub fn last_melt_spec(&self) -> Option<&MeltSpec> {
434        self.view.last_melt_spec.as_ref()
435    }
436
437    /// What the pivot or melt in effect ran over. See the field.
438    pub fn reshape_source(&self) -> Option<&ReshapeSource> {
439        self.view.reshape_source.as_ref()
440    }
441
442    /// Whether the view is a grouping's result (a `by` query, SQL GROUP BY) whose rows drill
443    /// into groups; recorded by the query, never inferred from list columns.
444    pub fn is_grouped(&self) -> bool {
445        self.view.group_source.is_some()
446    }
447
448    /// Whether any column holds lists.
449    fn has_list_columns(&self) -> bool {
450        self.view
451            .schema
452            .iter()
453            .any(|(_, dtype)| matches!(dtype, DataType::List(_)))
454    }
455
456    fn group_key_columns(&self) -> Vec<String> {
457        self.view
458            .schema
459            .iter()
460            .filter(|(_, dtype)| !matches!(dtype, DataType::List(_)))
461            .map(|(name, _)| name.to_string())
462            .collect()
463    }
464
465    fn group_value_columns(&self) -> Vec<String> {
466        self.view
467            .schema
468            .iter()
469            .filter(|(_, dtype)| matches!(dtype, DataType::List(_)))
470            .map(|(name, _)| name.to_string())
471            .collect()
472    }
473
474    /// Names of binary columns in the source schema. Their values are not read into the display
475    /// buffer (the `‹binary›` stub stands in); the renderer uses this to style those cells.
476    pub fn binary_column_names(&self) -> std::collections::HashSet<String> {
477        self.view
478            .schema
479            .iter()
480            .filter(|(_, dtype)| matches!(dtype, DataType::Binary))
481            .map(|(name, _)| name.to_string())
482            .collect()
483    }
484
485    /// Estimated heap size in bytes of the currently buffered slice (locked + scrollable), if collected.
486    pub fn buffered_memory_bytes(&self) -> Option<usize> {
487        let locked = self
488            .view
489            .locked_df
490            .as_ref()
491            .map(|df| df.estimated_size())
492            .unwrap_or(0);
493        let scroll = self
494            .view
495            .df
496            .as_ref()
497            .map(|df| df.estimated_size())
498            .unwrap_or(0);
499        if locked == 0 && scroll == 0 {
500            None
501        } else {
502            Some(locked + scroll)
503        }
504    }
505
506    /// The rows on hand, first to past the last.
507    pub fn buffered_span(&self) -> (usize, usize) {
508        (self.view.buffered_start_row, self.view.buffered_end_row)
509    }
510
511    /// Rows in the buffer; 0 when none is loaded.
512    pub fn buffered_rows(&self) -> usize {
513        self.view
514            .buffered_end_row
515            .saturating_sub(self.view.buffered_start_row)
516    }
517
518    /// The first `limit` distinct non-null values of `column` in the display buffer, as
519    /// text, a long string by its start. Reads nothing; a column not buffered gives none.
520    pub(crate) fn buffered_values(&self, column: &str, limit: usize) -> Vec<String> {
521        let Some(series) = [self.view.df.as_ref(), self.view.locked_df.as_ref()]
522            .into_iter()
523            .flatten()
524            .find_map(|df| df.column(column).ok())
525        else {
526            return Vec::new();
527        };
528        let series = series.as_materialized_series();
529        let mut values = Vec::new();
530        for value in (0..series.len()).filter_map(|index| series.get(index).ok()) {
531            if values.len() == limit {
532                break;
533            }
534            let text = match value {
535                AnyValue::Null => continue,
536                // Drawn each frame beside a column's name: its start, not a whole
537                // article.
538                AnyValue::String(text) => {
539                    crate::exact::prefix(text, crate::exact::CELL_PREVIEW_BYTES).to_string()
540                }
541                AnyValue::List(items) => crate::exact::list_preview(&items),
542                // Drawn on the UI thread: Polars' display panics on a date past
543                // the calendar.
544                value => {
545                    crate::exact::past_calendar_text(&value).unwrap_or_else(|| value.to_string())
546                }
547            };
548            if !values.contains(&text) {
549                values.push(text);
550            }
551        }
552        values
553    }
554
555    /// Current scrollable display buffer. None until first collect().
556    pub fn display_df(&self) -> Option<&DataFrame> {
557        self.view.df.as_ref()
558    }
559
560    /// The display buffer's rows on screen, as the table draws them.
561    pub fn display_slice_df(&self) -> Option<DataFrame> {
562        let df = self.view.df.as_ref()?;
563        let offset = self
564            .view
565            .start_row
566            .saturating_sub(self.view.buffered_start_row);
567        let slice_len = self.visible_rows.min(df.height().saturating_sub(offset));
568        if offset < df.height() && slice_len > 0 {
569            Some(df.slice(offset as i64, slice_len))
570        } else {
571            None
572        }
573    }
574
575    /// The selected row with every display column, raw and in display order:
576    /// what a row copy carries.
577    pub fn copy_row_df(&self) -> Option<DataFrame> {
578        let df = self.view.buffered_df.as_ref()?;
579        let absolute = self.view.start_row + self.table_state.selected()?;
580        let offset = absolute.checked_sub(self.view.buffered_start_row)?;
581        if offset >= df.height() {
582            return None;
583        }
584        let names: Vec<&str> = self.view.column_order.iter().map(|s| s.as_str()).collect();
585        df.select(names).ok().map(|d| d.slice(offset as i64, 1))
586    }
587
588    /// The rows on screen with every display column, raw, in display order, ignoring
589    /// column scroll (scrolled-past identifiers make the rows readable).
590    pub fn copy_view_df(&self) -> Option<DataFrame> {
591        let df = self.view.buffered_df.as_ref()?;
592        let names: Vec<&str> = self.view.column_order.iter().map(|s| s.as_str()).collect();
593        let selected = df.select(names).ok()?;
594        let offset = self
595            .view
596            .start_row
597            .saturating_sub(self.view.buffered_start_row);
598        let len = self
599            .visible_rows
600            .min(selected.height().saturating_sub(offset));
601        (len > 0).then(|| selected.slice(offset as i64, len))
602    }
603
604    /// The selected row's value in `column`, exactly as stored ([`crate::exact`]): floats
605    /// as their round-tripping decimal, null as empty (as in exports), lists and structs as
606    /// JSON.
607    pub fn copy_cell_value(&self, column: &str) -> Option<String> {
608        let row = self.copy_row_df()?;
609        crate::exact::copy_text(row.column(column).ok()?).ok()
610    }
611
612    /// The selected row's number as the row-numbers column would print it.
613    pub fn selected_display_row(&self) -> Option<usize> {
614        Some(self.view.start_row + self.table_state.selected()? + self.row_start_index)
615    }
616
617    /// Rows times estimated row width (binary at base64 size), for the copy guard. `None`
618    /// until counted, or while a binary column's width is unknown (only a footer says).
619    pub fn estimated_copy_bytes(&self) -> Option<usize> {
620        let rows = self.num_rows_if_valid()?;
621        if rows == 0 {
622            return Some(0);
623        }
624        let base64 = |bytes: usize| bytes.div_ceil(3) * 4;
625        let footer_width = |name: &str| {
626            self.column_bytes
627                .iter()
628                .find(|(n, _)| n == name)
629                .map(|(_, w)| *w)
630        };
631        let mut row = self.bytes_per_row();
632        for name in &self.view.column_order {
633            match self.view.schema.get(name.as_str()) {
634                Some(DataType::Binary) => row += base64(footer_width(name)?),
635                // Buffered whole, so the buffer measured it with the row; base64 adds
636                // a third on top.
637                Some(dtype) if crate::export::nested_json::has_binary(dtype) => {
638                    let buffered = self.view.buffered_df.as_ref().and_then(|df| {
639                        let column = df.column(name).ok()?;
640                        (df.height() > 0)
641                            .then(|| column.as_materialized_series().estimated_size() / df.height())
642                    });
643                    row += buffered.or_else(|| footer_width(name)).unwrap_or(0) / 3;
644                }
645                _ => {}
646            }
647        }
648        Some(rows.saturating_mul(row))
649    }
650
651    /// Each row's drift group from the top of the view, for a frame `frame_rows` tall (the
652    /// whole frame, header included: extra entries are ignored, and `visible_rows` is 0
653    /// before the first frame). Empty once a query or reshape replaced the frame (its rows
654    /// stand for no file).
655    pub fn display_drift(&self, frame_rows: usize) -> Vec<u32> {
656        if !self.view.drift_column_present {
657            return Vec::new();
658        }
659        let Some(df) = self.view.buffered_df.as_ref() else {
660            return Vec::new();
661        };
662        let Ok(column) = df.column(crate::formats::schema_union::DRIFT_COLUMN) else {
663            return Vec::new();
664        };
665        let offset = self
666            .view
667            .start_row
668            .saturating_sub(self.view.buffered_start_row);
669        let len = frame_rows.min(column.len().saturating_sub(offset));
670        if len == 0 {
671            return Vec::new();
672        }
673        let slice = column.slice(offset as i64, len);
674        let Ok(rows) = slice.u32() else {
675            return Vec::new();
676        };
677        // The column holds each row's place in the dataset. The file it came from is
678        // the last one starting at or before it, and the file says what it is missing.
679        rows.iter()
680            .map(|row| self.file_group_of(row.unwrap_or(0) as usize))
681            .collect()
682    }
683
684    /// Maximum buffer size in rows (0 = no limit).
685    pub fn max_buffered_rows(&self) -> usize {
686        self.max_buffered_rows
687    }
688
689    /// Maximum buffer size in MiB (0 = no limit).
690    pub fn max_buffered_mb(&self) -> usize {
691        self.max_buffered_mb
692    }
693
694    /// Whether Enter on a row drills into its group: the view is a grouped result and
695    /// not already a group's rows.
696    pub fn can_drill_down(&self) -> bool {
697        !self.is_drilled_down() && self.is_grouped()
698    }
699
700    /// Whether a drill shows the lists of a row as the group's rows rather than
701    /// filtering the source: a `by` result that holds its groups as lists.
702    fn drills_lists(&self) -> bool {
703        self.has_list_columns()
704            && self
705                .view
706                .group_source
707                .as_ref()
708                .is_some_and(|s| s.rows_in_lists)
709    }
710
711    /// The columns of a row that a drill into its group reads: every column of a result
712    /// holding its groups as lists, the keys of one holding aggregates.
713    fn drill_columns(&self) -> Vec<String> {
714        if self.drills_lists() {
715            return self
716                .view
717                .schema
718                .iter_names()
719                .map(|n| n.to_string())
720                .collect();
721        }
722        self.view
723            .group_source
724            .iter()
725            .flat_map(|source| source.keys.iter().map(|(name, _)| name.to_string()))
726            .collect()
727    }
728
729    /// The columns the inspector lists for a row: the table's, in its order, then
730    /// the ones hidden from it, in schema order. Never the scan's own row index.
731    pub fn inspect_fields(&self) -> Vec<InspectField> {
732        let shown = self.view.column_order.iter().filter_map(|name| {
733            Some(InspectField {
734                name: name.clone(),
735                dtype: self.view.schema.get(name.as_str())?.clone(),
736                hidden: false,
737            })
738        });
739        let hidden = self
740            .view
741            .schema
742            .iter()
743            .filter(|(name, _)| {
744                name.as_str() != crate::formats::schema_union::DRIFT_COLUMN
745                    && !self.view.column_order.iter().any(|c| c == name.as_str())
746            })
747            .map(|(name, dtype)| InspectField {
748                name: name.to_string(),
749                dtype: dtype.clone(),
750                hidden: true,
751            });
752        shown.chain(hidden).collect()
753    }
754
755    /// The selected row as the buffer holds it, with the file group that says what
756    /// its nulls are. Reads nothing; `None` while the row is not on hand.
757    pub fn inspect_row(&self) -> Option<InspectRow> {
758        self.inspect_row_at(self.view.start_row + self.table_state.selected()?)
759    }
760
761    /// Row `row` of the view as the buffer holds it, as [`Self::inspect_row`] does
762    /// the selected one: Compare's next row. `None` while it is not on hand.
763    pub fn inspect_row_at(&self, row: usize) -> Option<InspectRow> {
764        let df = self.view.buffered_df.as_ref()?;
765        let offset = row.checked_sub(self.view.buffered_start_row)?;
766        if offset >= df.height() {
767            return None;
768        }
769        let names: Vec<&str> = self.view.column_order.iter().map(|s| s.as_str()).collect();
770        let values = df.select(names).ok()?.slice(offset as i64, 1);
771        let drift_group = self
772            .view
773            .drift_column_present
774            .then(|| df.column(crate::formats::schema_union::DRIFT_COLUMN).ok())
775            .flatten()
776            .and_then(|c| c.get(offset).ok())
777            .and_then(|v| v.extract::<usize>())
778            .map(|place| self.file_group_of(place));
779        Some(InspectRow {
780            row,
781            frame: self.view.len_generation,
782            display_row: row + self.row_start_index,
783            values,
784            drift_group,
785        })
786    }
787
788    /// The drift group of the file holding the dataset's row `place`.
789    fn file_group_of(&self, place: usize) -> u32 {
790        let file = self
791            .drift_file_starts
792            .partition_point(|&start| start <= place)
793            .saturating_sub(1);
794        self.drift_file_group.get(file).copied().unwrap_or(0)
795    }
796
797    /// What a null in `column` is, for a row of file group `group`: the data's own,
798    /// a file without the column, or a file holding it in another type.
799    pub fn null_kind(&self, column: &str, group: Option<u32>) -> NullKind {
800        let Some(group) = group.and_then(|g| self.view.drift_groups.get(g as usize)) else {
801            return NullKind::Null;
802        };
803        if group.absent.iter().any(|c| c == column) {
804            NullKind::Absent
805        } else if group.unread.iter().any(|c| c == column) {
806            NullKind::Conflict
807        } else {
808            NullKind::Null
809        }
810    }
811
812    /// The frame reading `columns` of view row `row` for the inspector, through the
813    /// buffer's window (a remote dataset reads only that file). Run off this thread.
814    pub fn inspect_read_lf(&self, row: usize, columns: &[String]) -> PolarsResult<LazyFrame> {
815        let exprs = columns.iter().map(|c| col(c.as_str())).collect();
816        self.window_lf(row, 1, exprs)
817    }
818
819    /// What drilling into the group on view row `group_index` reads, or `None` when not
820    /// grouped. The buffer holds the row; a missing column (hidden, binary stub) means
821    /// reading the row, which for an aggregate is the whole aggregate, so off the UI thread.
822    pub fn drill_row(&self, group_index: usize) -> Option<DrillRow> {
823        if !self.can_drill_down() {
824            return None;
825        }
826        let columns = self.drill_columns();
827        let buffered = self
828            .view
829            .buffered_df
830            .as_ref()
831            .filter(|_| {
832                (self.view.buffered_start_row..self.view.buffered_end_row).contains(&group_index)
833            })
834            .filter(|_| {
835                columns
836                    .iter()
837                    .all(|c| !matches!(self.view.schema.get(c.as_str()), Some(DataType::Binary)))
838            })
839            .and_then(|df| df.select(columns.iter().map(|c| c.as_str())).ok())
840            .map(|df| df.slice((group_index - self.view.buffered_start_row) as i64, 1))
841            .filter(|row| row.height() == 1);
842        Some(match buffered {
843            Some(row) => DrillRow::Buffered(row),
844            None => DrillRow::Read(Box::new(
845                self.visible_lf()
846                    .select(columns.iter().map(|c| col(c.as_str())).collect::<Vec<_>>())
847                    .slice(group_index as i64, 1),
848            )),
849        })
850    }
851
852    /// Show the group on view row `group_index`, reading the row on this thread if not
853    /// buffered; the app uses [`Self::drill_row`] so keys never wait.
854    pub fn drill_down_into_group(&mut self, group_index: usize) -> Result<()> {
855        let row = match self.drill_row(group_index) {
856            None => return Ok(()),
857            Some(DrillRow::Buffered(row)) => row,
858            Some(DrillRow::Read(lf)) => collect_lazy(*lf, self.polars_streaming)?,
859        };
860        self.drill_down_with_row(group_index, &row)
861    }
862
863    /// Show the group whose row `group_index` is `row` (from [`Self::drill_row`]): list
864    /// columns as rows, or for aggregates the source rows sharing the group's keys, key
865    /// columns first.
866    pub fn drill_down_with_row(&mut self, group_index: usize, row: &DataFrame) -> Result<()> {
867        if !self.can_drill_down() {
868            return Ok(());
869        }
870        if row.height() == 0 {
871            return Err(color_eyre::eyre::eyre!("Group index out of bounds"));
872        }
873        let mut group = if self.drills_lists() {
874            Self::group_from_lists(
875                row,
876                self.group_key_columns(),
877                self.group_value_columns(),
878                self.view.lineage.clone(),
879            )?
880        } else if let Some(source) = &self.view.group_source {
881            Self::group_from_source(source, row)?
882        } else {
883            return Ok(());
884        };
885        // A list form that also aggregates (`select a, n: count a by k`) holds its
886        // aggregates beside the keys; the query knows which columns are keys.
887        if let Some(source) = self
888            .view
889            .group_source
890            .as_ref()
891            .filter(|_| self.drills_lists())
892        {
893            let keys: Vec<&str> = source.keys.iter().map(|(n, _)| n.as_str()).collect();
894            (group.key_columns, group.key_values) = group
895                .key_columns
896                .into_iter()
897                .zip(group.key_values)
898                .filter(|(name, _)| keys.contains(&name.as_str()))
899                .unzip();
900        }
901        let keys = row.select(group.key_columns.iter().map(String::as_str))?;
902        self.enter_group(group, group_index, false, keys)
903    }
904
905    /// Show the view's rows with `value` in `column` (null matches null), like a group
906    /// drill: breadcrumb names the value, Esc returns. Inside a group, it narrows that group
907    /// and Esc returns to the view it was drilled from.
908    pub fn drill_into_value(&mut self, column: &str, value: AnyValue<'static>) -> Result<()> {
909        let (dtype, value) = self
910            .value_in_column(column, value)
911            .ok_or_else(|| color_eyre::eyre::eyre!("{column} has no values of that type"))?;
912        let label = crate::exact::str_value(&value).to_string();
913        let mut steps = self.view_steps();
914        steps.push(match crate::export::python_script::py_value(&value) {
915            Some(literal) => Step::Matching(vec![(
916                format!("pl.col({})", crate::export::python_script::py_str(column)),
917                literal,
918            )]),
919            None => Step::Unreproducible(format!(
920                "drilled down to the rows where {column} is {label}, a value of a type not written as Python"
921            )),
922        });
923        let keys = DataFrame::new(
924            1,
925            vec![Column::new_scalar(
926                column.into(),
927                Scalar::new(dtype.clone(), value.clone()),
928                1,
929            )],
930        )?;
931        let matches = col(column).eq_missing(lit(Scalar::new(dtype, value)));
932        let group = GroupRows {
933            lf: self.visible_lf().filter(matches),
934            key_columns: vec![column.to_string()],
935            key_values: vec![label],
936            lead: vec![column.to_string()],
937            steps,
938            lineage: self.view.lineage.clone(),
939        };
940        if !self.is_drilled_down() {
941            let index = self.view.start_row + self.table_state.selected().unwrap_or(0);
942            return self.enter_group(group, index, true, keys);
943        }
944        let schema = group.lf.clone().collect_schema()?;
945        let order = std::mem::take(&mut self.view.column_order);
946        if let Some(keys) = self.view.drilled_down_group_key_columns.as_mut() {
947            keys.extend(group.key_columns);
948        }
949        if let Some(values) = self.view.drilled_down_group_key.as_mut() {
950            values.extend(group.key_values);
951        }
952        if let Some(grouped) = self.view.grouped.as_mut() {
953            grouped.steps.push(keys);
954        }
955        // The group's filters and sort are in the frame now.
956        self.view.filters.clear();
957        self.view.sort_columns.clear();
958        self.view.sort_descending.clear();
959        self.view.sort_ascending = true;
960        self.install_base(group.lf, schema);
961        self.view.base_steps = group.steps;
962        self.view.lineage = group.lineage;
963        self.view.column_order = order;
964        self.view.start_row = 0;
965        self.termcol_index = 0;
966        self.clear_column_moves();
967        self.settle_cursor();
968        self.table_state.select(Some(0));
969        self.collect();
970        Ok(())
971    }
972
973    /// Whether the drill on screen came from Value Counts.
974    pub fn drilled_into_value(&self) -> bool {
975        self.view.grouped.as_ref().is_some_and(|view| view.by_value)
976    }
977
978    /// Show `group`, the group on row `group_index` of the view, keeping the view to
979    /// come back to.
980    fn enter_group(
981        &mut self,
982        group: GroupRows,
983        group_index: usize,
984        by_value: bool,
985        keys: DataFrame,
986    ) -> Result<()> {
987        let schema = group.lf.clone().collect_schema()?;
988        self.view.drilled_down_group_key = Some(group.key_values);
989        self.view.drilled_down_group_key_columns = Some(group.key_columns);
990
991        // The group becomes the pipeline root while drilled in, so a sidebar filter or
992        // sort applies within it instead of rebuilding the grouped view underneath.
993        self.view.grouped = Some(GroupedView {
994            lf: self.view.lf.clone(),
995            base_lf: self.view.base_lf.clone(),
996            base_schema: self.view.base_schema.clone(),
997            schema: self.view.schema.clone(),
998            filters: std::mem::take(&mut self.view.filters),
999            sort_columns: std::mem::take(&mut self.view.sort_columns),
1000            sort_descending: std::mem::take(&mut self.view.sort_descending),
1001            sort_ascending: self.view.sort_ascending,
1002            drift: self.view.drift_column_present,
1003            drift_groups: self.view.drift_groups.clone(),
1004            view_numbered: self.view.view_numbered,
1005            notes: self.view.notes.clone(),
1006            group_source: self.view.group_source.take(),
1007            column_order: self.view.column_order.clone(),
1008            locked_columns_count: self.view.locked_columns_count,
1009            start_row: self.view.start_row,
1010            termcol_index: self.termcol_index,
1011            cursor_column: self.view.cursor_column.clone(),
1012            selected: self.table_state.selected(),
1013            by_value,
1014            steps: vec![keys],
1015            base_steps: std::mem::take(&mut self.view.base_steps),
1016            lineage: self.view.lineage.clone(),
1017        });
1018        self.view.sort_ascending = true;
1019        self.install_base(group.lf, schema);
1020        self.view.base_steps = group.steps;
1021        self.view.lineage = group.lineage;
1022        // Led by the keys, as a group drilled from lists is.
1023        let rest: Vec<String> = std::mem::take(&mut self.view.column_order)
1024            .into_iter()
1025            .filter(|c| !group.lead.contains(c))
1026            .collect();
1027        self.view.column_order = group.lead.into_iter().chain(rest).collect();
1028        self.view.drilled_down_group_index = Some(group_index);
1029        self.view.start_row = 0;
1030        self.termcol_index = 0;
1031        self.clear_column_moves();
1032        self.view.locked_columns_count = 0;
1033        self.settle_cursor();
1034        self.table_state.select(Some(0));
1035        self.collect();
1036
1037        Ok(())
1038    }
1039
1040    /// A group held as lists in `row`: the keys repeated beside the lists as columns.
1041    fn group_from_lists(
1042        row: &DataFrame,
1043        key_columns: Vec<String>,
1044        value_columns: Vec<String>,
1045        lineage: Lineage,
1046    ) -> Result<GroupRows> {
1047        if value_columns.is_empty() {
1048            return Err(color_eyre::eyre::eyre!("No value columns in grouped data"));
1049        }
1050        let row_count = match row.column(&value_columns[0])?.get(0)? {
1051            AnyValue::List(list_series) => list_series.len(),
1052            _ => 0,
1053        };
1054
1055        let mut columns = Vec::new();
1056        let mut key_values = Vec::new();
1057        for col_name in &key_columns {
1058            let key = row.column(col_name)?;
1059            key_values.push(crate::exact::str_value(&key.get(0)?).to_string());
1060            // Repeated in its own type, so a date stays a date and a null key null.
1061            columns.push(key.new_from_index(0, row_count));
1062        }
1063        for col_name in &value_columns {
1064            if let AnyValue::List(list_series) = row.column(col_name)?.get(0)? {
1065                columns.push(list_series.with_name(col_name.as_str().into()).into());
1066            }
1067        }
1068        let group = key_columns
1069            .iter()
1070            .zip(&key_values)
1071            .map(|(c, v)| format!("{c} = {v}"))
1072            .collect::<Vec<_>>()
1073            .join(", ");
1074        Ok(GroupRows {
1075            lf: DataFrame::new_infer_height(columns)?.lazy(),
1076            key_columns,
1077            key_values,
1078            // Already first.
1079            lead: Vec::new(),
1080            steps: vec![Step::Unreproducible(format!(
1081                "drilled down into the group {group}, read from the grouped result's lists: \
1082                 not written as Python"
1083            ))],
1084            // The lists keep the result's names.
1085            lineage,
1086        })
1087    }
1088
1089    /// A group of an aggregated result in `row`: the source rows whose keys equal the
1090    /// row's, a null key matching nulls.
1091    fn group_from_source(source: &GroupSource, row: &DataFrame) -> Result<GroupRows> {
1092        let mut predicate: Option<Expr> = None;
1093        let mut key_columns = Vec::new();
1094        let mut key_values = Vec::new();
1095        let mut lead = Vec::new();
1096        let mut matching = Vec::new();
1097        for (i, (name, expr)) in source.keys.iter().enumerate() {
1098            let column = row.column(name)?;
1099            let value = column.get(0)?.into_static();
1100            matching.push(
1101                source
1102                    .python_keys
1103                    .get(i)
1104                    .cloned()
1105                    .flatten()
1106                    .zip(crate::export::python_script::py_value(&value)),
1107            );
1108            key_columns.push(name.to_string());
1109            key_values.push(crate::exact::str_value(&value).to_string());
1110            // The key as the query computed it, against the value it produced; the alias
1111            // only named the result's column.
1112            let key = expr.clone().meta().undo_aliases();
1113            if let Expr::Column(source_column) = &key
1114                && !source.scratch.contains(source_column)
1115            {
1116                lead.push(source_column.to_string());
1117            }
1118            let matches = key.eq_missing(lit(Scalar::new(column.dtype().clone(), value)));
1119            predicate = Some(match predicate {
1120                Some(all) => all.and(matches),
1121                None => matches,
1122            });
1123        }
1124        let rows = source.rows.clone();
1125        let mut lf = match predicate {
1126            Some(predicate) => rows.filter(predicate),
1127            None => rows,
1128        };
1129        if !source.scratch.is_empty() {
1130            lf = lf.drop(by_name(source.scratch.iter().cloned(), true, false));
1131        }
1132        let matching: Option<Vec<(String, String)>> = matching.into_iter().collect();
1133        let steps = match (&source.python_rows, matching) {
1134            (Some(rows), Some(matching)) => {
1135                let mut steps = rows.clone();
1136                steps.push(Step::Matching(matching));
1137                if !source.scratch.is_empty() {
1138                    steps.push(Step::Drop(
1139                        source.scratch.iter().map(|c| c.to_string()).collect(),
1140                    ));
1141                }
1142                steps
1143            }
1144            _ => vec![Step::Unreproducible(format!(
1145                "drilled down into the group {}: not written as Python",
1146                key_columns
1147                    .iter()
1148                    .zip(&key_values)
1149                    .map(|(c, v)| format!("{c} = {v}"))
1150                    .collect::<Vec<_>>()
1151                    .join(", ")
1152            ))],
1153        };
1154        Ok(GroupRows {
1155            lf,
1156            key_columns,
1157            key_values,
1158            lead,
1159            steps,
1160            lineage: source.lineage.clone(),
1161        })
1162    }
1163
1164    pub fn drill_up(&mut self) -> Result<()> {
1165        let Some(view) = self.view.grouped.take() else {
1166            return Err(color_eyre::eyre::eyre!("Not in drill-down mode"));
1167        };
1168        self.invalidate_num_rows();
1169        // The buffer holds the group's rows; kept, it would stand in for the grouped
1170        // view wherever the view fits inside it.
1171        self.drop_buffer();
1172        self.view.observed_bytes_per_row = None;
1173        self.widths.relearn();
1174        self.view.lf = view.lf;
1175        self.view.unsorted_lf = None;
1176        self.view.base_lf = view.base_lf;
1177        self.view.base_schema = view.base_schema;
1178        self.view.base_steps = view.base_steps;
1179        self.view.lineage = view.lineage;
1180        self.view.filters = view.filters;
1181        self.view.sort_columns = view.sort_columns;
1182        self.view.sort_descending = view.sort_descending;
1183        self.view.sort_ascending = view.sort_ascending;
1184        self.view.drift_column_present = view.drift;
1185        self.view.drift_groups = view.drift_groups;
1186        self.view.view_numbered = view.view_numbered;
1187        self.view.notes = view.notes;
1188        self.view.group_source = view.group_source;
1189        // The restored frame already excludes what its filter and sort excluded; rederive
1190        // those notes so they cannot be stale.
1191        self.view.view_notes = self.view_notes_only();
1192        self.view.schema = view.schema;
1193        self.view.column_order = view.column_order;
1194        self.view.locked_columns_count = view.locked_columns_count;
1195        self.view.drilled_down_group_index = None;
1196        self.view.drilled_down_group_key = None;
1197        self.view.drilled_down_group_key_columns = None;
1198        self.view.start_row = view.start_row;
1199        self.termcol_index = view.termcol_index;
1200        self.clear_column_moves();
1201        self.view.cursor_column = view.cursor_column;
1202        self.settle_cursor();
1203        self.table_state.select(view.selected);
1204        self.collect();
1205        Ok(())
1206    }
1207}