Skip to main content

datui_lib/table/
copy.rs

1//! Drilling into groups and values, the inspector's rows, and what a copy takes.
2
3use super::*;
4
5/// The grouped view and pipeline state a drill-down saved, so filters and sort inside
6/// the group apply to it and `drill_up` restores the grouped view.
7#[derive(Clone)]
8pub(super) struct GroupedView {
9    pub(super) lf: LazyFrame,
10    pub(super) base_lf: LazyFrame,
11    /// The schemas of `base_lf` and of the view, so coming back resolves neither.
12    base_schema: Arc<Schema>,
13    schema: Arc<Schema>,
14    pub(super) filters: Vec<FilterStatement>,
15    pub(super) sort_columns: Vec<String>,
16    pub(super) sort_descending: Vec<bool>,
17    pub(super) sort_ascending: bool,
18    /// Whether `lf` has the hidden drift column and its groups' meaning, so drilling up
19    /// restores those cells and their notes.
20    drift: bool,
21    drift_groups: Arc<Vec<crate::formats::schema_union::DriftGroup>>,
22    /// Whether `lf` numbers its rows itself (`#`).
23    view_numbered: bool,
24    notes: Vec<crate::notes::Note>,
25    pub(super) group_source: Option<GroupSource>,
26    /// Where the user was, so coming back puts the cursor on the group drilled into
27    /// with the key columns still frozen.
28    column_order: Vec<String>,
29    locked_columns_count: usize,
30    start_row: usize,
31    termcol_index: usize,
32    cursor_column: Option<String>,
33    selected: Option<usize>,
34    /// Drilled into from Value Counts rather than from a grouped row.
35    by_value: bool,
36    /// How `base_lf` was built, for Copy as Python.
37    base_steps: Vec<Step>,
38    lineage: Lineage,
39}
40
41/// The rows a grouped result came from and how its keys were computed, so a drill-down
42/// finds a group's rows even from aggregates. Recorded by the grouping query, not
43/// inferred from (possibly renamed) columns.
44#[derive(Clone)]
45pub(super) struct GroupSource {
46    /// The rows before grouping, after any filter the query applied first.
47    pub(super) rows: LazyFrame,
48    /// Each key's column in the result, with the expression that computes it from `rows`.
49    pub(super) keys: Vec<(PlSmallStr, Expr)>,
50    /// Columns `rows` carries only to compute keys, left out of a drill.
51    pub(super) scratch: Vec<PlSmallStr>,
52    /// Whether the result's list columns are each group's rows, as a `by` query's are.
53    /// A SQL result's lists are values it computed, such as `ARRAY_AGG`.
54    pub(super) rows_in_lists: bool,
55    /// The same as Copy as Python steps: how `rows` was built, and each key as
56    /// Python code, aliases undone. None where the script cannot say.
57    pub(super) python_rows: Option<Vec<Step>>,
58    pub(super) python_keys: Vec<Option<String>>,
59    /// Which loaded column each column of `rows` is.
60    pub(super) lineage: Lineage,
61}
62
63/// One field the row inspector lists: a column of the frame on screen.
64#[derive(Debug, Clone, PartialEq)]
65pub struct InspectField {
66    pub name: String,
67    pub dtype: DataType,
68    /// Hidden from the table, so not among the rows read for it.
69    pub hidden: bool,
70}
71
72impl InspectField {
73    /// Whether the table's rows hold this field's value: a hidden column is not
74    /// read for them, and a binary one is read as a stub.
75    pub fn buffered(&self) -> bool {
76        !self.hidden && !matches!(self.dtype, DataType::Binary)
77    }
78}
79
80/// The selected row as the buffer holds it; see [`DataTableState::inspect_row`].
81#[derive(Clone)]
82pub struct InspectRow {
83    /// The row's index in the view.
84    pub row: usize,
85    /// The frame it is a row of (`len_generation`), which a sort, a filter or a
86    /// query replaces.
87    pub frame: u64,
88    /// The row number the table shows.
89    pub display_row: usize,
90    /// The row, one column per table column, raw.
91    pub values: DataFrame,
92    /// Which drift group its file is in, when the files differ.
93    pub drift_group: Option<u32>,
94}
95
96/// What a null means where the files of a dataset differ.
97#[derive(Debug, Clone, Copy, PartialEq, Eq)]
98pub enum NullKind {
99    Null,
100    Absent,
101    Conflict,
102}
103
104/// The row a drill into a group reads, from [`DataTableState::drill_row`].
105pub enum DrillRow {
106    /// Taken from the rows on screen.
107    Buffered(DataFrame),
108    /// Not on hand: the one-row frame to collect, off the UI thread.
109    Read(Box<LazyFrame>),
110}
111
112/// A group's rows, drilled from its row of a grouped view.
113pub(super) struct GroupRows {
114    lf: LazyFrame,
115    /// The key columns of the grouped view and the row's values in them, as text.
116    key_columns: Vec<String>,
117    key_values: Vec<String>,
118    /// Columns of `lf` that hold the keys as they stand, to lead the view.
119    lead: Vec<String>,
120    /// How `lf` is built, as Copy as Python steps.
121    steps: Vec<Step>,
122    /// Which loaded column each column of `lf` is.
123    lineage: Lineage,
124}
125
126/// The rows an export writes. See [`DataTableState::export_frame`].
127pub struct ExportFrame {
128    pub(super) lf: LazyFrame,
129    pub(super) files: Option<SourceFiles>,
130}
131
132/// The dataset's files in scan order, and the row each starts at.
133pub(super) struct SourceFiles {
134    pub(super) names: Arc<Vec<String>>,
135    pub(super) starts: Arc<Vec<usize>>,
136}
137
138impl ExportFrame {
139    /// Rows that are not the view's, such as a column's value counts.
140    pub fn of(lf: LazyFrame) -> Self {
141        Self { lf, files: None }
142    }
143}
144
145impl ExportFrame {
146    /// The name of the column an export adds when asked to say where each row is from.
147    pub const SOURCE_FILE_COLUMN: &'static str = "source_file";
148
149    /// The plan. Naming files reads the schema (possibly resolving the scan), so off the UI
150    /// thread; names map from the row index per batch, so streamed exports never hold
151    /// every row.
152    pub fn into_lazy(self) -> PolarsResult<LazyFrame> {
153        let Some(SourceFiles { names, starts }) = self.files else {
154            return Ok(self.lf);
155        };
156        let mut lf = self.lf;
157        let schema = lf.collect_schema()?;
158        let name = Self::free_name(schema.iter_names().map(|n| n.as_str()));
159        let index = crate::formats::schema_union::DRIFT_COLUMN;
160        let file_of = move |rows: Column| -> PolarsResult<Column> {
161            let rows = rows.strict_cast(&DataType::UInt64)?;
162            let named: StringChunked = rows
163                .u64()?
164                .iter()
165                .map(|row| {
166                    let row = row? as usize;
167                    let file = starts
168                        .partition_point(|&start| start <= row)
169                        .saturating_sub(1);
170                    names.get(file).map(String::as_str)
171                })
172                .collect();
173            Ok(named.with_name(rows.name().clone()).into_column())
174        };
175        // Added last, after the dataset's own columns, as the index comes off.
176        Ok(lf
177            .with_column(
178                col(index)
179                    .map(file_of, |_, field| {
180                        Ok(Field::new(field.name().clone(), DataType::String))
181                    })
182                    .alias(name),
183            )
184            .drop(by_name([index], true, false)))
185    }
186
187    /// A name for the source-file column no column has: `source_file` may exist already,
188    /// and adding a same-named column would silently replace it.
189    fn free_name<'a>(taken: impl Iterator<Item = &'a str>) -> String {
190        let taken: HashSet<&str> = taken.collect();
191        std::iter::once(Self::SOURCE_FILE_COLUMN.to_string())
192            .chain((1..).map(|n| format!("{}_{n}", Self::SOURCE_FILE_COLUMN)))
193            .find(|candidate| !taken.contains(candidate.as_str()))
194            .expect("some suffix is free")
195    }
196}
197
198impl DataTableState {
199    /// The group drilled into, as its key columns and their values.
200    pub fn drilled_group_key(&self) -> Option<(&[String], &[String])> {
201        let values = self.view.drilled_down_group_key.as_deref()?;
202        let columns = self
203            .view
204            .drilled_down_group_key_columns
205            .as_deref()
206            .unwrap_or_default();
207        Some((columns, values))
208    }
209
210    /// The columns and types of `df`, the table SQL runs against, from the known schema; a
211    /// drilled group or reshape only resolves its plan, reading nothing.
212    pub fn sql_table_columns(&self) -> Vec<(String, DataType)> {
213        let schema = if self.view.grouped.is_none() && self.view.reshaped_lf.is_none() {
214            Some(self.original_schema.clone())
215        } else {
216            self.query_root().collect_schema().ok()
217        };
218        schema
219            .map(|schema| {
220                schema
221                    .iter()
222                    .filter(|(name, _)| name.as_str() != crate::formats::schema_union::DRIFT_COLUMN)
223                    .map(|(name, dtype)| (name.to_string(), dtype.clone()))
224                    .collect()
225            })
226            .unwrap_or_default()
227    }
228
229    /// Rows `df` holds, when that is known without counting: the data as loaded,
230    /// once its count has come back.
231    pub fn sql_table_rows(&self) -> Option<usize> {
232        if self.view.grouped.is_some() || self.view.reshaped_lf.is_some() {
233            return None;
234        }
235        self.pristine_rows
236    }
237
238    pub fn get_active_fuzzy_query(&self) -> &str {
239        &self.view.active_fuzzy_query
240    }
241
242    pub fn last_pivot_spec(&self) -> Option<&PivotSpec> {
243        self.view.last_pivot_spec.as_ref()
244    }
245
246    pub fn last_melt_spec(&self) -> Option<&MeltSpec> {
247        self.view.last_melt_spec.as_ref()
248    }
249
250    /// What the pivot or melt in effect ran over. See the field.
251    pub fn reshape_source(&self) -> Option<&ReshapeSource> {
252        self.view.reshape_source.as_ref()
253    }
254
255    /// Whether the view is a grouping's result (a `by` query, SQL GROUP BY) whose rows drill
256    /// into groups; recorded by the query, never inferred from list columns.
257    pub fn is_grouped(&self) -> bool {
258        self.view.group_source.is_some()
259    }
260
261    /// Whether any column holds lists.
262    fn has_list_columns(&self) -> bool {
263        self.view
264            .schema
265            .iter()
266            .any(|(_, dtype)| matches!(dtype, DataType::List(_)))
267    }
268
269    fn group_key_columns(&self) -> Vec<String> {
270        self.view
271            .schema
272            .iter()
273            .filter(|(_, dtype)| !matches!(dtype, DataType::List(_)))
274            .map(|(name, _)| name.to_string())
275            .collect()
276    }
277
278    fn group_value_columns(&self) -> Vec<String> {
279        self.view
280            .schema
281            .iter()
282            .filter(|(_, dtype)| matches!(dtype, DataType::List(_)))
283            .map(|(name, _)| name.to_string())
284            .collect()
285    }
286
287    /// Names of binary columns in the source schema. Their values are not read into the display
288    /// buffer (the `‹binary›` stub stands in); the renderer uses this to style those cells.
289    pub fn binary_column_names(&self) -> std::collections::HashSet<String> {
290        self.view
291            .schema
292            .iter()
293            .filter(|(_, dtype)| matches!(dtype, DataType::Binary))
294            .map(|(name, _)| name.to_string())
295            .collect()
296    }
297
298    /// Estimated heap size in bytes of the currently buffered slice (locked + scrollable), if collected.
299    pub fn buffered_memory_bytes(&self) -> Option<usize> {
300        let locked = self
301            .view
302            .locked_df
303            .as_ref()
304            .map(|df| df.estimated_size())
305            .unwrap_or(0);
306        let scroll = self
307            .view
308            .df
309            .as_ref()
310            .map(|df| df.estimated_size())
311            .unwrap_or(0);
312        if locked == 0 && scroll == 0 {
313            None
314        } else {
315            Some(locked + scroll)
316        }
317    }
318
319    /// The rows on hand, first to past the last.
320    pub fn buffered_span(&self) -> (usize, usize) {
321        (self.view.buffered_start_row, self.view.buffered_end_row)
322    }
323
324    /// Rows in the buffer; 0 when none is loaded.
325    pub fn buffered_rows(&self) -> usize {
326        self.view
327            .buffered_end_row
328            .saturating_sub(self.view.buffered_start_row)
329    }
330
331    /// The first `limit` distinct non-null values of `column` in the display buffer, as
332    /// text. Reads nothing; a column not buffered gives none.
333    pub(crate) fn buffered_values(&self, column: &str, limit: usize) -> Vec<String> {
334        let Some(series) = [self.view.df.as_ref(), self.view.locked_df.as_ref()]
335            .into_iter()
336            .flatten()
337            .find_map(|df| df.column(column).ok())
338        else {
339            return Vec::new();
340        };
341        let series = series.as_materialized_series();
342        let mut values = Vec::new();
343        for value in (0..series.len()).filter_map(|index| series.get(index).ok()) {
344            if values.len() == limit {
345                break;
346            }
347            let text = match value {
348                AnyValue::Null => continue,
349                AnyValue::String(text) => text.to_string(),
350                AnyValue::List(items) => crate::exact::list_preview(&items),
351                // Drawn on the UI thread: Polars' display panics on a date past
352                // the calendar.
353                value => {
354                    crate::exact::past_calendar_text(&value).unwrap_or_else(|| value.to_string())
355                }
356            };
357            if !values.contains(&text) {
358                values.push(text);
359            }
360        }
361        values
362    }
363
364    /// Current scrollable display buffer. None until first collect().
365    pub fn display_df(&self) -> Option<&DataFrame> {
366        self.view.df.as_ref()
367    }
368
369    /// The display buffer's rows on screen, as the table draws them.
370    pub fn display_slice_df(&self) -> Option<DataFrame> {
371        let df = self.view.df.as_ref()?;
372        let offset = self
373            .view
374            .start_row
375            .saturating_sub(self.view.buffered_start_row);
376        let slice_len = self.visible_rows.min(df.height().saturating_sub(offset));
377        if offset < df.height() && slice_len > 0 {
378            Some(df.slice(offset as i64, slice_len))
379        } else {
380            None
381        }
382    }
383
384    /// The selected row with every display column, raw and in display order:
385    /// what a row copy carries.
386    pub fn copy_row_df(&self) -> Option<DataFrame> {
387        let df = self.view.buffered_df.as_ref()?;
388        let absolute = self.view.start_row + self.table_state.selected()?;
389        let offset = absolute.checked_sub(self.view.buffered_start_row)?;
390        if offset >= df.height() {
391            return None;
392        }
393        let names: Vec<&str> = self.view.column_order.iter().map(|s| s.as_str()).collect();
394        df.select(names).ok().map(|d| d.slice(offset as i64, 1))
395    }
396
397    /// The rows on screen with every display column, raw, in display order, ignoring
398    /// column scroll (scrolled-past identifiers make the rows readable).
399    pub fn copy_view_df(&self) -> Option<DataFrame> {
400        let df = self.view.buffered_df.as_ref()?;
401        let names: Vec<&str> = self.view.column_order.iter().map(|s| s.as_str()).collect();
402        let selected = df.select(names).ok()?;
403        let offset = self
404            .view
405            .start_row
406            .saturating_sub(self.view.buffered_start_row);
407        let len = self
408            .visible_rows
409            .min(selected.height().saturating_sub(offset));
410        (len > 0).then(|| selected.slice(offset as i64, len))
411    }
412
413    /// The selected row's value in `column`, exactly as stored ([`crate::exact`]): floats
414    /// as their round-tripping decimal, null as empty (as in exports), lists and structs as
415    /// JSON.
416    pub fn copy_cell_value(&self, column: &str) -> Option<String> {
417        let row = self.copy_row_df()?;
418        crate::exact::copy_text(row.column(column).ok()?).ok()
419    }
420
421    /// The selected row's number as the row-numbers column would print it.
422    pub fn selected_display_row(&self) -> Option<usize> {
423        Some(self.view.start_row + self.table_state.selected()? + self.row_start_index)
424    }
425
426    /// Rows times estimated row width (binary at base64 size), for the copy guard. `None`
427    /// until counted, or while a binary column's width is unknown (only a footer says).
428    pub fn estimated_copy_bytes(&self) -> Option<usize> {
429        let rows = self.num_rows_if_valid()?;
430        if rows == 0 {
431            return Some(0);
432        }
433        let base64 = |bytes: usize| bytes.div_ceil(3) * 4;
434        let footer_width = |name: &str| {
435            self.column_bytes
436                .iter()
437                .find(|(n, _)| n == name)
438                .map(|(_, w)| *w)
439        };
440        let mut row = self.bytes_per_row();
441        for name in &self.view.column_order {
442            match self.view.schema.get(name.as_str()) {
443                Some(DataType::Binary) => row += base64(footer_width(name)?),
444                // Buffered whole, so the buffer measured it with the row; base64 adds
445                // a third on top.
446                Some(dtype) if crate::export::nested_json::has_binary(dtype) => {
447                    let buffered = self.view.buffered_df.as_ref().and_then(|df| {
448                        let column = df.column(name).ok()?;
449                        (df.height() > 0)
450                            .then(|| column.as_materialized_series().estimated_size() / df.height())
451                    });
452                    row += buffered.or_else(|| footer_width(name)).unwrap_or(0) / 3;
453                }
454                _ => {}
455            }
456        }
457        Some(rows.saturating_mul(row))
458    }
459
460    /// Each row's drift group from the top of the view, for a frame `frame_rows` tall (the
461    /// whole frame, header included: extra entries are ignored, and `visible_rows` is 0
462    /// before the first frame). Empty once a query or reshape replaced the frame (its rows
463    /// stand for no file).
464    pub fn display_drift(&self, frame_rows: usize) -> Vec<u32> {
465        if !self.view.drift_column_present {
466            return Vec::new();
467        }
468        let Some(df) = self.view.buffered_df.as_ref() else {
469            return Vec::new();
470        };
471        let Ok(column) = df.column(crate::formats::schema_union::DRIFT_COLUMN) else {
472            return Vec::new();
473        };
474        let offset = self
475            .view
476            .start_row
477            .saturating_sub(self.view.buffered_start_row);
478        let len = frame_rows.min(column.len().saturating_sub(offset));
479        if len == 0 {
480            return Vec::new();
481        }
482        let slice = column.slice(offset as i64, len);
483        let Ok(rows) = slice.u32() else {
484            return Vec::new();
485        };
486        // The column holds each row's place in the dataset. The file it came from is
487        // the last one starting at or before it, and the file says what it is missing.
488        rows.iter()
489            .map(|row| self.file_group_of(row.unwrap_or(0) as usize))
490            .collect()
491    }
492
493    /// Maximum buffer size in rows (0 = no limit).
494    pub fn max_buffered_rows(&self) -> usize {
495        self.max_buffered_rows
496    }
497
498    /// Maximum buffer size in MiB (0 = no limit).
499    pub fn max_buffered_mb(&self) -> usize {
500        self.max_buffered_mb
501    }
502
503    /// Whether Enter on a row drills into its group: the view is a grouped result and
504    /// not already a group's rows.
505    pub fn can_drill_down(&self) -> bool {
506        !self.is_drilled_down() && self.is_grouped()
507    }
508
509    /// Whether a drill shows the lists of a row as the group's rows rather than
510    /// filtering the source: a `by` result that holds its groups as lists.
511    fn drills_lists(&self) -> bool {
512        self.has_list_columns()
513            && self
514                .view
515                .group_source
516                .as_ref()
517                .is_some_and(|s| s.rows_in_lists)
518    }
519
520    /// The columns of a row that a drill into its group reads: every column of a result
521    /// holding its groups as lists, the keys of one holding aggregates.
522    fn drill_columns(&self) -> Vec<String> {
523        if self.drills_lists() {
524            return self
525                .view
526                .schema
527                .iter_names()
528                .map(|n| n.to_string())
529                .collect();
530        }
531        self.view
532            .group_source
533            .iter()
534            .flat_map(|source| source.keys.iter().map(|(name, _)| name.to_string()))
535            .collect()
536    }
537
538    /// The columns the inspector lists for a row: the table's, in its order, then
539    /// the ones hidden from it, in schema order. Never the scan's own row index.
540    pub fn inspect_fields(&self) -> Vec<InspectField> {
541        let shown = self.view.column_order.iter().filter_map(|name| {
542            Some(InspectField {
543                name: name.clone(),
544                dtype: self.view.schema.get(name.as_str())?.clone(),
545                hidden: false,
546            })
547        });
548        let hidden = self
549            .view
550            .schema
551            .iter()
552            .filter(|(name, _)| {
553                name.as_str() != crate::formats::schema_union::DRIFT_COLUMN
554                    && !self.view.column_order.iter().any(|c| c == name.as_str())
555            })
556            .map(|(name, dtype)| InspectField {
557                name: name.to_string(),
558                dtype: dtype.clone(),
559                hidden: true,
560            });
561        shown.chain(hidden).collect()
562    }
563
564    /// The selected row as the buffer holds it, with the file group that says what
565    /// its nulls are. Reads nothing; `None` while the row is not on hand.
566    pub fn inspect_row(&self) -> Option<InspectRow> {
567        self.inspect_row_at(self.view.start_row + self.table_state.selected()?)
568    }
569
570    /// Row `row` of the view as the buffer holds it, as [`Self::inspect_row`] does
571    /// the selected one: Compare's next row. `None` while it is not on hand.
572    pub fn inspect_row_at(&self, row: usize) -> Option<InspectRow> {
573        let df = self.view.buffered_df.as_ref()?;
574        let offset = row.checked_sub(self.view.buffered_start_row)?;
575        if offset >= df.height() {
576            return None;
577        }
578        let names: Vec<&str> = self.view.column_order.iter().map(|s| s.as_str()).collect();
579        let values = df.select(names).ok()?.slice(offset as i64, 1);
580        let drift_group = self
581            .view
582            .drift_column_present
583            .then(|| df.column(crate::formats::schema_union::DRIFT_COLUMN).ok())
584            .flatten()
585            .and_then(|c| c.get(offset).ok())
586            .and_then(|v| v.extract::<usize>())
587            .map(|place| self.file_group_of(place));
588        Some(InspectRow {
589            row,
590            frame: self.view.len_generation,
591            display_row: row + self.row_start_index,
592            values,
593            drift_group,
594        })
595    }
596
597    /// The drift group of the file holding the dataset's row `place`.
598    fn file_group_of(&self, place: usize) -> u32 {
599        let file = self
600            .drift_file_starts
601            .partition_point(|&start| start <= place)
602            .saturating_sub(1);
603        self.drift_file_group.get(file).copied().unwrap_or(0)
604    }
605
606    /// What a null in `column` is, for a row of file group `group`: the data's own,
607    /// a file without the column, or a file holding it in another type.
608    pub fn null_kind(&self, column: &str, group: Option<u32>) -> NullKind {
609        let Some(group) = group.and_then(|g| self.view.drift_groups.get(g as usize)) else {
610            return NullKind::Null;
611        };
612        if group.absent.iter().any(|c| c == column) {
613            NullKind::Absent
614        } else if group.unread.iter().any(|c| c == column) {
615            NullKind::Conflict
616        } else {
617            NullKind::Null
618        }
619    }
620
621    /// The frame reading `columns` of view row `row` for the inspector, through the
622    /// buffer's window (a remote dataset reads only that file). Run off this thread.
623    pub fn inspect_read_lf(&self, row: usize, columns: &[String]) -> PolarsResult<LazyFrame> {
624        let exprs = columns.iter().map(|c| col(c.as_str())).collect();
625        self.window_lf(row, 1, exprs)
626    }
627
628    /// What drilling into the group on view row `group_index` reads, or `None` when not
629    /// grouped. The buffer holds the row; a missing column (hidden, binary stub) means
630    /// reading the row, which for an aggregate is the whole aggregate, so off the UI thread.
631    pub fn drill_row(&self, group_index: usize) -> Option<DrillRow> {
632        if !self.can_drill_down() {
633            return None;
634        }
635        let columns = self.drill_columns();
636        let buffered = self
637            .view
638            .buffered_df
639            .as_ref()
640            .filter(|_| {
641                (self.view.buffered_start_row..self.view.buffered_end_row).contains(&group_index)
642            })
643            .filter(|_| {
644                columns
645                    .iter()
646                    .all(|c| !matches!(self.view.schema.get(c.as_str()), Some(DataType::Binary)))
647            })
648            .and_then(|df| df.select(columns.iter().map(|c| c.as_str())).ok())
649            .map(|df| df.slice((group_index - self.view.buffered_start_row) as i64, 1))
650            .filter(|row| row.height() == 1);
651        Some(match buffered {
652            Some(row) => DrillRow::Buffered(row),
653            None => DrillRow::Read(Box::new(
654                self.visible_lf()
655                    .select(columns.iter().map(|c| col(c.as_str())).collect::<Vec<_>>())
656                    .slice(group_index as i64, 1),
657            )),
658        })
659    }
660
661    /// Show the group on view row `group_index`, reading the row on this thread if not
662    /// buffered; the app uses [`Self::drill_row`] so keys never wait.
663    pub fn drill_down_into_group(&mut self, group_index: usize) -> Result<()> {
664        let row = match self.drill_row(group_index) {
665            None => return Ok(()),
666            Some(DrillRow::Buffered(row)) => row,
667            Some(DrillRow::Read(lf)) => collect_lazy(*lf, self.polars_streaming)?,
668        };
669        self.drill_down_with_row(group_index, &row)
670    }
671
672    /// Show the group whose row `group_index` is `row` (from [`Self::drill_row`]): list
673    /// columns as rows, or for aggregates the source rows sharing the group's keys, key
674    /// columns first.
675    pub fn drill_down_with_row(&mut self, group_index: usize, row: &DataFrame) -> Result<()> {
676        if !self.can_drill_down() {
677            return Ok(());
678        }
679        if row.height() == 0 {
680            return Err(color_eyre::eyre::eyre!("Group index out of bounds"));
681        }
682        let mut group = if self.drills_lists() {
683            Self::group_from_lists(
684                row,
685                self.group_key_columns(),
686                self.group_value_columns(),
687                self.view.lineage.clone(),
688            )?
689        } else if let Some(source) = &self.view.group_source {
690            Self::group_from_source(source, row)?
691        } else {
692            return Ok(());
693        };
694        // A list form that also aggregates (`select a, n: count a by k`) holds its
695        // aggregates beside the keys; the query knows which columns are keys.
696        if let Some(source) = self
697            .view
698            .group_source
699            .as_ref()
700            .filter(|_| self.drills_lists())
701        {
702            let keys: Vec<&str> = source.keys.iter().map(|(n, _)| n.as_str()).collect();
703            (group.key_columns, group.key_values) = group
704                .key_columns
705                .into_iter()
706                .zip(group.key_values)
707                .filter(|(name, _)| keys.contains(&name.as_str()))
708                .unzip();
709        }
710        self.enter_group(group, group_index, false)
711    }
712
713    /// Show the view's rows with `value` in `column` (null matches null), like a group
714    /// drill: breadcrumb names the value, Esc returns. Inside a group, it narrows that group
715    /// and Esc returns to the view it was drilled from.
716    pub fn drill_into_value(&mut self, column: &str, value: AnyValue<'static>) -> Result<()> {
717        let dtype = self
718            .view
719            .schema
720            .get(column)
721            .cloned()
722            .ok_or_else(|| color_eyre::eyre::eyre!("no column {column}"))?;
723        let label = crate::exact::str_value(&value).to_string();
724        let mut steps = self.view_steps();
725        steps.push(match crate::export::python_script::py_value(&value) {
726            Some(literal) => Step::Matching(vec![(
727                format!("pl.col({})", crate::export::python_script::py_str(column)),
728                literal,
729            )]),
730            None => Step::Unreproducible(format!(
731                "drilled down to the rows where {column} is {label}, a value of a type not written as Python"
732            )),
733        });
734        let matches = col(column).eq_missing(lit(Scalar::new(dtype, value)));
735        let group = GroupRows {
736            lf: self.visible_lf().filter(matches),
737            key_columns: vec![column.to_string()],
738            key_values: vec![label],
739            lead: vec![column.to_string()],
740            steps,
741            lineage: self.view.lineage.clone(),
742        };
743        if !self.is_drilled_down() {
744            let index = self.view.start_row + self.table_state.selected().unwrap_or(0);
745            return self.enter_group(group, index, true);
746        }
747        let schema = group.lf.clone().collect_schema()?;
748        let order = std::mem::take(&mut self.view.column_order);
749        if let Some(keys) = self.view.drilled_down_group_key_columns.as_mut() {
750            keys.extend(group.key_columns);
751        }
752        if let Some(values) = self.view.drilled_down_group_key.as_mut() {
753            values.extend(group.key_values);
754        }
755        // The group's filters and sort are in the frame now.
756        self.view.filters.clear();
757        self.view.sort_columns.clear();
758        self.view.sort_descending.clear();
759        self.view.sort_ascending = true;
760        self.install_base(group.lf, schema);
761        self.view.base_steps = group.steps;
762        self.view.lineage = group.lineage;
763        self.view.column_order = order;
764        self.view.start_row = 0;
765        self.termcol_index = 0;
766        self.clear_column_moves();
767        self.settle_cursor();
768        self.table_state.select(Some(0));
769        self.collect();
770        Ok(())
771    }
772
773    /// Whether the drill on screen came from Value Counts.
774    pub fn drilled_into_value(&self) -> bool {
775        self.view.grouped.as_ref().is_some_and(|view| view.by_value)
776    }
777
778    /// Show `group`, the group on row `group_index` of the view, keeping the view to
779    /// come back to.
780    fn enter_group(&mut self, group: GroupRows, group_index: usize, by_value: bool) -> Result<()> {
781        let schema = group.lf.clone().collect_schema()?;
782        self.view.drilled_down_group_key = Some(group.key_values);
783        self.view.drilled_down_group_key_columns = Some(group.key_columns);
784
785        // The group becomes the pipeline root while drilled in, so a sidebar filter or
786        // sort applies within it instead of rebuilding the grouped view underneath.
787        self.view.grouped = Some(GroupedView {
788            lf: self.view.lf.clone(),
789            base_lf: self.view.base_lf.clone(),
790            base_schema: self.view.base_schema.clone(),
791            schema: self.view.schema.clone(),
792            filters: std::mem::take(&mut self.view.filters),
793            sort_columns: std::mem::take(&mut self.view.sort_columns),
794            sort_descending: std::mem::take(&mut self.view.sort_descending),
795            sort_ascending: self.view.sort_ascending,
796            drift: self.view.drift_column_present,
797            drift_groups: self.view.drift_groups.clone(),
798            view_numbered: self.view.view_numbered,
799            notes: self.view.notes.clone(),
800            group_source: self.view.group_source.take(),
801            column_order: self.view.column_order.clone(),
802            locked_columns_count: self.view.locked_columns_count,
803            start_row: self.view.start_row,
804            termcol_index: self.termcol_index,
805            cursor_column: self.view.cursor_column.clone(),
806            selected: self.table_state.selected(),
807            by_value,
808            base_steps: std::mem::take(&mut self.view.base_steps),
809            lineage: self.view.lineage.clone(),
810        });
811        self.view.sort_ascending = true;
812        self.install_base(group.lf, schema);
813        self.view.base_steps = group.steps;
814        self.view.lineage = group.lineage;
815        // Led by the keys, as a group drilled from lists is.
816        let rest: Vec<String> = std::mem::take(&mut self.view.column_order)
817            .into_iter()
818            .filter(|c| !group.lead.contains(c))
819            .collect();
820        self.view.column_order = group.lead.into_iter().chain(rest).collect();
821        self.view.drilled_down_group_index = Some(group_index);
822        self.view.start_row = 0;
823        self.termcol_index = 0;
824        self.clear_column_moves();
825        self.view.locked_columns_count = 0;
826        self.settle_cursor();
827        self.table_state.select(Some(0));
828        self.collect();
829
830        Ok(())
831    }
832
833    /// A group held as lists in `row`: the keys repeated beside the lists as columns.
834    fn group_from_lists(
835        row: &DataFrame,
836        key_columns: Vec<String>,
837        value_columns: Vec<String>,
838        lineage: Lineage,
839    ) -> Result<GroupRows> {
840        if value_columns.is_empty() {
841            return Err(color_eyre::eyre::eyre!("No value columns in grouped data"));
842        }
843        let row_count = match row.column(&value_columns[0])?.get(0)? {
844            AnyValue::List(list_series) => list_series.len(),
845            _ => 0,
846        };
847
848        let mut columns = Vec::new();
849        let mut key_values = Vec::new();
850        for col_name in &key_columns {
851            let key = row.column(col_name)?;
852            key_values.push(crate::exact::str_value(&key.get(0)?).to_string());
853            // Repeated in its own type, so a date stays a date and a null key null.
854            columns.push(key.new_from_index(0, row_count));
855        }
856        for col_name in &value_columns {
857            if let AnyValue::List(list_series) = row.column(col_name)?.get(0)? {
858                columns.push(list_series.with_name(col_name.as_str().into()).into());
859            }
860        }
861        let group = key_columns
862            .iter()
863            .zip(&key_values)
864            .map(|(c, v)| format!("{c} = {v}"))
865            .collect::<Vec<_>>()
866            .join(", ");
867        Ok(GroupRows {
868            lf: DataFrame::new_infer_height(columns)?.lazy(),
869            key_columns,
870            key_values,
871            // Already first.
872            lead: Vec::new(),
873            steps: vec![Step::Unreproducible(format!(
874                "drilled down into the group {group}, read from the grouped result's lists: \
875                 not written as Python"
876            ))],
877            // The lists keep the result's names.
878            lineage,
879        })
880    }
881
882    /// A group of an aggregated result in `row`: the source rows whose keys equal the
883    /// row's, a null key matching nulls.
884    fn group_from_source(source: &GroupSource, row: &DataFrame) -> Result<GroupRows> {
885        let mut predicate: Option<Expr> = None;
886        let mut key_columns = Vec::new();
887        let mut key_values = Vec::new();
888        let mut lead = Vec::new();
889        let mut matching = Vec::new();
890        for (i, (name, expr)) in source.keys.iter().enumerate() {
891            let column = row.column(name)?;
892            let value = column.get(0)?.into_static();
893            matching.push(
894                source
895                    .python_keys
896                    .get(i)
897                    .cloned()
898                    .flatten()
899                    .zip(crate::export::python_script::py_value(&value)),
900            );
901            key_columns.push(name.to_string());
902            key_values.push(crate::exact::str_value(&value).to_string());
903            // The key as the query computed it, against the value it produced; the alias
904            // only named the result's column.
905            let key = expr.clone().meta().undo_aliases();
906            if let Expr::Column(source_column) = &key
907                && !source.scratch.contains(source_column)
908            {
909                lead.push(source_column.to_string());
910            }
911            let matches = key.eq_missing(lit(Scalar::new(column.dtype().clone(), value)));
912            predicate = Some(match predicate {
913                Some(all) => all.and(matches),
914                None => matches,
915            });
916        }
917        let rows = source.rows.clone();
918        let mut lf = match predicate {
919            Some(predicate) => rows.filter(predicate),
920            None => rows,
921        };
922        if !source.scratch.is_empty() {
923            lf = lf.drop(by_name(source.scratch.iter().cloned(), true, false));
924        }
925        let matching: Option<Vec<(String, String)>> = matching.into_iter().collect();
926        let steps = match (&source.python_rows, matching) {
927            (Some(rows), Some(matching)) => {
928                let mut steps = rows.clone();
929                steps.push(Step::Matching(matching));
930                if !source.scratch.is_empty() {
931                    steps.push(Step::Drop(
932                        source.scratch.iter().map(|c| c.to_string()).collect(),
933                    ));
934                }
935                steps
936            }
937            _ => vec![Step::Unreproducible(format!(
938                "drilled down into the group {}: not written as Python",
939                key_columns
940                    .iter()
941                    .zip(&key_values)
942                    .map(|(c, v)| format!("{c} = {v}"))
943                    .collect::<Vec<_>>()
944                    .join(", ")
945            ))],
946        };
947        Ok(GroupRows {
948            lf,
949            key_columns,
950            key_values,
951            lead,
952            steps,
953            lineage: source.lineage.clone(),
954        })
955    }
956
957    pub fn drill_up(&mut self) -> Result<()> {
958        let Some(view) = self.view.grouped.take() else {
959            return Err(color_eyre::eyre::eyre!("Not in drill-down mode"));
960        };
961        self.invalidate_num_rows();
962        // The buffer holds the group's rows; kept, it would stand in for the grouped
963        // view wherever the view fits inside it.
964        self.drop_buffer();
965        self.view.observed_bytes_per_row = None;
966        self.widths.relearn();
967        self.view.lf = view.lf;
968        self.view.unsorted_lf = None;
969        self.view.base_lf = view.base_lf;
970        self.view.base_schema = view.base_schema;
971        self.view.base_steps = view.base_steps;
972        self.view.lineage = view.lineage;
973        self.view.filters = view.filters;
974        self.view.sort_columns = view.sort_columns;
975        self.view.sort_descending = view.sort_descending;
976        self.view.sort_ascending = view.sort_ascending;
977        self.view.drift_column_present = view.drift;
978        self.view.drift_groups = view.drift_groups;
979        self.view.view_numbered = view.view_numbered;
980        self.view.notes = view.notes;
981        self.view.group_source = view.group_source;
982        // The restored frame already excludes what its filter and sort excluded; rederive
983        // those notes so they cannot be stale.
984        self.view.view_notes = self.view_notes_only();
985        self.view.schema = view.schema;
986        self.view.column_order = view.column_order;
987        self.view.locked_columns_count = view.locked_columns_count;
988        self.view.drilled_down_group_index = None;
989        self.view.drilled_down_group_key = None;
990        self.view.drilled_down_group_key_columns = None;
991        self.view.start_row = view.start_row;
992        self.termcol_index = view.termcol_index;
993        self.clear_column_moves();
994        self.view.cursor_column = view.cursor_column;
995        self.settle_cursor();
996        self.table_state.select(view.selected);
997        self.collect();
998        Ok(())
999    }
1000}