Skip to main content

datui_lib/table/
buffer.rs

1//! The row buffer: planning which rows to read around the view, reading them (on a
2//! worker, or here for tests), installing them, and the scroll that walks through them.
3
4use super::*;
5
6/// Parameters for a background buffer load. Produced by `prepare_async_collect()`.
7pub struct CollectRequest {
8    /// LazyFrame to collect (sliced to the buffer range, with column selection applied).
9    pub lf: LazyFrame,
10    /// Whether to use Polars streaming engine.
11    pub polars_streaming: bool,
12    /// Buffer start row in the full dataset.
13    pub buffer_start: usize,
14    /// Buffer end row in the full dataset.
15    pub buffer_end: usize,
16    /// How the worker fits the rows it reads to the buffer: [`FillPlan::fit`].
17    pub plan: FillPlan,
18}
19
20/// What the fill-reading worker needs to make it the buffer: the adjoining rows on
21/// hand, the view and the caps, as planned. A trim that would copy (see
22/// `trim_rows`) runs here, off the UI thread; `apply_async_collect` installs the
23/// result as is.
24pub struct FillPlan {
25    buffer_start: usize,
26    buffer_end: usize,
27    num_rows: usize,
28    count_known: bool,
29    /// Lines were still being indexed when the read was planned: a short read ends
30    /// where the indexing had got to, not the file.
31    indexing: bool,
32    /// The rows on hand and their first row, when the fill is planned to be stitched
33    /// on to them. Shared, not copied.
34    held: Option<(DataFrame, usize)>,
35    view_start: usize,
36    view_len: usize,
37    max_rows: usize,
38    max_mb: usize,
39}
40
41impl FillPlan {
42    /// Make the buffer from `df`, the planned range: stitched to adjoining rows on hand,
43    /// then cut to the caps around the view.
44    pub fn fit(mut self, df: DataFrame) -> CollectResult {
45        let returned = df.height();
46        let bytes_per_row = (returned > 0).then(|| (df.estimated_size() / returned).max(1));
47        // A shape mismatch (the columns changed underneath) keeps the fetched rows alone.
48        let (df, start, seam) = match self.held.take() {
49            Some((mut held, held_start)) if held_start + held.height() == self.buffer_start => {
50                let seam = held.height();
51                match held.vstack_mut(&df) {
52                    Ok(_) => (held, held_start, Some(seam)),
53                    Err(_) => (df, self.buffer_start, None),
54                }
55            }
56            Some((held, held_start))
57                if returned > 0 && self.buffer_start + returned == held_start =>
58            {
59                match df.vstack(&held) {
60                    Ok(joined) => (joined, self.buffer_start, Some(returned)),
61                    Err(_) => (df, self.buffer_start, None),
62                }
63            }
64            _ => (df, self.buffer_start, None),
65        };
66        let (df, start) = self.cut_to_caps(df, start, seam);
67        CollectResult {
68            df,
69            start,
70            returned,
71            bytes_per_row,
72            buffer_start: self.buffer_start,
73            buffer_end: self.buffer_end,
74            num_rows: self.num_rows,
75            count_known: self.count_known,
76            indexing: self.indexing,
77        }
78    }
79
80    /// Cut `df` (rows `[start, start + df.height())`) to the row cap and byte budget,
81    /// centered on the view (a head cut would drop a late jump's rows). Returns the kept
82    /// rows and their first row. The budget bounds rows held between collects, not the
83    /// collect.
84    fn cut_to_caps(&self, df: DataFrame, start: usize, seam: Option<usize>) -> (DataFrame, usize) {
85        let total = df.height();
86        if total == 0 {
87            return (df, start);
88        }
89        // The row cap as well: a row group stitched on to the rows on hand can run over it.
90        let mut max_rows = total;
91        if self.max_rows > 0 {
92            max_rows = max_rows.min(self.max_rows);
93        }
94        if self.max_mb > 0 {
95            let bytes_per_row = (df.estimated_size() / total).max(1);
96            max_rows = max_rows.min(self.max_mb * 1024 * 1024 / bytes_per_row);
97        }
98        let max_rows = max_rows.max(1);
99        if max_rows >= total {
100            return (df, start);
101        }
102        let view_off = self.view_start.saturating_sub(start).min(total);
103        let view_len = self.view_len.max(1).min(total);
104        let view_center = view_off + view_len / 2;
105        let mut keep_start = view_center.saturating_sub(max_rows / 2);
106        if keep_start + max_rows > total {
107            keep_start = total - max_rows;
108        }
109        let kept = max_rows.min(total - keep_start);
110        (trim_rows(df, keep_start, kept, seam), start + keep_start)
111    }
112}
113
114/// Result of a background buffer load, made by [`FillPlan::fit`] on the worker and
115/// installed as it is by `apply_async_collect()`.
116pub struct CollectResult {
117    /// The buffer: the rows read, stitched and cut to the caps.
118    pub(super) df: DataFrame,
119    /// The first row of `df`.
120    pub(super) start: usize,
121    /// Rows the read returned, before the stitch and the cut.
122    returned: usize,
123    /// Bytes per row of the rows read, to plan the next fill by.
124    bytes_per_row: Option<usize>,
125    /// The range the read was planned for.
126    buffer_start: usize,
127    buffer_end: usize,
128    /// Row count for the full (unsliced) dataset. Only meaningful when `count_known`.
129    num_rows: usize,
130    /// Whether `num_rows` is the true total. False for a first buffer read before the
131    /// background `len()` count has resolved; `num_rows` is then provisional.
132    count_known: bool,
133    /// See `FillPlan::indexing`.
134    indexing: bool,
135}
136
137impl CollectResult {
138    /// The rows read, as the buffer will hold them.
139    pub(crate) fn rows(&self) -> &DataFrame {
140        &self.df
141    }
142}
143
144/// A string's in-memory width when nothing says otherwise: the view plus a short value.
145pub(super) const STRING_BYTES_GUESS: usize = 40;
146
147/// Estimated in-memory bytes per row of `columns`: fixed widths, strings by the footer
148/// average in `column_bytes` (or a guess) plus their view, nested by footer average or
149/// guess. Binary columns are buffered as a stub (`binary_stub_exprs`).
150pub(super) fn estimate_bytes_per_row(
151    schema: &Schema,
152    columns: &[String],
153    column_bytes: &[(String, usize)],
154) -> usize {
155    let footer_width = |name: &String| {
156        column_bytes
157            .iter()
158            .find(|(n, _)| n == name)
159            .map(|(_, w)| *w)
160    };
161    columns
162        .iter()
163        .map(|name| match schema.get(name.as_str()) {
164            Some(DataType::String) => 16 + footer_width(name).unwrap_or(STRING_BYTES_GUESS - 16),
165            Some(DataType::Binary) => 16 + binary_stub().len(),
166            Some(DataType::Boolean) => 1,
167            Some(DataType::Null) => 0,
168            Some(dtype) if dtype.is_primitive_numeric() || dtype.is_temporal() => {
169                match dtype.to_physical() {
170                    DataType::Int8 | DataType::UInt8 => 1,
171                    DataType::Int16 | DataType::UInt16 => 2,
172                    DataType::Int32 | DataType::UInt32 | DataType::Float32 => 4,
173                    DataType::Int128 => 16,
174                    _ => 8,
175                }
176            }
177            Some(DataType::Decimal(..)) => 16,
178            _ => footer_width(name).unwrap_or(64),
179        })
180        .sum::<usize>()
181        .max(1)
182}
183
184/// Rows `[offset, offset + len)` of `df`, copied when a slice would keep much more
185/// allocated (a single-chunk fill, a stitched union, shared string data), keeping a
186/// `seam` as a chunk boundary (see [`compact_rows`]). A chunk that is itself a slice is
187/// not seen through.
188pub(super) fn trim_rows(
189    df: DataFrame,
190    offset: usize,
191    len: usize,
192    seam: Option<usize>,
193) -> DataFrame {
194    if backing_rows(&df, offset, len) > len + len / 4 {
195        compact_rows(df, offset, len, seam)
196    } else {
197        df.slice(offset as i64, len)
198    }
199}
200
201/// The most rows any column of `df` keeps allocated behind the slice `[offset, offset
202/// + len)`: every chunk the slice touches, whole.
203pub(super) fn backing_rows(df: &DataFrame, offset: usize, len: usize) -> usize {
204    let end = offset + len;
205    df.columns()
206        .iter()
207        .filter_map(Column::as_series)
208        .map(|s| {
209            let mut start = 0;
210            let mut touched = 0;
211            for chunk in s.chunks() {
212                let chunk_end = start + chunk.len();
213                if start < end && offset < chunk_end {
214                    touched += chunk.len();
215                }
216                start = chunk_end;
217            }
218            touched
219        })
220        .max()
221        .unwrap_or(len)
222}
223
224/// Rows `[offset, offset + len)` in their own storage: one chunk per column, two when
225/// `seam` falls inside, so a later cut to one side lets the other go. Neither `rechunk`
226/// nor `take` reliably releases the parent; builders with `ShareStrategy::Never` copy
227/// everything. Constant columns stay one value. Each source column is released once
228/// copied, bounding the extra memory to about one column.
229pub(super) fn compact_rows(
230    df: DataFrame,
231    offset: usize,
232    len: usize,
233    seam: Option<usize>,
234) -> DataFrame {
235    use polars::series::builder::SeriesBuilder;
236    use polars_arrow::array::builder::ShareStrategy;
237    #[cfg(test)]
238    tests::COMPACTIONS.with(|count| count.set(count.get() + 1));
239    let len = len.min(df.height().saturating_sub(offset));
240    let pieces = match seam.filter(|&seam| offset < seam && seam < offset + len) {
241        Some(seam) => vec![(offset, seam - offset), (seam, offset + len - seam)],
242        None => vec![(offset, len)],
243    };
244    let copy = |series: &Series, (offset, len): (usize, usize)| {
245        let mut builder = SeriesBuilder::new(series.dtype().clone());
246        builder.reserve(len);
247        builder.subslice_extend(series, offset, len, ShareStrategy::Never);
248        builder.freeze(series.name().clone())
249    };
250    let columns = df
251        .into_columns()
252        .into_iter()
253        .map(|column| match column {
254            Column::Scalar(constant) => {
255                Column::new_scalar(constant.name().clone(), constant.scalar().clone(), len)
256            }
257            Column::Series(series) => {
258                let mut kept = copy(&series, pieces[0]);
259                for &piece in &pieces[1..] {
260                    if kept.append_owned(copy(&series, piece)).is_err() {
261                        kept = copy(&series, (offset, len));
262                        break;
263                    }
264                }
265                kept.into_column()
266            }
267        })
268        .collect();
269    // Cannot fail: the names are one frame's and every column was built to `len` rows.
270    DataFrame::new(len, columns).unwrap_or_else(|_| DataFrame::empty_with_height(len))
271}
272
273/// Shrink `[buffer_start, buffer_end)` to at most `max_len` rows, kept around the view
274/// `[view_start, view_end)` and inside `[floor, ceil)`.
275pub(super) fn shrink_around_view(
276    view_start: usize,
277    view_end: usize,
278    max_len: usize,
279    floor: usize,
280    ceil: usize,
281    buffer_start: &mut usize,
282    buffer_end: &mut usize,
283) {
284    if buffer_end.saturating_sub(*buffer_start) <= max_len {
285        return;
286    }
287    let view_len = view_end.saturating_sub(view_start);
288    if view_len >= max_len {
289        *buffer_start = view_start;
290        *buffer_end = (view_start + max_len).min(ceil);
291        return;
292    }
293    let half = (max_len - view_len) / 2;
294    *buffer_end = (view_end + half).min(ceil);
295    *buffer_start = buffer_end.saturating_sub(max_len).max(floor);
296    if *buffer_start > view_start {
297        *buffer_start = view_start;
298    }
299    *buffer_end = (*buffer_start + max_len).min(ceil);
300}
301
302/// The most files one buffer read opens, beyond those the view itself spans.
303pub(super) const MAX_FILES_PER_BUFFER: usize = 16;
304
305/// Narrow `[start, end)` to at most `max_files` files, keeping every file the view
306/// `[view_start, view_end)` lies in and adding the ones after it first.
307pub(super) fn limit_files(
308    offsets: &[usize],
309    view_start: usize,
310    view_end: usize,
311    start: usize,
312    end: usize,
313    max_files: usize,
314) -> (usize, usize) {
315    let (Some((first, last)), Some((view_first, view_last))) = (
316        files_holding(offsets, start, end.saturating_sub(start)),
317        files_holding(
318            offsets,
319            view_start,
320            view_end.saturating_sub(view_start).max(1),
321        ),
322    ) else {
323        return (start, end);
324    };
325    // An empty file is not opened (see `window_of`), so it costs nothing to reach past.
326    let opened = |from: usize, to: usize| (from..=to).filter(|&i| holds_rows(offsets, i)).count();
327    if opened(first, last) <= max_files {
328        return (start, end);
329    }
330    let (mut lo, mut hi) = (view_first.max(first), view_last.min(last));
331    let mut files = opened(lo, hi);
332    while files < max_files && (hi < last || lo > first) {
333        if hi < last {
334            hi += 1;
335            files += usize::from(holds_rows(offsets, hi));
336        }
337        if files < max_files && lo > first {
338            lo -= 1;
339            files += usize::from(holds_rows(offsets, lo));
340        }
341    }
342    (start.max(offsets[lo]), end.min(offsets[hi + 1]))
343}
344
345/// Whether file `i` has any rows, given where each file's rows start.
346pub(super) fn holds_rows(offsets: &[usize], i: usize) -> bool {
347    offsets[i + 1] > offsets[i]
348}
349
350/// The files from `first` to `last` that hold rows: a window reads these and passes
351/// over the empty ones, which a dataset written a file a day can be mostly made of.
352pub(super) fn files_with_rows(offsets: &[usize], first: usize, last: usize) -> Vec<usize> {
353    (first..=last).filter(|&i| holds_rows(offsets, i)).collect()
354}
355
356/// The first and last files holding rows `[start, start + len)`, given where each file's
357/// rows start (`offsets`, with the total last). `None` when the rows lie past the end.
358pub(super) fn files_holding(offsets: &[usize], start: usize, len: usize) -> Option<(usize, usize)> {
359    let files = offsets.len().checked_sub(1)?;
360    let total = *offsets.last()?;
361    if files == 0 || len == 0 || start >= total {
362        return None;
363    }
364    let end = (start + len).min(total);
365    // The file a row is in: the last one starting at or before it. Empty files start
366    // where the next one does and are skipped over.
367    let file_of = |row: usize| offsets.partition_point(|&o| o <= row).saturating_sub(1);
368    Some((file_of(start), file_of(end - 1).min(files - 1)))
369}
370
371/// Rows `[start, start + len)` of `lf` as `all_columns`: with counted `files`, a scan of
372/// only the files holding them; with `records`, read straight from the source.
373pub(super) fn window_of(
374    lf: &LazyFrame,
375    files: Option<&RemoteFiles>,
376    records: Option<&dyn crate::formats::pushdown::Windowed>,
377    read_as_text: &[PlSmallStr],
378    start: usize,
379    len: usize,
380    all_columns: Vec<Expr>,
381) -> PolarsResult<LazyFrame> {
382    // Polars gives an anonymous scan no row offset, so a slice deep in the view would
383    // read every row before it; the source starts the window there instead.
384    if let Some(records) = records {
385        return Ok(records.window(start, len)?.select(all_columns));
386    }
387    if let Some((files, offsets)) = files.and_then(|f| f.offsets.as_ref().map(|o| (f, o)))
388        && let Some((first, last)) = files_holding(offsets, start, len)
389    {
390        // The window's first file holds its first row, so leaving out the empty files
391        // after it does not move the slice.
392        let urls: Vec<String> = files_with_rows(offsets, first, last)
393            .into_iter()
394            .map(|i| files.urls[i].clone())
395            .collect();
396        let lf = (files.scan)(&urls, read_as_text)?;
397        return Ok(lf
398            .select(all_columns)
399            .slice((start - offsets[first]) as i64, len as u32));
400    }
401    Ok(lf
402        .clone()
403        .select(all_columns)
404        .slice(start as i64, len as u32))
405}
406
407/// The rows of a view, for a reader off the UI thread: read a window at a time as a
408/// page is, or from the buffer the table already holds.
409#[derive(Clone)]
410pub(crate) struct ViewRows {
411    lf: LazyFrame,
412    files: Option<RemoteFiles>,
413    /// See [`DataTableState::window_now`].
414    records: Option<Arc<dyn crate::formats::pushdown::Windowed>>,
415    read_as_text: Vec<PlSmallStr>,
416    /// The buffer on hand and the view row it starts at.
417    pub(crate) buffer: Option<(DataFrame, usize)>,
418    /// The view's row count, when it is known.
419    pub(crate) num_rows: Option<usize>,
420    pub(crate) streaming: bool,
421    /// Any window of the view reads all of it: see [`sees_every_row_first`].
422    pub(crate) whole: bool,
423    /// A window of the view reads every row before it: see [`reads_up_to_a_window`].
424    pub(crate) reads_up_to: bool,
425}
426
427/// Whether `lf` has to see every row before it gives its first: a sort, a group by or a
428/// pivot under it. Then a window of it costs as much as all of it.
429pub(crate) fn sees_every_row_first(lf: &LazyFrame) -> bool {
430    use polars::lazy::dsl::DslPlan;
431    lf.logical_plan.into_iter().any(|node| {
432        matches!(
433            node,
434            DslPlan::Sort { .. } | DslPlan::GroupBy { .. } | DslPlan::Pivot { .. }
435        )
436    })
437}
438
439/// Whether a window of `lf` reads every row before it: a filter (to find the first
440/// match) or a scan with no row index to skip by (CSV). Parquet and IPC skip.
441pub(crate) fn reads_up_to_a_window(lf: &LazyFrame) -> bool {
442    use polars::lazy::dsl::{DslPlan, FileScanDsl};
443    lf.logical_plan.into_iter().any(|node| match node {
444        DslPlan::Filter { .. } => true,
445        DslPlan::Scan { scan_type, .. } => !matches!(
446            **scan_type,
447            FileScanDsl::Parquet { .. } | FileScanDsl::Ipc { .. }
448        ),
449        _ => false,
450    })
451}
452
453impl ViewRows {
454    /// Rows `[start, start + len)` of the view as `exprs`.
455    pub(crate) fn window(
456        &self,
457        start: usize,
458        len: usize,
459        exprs: Vec<Expr>,
460    ) -> PolarsResult<LazyFrame> {
461        window_of(
462            &self.lf,
463            self.files.as_ref(),
464            self.records.as_deref(),
465            &self.read_as_text,
466            start,
467            len,
468            exprs,
469        )
470    }
471
472    /// The view `lf`, with `buffer` on hand from row `buffer_start`.
473    #[cfg(test)]
474    pub(crate) fn of(lf: LazyFrame, buffer: Option<(DataFrame, usize)>) -> Self {
475        Self {
476            whole: sees_every_row_first(&lf),
477            reads_up_to: reads_up_to_a_window(&lf),
478            lf,
479            files: None,
480            records: None,
481            read_as_text: Vec::new(),
482            buffer,
483            num_rows: None,
484            streaming: false,
485        }
486    }
487}
488
489/// Snap `[start, end)` outward to whole row groups (`offsets`, total last). The groups
490/// the view lies in are always whole (Polars fetches whole groups, so paging inside is
491/// free); others the window reaches are added within `cap` rows (0: none), ahead of the
492/// view first.
493pub(super) fn align_to_row_groups(
494    offsets: &[usize],
495    view_start: usize,
496    view_end: usize,
497    start: usize,
498    end: usize,
499    cap: usize,
500) -> (usize, usize) {
501    let Some(groups) = offsets.len().checked_sub(1).filter(|n| *n > 0) else {
502        return (start, end);
503    };
504    let group_of = |row: usize| {
505        offsets
506            .partition_point(|&o| o <= row)
507            .saturating_sub(1)
508            .min(groups - 1)
509    };
510    let last_row = |s: usize, e: usize| e.saturating_sub(1).max(s);
511    let (mut lo, mut hi) = (
512        group_of(view_start),
513        group_of(last_row(view_start, view_end)),
514    );
515    let (want_lo, want_hi) = (group_of(start), group_of(last_row(start, end)));
516    let fits = |lo: usize, hi: usize| cap == 0 || offsets[hi + 1] - offsets[lo] <= cap;
517    loop {
518        if hi < want_hi && fits(lo, hi + 1) {
519            hi += 1;
520        } else if lo > want_lo && fits(lo - 1, hi) {
521            lo -= 1;
522        } else {
523            break;
524        }
525    }
526    (offsets[lo], offsets[hi + 1])
527}
528
529impl DataTableState {
530    /// Returns true if a scroll by `rows` would trigger a collect (view would leave the buffer).
531    /// Used so the UI only shows the throbber when actual data loading will occur.
532    pub fn scroll_would_trigger_collect(&self, rows: i64) -> bool {
533        if rows < 0 && self.view.start_row == 0 {
534            return false;
535        }
536        let new_start_row = if self.view.start_row as i64 + rows <= 0 {
537            0
538        } else {
539            if let Some(df) = self.view.df.as_ref()
540                && rows > 0
541                && df.shape().0 <= self.visible_rows
542            {
543                return false;
544            }
545            let unclamped = (self.view.start_row as i64 + rows) as usize;
546            if rows > 0 {
547                unclamped.min(self.view.num_rows.saturating_sub(self.visible_rows))
548            } else {
549                unclamped
550            }
551        };
552        if new_start_row == self.view.start_row {
553            return false;
554        }
555        let view_end = new_start_row
556            + self
557                .visible_rows
558                .min(self.view.num_rows.saturating_sub(new_start_row));
559        let within_buffer = new_start_row >= self.view.buffered_start_row
560            && view_end <= self.view.buffered_end_row
561            && self.view.buffered_end_row > 0;
562        !within_buffer
563    }
564
565    /// Scroll by `rows`: within the buffer, re-slice the display; outside it, set the
566    /// position and return true so the caller collects.
567    pub fn slide_table(&mut self, rows: i64) -> bool {
568        if rows < 0 && self.view.start_row == 0 {
569            return false;
570        }
571
572        let new_start_row = if self.view.start_row as i64 + rows <= 0 {
573            0
574        } else {
575            if let Some(df) = self.view.df.as_ref()
576                && rows > 0
577                && df.shape().0 <= self.visible_rows
578            {
579                return false;
580            }
581            let unclamped = (self.view.start_row as i64 + rows) as usize;
582            if rows > 0 {
583                // Keep a screen of data in view: otherwise held PageDown at the bottom pushes past
584                // `num_rows`, repeatedly asking for no-op collects.
585                unclamped.min(self.view.num_rows.saturating_sub(self.visible_rows))
586            } else {
587                unclamped
588            }
589        };
590
591        if new_start_row == self.view.start_row {
592            return false;
593        }
594
595        let view_end = new_start_row
596            + self
597                .visible_rows
598                .min(self.view.num_rows.saturating_sub(new_start_row));
599        let within_buffer = new_start_row >= self.view.buffered_start_row
600            && view_end <= self.view.buffered_end_row
601            && self.view.buffered_end_row > 0;
602
603        self.view.start_row = new_start_row;
604
605        if within_buffer {
606            if self.table_state.selected().is_none() {
607                self.table_state.select(Some(0));
608            }
609            false
610        } else {
611            true // caller must collect
612        }
613    }
614
615    /// Read the view's rows synchronously, as a job does for the app; for tests only.
616    #[cfg(test)]
617    pub fn collect(&mut self) {
618        if self.defer_collect {
619            return;
620        }
621        if !self.view.num_rows_valid {
622            // A count that fails means the frame itself is broken: say so rather than
623            // draw it as empty.
624            match collect_lazy(row_count_lf(&self.view.lf), self.polars_streaming) {
625                Ok(df) => {
626                    self.error = None;
627                    let n = match df.get(0).as_deref().and_then(|row| row.first()) {
628                        Some(AnyValue::UInt64(len)) => *len as usize,
629                        _ => 0,
630                    };
631                    self.set_num_rows(n);
632                }
633                Err(e) => {
634                    self.error = Some(e);
635                    self.set_num_rows(0);
636                }
637            }
638        }
639        let Some(request) = self.prepare_async_collect(None) else {
640            return;
641        };
642        match collect_lazy(request.lf, request.polars_streaming) {
643            Ok(df) => self.apply_async_collect(request.plan.fit(df)),
644            Err(e) => self.error = Some(e),
645        }
646    }
647
648    /// In the app a mutation asks for its rows: the event loop reads them on a job
649    /// after the next frame. Under [`Self::deferred`] the caller reads them itself.
650    #[cfg(not(test))]
651    pub(super) fn collect(&mut self) {
652        if !self.defer_collect {
653            self.needs_recollect = true;
654        }
655    }
656
657    /// Expressions for every column in `column_order`, binary columns replaced by a stub
658    /// ([`binary_stub`]) so blobs are never read, for the display buffer and analysis
659    /// (multi-GB blobs would exhaust memory). `lf` keeps the bytes for export.
660    pub(crate) fn binary_stub_exprs(&self) -> Vec<Expr> {
661        self.view
662            .column_order
663            .iter()
664            .map(|name| {
665                if matches!(self.view.schema.get(name.as_str()), Some(DataType::Binary)) {
666                    lit(binary_stub()).alias(name.as_str())
667                } else {
668                    col(name.as_str())
669                }
670            })
671            .collect()
672    }
673
674    /// Plan an async collect without blocking: clamp the start row, then a `CollectRequest`
675    /// if the rows on screen need a new buffer, or `None` (display slices updated). Without
676    /// `num_rows_override`, `Self::num_rows_bound` lets the first rows skip the count.
677    pub fn prepare_async_collect(
678        &mut self,
679        num_rows_override: Option<usize>,
680    ) -> Option<CollectRequest> {
681        if self.visible_rows > 0 {
682            self.proximity_threshold = self.proximity();
683        }
684
685        if let Some(n) = num_rows_override {
686            self.view.num_rows = n;
687            self.view.num_rows_valid = true;
688        }
689
690        // `bound` is the total, or `usize::MAX` while `len()` runs, planning a top-of-data
691        // window without waiting. See `num_rows_bound`.
692        let count_known = self.view.num_rows_valid;
693        let bound = self.num_rows_bound();
694
695        if count_known {
696            if self.view.num_rows > 0 {
697                let max_start = self.view.num_rows.saturating_sub(1);
698                if self.view.start_row > max_start {
699                    self.view.start_row = max_start;
700                }
701            } else {
702                // Confirmed-empty dataset: clear everything.
703                self.view.start_row = 0;
704                self.drop_buffer();
705                self.view.df = None;
706                self.view.locked_df = None;
707                return None;
708            }
709        }
710
711        // No column shown: there are no rows to read, and a read of none would come
712        // back empty and ask again.
713        if self.view.column_order.is_empty() {
714            self.drop_buffer();
715            self.view.df = None;
716            self.view.locked_df = None;
717            return None;
718        }
719
720        let view_start = self.view.start_row;
721        let view_end = self.view.start_row + self.visible_rows.min(bound - self.view.start_row);
722        let within_buffer = view_start >= self.view.buffered_start_row
723            && view_end <= self.view.buffered_end_row
724            && self.view.buffered_end_row > 0;
725
726        // Compute the buffer range using the same logic as collect().
727        let (new_buffer_start, new_buffer_end) = if within_buffer {
728            let dist_to_start = view_start.saturating_sub(self.view.buffered_start_row);
729            let dist_to_end = self.view.buffered_end_row.saturating_sub(view_end);
730            let needs_expansion_back =
731                dist_to_start <= self.proximity_threshold && self.view.buffered_start_row > 0;
732            let needs_expansion_forward =
733                dist_to_end <= self.proximity_threshold && self.view.buffered_end_row < bound;
734
735            if !needs_expansion_back && !needs_expansion_forward {
736                // Buffer is fine, just re-slice display.
737                (self.view.buffered_start_row, self.view.buffered_end_row)
738            } else {
739                let mut s = if needs_expansion_back {
740                    view_start.saturating_sub(self.reach_rows(self.pages_lookback))
741                } else {
742                    self.view.buffered_start_row
743                };
744                let mut e = if needs_expansion_forward {
745                    (view_end + self.reach_rows(self.pages_lookahead)).min(bound)
746                } else {
747                    self.view.buffered_end_row
748                };
749                self.fit_window(view_start, view_end, &mut s, &mut e);
750                (s, e)
751            }
752        } else {
753            let had_buffer = self.view.buffered_end_row > 0;
754            let scrolled_past_end = had_buffer && view_start >= self.view.buffered_end_row;
755            let scrolled_past_start = had_buffer && view_end <= self.view.buffered_start_row;
756            let extend_forward_ok = scrolled_past_end
757                && (view_start - self.view.buffered_end_row)
758                    <= self.reach_rows(self.pages_lookahead);
759            let extend_backward_ok = scrolled_past_start
760                && (self.view.buffered_start_row - view_end)
761                    <= self.reach_rows(self.pages_lookback);
762
763            let mut s;
764            let mut e;
765            if extend_forward_ok {
766                s = self.view.buffered_start_row;
767                e = (view_end + self.reach_rows(self.pages_lookahead)).min(bound);
768            } else if extend_backward_ok {
769                s = view_start.saturating_sub(self.reach_rows(self.pages_lookback));
770                e = self.view.buffered_end_row;
771            } else {
772                s = view_start.saturating_sub(self.reach_rows(self.pages_lookback));
773                e = (view_end + self.reach_rows(self.pages_lookahead)).min(bound);
774                let min_initial_len = self.min_buffer_len();
775                let current_len = e.saturating_sub(s);
776                if current_len < min_initial_len {
777                    let need = min_initial_len.saturating_sub(current_len);
778                    let can_extend_end = bound.saturating_sub(e);
779                    let can_extend_start = s;
780                    if can_extend_end >= need {
781                        e = (e + need).min(bound);
782                    } else if can_extend_start >= need {
783                        s = s.saturating_sub(need);
784                    } else {
785                        e = (e + can_extend_end).min(bound);
786                        s = s.saturating_sub(need.saturating_sub(can_extend_end));
787                    }
788                }
789            }
790            self.fit_window(view_start, view_end, &mut s, &mut e);
791            (s, e)
792        };
793
794        let buffer_size = new_buffer_end.saturating_sub(new_buffer_start);
795        if buffer_size == 0 {
796            return None;
797        }
798        // Already held: the view fits, or fitting the expansion to whole row groups
799        // gave back the group on hand.
800        if self.holds_buffer(new_buffer_start, new_buffer_end) {
801            self.slice_buffer_into_display();
802            if self.table_state.selected().is_none() {
803                self.table_state.select(Some(0));
804            }
805            return None;
806        }
807
808        let lf = match self.buffer_lf(new_buffer_start, buffer_size) {
809            Ok(lf) => lf,
810            Err(e) => {
811                self.error = Some(e);
812                return None;
813            }
814        };
815
816        // Unknown count: `num_rows` is provisional (this buffer's end); the background `len()`
817        // corrects it unless a short read reveals the end.
818        let num_rows = if count_known {
819            self.view.num_rows
820        } else {
821            new_buffer_end
822        };
823        Some(CollectRequest {
824            lf,
825            polars_streaming: self.polars_streaming,
826            buffer_start: new_buffer_start,
827            buffer_end: new_buffer_end,
828            plan: self.fill_plan(new_buffer_start, new_buffer_end, num_rows, count_known),
829        })
830    }
831
832    /// How a fill of `[buffer_start, buffer_end)` is to be made the buffer, from what
833    /// is held and shown now. See [`FillPlan`].
834    pub(super) fn fill_plan(
835        &self,
836        buffer_start: usize,
837        buffer_end: usize,
838        num_rows: usize,
839        count_known: bool,
840    ) -> FillPlan {
841        let held = self
842            .abuts_buffer(buffer_start, buffer_end.saturating_sub(buffer_start))
843            .then(|| self.view.buffered_df.clone())
844            .flatten()
845            .map(|df| (df, self.view.buffered_start_row));
846        FillPlan {
847            buffer_start,
848            buffer_end,
849            num_rows,
850            count_known,
851            indexing: self.indexing().is_some(),
852            held,
853            view_start: self.view.start_row,
854            view_len: self.visible_rows,
855            max_rows: self.max_buffered_rows,
856            max_mb: self.max_buffered_mb,
857        }
858    }
859
860    /// Apply the result of a background buffer load. The worker has already stitched
861    /// and cut it ([`FillPlan::fit`]): installing it copies nothing.
862    pub fn apply_async_collect(&mut self, result: CollectResult) {
863        let CollectResult {
864            df,
865            start,
866            returned: returned_rows,
867            bytes_per_row,
868            buffer_start,
869            buffer_end,
870            num_rows,
871            count_known,
872            indexing,
873        } = result;
874        let requested_rows = buffer_end.saturating_sub(buffer_start);
875
876        if count_known {
877            self.view.num_rows = num_rows;
878            self.view.num_rows_valid = true;
879        } else if returned_rows < requested_rows
880            && (buffer_start == 0 || returned_rows > 0)
881            // Lines still being indexed end where the indexing has got to, not the file.
882            && !indexing
883            && self.indexing().is_none()
884        {
885            // A short read that began inside the data gives the exact total; an empty slice deep
886            // in the frame may lie past the data, so only the count can say.
887            self.view.num_rows = buffer_start + returned_rows;
888            self.view.num_rows_valid = true;
889        } else if !self.view.num_rows_valid {
890            // A full buffer with the count pending: a provisional total (at least this buffer's
891            // end), corrected by `count_landed()`.
892            self.view.num_rows = self.view.num_rows.max(buffer_end);
893        }
894        // else: the background len() already resolved the exact count between this
895        // buffer being requested and applied — keep it; don't downgrade to provisional.
896        self.error = None;
897        self.remember_pristine_count();
898
899        if bytes_per_row.is_some() {
900            self.view.observed_bytes_per_row = bytes_per_row;
901        }
902        // A fill without the view's first row was planned for replaced rows: keep what is held
903        // and replan. One holding the first row but not the whole view (a resize) is kept and
904        // the rest fetched. After a short read, a view past the end shows only rows up to it.
905        let end = start + df.height();
906        let view_end = self.view.start_row + self.visible_rows.max(1);
907        let reaches_end = end >= buffer_start + returned_rows;
908        let shows_view = start <= self.view.start_row
909            && (self.view.start_row < end || (returned_rows < requested_rows && reaches_end));
910        if !shows_view {
911            self.needs_recollect = true;
912            return;
913        }
914        self.release_display_buffer();
915        self.view.buffered_start_row = start;
916        self.view.buffered_end_row = end;
917        self.view.buffered_df = Some(df);
918        // Slice the buffered DataFrame into display DataFrames (locked + scroll columns).
919        self.slice_buffer_into_display();
920        if self.table_state.selected().is_none() {
921            self.table_state.select(Some(0));
922        }
923        if view_end > end && end < self.view.num_rows {
924            self.needs_recollect = true;
925        }
926    }
927
928    /// True when `rows` rows fetched from `start` run on from the rows on hand or up to
929    /// them, so a fill of them is planned to be stitched on (see [`FillPlan`]).
930    fn abuts_buffer(&self, start: usize, rows: usize) -> bool {
931        self.stitches_buffer()
932            && (start == self.view.buffered_end_row || start + rows == self.view.buffered_start_row)
933    }
934
935    /// A view of `sample`, drawn from `source` into `rows`, with `schema`'s columns. It
936    /// starts empty and grows via [`Self::sample_grew`]. `through`: drawn from the view's
937    /// query or filters rather than the source.
938    pub(crate) fn sampled_from(
939        source: DataTableState,
940        sample: crate::analysis::sampling::Sample,
941        schema: &Schema,
942        rows: Arc<crate::analysis::table_sample::SampleRows>,
943        through: bool,
944        path: Option<crate::analysis::table_sample::DrawPath>,
945    ) -> Result<Self> {
946        let mut view = source.sample_view(DataFrame::empty_with_schema(schema))?;
947        let frame = scanned_frame(&view.original_lf)
948            .ok_or_else(|| color_eyre::eyre::eyre!("a sample's frame has no rows to scan"))?;
949        view.sampled = Some(Box::new(Sampled {
950            source: Box::new(source),
951            sample,
952            rows,
953            frame,
954            through,
955            drawn: None,
956            path,
957        }));
958        Ok(view)
959    }
960
961    /// The view's sample, while it has one.
962    pub fn sampled(&self) -> Option<&Sampled> {
963        self.sampled.as_deref()
964    }
965
966    /// The view the sample was drawn from, or this one when it has none: where a new
967    /// sample is drawn from.
968    pub fn unsampled(&self) -> &DataTableState {
969        self.sampled
970            .as_ref()
971            .map_or(self, |sampled| sampled.source.as_ref())
972    }
973
974    /// The view the sample was drawn from, putting the sample down; `self` when it
975    /// has none.
976    pub(crate) fn into_unsampled(mut self) -> DataTableState {
977        match self.sampled.take() {
978            Some(sampled) => *sampled.source,
979            None => self,
980        }
981    }
982
983    /// Take the chunks drawn since the last call into every frame (so query, filters and
984    /// sort run over them). `None` if none; otherwise whether the rows on hand still stand
985    /// (they do while nothing reorders, since new rows come after).
986    pub(crate) fn sample_grew(&mut self) -> Option<bool> {
987        let sampled = self.sampled.as_ref()?;
988        let chunks = sampled.rows.take_new();
989        if chunks.is_empty() {
990            return None;
991        }
992        // On the same buffers: each column takes the chunks' arrays, nothing copied.
993        let mut frame = (*sampled.frame).clone();
994        for chunk in &chunks {
995            frame.vstack_mut(chunk).ok()?;
996        }
997        Some(self.rebind_sample(Arc::new(frame), false))
998    }
999
1000    /// The draw ended, having read what `drawn` says: the rows go into the order the
1001    /// source holds them, once.
1002    pub(crate) fn sample_drawn(&mut self, drawn: crate::analysis::table_sample::Drawn) {
1003        let Some(sampled) = self.sampled.as_mut() else {
1004            return;
1005        };
1006        // One chunk per column from here: the many the draw left would slow every
1007        // read, and the chunks are let go so the rows are held once.
1008        let ordered = sampled.rows.take_in_source_order().ok().flatten();
1009        // A seeded read of one file needs no path; it was not one, then.
1010        sampled.path = drawn.path;
1011        sampled.drawn = Some(drawn);
1012        if let Some(frame) = ordered {
1013            self.rebind_sample(Arc::new(frame), true);
1014        }
1015    }
1016
1017    /// Every frame scans `frame` in place of the sample's last one. `reordered` when
1018    /// the rows already shown changed places. Returns whether the rows on hand stand.
1019    fn rebind_sample(&mut self, frame: Arc<DataFrame>, reordered: bool) -> bool {
1020        let Some(old) = self.sampled.as_ref().map(|sampled| sampled.frame.clone()) else {
1021            return false;
1022        };
1023        let rows_stand = !reordered
1024            && self.view.sort_columns.is_empty()
1025            && self.view.sort_ascending
1026            && self.scan_is_the_root();
1027        let rows = frame.height();
1028        self.each_frame(|lf| {
1029            crate::analysis::table_sample::rebind(&mut lf.logical_plan, &old, &frame)
1030        });
1031        if let Some(sampled) = self.sampled.as_mut() {
1032            sampled.frame = frame;
1033        }
1034        self.invalidate_num_rows();
1035        if self.is_pristine() {
1036            self.set_num_rows(rows);
1037        } else if self.scan_is_the_root() {
1038            self.pristine_rows = Some(rows);
1039        }
1040        if !rows_stand {
1041            self.drop_buffer();
1042        }
1043        self.needs_recollect = true;
1044        rows_stand
1045    }
1046
1047    /// Bytes per row of a sample of this view: every column of the source (from source) or
1048    /// the view, shown or not.
1049    pub(crate) fn sample_row_bytes(&self, from_source: bool) -> usize {
1050        let schema = if from_source {
1051            &self.original_schema
1052        } else {
1053            &self.view.schema
1054        };
1055        let columns: Vec<String> = schema
1056            .iter_names()
1057            .filter(|name| name.as_str() != crate::formats::schema_union::DRIFT_COLUMN)
1058            .map(|name| name.to_string())
1059            .collect();
1060        // What the table measured, when it measured these columns.
1061        if !from_source && columns.len() == self.view.column_order.len() {
1062            return self.bytes_per_row();
1063        }
1064        estimate_bytes_per_row(schema, &columns, &self.column_bytes)
1065    }
1066
1067    /// How many files a page at `start` would read: only for a windowed remote scan;
1068    /// `None` (not zero) when Polars reads the whole scan as it decides.
1069    pub fn files_a_page_reads(&self, start: usize, len: usize) -> Option<usize> {
1070        let offsets = self.files_window().and_then(|f| f.offsets.as_ref())?;
1071        let (first, last) = files_holding(offsets, start, len)?;
1072        Some(files_with_rows(offsets, first, last).len())
1073    }
1074
1075    /// The frame for buffer rows `[start, start + len)`, columns in display order. For a
1076    /// remote dataset whose files are counted, a scan of only the files holding them.
1077    pub(super) fn buffer_lf(&self, start: usize, len: usize) -> PolarsResult<LazyFrame> {
1078        let mut all_columns = self.binary_stub_exprs();
1079        if self.carries_source_rows() {
1080            all_columns.push(col(crate::formats::schema_union::DRIFT_COLUMN));
1081        }
1082        self.window_lf(start, len, all_columns)
1083    }
1084
1085    /// Whether rows carry their source position for `#`: file-traced rows or lines while
1086    /// the frame is the scan's (query, reshape and group rows do not).
1087    pub(crate) fn carries_source_rows(&self) -> bool {
1088        self.view.drift_column_present
1089            || (self.scan_is_the_root() && (self.source_rows_at_open || self.view.view_numbered))
1090    }
1091
1092    /// What `#` shows for `rows` rows from `start`: source positions where carried, else
1093    /// view positions from `row_start_index` (the same when pristine).
1094    pub fn row_numbers_from(&self, start: usize, rows: usize) -> Vec<usize> {
1095        let view = |i: usize| start + i + self.row_start_index;
1096        let places = self
1097            .view
1098            .buffered_df
1099            .as_ref()
1100            .filter(|_| self.carries_source_rows())
1101            .and_then(|df| df.column(crate::formats::schema_union::DRIFT_COLUMN).ok())
1102            .and_then(|column| {
1103                let offset = start.checked_sub(self.view.buffered_start_row)?;
1104                let len = rows.min(column.len().saturating_sub(offset));
1105                let slice = column.slice(offset as i64, len);
1106                let places = slice.u32().ok()?;
1107                // Several files' lines are numbered in their own file.
1108                let place = |p: usize| {
1109                    self.numbering
1110                        .as_ref()
1111                        .and_then(|lines| lines.line_in_file(p))
1112                        .unwrap_or(p)
1113                };
1114                Some(
1115                    places
1116                        .iter()
1117                        .map(|p| p.map(|p| place(p as usize) + self.row_start_index))
1118                        .collect::<Vec<_>>(),
1119                )
1120            });
1121        (0..rows)
1122            .map(|i| {
1123                places
1124                    .as_ref()
1125                    .and_then(|p| p.get(i).copied().flatten())
1126                    .unwrap_or_else(|| view(i))
1127            })
1128            .collect()
1129    }
1130
1131    /// The frame for rows `[start, start + len)` of the view, as `all_columns`. For a
1132    /// remote dataset whose files are counted, a scan of only the files holding them.
1133    pub(super) fn window_lf(
1134        &self,
1135        start: usize,
1136        len: usize,
1137        all_columns: Vec<Expr>,
1138    ) -> PolarsResult<LazyFrame> {
1139        window_of(
1140            &self.view.lf,
1141            self.files_window(),
1142            self.window_now().as_deref(),
1143            &self.read_as_text,
1144            start,
1145            len,
1146            all_columns,
1147        )
1148    }
1149
1150    /// The view's rows as a find reads them: a window at a time, the way a page is
1151    /// read, and the buffer already on hand.
1152    pub(crate) fn view_rows(&self) -> ViewRows {
1153        ViewRows {
1154            lf: self.view.lf.clone(),
1155            files: self.files_window().cloned(),
1156            // A find reads every row it can reach: lines still being indexed are read
1157            // through the frame, which waits for them, not the window of those so far.
1158            records: self.window_now().filter(|_| self.indexing().is_none()),
1159            read_as_text: self.read_as_text.clone(),
1160            buffer: self
1161                .view
1162                .buffered_df
1163                .as_ref()
1164                .filter(|_| self.buffer_on_hand())
1165                .map(|df| (df.clone(), self.view.buffered_start_row)),
1166            num_rows: self.view.num_rows_valid.then_some(self.view.num_rows),
1167            streaming: self.polars_streaming,
1168            whole: sees_every_row_first(&self.view.lf),
1169            reads_up_to: reads_up_to_a_window(&self.view.lf),
1170        }
1171    }
1172
1173    /// Center the cursor on view row `row`, where a find matched; true if a collect is
1174    /// needed. A row past a provisional total extends it until the count lands.
1175    pub(crate) fn go_to_found_row(&mut self, row: usize) -> bool {
1176        if !self.view.num_rows_valid && self.view.num_rows <= row {
1177            self.view.num_rows = row + 1;
1178        }
1179        self.scroll_to_row_centered(row)
1180    }
1181
1182    /// The view row the cursor is on.
1183    pub(crate) fn cursor_row(&self) -> usize {
1184        self.view.start_row + self.table_state.selected().unwrap_or(0)
1185    }
1186
1187    /// Bytes a buffered row takes: measured on the last buffer collected, or until
1188    /// then estimated from the schema.
1189    pub(super) fn bytes_per_row(&self) -> usize {
1190        self.view.observed_bytes_per_row.unwrap_or_else(|| {
1191            estimate_bytes_per_row(
1192                &self.view.schema,
1193                &self.view.column_order,
1194                &self.column_bytes,
1195            )
1196        })
1197    }
1198
1199    /// The best in-memory width estimate per row, shared by buffer planning and Data
1200    /// Quality's (approximate) preflight so they agree.
1201    pub fn estimated_row_bytes(&self) -> usize {
1202        self.bytes_per_row()
1203    }
1204
1205    /// Number of source files known to participate in the pristine dataset scan.
1206    /// Returns `None` after a query or reshape has broken the row-to-file mapping.
1207    pub fn source_file_count(&self) -> Option<usize> {
1208        self.is_pristine().then(|| self.loaded_file_count())
1209    }
1210
1211    /// Files the dataset was loaded from, whatever the view does with their rows.
1212    pub(crate) fn loaded_file_count(&self) -> usize {
1213        if !self.drift_files.is_empty() {
1214            self.drift_files.len()
1215        } else if let Some(remote) = &self.remote_files {
1216            remote.urls.len()
1217        } else {
1218            1
1219        }
1220    }
1221
1222    /// Rows the `max_buffered_mb` budget allows, at least a screen; 0 for none. Planned to,
1223    /// so a wide window is never materialized only to be cut.
1224    pub(super) fn byte_cap_rows(&self) -> usize {
1225        if self.max_buffered_mb == 0 {
1226            return 0;
1227        }
1228        let max_bytes = self.max_buffered_mb * 1024 * 1024;
1229        (max_bytes / self.bytes_per_row()).max(self.visible_rows.max(1))
1230    }
1231
1232    /// Whether the buffer is a remote window: a pristine object-store scan. With anything
1233    /// applied, `slice(0, N)` stops at N matches, so page windows cost a row group, not
1234    /// forty.
1235    pub(super) fn remote_window(&self) -> bool {
1236        self.remote_source && self.is_pristine()
1237    }
1238
1239    /// The files a page reads by, while the frame is the scan as loaded: a filter or
1240    /// sort reads every file before its window, so it goes through the whole scan.
1241    fn files_window(&self) -> Option<&RemoteFiles> {
1242        self.remote_files.as_ref().filter(|_| self.is_pristine())
1243    }
1244
1245    /// Rows the buffer reaches past the view one way: `pages` of it locally, for remote
1246    /// scans with something applied, or many-file datasets (where cost is files opened, so
1247    /// a few at a time); half the window for a pristine remote object (`fit_window` trims
1248    /// to the cap).
1249    fn reach_rows(&self, pages: usize) -> usize {
1250        if !self.remote_window() || self.remote_files.is_some() {
1251            return pages * self.visible_rows.max(1);
1252        }
1253        let window = if self.max_buffered_rows > 0 {
1254            self.max_buffered_rows
1255        } else {
1256            DEFAULT_MAX_BUFFERED_ROWS
1257        };
1258        window / 2
1259    }
1260
1261    /// The smallest buffer worth filling: a page plus the reach either side.
1262    fn min_buffer_len(&self) -> usize {
1263        self.visible_rows.max(1)
1264            + self.reach_rows(self.pages_lookahead)
1265            + self.reach_rows(self.pages_lookback)
1266    }
1267
1268    /// True when the view already shows the last page, so End has nothing to load.
1269    pub fn at_end(&self) -> bool {
1270        self.view.start_row == self.view.num_rows.saturating_sub(self.visible_rows)
1271    }
1272
1273    /// Fit `[buffer_start, buffer_end)` to the caps around the view, then for a remote
1274    /// object with a known footer to the view's row groups, cut back to the caps inside.
1275    fn fit_window(
1276        &self,
1277        view_start: usize,
1278        view_end: usize,
1279        buffer_start: &mut usize,
1280        buffer_end: &mut usize,
1281    ) {
1282        let byte_cap = self.byte_cap_rows();
1283        let cap = match (self.max_buffered_rows, byte_cap) {
1284            (0, cap) | (cap, 0) => cap,
1285            (rows, bytes) => rows.min(bytes),
1286        };
1287        if cap > 0 {
1288            shrink_around_view(
1289                view_start,
1290                view_end,
1291                cap,
1292                0,
1293                self.num_rows_bound(),
1294                buffer_start,
1295                buffer_end,
1296            );
1297        }
1298        let Some(offsets) = self
1299            .row_group_offsets
1300            .as_deref()
1301            .filter(|_| self.remote_window())
1302        else {
1303            return;
1304        };
1305        (*buffer_start, *buffer_end) = align_to_row_groups(
1306            offsets,
1307            view_start,
1308            view_end,
1309            *buffer_start,
1310            *buffer_end,
1311            cap,
1312        );
1313        // Caps hold inside a group: an oversized group is read a window at a time, never
1314        // pulling the next group early.
1315        if cap > 0 {
1316            let (floor, ceil) = (*buffer_start, *buffer_end);
1317            shrink_around_view(
1318                view_start,
1319                view_end,
1320                cap,
1321                floor,
1322                ceil,
1323                buffer_start,
1324                buffer_end,
1325            );
1326        }
1327        // Over many files, at most a few of them, around the view's.
1328        if let Some(file_offsets) = self.remote_files.as_ref().and_then(|f| f.offsets.as_ref()) {
1329            (*buffer_start, *buffer_end) = limit_files(
1330                file_offsets,
1331                view_start,
1332                view_end,
1333                *buffer_start,
1334                *buffer_end,
1335                MAX_FILES_PER_BUFFER,
1336            );
1337        }
1338        // A view straddling two groups needs both, but one is on hand: fetch the other
1339        // alone and stitch it on (see `apply_async_collect`).
1340        if self.buffer_on_hand() {
1341            let (held_start, held_end) = (self.view.buffered_start_row, self.view.buffered_end_row);
1342            if held_start <= *buffer_start && *buffer_start < held_end && held_end < *buffer_end {
1343                *buffer_start = held_end;
1344            } else if *buffer_start < held_start
1345                && held_start < *buffer_end
1346                && *buffer_end <= held_end
1347            {
1348                *buffer_end = held_start;
1349            }
1350        }
1351    }
1352
1353    /// Release the replaced buffer and its display frames (a stitch already took what it
1354    /// keeps); a requested relearn applies to the next rows.
1355    fn release_display_buffer(&mut self) {
1356        self.widths.rows_arrived();
1357        self.view.buffered_df = None;
1358        self.view.locked_df = None;
1359        self.view.df = None;
1360    }
1361
1362    /// Recompute locked_df and df from the cached full buffer. Used when only termcol_index (or locked columns) changed.
1363    pub(super) fn slice_buffer_into_display(&mut self) {
1364        let full_df = match self.view.buffered_df.as_ref() {
1365            Some(df) => df,
1366            None => return,
1367        };
1368
1369        if self.view.locked_columns_count > 0 {
1370            let locked_names: Vec<&str> = self
1371                .view
1372                .column_order
1373                .iter()
1374                .take(self.view.locked_columns_count)
1375                .map(|s| s.as_str())
1376                .collect();
1377            if let Ok(locked_df) = full_df.select(locked_names) {
1378                self.view.locked_df = Some(locked_df);
1379            }
1380        } else {
1381            self.view.locked_df = None;
1382        }
1383
1384        let scroll_names: Vec<&str> = self
1385            .view
1386            .column_order
1387            .iter()
1388            .skip(self.frozen_shown() + self.termcol_index)
1389            .map(|s| s.as_str())
1390            .collect();
1391        if scroll_names.is_empty() {
1392            self.view.df = None;
1393        } else {
1394            if let Ok(scroll_df) = full_df.select(scroll_names) {
1395                self.view.df = Some(scroll_df);
1396            }
1397        }
1398    }
1399
1400    /// Whether the view is inside the buffer within a page of an end with more data past
1401    /// it: where to grow the buffer ahead, before an in-buffer scroll leaves it blank.
1402    pub fn wants_to_load_ahead(&self) -> bool {
1403        if self.visible_rows == 0
1404            || self.view.buffered_df.is_none()
1405            || !self.page_on_hand(self.view.start_row)
1406        {
1407            return false;
1408        }
1409        let near = self.proximity();
1410        let view_end = self.view.start_row
1411            + self
1412                .visible_rows
1413                .min(self.num_rows_bound().saturating_sub(self.view.start_row));
1414        let behind = self.view.start_row - self.view.buffered_start_row <= near
1415            && self.view.buffered_start_row > 0;
1416        let ahead = self.view.buffered_end_row - view_end <= near
1417            && self.view.buffered_end_row < self.num_rows_bound();
1418        behind || ahead
1419    }
1420
1421    /// How near an end the view comes before the buffer grows: half the reach, at least a
1422    /// page (a cloud fetch outlasts a PageDown).
1423    fn proximity(&self) -> usize {
1424        (self.reach_rows(self.pages_lookahead) / 2).max(self.visible_rows)
1425    }
1426
1427    /// Where the view and the buffer are, to tell one load-ahead attempt from the next.
1428    pub fn buffer_position(&self) -> (u64, usize, usize, usize) {
1429        (
1430            self.len_generation(),
1431            self.view.start_row,
1432            self.view.buffered_start_row,
1433            self.view.buffered_end_row,
1434        )
1435    }
1436
1437    /// Whether every row of the page starting at `start` is in the buffer.
1438    pub(crate) fn page_on_hand(&self, start: usize) -> bool {
1439        let bound = self.num_rows_bound();
1440        let end = start + self.visible_rows.min(bound.saturating_sub(start));
1441        self.view.buffered_df.is_some()
1442            && self.view.buffered_end_row > 0
1443            && start >= self.view.buffered_start_row
1444            && end <= self.view.buffered_end_row
1445    }
1446
1447    /// The first row to draw: the view's own once its rows are on hand, else the last page
1448    /// drawn whole, so a pending fetch never shows half a page of nothing.
1449    pub(crate) fn start_to_draw(&mut self) -> usize {
1450        if self.page_on_hand(self.view.start_row) {
1451            self.view.drawn_start = self.view.start_row;
1452            self.view.start_row
1453        } else if self.page_on_hand(self.view.drawn_start) {
1454            self.view.drawn_start
1455        } else {
1456            self.view.start_row
1457        }
1458    }
1459}