Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::cmp::Ordering;
36use std::collections::{HashMap, VecDeque};
37use std::fs::{File, OpenOptions};
38use std::io::{Read, Seek, SeekFrom};
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Mutex, OnceLock};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::{bitpack, chooser, integer, string};
48use rudb_storage::sieve::Sieve;
49use rudb_storage::{Probe, Range, Zone};
50use rudb_vector::string::StringColumn;
51use rudb_vector::validity::Validity;
52use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
53
54pub mod graph;
55pub mod section;
56pub mod stats;
57mod zones;
58
59pub use section::Section;
60pub use zones::{Common, Stripes, distincts};
61
62const MAGIC: &[u8; 8] = b"RUDBNV10";
63const DIRECTORY: &[u8; 8] = b"RUDBDI10";
64const CATALOG: &[u8; 8] = b"RUDBCA10";
65const FORMAT: u32 = 25;
66
67/// Formats this build can open.
68///
69/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
70/// criterion: a build with the section table in it has to open a file written before the section
71/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
72/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
73/// graph sections is.
74///
75/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
76/// was tags for fourteen more column types, and a file written before that has none of them in it,
77/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
78/// section table, which a file written before it simply does not have. What takes it from 24 to 25
79/// is the view section on the end of the catalog, which an older file does not have either, and a
80/// catalog that ends where the tables end reads as a catalog with no views in it.
81///
82/// This is not a general compatibility promise. Four formats are readable because there was a
83/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
84/// carrying.
85const READABLE: &[u32] = &[22, 23, 24, FORMAT];
86
87const HEADER: u64 = 80;
88const SLOT_BYTES: usize = 28;
89const MAX_PAGE: usize = 256 * 1024 * 1024;
90const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
91const FREQUENCIES: &[u8; 8] = b"RUDBFQ2\0";
92/// The clustering declaration, written after the frequencies and only when there is one.
93///
94/// No format bump for this, which is the convention the frequency section set in #728: a new
95/// optional trailing section with its own magic leaves every file that does not use it byte for
96/// byte what it was, and the version is bumped for a change to a layout that already exists, as
97/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
98///
99/// The width byte in this block gained a fifth value for #1285, for a declaration that leaves the
100/// bucket to the row count, and that did not bump the format either. It is the one case where the
101/// reasoning needs saying out loud, because it is a new value in a layout that already exists
102/// rather than a new section. A build without it reading one of these says `clustering width
103/// tag differs` and refuses the table, which is what that message was written for. Bumping the
104/// format instead would have made every file this build writes unreadable to an older one, whether
105/// it has a declaration in it or not, to warn about a case that only arises when it does.
106const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
107/// The graph section table, written after the clustering declaration and written even when empty.
108///
109/// Same convention and the same reason as the block above it, with one difference: this one is
110/// always there, so a file written by this build says which sections it has rather than leaving a
111/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
112/// that safe to add without a format bump, because a table with no sections answers every query
113/// the way it did before, only without the graph path.
114const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
115
116/// The most sections one table's directory may name.
117///
118/// A relationship contributes at most three sections, so this bounds a table at a few thousand
119/// relationships, which is far past anything a schema has. The bound is here so that a torn
120/// directory naming four billion of them is refused at decode rather than turned into an
121/// allocation, the same reason the extent count has one.
122const MAX_SECTIONS: usize = 4096;
123const FREQUENCY_CANDIDATES: usize = 32_768;
124const FREQUENCY_ENTRIES: usize = 512;
125const FREQUENCY_BUILD_RANK: usize = 10;
126const FREQUENCY_ORDINALS: usize = 65_536;
127/// The most threads the two per column passes at the end of a commit are spread over.
128///
129/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
130/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
131/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
132/// on a narrow machine would be worse than waiting.
133const MAX_FREQUENCY_WORKERS: usize = 32;
134
135/// The most threads one stripe's encode is spread over.
136///
137/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
138/// it, and the work is one column of sixty four parts, which is large enough that a thread that
139/// takes one is not a thread that was started for nothing. A machine with more cores than this has
140/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
141const MAX_ENCODE_WORKERS: usize = 32;
142
143/// The most bytes one column of one part may spend on a membership sieve.
144///
145/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
146/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
147/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
148/// rule in `encode_column` that a sieve may not be as large as the part it indexes, which is a cap
149/// per column rather than one number for the whole file.
150const SIEVE_BUDGET: usize = 8 * 1024;
151
152/// The most bytes one end of a per part range may spend on a string.
153///
154/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
155/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
156/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
157/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
158/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
159/// where two URLs of the same site still look alike.
160const PART_BOUND_BYTES: usize = 24;
161
162fn io(error: std::io::Error) -> Error {
163    Error::io(error.to_string())
164}
165
166fn invalid(message: &str) -> Error {
167    Error::invalid_input(format!("invalid rudb native file: {message}"))
168}
169
170/// Adds a sequence of byte counts without an overflow the caller has to think about.
171fn sum(counts: impl Iterator<Item = u64>) -> u64 {
172    counts.fold(0, u64::saturating_add)
173}
174
175/// One column's span out of a per column list, or zero when the list is shorter than the column.
176fn span_bytes(spans: &[Span], at: usize) -> u64 {
177    spans.get(at).map_or(0, |span| u64::from(span.length))
178}
179
180/// One column's page out of a per column list, or zero when that column has no page at all.
181fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
182    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
183}
184
185/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
186///
187/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
188/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
189/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
190/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
191/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
192/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
193/// 8 is about five percent of the query.
194fn checksum(bytes: &[u8]) -> u64 {
195    const P1: u64 = 11_400_714_785_074_694_791;
196    const P2: u64 = 14_029_467_366_897_019_727;
197    const P3: u64 = 1_609_587_929_392_839_161;
198    const P4: u64 = 9_650_029_242_287_828_579;
199    const P5: u64 = 2_870_177_450_012_600_261;
200    let round = |state: u64, word: u64| {
201        state.wrapping_add(word.wrapping_mul(P2)).rotate_left(31).wrapping_mul(P1)
202    };
203    let merge = |state: u64, lane: u64| (state ^ round(0, lane)).wrapping_mul(P1).wrapping_add(P4);
204    let word = |chunk: &[u8]| u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"));
205
206    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
207    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
208    let mut blocks = bytes.chunks_exact(32);
209    let mut rest = blocks.remainder();
210    let mut hash = if bytes.len() >= 32 {
211        let mut one = P1.wrapping_add(P2);
212        let mut two = P2;
213        let mut three = 0;
214        let mut four = 0_u64.wrapping_sub(P1);
215        for block in blocks.by_ref() {
216            one = round(one, word(&block[..8]));
217            two = round(two, word(&block[8..16]));
218            three = round(three, word(&block[16..24]));
219            four = round(four, word(&block[24..]));
220        }
221        let combined = one
222            .rotate_left(1)
223            .wrapping_add(two.rotate_left(7))
224            .wrapping_add(three.rotate_left(12))
225            .wrapping_add(four.rotate_left(18));
226        merge(merge(merge(merge(combined, one), two), three), four)
227    } else {
228        P5
229    };
230    hash = hash.wrapping_add(bytes.len() as u64);
231    let mut words = rest.chunks_exact(8);
232    for chunk in words.by_ref() {
233        hash ^= round(0, word(chunk));
234        hash = hash.rotate_left(27).wrapping_mul(P1).wrapping_add(P4);
235    }
236    rest = words.remainder();
237    if rest.len() >= 4 {
238        let (head, tail) = rest.split_at(4);
239        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
240        hash ^= u64::from(quarter).wrapping_mul(P1);
241        hash = hash.rotate_left(23).wrapping_mul(P2).wrapping_add(P3);
242        rest = tail;
243    }
244    for &byte in rest {
245        hash ^= u64::from(byte).wrapping_mul(P5);
246        hash = hash.rotate_left(11).wrapping_mul(P1);
247    }
248    hash ^= hash >> 33;
249    hash = hash.wrapping_mul(P2);
250    hash ^= hash >> 29;
251    hash = hash.wrapping_mul(P3);
252    hash ^ (hash >> 32)
253}
254
255#[derive(Debug, Clone, Copy)]
256struct Slot {
257    offset: u64,
258    length: u32,
259    generation: u64,
260    hash: u64,
261}
262
263impl Slot {
264    fn bytes(self) -> [u8; SLOT_BYTES] {
265        let mut result = [0; SLOT_BYTES];
266        result[..8].copy_from_slice(&self.offset.to_le_bytes());
267        result[8..12].copy_from_slice(&self.length.to_le_bytes());
268        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
269        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
270        result
271    }
272
273    fn read(bytes: &[u8]) -> Self {
274        Self {
275            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
276            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
277            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
278            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
279        }
280    }
281}
282
283#[derive(Debug, Clone, Copy)]
284struct Page {
285    offset: u64,
286    length: u32,
287    hash: u64,
288}
289
290impl Page {
291    /// How much of the file this page takes, for [`Reader::layout`].
292    fn bytes(&self) -> u64 {
293        u64::from(self.length)
294    }
295}
296
297#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
298enum FrequencyValue {
299    Null,
300    Integer(i128),
301    Code(u32),
302}
303
304#[derive(Debug, Clone)]
305struct FrequencyEntry {
306    value: FrequencyValue,
307    count: u64,
308}
309
310/// Exact leading frequencies for one column.
311///
312/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
313/// use the synopsis only when its last winner is strictly above every omitted value.
314#[derive(Debug, Clone)]
315struct FrequencySummary {
316    entries: Vec<FrequencyEntry>,
317    omitted_max: u64,
318    ordinals: Vec<u64>,
319}
320
321/// The values one column's frequency synopsis lists, with a bound on everything it left out.
322///
323/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
324/// rows any value not in the list can hold, which is zero when nothing was left out at all.
325#[derive(Debug, Clone)]
326pub struct FrequencyPrefix {
327    /// Every value the synopsis lists, with the number of rows holding it, count descending.
328    pub entries: Vec<(Value, u64)>,
329    /// How many rows the most common value outside the list holds, and zero for a complete list.
330    pub omitted_max: u64,
331}
332
333/// Sparse row ordinals covered by a numeric frequency candidate set.
334#[derive(Debug, Clone, PartialEq, Eq)]
335pub struct FrequencyOccurrences {
336    /// Upper bound for the frequency of every value absent from the fetched rows.
337    pub omitted_max: u64,
338    /// Table-wide row ordinals in ascending order.
339    pub ordinals: Vec<u64>,
340}
341
342/// Where one column's page for one stripe sits in the file.
343///
344/// A column page has no checksum of its own because every part inside it carries one, and the
345/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
346/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
347/// or pulled one part out of the middle of it.
348#[derive(Debug, Clone, Copy, Default)]
349struct Span {
350    offset: u64,
351    length: u32,
352}
353
354/// One independently readable stripe of a table.
355#[derive(Debug, Clone)]
356pub struct Stripe {
357    rows: usize,
358    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
359    /// part, which every sparse fetch does, never reads the file.
360    parts: Vec<u32>,
361    /// The index page: one section per column, holding a length and a checksum for every part and
362    /// then a checksum of the section itself, so that a reader can pread one column's section and
363    /// still know it is intact.
364    index: Span,
365    pages: Vec<Span>,
366    memberships: Vec<Option<Page>>,
367    /// One page per column holding the membership sieve of every part of the stripe, for the
368    /// columns that have one. A column whose parts all declined a sieve has no page at all.
369    sieves: Vec<Option<Page>>,
370    /// One page per column holding the two ends and the null count of every part of the stripe.
371    ///
372    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
373    /// not the one the rows are ordered by that is the difference between skipping half the file and
374    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
375    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
376    ///
377    /// A page per column rather than one page for the stripe, so that a query that compares one
378    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
379    /// for the same reason, like the sieves.
380    part_ranges: Vec<Option<Page>>,
381    zone: Zone,
382}
383
384impl Stripe {
385    /// Number of rows in this stripe.
386    #[must_use]
387    pub fn rows(&self) -> usize {
388        self.rows
389    }
390
391    /// Number of parts in this stripe.
392    #[must_use]
393    pub fn parts(&self) -> usize {
394        self.parts.len()
395    }
396
397    /// The two ends and the null count of every column over the whole stripe.
398    ///
399    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
400    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
401    /// scan wants to know which parts to open.
402    #[must_use]
403    pub fn zone(&self) -> &Zone {
404        &self.zone
405    }
406}
407
408/// The committed table directory.
409#[derive(Debug, Clone)]
410pub struct Table {
411    name: String,
412    fields: Vec<Field>,
413    stripes: Vec<Stripe>,
414    rows: usize,
415    dictionaries: Vec<Option<Page>>,
416    frequencies: Vec<Option<FrequencySummary>>,
417    /// How many distinct values each column holds, for the columns that know.
418    ///
419    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
420    /// the size of the dictionary is the number of distinct values in the column. That is the whole
421    /// story for a column with no null in it, and the wrong number by one for a column with a null
422    /// in it, because a null row is written as the code for the empty string and makes an entry the
423    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
424    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
425    /// work it out from the dictionary alone. So the writer settles it here.
426    distincts: Vec<Option<u64>>,
427    /// The order the rows of this table are meant to be stored in, if anybody declared one.
428    ///
429    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
430    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
431    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
432    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
433    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
434    clustering: Option<Clustering>,
435    /// The file generation of the commit that last wrote this table's column pages.
436    ///
437    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
438    /// the definition is deliberately about the pages rather than about the directory. A graph
439    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
440    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
441    /// section to this one, commits a new file generation without touching a single row of this
442    /// table, and a definition that moved with those would declare every section in the file stale
443    /// for no reason.
444    ///
445    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
446    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
447    /// sections for it to match anyway.
448    generation: u64,
449    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
450    ///
451    /// Empty for every table written before the section table existed, and empty is not a
452    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
453    /// only the time, so a table with none here answers every query the same way and slower. That
454    /// is what lets this field arrive without a migration.
455    sections: Vec<Section>,
456}
457
458impl Table {
459    /// The SQL table name held by this snapshot.
460    #[must_use]
461    pub fn name(&self) -> &str {
462        &self.name
463    }
464
465    /// Columns in their SQL order.
466    #[must_use]
467    pub fn fields(&self) -> &[Field] {
468        &self.fields
469    }
470
471    /// Committed row count.
472    #[must_use]
473    pub fn rows(&self) -> usize {
474        self.rows
475    }
476
477    /// Independently readable stripes.
478    #[must_use]
479    pub fn stripes(&self) -> &[Stripe] {
480        &self.stripes
481    }
482
483    /// The order the rows are meant to be stored in, if this table was declared with one.
484    #[must_use]
485    pub fn clustering(&self) -> Option<&Clustering> {
486        self.clustering.as_ref()
487    }
488
489    /// The generation every section of this table is judged against.
490    ///
491    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
492    /// this.
493    #[must_use]
494    pub fn generation(&self) -> u64 {
495        self.generation
496    }
497
498    /// Every graph section this table names, including the kinds this build does not know.
499    ///
500    /// Including them is the point. A caller that wants only the ones it can use asks
501    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
502    /// file opened by an older build and written again does not silently lose a section that build
503    /// had no name for.
504    #[must_use]
505    pub fn sections(&self) -> &[Section] {
506        &self.sections
507    }
508}
509
510/// One table's line in the catalog directory.
511///
512/// The small level of the two. It holds what opening a database needs and nothing else: the name to
513/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
514/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
515/// thousand rows or a billion.
516///
517/// The name, the fields and the row count are repeated here rather than pointed at inside the table
518/// directory, which is the entire point of having two levels. A catalog that pointed at them would
519/// have to read every table directory at open to answer what tables there are, which is the cost
520/// this level exists to avoid.
521#[derive(Debug, Clone)]
522struct Entry {
523    name: String,
524    fields: Vec<Field>,
525    rows: usize,
526    /// Where this table's own directory sits, with the checksum it was committed under.
527    directory: Page,
528}
529
530/// One view's line in the catalog directory.
531///
532/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
533/// What it is made of is text: the body the binder binds again at every reference, and the whole
534/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
535///
536/// The columns are a cache and they are written down anyway, which is worth saying out loud because
537/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
538/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
539/// true without anything having bound the body, so the list survived the write. Not writing it
540/// would answer null and false there, and the only way back would be to bind every view at open,
541/// which is the thing the cache exists to avoid.
542#[derive(Debug, Clone, PartialEq, Eq)]
543pub struct ViewEntry {
544    /// The view's own name, without the schema, the way a table entry holds its name.
545    pub name: String,
546    /// The query the view stands for, as the text that was written.
547    pub sql: String,
548    /// The whole `CREATE VIEW` written back out.
549    pub statement: String,
550    /// The column names the statement gave, which rename a prefix of what the body produces.
551    pub aliases: Vec<String>,
552    /// The columns the last bind of the body produced.
553    pub columns: Vec<Field>,
554}
555
556/// Where one column's bytes went, taken from the directory rather than by reading pages.
557#[derive(Debug, Clone)]
558pub struct ColumnLayout {
559    /// The column's name, so a report does not have to carry the field list beside this.
560    pub name: String,
561    /// The type, spelled the way the catalog spells it.
562    pub kind: String,
563    /// Every stripe's page of this column added up, which is the encoded data itself.
564    pub pages: u64,
565    /// Every stripe's exact code membership page for this column.
566    pub memberships: u64,
567    /// Every stripe's membership sieve page for this column.
568    pub sieves: u64,
569    /// Every stripe's per part range page for this column.
570    pub part_ranges: u64,
571    /// The table wide dictionary of this column, if it has one.
572    pub dictionary: u64,
573}
574
575impl ColumnLayout {
576    /// Everything this column costs, which is what the file would lose if the column went.
577    #[must_use]
578    pub fn total(&self) -> u64 {
579        self.pages
580            .saturating_add(self.memberships)
581            .saturating_add(self.sieves)
582            .saturating_add(self.part_ranges)
583            .saturating_add(self.dictionary)
584    }
585}
586
587/// Where a whole file's bytes went.
588///
589/// Every number here comes out of the committed directory, so taking it costs one directory read
590/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
591/// without being read, or nobody will ask.
592///
593/// The parts that are not a column are kept apart rather than shared out over the columns. The
594/// stripe index page holds a section per column and could be split, and the directory and the
595/// header cannot be, so splitting one of the three and not the others would read as if the columns
596/// accounted for everything. They do not, and the gap is the thing worth looking at.
597#[derive(Debug, Clone)]
598pub struct Layout {
599    /// The size of the file on disk.
600    pub file: u64,
601    /// Committed rows.
602    pub rows: usize,
603    /// Committed stripes.
604    pub stripes: usize,
605    /// Committed parts, which is how many chunks a scan reads.
606    pub parts: usize,
607    /// One entry per column, in the table's column order.
608    pub columns: Vec<ColumnLayout>,
609    /// Every stripe's index page, which carries a length and a checksum for every part of every
610    /// column and is charged per stripe rather than per column.
611    pub indexes: u64,
612    /// The committed directory itself, the one that was read to build this.
613    pub directory: u64,
614    /// The fixed header, which holds the magic, the format and the two directory slots.
615    pub header: u64,
616}
617
618impl Layout {
619    /// Everything the columns cost together.
620    #[must_use]
621    pub fn columns_total(&self) -> u64 {
622        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
623    }
624
625    /// What the file holds that this does not account for.
626    ///
627    /// A committed file is written once and never rewritten in place, so an earlier directory and
628    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
629    /// are bytes on disk that no column owns.
630    #[must_use]
631    pub fn unaccounted(&self) -> u64 {
632        self.file
633            .saturating_sub(self.columns_total())
634            .saturating_sub(self.indexes)
635            .saturating_sub(self.directory)
636            .saturating_sub(self.header)
637    }
638}
639
640/// How one part of one column is stored, which is one row of `pragma_storage_info`.
641///
642/// Everything here is read off the file rather than worked out from the schema, because the whole
643/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
644/// holding the same rows in a different order give different answers and that difference is the
645/// reason to ask.
646///
647/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
648/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
649/// of a page that is a quarter of a megabyte.
650#[derive(Debug, Clone)]
651pub struct StoredPart {
652    /// Which stripe the part belongs to.
653    pub stripe: usize,
654    /// Which part of that stripe it is, counting from zero inside the stripe.
655    pub part: usize,
656    /// The table wide row number the part starts at.
657    pub row: usize,
658    /// How many rows it holds.
659    pub rows: usize,
660    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
661    pub encoding: String,
662    /// The stored bytes of the part, which is what it costs in the file.
663    pub bytes: u64,
664    /// Where in the file the column page holding this part starts.
665    pub page: u64,
666    /// Where in that page the part starts.
667    pub offset: u64,
668    /// The smallest value the part holds, when the stored ranges say.
669    pub low: Option<Value>,
670    /// The largest, same.
671    pub high: Option<Value>,
672    /// How many of its rows are null, when the stored ranges say.
673    pub nulls: Option<usize>,
674}
675
676/// Appends pages and commits a new directory for one table.
677#[derive(Debug)]
678struct GlobalDictionary {
679    primary: HashMap<u64, u32>,
680    collisions: HashMap<u64, Vec<u32>>,
681    offsets: Vec<u32>,
682    payload: Vec<u8>,
683    counts: Vec<u64>,
684    nulls: u64,
685}
686
687impl GlobalDictionary {
688    fn new() -> Self {
689        Self {
690            primary: HashMap::new(),
691            collisions: HashMap::new(),
692            offsets: vec![0],
693            payload: Vec::new(),
694            counts: Vec::new(),
695            nulls: 0,
696        }
697    }
698
699    fn bytes(&self, code: u32) -> Option<&[u8]> {
700        let start = *self.offsets.get(code as usize)? as usize;
701        let end = *self.offsets.get(code as usize + 1)? as usize;
702        self.payload.get(start..end)
703    }
704
705    fn code(&mut self, text: &str) -> Result<u32> {
706        let hash = checksum(text.as_bytes());
707        if let Some(&code) = self.primary.get(&hash) {
708            if self.bytes(code) == Some(text.as_bytes()) {
709                return Ok(code);
710            }
711            if let Some(codes) = self.collisions.get(&hash) {
712                if let Some(code) =
713                    codes.iter().copied().find(|&code| self.bytes(code) == Some(text.as_bytes()))
714                {
715                    return Ok(code);
716                }
717            }
718            let code = self.insert(text)?;
719            self.collisions.entry(hash).or_default().push(code);
720            return Ok(code);
721        }
722        let code = self.insert(text)?;
723        self.primary.insert(hash, code);
724        Ok(code)
725    }
726
727    fn insert(&mut self, text: &str) -> Result<u32> {
728        let code = u32::try_from(self.offsets.len() - 1)
729            .map_err(|_| invalid("global dictionary has too many values"))?;
730        self.payload.extend_from_slice(text.as_bytes());
731        self.offsets.push(
732            u32::try_from(self.payload.len())
733                .map_err(|_| invalid("global dictionary payload exceeds 4 GiB"))?,
734        );
735        self.counts.push(0);
736        Ok(code)
737    }
738
739    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
740    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
741    /// are sorted by their bytes.
742    ///
743    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
744    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
745    /// stripe's codes close together because the data is clustered. This is what puts the values
746    /// back in order for anything that needs it, and it is separate from the codes so that getting
747    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
748    ///
749    /// The sort compares the first eight bytes as one integer before it compares the values, which
750    /// settles almost every pair without touching the payload. Padding with zero on the right is
751    /// order preserving for byte strings, because a shorter value differs from a longer one that
752    /// starts the same way at a position where the shorter one has run out, and zero is below every
753    /// byte that could be there. A pair the head cannot settle falls through to the bytes.
754    ///
755    /// The heads are kept rather than thrown away once the sort is over, because a reader searching
756    /// this order wants exactly the same comparison and for exactly the same reason. Eight bytes an
757    /// entry of file is what buys a binary search that reads no values at all in the ordinary case.
758    fn ranked(&self) -> Vec<(u64, u32)> {
759        let count = self.offsets.len() - 1;
760        let mut ranked = (0..count)
761            .map(|code| {
762                let code = code as u32;
763                (head(self.bytes(code).unwrap_or_default()), code)
764            })
765            .collect::<Vec<_>>();
766        ranked.sort_unstable_by(|left, right| {
767            left.0.cmp(&right.0).then_with(|| self.bytes(left.1).cmp(&self.bytes(right.1)))
768        });
769        ranked
770    }
771
772    fn observe(&mut self, code: u32, null: bool) -> Result<()> {
773        if null {
774            self.nulls = self.nulls.saturating_add(1);
775            return Ok(());
776        }
777        let count = self
778            .counts
779            .get_mut(code as usize)
780            .ok_or_else(|| invalid("global dictionary count code is out of range"))?;
781        *count = count.saturating_add(1);
782        Ok(())
783    }
784}
785
786/// Appends pages and commits a new directory.
787///
788/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
789/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
790/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
791/// the end of it and a reader sees every table at the generation before it or every table at the
792/// generation after it.
793#[derive(Debug)]
794pub struct Writer {
795    file: File,
796    /// Where the next write goes, counted here rather than asked of the file.
797    ///
798    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
799    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
800    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
801    /// it read. A writer that asked the file where it was would then write the directory over a
802    /// page it had already written, which is what it did.
803    at: u64,
804    table: Table,
805    generation: u64,
806    /// The first and the last source position in every stripe, in the order the stripes were
807    /// written.
808    order: Vec<((u64, u64), (u64, u64))>,
809    next_order: u64,
810    dictionaries: Vec<Option<GlobalDictionary>>,
811    pending: Vec<PendingChunk>,
812    /// The tables already closed in this generation, in the order they were written.
813    closed: Vec<Entry>,
814    /// The views the next commit writes down, which [`Writer::with_views`] sets.
815    ///
816    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
817    /// opened to append a table does not have to know about views to avoid dropping them.
818    views: Vec<ViewEntry>,
819}
820
821/// A chunk that has arrived and is waiting for the rest of its stripe.
822///
823/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
824/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
825/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
826/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
827/// that share nothing.
828#[derive(Debug)]
829struct PendingChunk {
830    order: (u64, u64),
831    chunk: Chunk,
832}
833
834/// One column's share of a stripe, which is what one encode worker produces.
835///
836/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
837/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
838/// parts next to each other, and it used to reach across a row of parts to do it.
839#[derive(Debug)]
840struct ColumnStripe {
841    pages: Vec<Vec<u8>>,
842    codes: Vec<Option<Vec<u32>>>,
843    sieves: Vec<Option<Sieve>>,
844    ranges: Vec<Range>,
845}
846
847/// Roughly what encoding a column of this type costs, for ordering the encode queue.
848///
849/// Only the order matters and only roughly. A string column hashes and copies every value into a
850/// dictionary and is in a different class from everything else, and among the fixed widths the wide
851/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
852/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
853/// a column nobody else can help with.
854fn weight(ty: &LogicalType) -> usize {
855    match ty {
856        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
857        LogicalType::HugeInt
858        | LogicalType::UHugeInt
859        | LogicalType::Uuid
860        | LogicalType::Interval => 16,
861        LogicalType::BigInt
862        | LogicalType::UBigInt
863        | LogicalType::Timestamp
864        | LogicalType::Time
865        | LogicalType::TimeTz
866        | LogicalType::TimestampTz
867        | LogicalType::TimestampS
868        | LogicalType::TimestampMs
869        | LogicalType::TimestampNs
870        | LogicalType::Double
871        | LogicalType::Decimal { .. } => 8,
872        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
873        LogicalType::SmallInt | LogicalType::USmallInt => 2,
874        _ => 1,
875    }
876}
877
878/// Parts in one stripe.
879///
880/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
881/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
882/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
883/// and cost a sparse fetch, which has to read a page index before it can reach one part.
884pub const STRIPE_PARTS: usize = 64;
885
886/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
887/// its global dictionary.
888///
889/// See [`Writer::encode_column`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
890/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
891/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
892/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
893const DICTIONARY_DECIDE_ROWS: usize = 4_096;
894
895/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
896/// first stripe held a value that stripe had not seen before.
897///
898/// See [`Writer::encode_column`]. Nine and not five, because the properties a dictionary buys are
899/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
900/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
901/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
902/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
903///
904/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
905/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
906/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
907/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
908/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
909/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
910const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
911
912/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
913const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
914
915/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
916fn index_section(parts: usize) -> Result<usize> {
917    parts
918        .checked_mul(INDEX_ENTRY)
919        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
920        .ok_or_else(|| invalid("index page length overflow"))
921}
922
923impl Writer {
924    /// Opens a committed file and starts a table in the generation after the one it holds.
925    ///
926    /// The tables already in the file are carried forward by name and by directory pointer, and
927    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
928    /// new catalog go on the end, past the catalog the committed generation points at, and the one
929    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
930    ///
931    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
932    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
933    /// still reads as the generation before it, and a slot torn across a write fails its checksum
934    /// and the reader falls back to the one beside it. This is what the second slot has always been
935    /// for.
936    ///
937    /// # Errors
938    ///
939    /// If the file has no valid committed directory, is not this build's format, repeats the name
940    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
941    /// written.
942    pub fn open(
943        path: impl AsRef<Path>,
944        name: impl Into<String>,
945        fields: Vec<Field>,
946    ) -> Result<Self> {
947        for field in &fields {
948            type_tag(&field.ty)?;
949        }
950        let name = name.into();
951        let path = path.as_ref();
952        let (_, size, slot, bytes, _) = slot_bytes(path)?;
953        let (mut closed, views) = decode_catalog(&bytes, size)?;
954        // A table already in the file under this name is only in the way if it holds rows. One that
955        // holds none has no pages for this generation to carry and no reader that could lose
956        // anything, so the table being started here takes its place in the catalog rather than
957        // colliding with it, and `finish` writes the new entry where the old one was.
958        //
959        // That is not a corner. It is the shape every loading script writes: the schema goes in one
960        // statement and the rows go in the next, and a checkpoint between them commits the empty
961        // table. Before this, the second statement had to build the whole table in memory because
962        // the first had already put the name in the file, which is how a load of a table larger
963        // than memory became a load that needed memory the size of the table.
964        if let Some(at) = closed.iter().position(|held| held.name == name) {
965            if closed[at].rows > 0 {
966                return Err(invalid("two tables in one native file have the same name"));
967            }
968            closed.remove(at);
969        }
970        // The generation of the slot whose bytes checksummed, and not the highest number in the
971        // header. A slot torn across a write can hold any number at all, and taking that one would
972        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
973        // half written commit gets to destroy the one good copy beside it.
974        let generation = slot
975            .generation
976            .checked_add(1)
977            .ok_or_else(|| invalid("native file generation overflow"))?;
978        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
979        Ok(Self {
980            file,
981            // The end of the file, so that the committed generation's catalog stays where its slot
982            // says it is and keeps naming a file a reader can still open.
983            at: size,
984            dictionaries: fields
985                .iter()
986                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
987                .collect(),
988            table: Table {
989                name,
990                dictionaries: vec![None; fields.len()],
991                distincts: vec![None; fields.len()],
992                fields,
993                stripes: Vec::new(),
994                rows: 0,
995                frequencies: Vec::new(),
996                clustering: None,
997                generation,
998                sections: Vec::new(),
999            },
1000            generation,
1001            order: Vec::new(),
1002            next_order: 0,
1003            pending: Vec::with_capacity(STRIPE_PARTS),
1004            closed,
1005            views,
1006        })
1007    }
1008
1009    /// Creates a new v10 file and its first table.
1010    ///
1011    /// # Errors
1012    ///
1013    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
1014    pub fn create(
1015        path: impl AsRef<Path>,
1016        name: impl Into<String>,
1017        fields: Vec<Field>,
1018    ) -> Result<Self> {
1019        for field in &fields {
1020            type_tag(&field.ty)?;
1021        }
1022        let file =
1023            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1024        let mut header = [0; HEADER as usize];
1025        header[..8].copy_from_slice(MAGIC);
1026        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1027        write_at(&file, 0, &header)?;
1028        Ok(Self {
1029            file,
1030            at: HEADER,
1031            dictionaries: fields
1032                .iter()
1033                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1034                .collect(),
1035            table: Table {
1036                name: name.into(),
1037                dictionaries: vec![None; fields.len()],
1038                distincts: vec![None; fields.len()],
1039                fields,
1040                stripes: Vec::new(),
1041                rows: 0,
1042                frequencies: Vec::new(),
1043                clustering: None,
1044                generation: 1,
1045                sections: Vec::new(),
1046            },
1047            generation: 1,
1048            order: Vec::new(),
1049            next_order: 0,
1050            pending: Vec::with_capacity(STRIPE_PARTS),
1051            closed: Vec::new(),
1052            views: Vec::new(),
1053        })
1054    }
1055
1056    /// Creates a new file that holds no table at all, committed and ready to open.
1057    ///
1058    /// A database somebody dropped the last table out of is still a database, and until this there
1059    /// was no way to write one down. Every other way into this file goes through a table, because
1060    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
1061    /// catalog with nothing in it could be read and not written. The format already allowed it: the
1062    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
1063    /// way every other count does, which is why nothing here is a version change.
1064    ///
1065    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
1066    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
1067    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
1068    /// wrote the same way it reads any other generation.
1069    ///
1070    /// It takes the views anyway, because a database with no table can still have views in it. A
1071    /// view over `range` or over another view names no table, so dropping the last table out of a
1072    /// database does not have to leave the catalog with nothing worth writing down.
1073    ///
1074    /// # Errors
1075    ///
1076    /// If the file exists or the path cannot be written.
1077    pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1078        let file =
1079            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1080        let mut header = [0; HEADER as usize];
1081        header[..8].copy_from_slice(MAGIC);
1082        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1083        write_at(&file, 0, &header)?;
1084        let catalog = encode_catalog(&[], views)?;
1085        write_at(&file, HEADER, &catalog)?;
1086        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
1087        // catalog is on the disk before the slot names it, so a file this is interrupted in the
1088        // middle of is a header with no valid slot rather than a slot pointing at nothing.
1089        file.sync_all().map_err(io)?;
1090        let slot = Slot {
1091            offset: HEADER,
1092            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1093            generation: 1,
1094            hash: checksum(&catalog),
1095        };
1096        write_at(&file, slot_offset(1), &slot.bytes())?;
1097        file.sync_all().map_err(io)?;
1098        Ok(())
1099    }
1100
1101    /// Closes the table this writer is on and starts another one in the same file.
1102    ///
1103    /// Nothing is published here. The closed table's directory is written so that the bytes are on
1104    /// disk and its span is known, and the catalog that names it is only written by
1105    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
1106    ///
1107    /// # Errors
1108    ///
1109    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
1110    /// being closed cannot be written.
1111    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
1112        for field in &fields {
1113            type_tag(&field.ty)?;
1114        }
1115        let name = name.into();
1116        let entry = self.close()?;
1117        if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
1118            return Err(invalid("two tables in one native file have the same name"));
1119        }
1120        let Self { file, at, generation, mut closed, views, .. } = self;
1121        closed.push(entry);
1122        Ok(Self {
1123            file,
1124            at,
1125            generation,
1126            closed,
1127            views,
1128            dictionaries: fields
1129                .iter()
1130                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1131                .collect(),
1132            table: Table {
1133                name,
1134                dictionaries: vec![None; fields.len()],
1135                distincts: vec![None; fields.len()],
1136                fields,
1137                stripes: Vec::new(),
1138                rows: 0,
1139                frequencies: Vec::new(),
1140                clustering: None,
1141                generation,
1142                sections: Vec::new(),
1143            },
1144            order: Vec::new(),
1145            next_order: 0,
1146            pending: Vec::with_capacity(STRIPE_PARTS),
1147        })
1148    }
1149
1150    /// Sets the views the next commit writes down, replacing whatever was carried forward.
1151    ///
1152    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
1153    /// writer does not. A view that was dropped is a view that is not in the list any more, and
1154    /// there is no other way for the writer to hear about that, since nothing else it is told about
1155    /// mentions views at all.
1156    ///
1157    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
1158    /// checkpoint that only had a table to append does not quietly drop them.
1159    #[must_use]
1160    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
1161        self.views = views;
1162        self
1163    }
1164
1165    /// Records the order this table's rows are meant to be stored in.
1166    ///
1167    /// The declaration goes in the table directory and comes back out of
1168    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
1169    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
1170    /// the thing that was missing was a place to write the order down, and a loader that honours
1171    /// the declaration is the next piece rather than this one.
1172    ///
1173    /// The declaration applies to the table the writer is currently on, so it is set after
1174    /// [`Writer::next`] rather than once for the file.
1175    ///
1176    /// # Errors
1177    ///
1178    /// If the declaration names a column this table does not have.
1179    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
1180        // Rebuilt against this table's own column count rather than trusted, because the caller
1181        // built it against a catalog entry and the two could have drifted.
1182        self.table.clustering = Some(Clustering::new(
1183            clustering.columns().to_vec(),
1184            clustering.width(),
1185            &self.table.fields,
1186        )?);
1187        Ok(self)
1188    }
1189
1190    /// Appends bytes at the end of the file and moves the writer's own offset past them.
1191    ///
1192    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
1193    /// anything is and the file's cursor is never consulted for it.
1194    fn put(&mut self, bytes: &[u8]) -> Result<()> {
1195        write_at(&self.file, self.at, bytes)?;
1196        self.at = self
1197            .at
1198            .checked_add(bytes.len() as u64)
1199            .ok_or_else(|| invalid("native file length overflow"))?;
1200        Ok(())
1201    }
1202
1203    /// Writes one chunk as independently readable column pages.
1204    ///
1205    /// # Errors
1206    ///
1207    /// If its width or types differ from the declared table, or a page exceeds its bound.
1208    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
1209        let order = (self.next_order, 0);
1210        self.next_order = self.next_order.saturating_add(1);
1211        self.append_at(order, chunk)
1212    }
1213
1214    /// Writes one chunk and records its source position for directory ordering.
1215    ///
1216    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
1217    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
1218    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
1219    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
1220    ///
1221    /// # Errors
1222    ///
1223    /// The same as [`Self::append`].
1224    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
1225        if chunk.is_empty() {
1226            return Ok(());
1227        }
1228        self.admit(chunk)?;
1229        if self.pending.last().is_some_and(|last| last.order > order) {
1230            self.flush_pending()?;
1231        }
1232        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
1233        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
1234        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
1235        // against the hundreds of seconds of encode this is what lets off one thread.
1236        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
1237        if self.pending.len() == STRIPE_PARTS {
1238            self.flush_pending()?;
1239        }
1240        Ok(())
1241    }
1242
1243    /// Writes a run of chunks as one stripe of its own.
1244    ///
1245    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
1246    /// when one caller hands over every chunk in source order and does not when several do. A
1247    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
1248    /// that ends every time two of them cross is a stripe of one or two parts.
1249    ///
1250    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
1251    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
1252    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
1253    /// so the runs from different callers may interleave with each other but may not overlap.
1254    ///
1255    /// # Errors
1256    ///
1257    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
1258    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
1259        if parts.len() > STRIPE_PARTS {
1260            return Err(invalid("a stripe was handed more parts than it holds"));
1261        }
1262        // Whatever an earlier caller left behind is its own stripe rather than the front of this
1263        // one, because the two runs are from different places in the source and a stripe is a run.
1264        self.flush_pending()?;
1265        for (order, chunk) in parts {
1266            if chunk.is_empty() {
1267                continue;
1268            }
1269            self.admit(&chunk)?;
1270            self.pending.push(PendingChunk { order, chunk });
1271        }
1272        self.flush_pending()
1273    }
1274
1275    /// Checks a chunk against the declared table and counts its rows in.
1276    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
1277        if chunk.width() != self.table.fields.len() {
1278            return Err(invalid("chunk width differs from table schema"));
1279        }
1280        for (index, field) in self.table.fields.iter().enumerate() {
1281            if chunk.column(index)?.logical_type() != &field.ty {
1282                return Err(invalid("chunk type differs from table schema"));
1283            }
1284        }
1285        self.table.rows = self
1286            .table
1287            .rows
1288            .checked_add(chunk.len())
1289            .ok_or_else(|| invalid("row count overflow"))?;
1290        Ok(())
1291    }
1292
1293    /// Encodes one column's parts of a stripe, and on the first stripe decides whether the column
1294    /// should have a dictionary at all.
1295    ///
1296    /// Every varchar column starts with one, because the writer cannot know what is in a column
1297    /// before it has seen some of it. A global dictionary is the right shape for a column of a few
1298    /// dozen values repeated down the table: the pages become small integers, a filter against a
1299    /// literal is one search of the sorted order rather than a comparison a row, and a group by is
1300    /// on the codes. It is the wrong shape for a column whose values are nearly all different.
1301    /// There the codes are as wide as row numbers, nothing is saved on the pages, and the
1302    /// membership index of a stripe is a list of very nearly every code in the column. On TPC-H the
1303    /// orders table written on its own goes from 52.3 MB to 41.4 MB, the load from 6.9 s to 5.8 s,
1304    /// and `select o_comment from orders` from 1.810 G instructions to 1.213 G, which is what the
1305    /// rudb parquet reader takes over the same values.
1306    ///
1307    /// So the first stripe of a column is the sample and the decision is made once on it. Once,
1308    /// rather than per stripe, because the codes of one column have to mean the same thing in every
1309    /// page of it, and a column that changed its mind halfway would need its earlier stripes
1310    /// rewritten. The first stripe is re-encoded when the answer comes out against the dictionary,
1311    /// which is the one stripe that pays for the decision.
1312    ///
1313    /// The threshold is deliberately near the top. [`DICTIONARY_DISTINCT_IN_TEN`] of the sample has
1314    /// to be values never seen before, which is a column with essentially no repeats. Everything
1315    /// with real repetition keeps its dictionary and keeps every property that hangs off it, and
1316    /// nothing is claimed here about where between the two the crossover really sits.
1317    ///
1318    /// Nothing here is shared with another column. The dictionary belongs to this one, the sieve
1319    /// reads only this one, and the page bytes go in a vector of this one's own. That is why the
1320    /// fan out below can hand a whole column to a thread and take a plain `&mut` on the dictionary
1321    /// rather than making it something several threads can grow at once, which is the harder half
1322    /// of #808 and is still open.
1323    fn encode_column(
1324        index: usize,
1325        held: &[PendingChunk],
1326        dictionary: &mut Option<GlobalDictionary>,
1327    ) -> Result<ColumnStripe> {
1328        // Empty means nothing has been written through it yet, so this is the column's first stripe
1329        // and the only stripe the decision below is allowed to be made on.
1330        let deciding = dictionary.as_ref().is_some_and(|held| held.offsets.len() == 1);
1331        let stripe = Self::encode_pages(index, held, dictionary.as_mut())?;
1332        if !deciding {
1333            return Ok(stripe);
1334        }
1335        let rows: usize = held.iter().map(|pending| pending.chunk.len()).sum();
1336        let distinct = dictionary.as_ref().map_or(0, |held| held.offsets.len() - 1);
1337        if rows < DICTIONARY_DECIDE_ROWS
1338            || distinct.saturating_mul(10) <= rows.saturating_mul(DICTIONARY_DISTINCT_IN_TEN)
1339        {
1340            return Ok(stripe);
1341        }
1342        *dictionary = None;
1343        Self::encode_pages(index, held, None)
1344    }
1345
1346    /// One column's parts of a stripe, with whatever dictionary it was given.
1347    fn encode_pages(
1348        index: usize,
1349        held: &[PendingChunk],
1350        mut dictionary: Option<&mut GlobalDictionary>,
1351    ) -> Result<ColumnStripe> {
1352        let mut stripe = ColumnStripe {
1353            pages: Vec::with_capacity(held.len()),
1354            codes: Vec::with_capacity(held.len()),
1355            sieves: Vec::with_capacity(held.len()),
1356            ranges: Vec::with_capacity(held.len()),
1357        };
1358        for pending in held {
1359            let column = pending.chunk.column(index)?;
1360            let (bytes, unique) = encode(column, dictionary.as_deref_mut())?;
1361            if bytes.len() > MAX_PAGE {
1362                return Err(invalid("column page exceeds the configured bound"));
1363            }
1364            // The range is built first because the sieve reads it rather than walking the column a
1365            // second time to find out how wide it is.
1366            let range = Range::of(column);
1367            // A column with a global dictionary already has an exact membership index per stripe,
1368            // so an approximate one beside it would cost a hash of every string in the table to
1369            // answer a question that is already answered. What it would buy is the finer grain, a
1370            // part rather than a stripe, and that is worth coming back for on its own.
1371            //
1372            // A sieve at least as large as the part it indexes is not written. A reader reads the
1373            // sieve to decide whether to read the part, so when the sieve is the larger of the two
1374            // it has already spent more than the read it is trying to avoid, and that holds even if
1375            // it rejects every time. It is a necessary condition rather than the whole rule, which
1376            // is that a sieve pays when its bytes are under the rejection rate times the part's,
1377            // but the rejection rate depends on what a query probes for and the writer does not
1378            // know that. The necessary half needs two numbers that are both in hand here.
1379            let sieve = match dictionary {
1380                Some(_) => None,
1381                None => Sieve::of(column, &range, SIEVE_BUDGET)
1382                    .filter(|sieve| sieve.len() < bytes.len()),
1383            };
1384            stripe.pages.push(bytes);
1385            stripe.codes.push(unique);
1386            stripe.sieves.push(sieve);
1387            stripe.ranges.push(range);
1388        }
1389        Ok(stripe)
1390    }
1391
1392    /// Encodes a whole stripe, one column to a worker.
1393    ///
1394    /// The columns are handed out through a queue rather than dealt in equal piles, because they
1395    /// are nothing like equal: `URL` on ClickBench is a global dictionary of sixty one million
1396    /// strings and `IsMobile` is a byte. A pile that happened to hold the four large string columns
1397    /// would be the whole stripe and the other workers would be waiting on it. The queue is sorted
1398    /// so the expensive ones are taken first, which is the classic answer to a last job that runs
1399    /// longer than everything after it.
1400    fn encode_columns(&mut self, held: &[PendingChunk]) -> Result<Vec<ColumnStripe>> {
1401        let width = self.table.fields.len();
1402        let workers = std::thread::available_parallelism()
1403            .map_or(1, usize::from)
1404            .min(MAX_ENCODE_WORKERS)
1405            .min(width);
1406        if workers <= 1 || held.len() <= 1 {
1407            return self
1408                .dictionaries
1409                .iter_mut()
1410                .enumerate()
1411                .map(|(index, dictionary)| Self::encode_column(index, held, dictionary))
1412                .collect();
1413        }
1414        // The dictionaries are moved out and back rather than borrowed, because a worker that takes
1415        // the next column off a queue cannot be holding a borrow of the vector the queue came from.
1416        let mut jobs: Vec<(usize, Option<GlobalDictionary>)> =
1417            std::mem::take(&mut self.dictionaries).into_iter().enumerate().collect();
1418        // Popped from the back, so the expensive columns go last in the vector.
1419        jobs.sort_by_key(|(index, _)| weight(&self.table.fields[*index].ty));
1420        let queue = Mutex::new(jobs);
1421        let pieces = std::thread::scope(|scope| {
1422            (0..workers)
1423                .map(|_| {
1424                    scope.spawn(|| {
1425                        let mut mine = Vec::new();
1426                        loop {
1427                            let taken = queue
1428                                .lock()
1429                                .map_err(|_| Error::internal("a native encode worker panicked"))?
1430                                .pop();
1431                            let Some((index, mut dictionary)) = taken else { break };
1432                            let encoded = Self::encode_column(index, held, &mut dictionary)?;
1433                            mine.push((index, dictionary, encoded));
1434                        }
1435                        Ok(mine)
1436                    })
1437                })
1438                .collect::<Vec<_>>()
1439                .into_iter()
1440                .map(|handle| {
1441                    handle.join().map_err(|_| Error::internal("a native encode worker panicked"))?
1442                })
1443                .collect::<Result<Vec<_>>>()
1444        })?;
1445        let mut dictionaries: Vec<Option<GlobalDictionary>> = (0..width).map(|_| None).collect();
1446        let mut encoded: Vec<Option<ColumnStripe>> = (0..width).map(|_| None).collect();
1447        for piece in pieces {
1448            for (index, dictionary, stripe) in piece {
1449                dictionaries[index] = dictionary;
1450                encoded[index] = Some(stripe);
1451            }
1452        }
1453        self.dictionaries = dictionaries;
1454        encoded
1455            .into_iter()
1456            .map(|stripe| stripe.ok_or_else(|| Error::internal("a column was never encoded")))
1457            .collect()
1458    }
1459
1460    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
1461    fn flush_pending(&mut self) -> Result<()> {
1462        if self.pending.is_empty() {
1463            return Ok(());
1464        }
1465        let width = self.table.fields.len();
1466        // Held here rather than read off the writer, because writing a page needs the writer and
1467        // the borrow checker is right that those are two different uses of it.
1468        let mut held = std::mem::take(&mut self.pending);
1469        let parts = held.len();
1470        let encoded = self.encode_columns(&held)?;
1471        let mut pages = Vec::with_capacity(width);
1472        let mut memberships = vec![None; width];
1473        let mut ranges = Vec::with_capacity(width);
1474        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
1475        for stripe in &encoded {
1476            let offset = self.at;
1477            let section = index.len();
1478            let mut length = 0_usize;
1479            for bytes in &stripe.pages {
1480                write_at(&self.file, self.at + length as u64, bytes)?;
1481                put_u32(
1482                    &mut index,
1483                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
1484                );
1485                put_u64(&mut index, checksum(bytes));
1486                length = length
1487                    .checked_add(bytes.len())
1488                    .ok_or_else(|| invalid("column page length overflow"))?;
1489            }
1490            let hash = checksum(&index[section..]);
1491            put_u64(&mut index, hash);
1492            if length > MAX_PAGE {
1493                return Err(invalid("column page exceeds the configured bound"));
1494            }
1495            self.at = self
1496                .at
1497                .checked_add(length as u64)
1498                .ok_or_else(|| invalid("native file length overflow"))?;
1499            pages.push(Span {
1500                offset,
1501                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
1502            });
1503            ranges.push(merged_range(stripe.ranges.iter().cloned()));
1504        }
1505        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
1506            if stripe.codes.iter().all(Option::is_none) {
1507                continue;
1508            }
1509            let lists = stripe
1510                .codes
1511                .iter()
1512                .map(|codes| codes.clone().unwrap_or_default())
1513                .collect::<Vec<_>>();
1514            let bytes = encode_membership(&merged_codes(lists));
1515            let offset = self.at;
1516            self.put(&bytes)?;
1517            *membership = Some(Page {
1518                offset,
1519                length: u32::try_from(bytes.len())
1520                    .map_err(|_| invalid("membership page length overflow"))?,
1521                hash: checksum(&bytes),
1522            });
1523        }
1524        let mut sieves = vec![None; width];
1525        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
1526            if stripe.sieves.iter().all(Option::is_none) {
1527                continue;
1528            }
1529            let bytes = encode_sieves(stripe.sieves.iter())?;
1530            let offset = self.at;
1531            self.put(&bytes)?;
1532            *page = Some(Page {
1533                offset,
1534                length: u32::try_from(bytes.len())
1535                    .map_err(|_| invalid("sieve page length overflow"))?,
1536                hash: checksum(&bytes),
1537            });
1538        }
1539        // A stripe of one part has the same rows in it as that part, so its own bounds are already
1540        // the part's and a page here would say what the directory says. Everywhere else the page is
1541        // written unless it comes to more than the column it indexes, which is the rule the sieves
1542        // go by and for the same reason: a reader reads this to decide whether to read the column,
1543        // so a page larger than the column has spent more than the read it is avoiding.
1544        let mut part_ranges = vec![None; width];
1545        if parts > 1 {
1546            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
1547                let bytes = encode_part_ranges(&stripe.ranges)?;
1548                if bytes.len() >= span.length as usize {
1549                    continue;
1550                }
1551                let offset = self.at;
1552                self.put(&bytes)?;
1553                *page = Some(Page {
1554                    offset,
1555                    length: u32::try_from(bytes.len())
1556                        .map_err(|_| invalid("part range page length overflow"))?,
1557                    hash: checksum(&bytes),
1558                });
1559            }
1560        }
1561        let offset = self.at;
1562        self.put(&index)?;
1563        let index = Span {
1564            offset,
1565            length: u32::try_from(index.len())
1566                .map_err(|_| invalid("index page length overflow"))?,
1567        };
1568        let mut rows = 0_usize;
1569        let mut lengths = Vec::with_capacity(parts);
1570        let mut span = None;
1571        for pending in held.drain(..) {
1572            let part = pending.chunk.len();
1573            rows = rows.checked_add(part).ok_or_else(|| invalid("row count overflow"))?;
1574            lengths.push(u32::try_from(part).map_err(|_| invalid("part row count overflow"))?);
1575            span = Some(
1576                span.map_or((pending.order, pending.order), |(first, _)| (first, pending.order)),
1577            );
1578        }
1579        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
1580        self.table.stripes.push(Stripe {
1581            rows,
1582            parts: lengths,
1583            index,
1584            pages,
1585            memberships,
1586            sieves,
1587            part_ranges,
1588            zone: Zone::from_ranges(ranges),
1589        });
1590        // Back where it came from, empty, so the next stripe buffers into the same allocation.
1591        self.pending = held;
1592        Ok(())
1593    }
1594
1595    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
1596    /// load is live. The pages are already in the target file, so one column at a time uses a
1597    /// bounded Misra-Gries candidate table and then recounts only those candidates.
1598    fn numeric_frequency(&self, column: usize) -> Result<Option<FrequencySummary>> {
1599        let ty = &self.table.fields[column].ty;
1600        if !matches!(
1601            ty,
1602            LogicalType::TinyInt
1603                | LogicalType::SmallInt
1604                | LogicalType::Integer
1605                | LogicalType::BigInt
1606                | LogicalType::UTinyInt
1607                | LogicalType::USmallInt
1608                | LogicalType::UInteger
1609                | LogicalType::UBigInt
1610                | LogicalType::Date
1611                | LogicalType::Timestamp
1612        ) {
1613            return Ok(None);
1614        }
1615        let mut candidates: HashMap<FrequencyValue, u32> = HashMap::new();
1616        let mut decrements = 0_u64;
1617        self.visit_numeric(column, |_, value| {
1618            if let Some(count) = candidates.get_mut(&value) {
1619                *count = count.saturating_add(1);
1620            } else if candidates.len() < FREQUENCY_CANDIDATES {
1621                candidates.insert(value, 1);
1622            } else {
1623                candidates.retain(|_, count| {
1624                    *count -= 1;
1625                    *count != 0
1626                });
1627                decrements = decrements.saturating_add(1);
1628            }
1629        })?;
1630        let (exact, ordinals) = if decrements == 0 {
1631            (
1632                candidates
1633                    .into_iter()
1634                    .map(|(value, count)| (value, u64::from(count)))
1635                    .collect::<HashMap<_, _>>(),
1636                Vec::new(),
1637            )
1638        } else {
1639            let mut lower = candidates.values().copied().collect::<Vec<_>>();
1640            lower.sort_unstable_by(|left, right| right.cmp(left));
1641            if lower.len() < FREQUENCY_BUILD_RANK
1642                || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
1643            {
1644                return Ok(None);
1645            }
1646            let mut exact =
1647                candidates.into_keys().map(|value| (value, 0_u64)).collect::<HashMap<_, _>>();
1648            let mut ordinals = Vec::new();
1649            let mut exceeded = false;
1650            self.visit_numeric(column, |ordinal, value| {
1651                if let Some(count) = exact.get_mut(&value) {
1652                    *count = count.saturating_add(1);
1653                    if !exceeded {
1654                        if ordinals.len() < FREQUENCY_ORDINALS {
1655                            ordinals.push(ordinal);
1656                        } else {
1657                            ordinals.clear();
1658                            exceeded = true;
1659                        }
1660                    }
1661                }
1662            })?;
1663            (exact, ordinals)
1664        };
1665        let mut entries = exact
1666            .into_iter()
1667            .map(|(value, count)| FrequencyEntry { value, count })
1668            .collect::<Vec<_>>();
1669        entries.sort_unstable_by(|left, right| {
1670            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
1671        });
1672        let omitted_max =
1673            entries.get(FREQUENCY_ENTRIES).map_or(decrements, |entry| decrements.max(entry.count));
1674        entries.truncate(FREQUENCY_ENTRIES);
1675        Ok(Some(FrequencySummary { entries, omitted_max, ordinals }))
1676    }
1677
1678    fn visit_numeric(
1679        &self,
1680        column: usize,
1681        mut visit: impl FnMut(u64, FrequencyValue),
1682    ) -> Result<()> {
1683        let ty = &self.table.fields[column].ty;
1684        let mut start = 0_u64;
1685        for stripe in &self.table.stripes {
1686            let spans = read_index(&self.file, stripe, column)?;
1687            let page = stripe.pages[column];
1688            let mut bytes = vec![0; page.length as usize];
1689            read_at(&self.file, page.offset, &mut bytes)?;
1690            for (span, &rows) in spans.iter().zip(&stripe.parts) {
1691                let part = part_bytes(&bytes, *span)?;
1692                if checksum(part) != span.hash {
1693                    return Err(invalid("column page checksum differs while building frequencies"));
1694                }
1695                let rows = rows as usize;
1696                let vector = decode(ty, rows, part, None)?;
1697                // row at a time: frequency construction visits decoded values to update bounded candidates.
1698                for row in 0..rows {
1699                    let value = if vector.is_null_at(row) {
1700                        FrequencyValue::Null
1701                    } else {
1702                        // An unsigned column has no signed reading, and the documented fallback is
1703                        // the value itself. Every unsigned width the format stores fits in the
1704                        // `i128` a candidate is keyed by, so nothing is lost on the way through.
1705                        let widened = match vector.signed_at(row) {
1706                            Some(value) => Some(value),
1707                            None => match vector.value_at(row) {
1708                                Value::UTinyInt(value) => Some(i128::from(value)),
1709                                Value::USmallInt(value) => Some(i128::from(value)),
1710                                Value::UInteger(value) => Some(i128::from(value)),
1711                                Value::UBigInt(value) => Some(i128::from(value)),
1712                                _ => None,
1713                            },
1714                        };
1715                        FrequencyValue::Integer(widened.ok_or_else(|| {
1716                            invalid("numeric frequency page did not contain an integer value")
1717                        })?)
1718                    };
1719                    visit(start.saturating_add(row as u64), value);
1720                }
1721                start = start.saturating_add(rows as u64);
1722            }
1723        }
1724        Ok(())
1725    }
1726
1727    /// Builds independent numeric synopses concurrently after all column pages are committed.
1728    ///
1729    /// The columns go through a queue rather than being cut into equal runs, because they are not
1730    /// equally expensive and they are not shuffled. A `BIGINT` column carries eight times the bytes
1731    /// of a `TINYINT` through the decode, and a run of them sits together in a schema the way it
1732    /// sits together in `hits`, so a worker that was handed the wrong six columns finishes long
1733    /// after one that was handed the right six and the whole phase waits for it.
1734    fn numeric_frequencies(&self) -> Result<Vec<Option<FrequencySummary>>> {
1735        let mut columns = self
1736            .table
1737            .fields
1738            .iter()
1739            .enumerate()
1740            .filter_map(|(column, field)| {
1741                matches!(
1742                    field.ty,
1743                    LogicalType::TinyInt
1744                        | LogicalType::SmallInt
1745                        | LogicalType::Integer
1746                        | LogicalType::BigInt
1747                        | LogicalType::UTinyInt
1748                        | LogicalType::USmallInt
1749                        | LogicalType::UInteger
1750                        | LogicalType::UBigInt
1751                        | LogicalType::Date
1752                        | LogicalType::Timestamp
1753                )
1754                .then_some(column)
1755            })
1756            .collect::<Vec<_>>();
1757        let workers = std::thread::available_parallelism()
1758            .map_or(1, usize::from)
1759            .min(MAX_FREQUENCY_WORKERS)
1760            .min(columns.len());
1761        if workers <= 1 {
1762            let mut frequencies = vec![None; self.table.fields.len()];
1763            for column in columns {
1764                frequencies[column] = self.numeric_frequency(column)?;
1765            }
1766            return Ok(frequencies);
1767        }
1768        // Popped from the back, so the expensive columns are the ones taken first and the cheap ones
1769        // are what is left to fill in behind them.
1770        columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
1771        let queue = Mutex::new(columns);
1772        let pieces = std::thread::scope(|scope| {
1773            (0..workers)
1774                .map(|_| {
1775                    scope.spawn(|| {
1776                        let mut mine = Vec::new();
1777                        loop {
1778                            let taken = queue
1779                                .lock()
1780                                .map_err(|_| Error::internal("a native frequency worker panicked"))?
1781                                .pop();
1782                            let Some(column) = taken else { break };
1783                            mine.push((column, self.numeric_frequency(column)?));
1784                        }
1785                        Ok(mine)
1786                    })
1787                })
1788                .collect::<Vec<_>>()
1789                .into_iter()
1790                .map(|handle| {
1791                    handle
1792                        .join()
1793                        .map_err(|_| Error::internal("a native frequency worker panicked"))?
1794                })
1795                .collect::<Result<Vec<_>>>()
1796        })?;
1797        let mut frequencies = vec![None; self.table.fields.len()];
1798        for piece in pieces {
1799            for (column, summary) in piece {
1800                frequencies[column] = summary;
1801            }
1802        }
1803        Ok(frequencies)
1804    }
1805
1806    /// Writes the directory of the table this writer is on and says where it went.
1807    ///
1808    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
1809    /// is what lets a second table follow a first: the bytes of a closed table are complete and
1810    /// addressable while nothing yet points at them, and the pointer is the last write of the
1811    /// commit.
1812    ///
1813    /// # Errors
1814    ///
1815    /// If directory encoding or writing fails.
1816    fn close(&mut self) -> Result<Entry> {
1817        self.flush_pending()?;
1818        let mut stripes = std::mem::take(&mut self.order)
1819            .into_iter()
1820            .zip(std::mem::take(&mut self.table.stripes))
1821            .collect::<Vec<_>>();
1822        stripes.sort_by_key(|(order, _)| order.0);
1823        let mut previous: Option<(u64, u64)> = None;
1824        for ((first, last), _) in &stripes {
1825            if previous.is_some_and(|previous| previous >= *first) {
1826                return Err(invalid("chunks did not arrive in source order"));
1827            }
1828            previous = Some(*last);
1829        }
1830        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
1831        self.table.frequencies = self.numeric_frequencies()?;
1832        let dictionaries = std::mem::take(&mut self.dictionaries);
1833        let orders = rankings(&dictionaries)?;
1834        for (index, (dictionary, order)) in dictionaries.into_iter().zip(orders).enumerate() {
1835            let Some(dictionary) = dictionary else { continue };
1836            // A code nothing counted is a code no non-null row of this column holds, which is the
1837            // empty string a null was written as and nothing else, because a code is only ever made
1838            // by a row asking for one.
1839            self.table.distincts[index] =
1840                Some(dictionary.counts.iter().filter(|count| **count != 0).count() as u64);
1841            self.table.frequencies[index] = Some(code_frequency(&dictionary));
1842            let encoded = encode_global_dictionary(dictionary, &order)?;
1843            let offset = self.at;
1844            self.put(&encoded.index)?;
1845            self.put(&encoded.ranks)?;
1846            for block in &encoded.payload {
1847                self.put(block)?;
1848            }
1849            let payload_len =
1850                encoded.payload.iter().try_fold(0_usize, |len, block| len.checked_add(block.len()));
1851            let length = payload_len
1852                .and_then(|len| len.checked_add(encoded.index.len()))
1853                .and_then(|len| len.checked_add(encoded.ranks.len()))
1854                .ok_or_else(|| invalid("dictionary page length overflow"))?;
1855            self.table.dictionaries[index] = Some(Page {
1856                offset,
1857                length: u32::try_from(length)
1858                    .map_err(|_| invalid("dictionary page length overflow"))?,
1859                hash: checksum(&encoded.index),
1860            });
1861        }
1862        let directory = encode_directory(&self.table)?;
1863        if directory.len() > MAX_DIRECTORY {
1864            return Err(invalid("directory exceeds the configured bound"));
1865        }
1866        let offset = self.at;
1867        self.put(&directory)?;
1868        Ok(Entry {
1869            name: self.table.name.clone(),
1870            fields: self.table.fields.clone(),
1871            rows: self.table.rows,
1872            directory: Page {
1873                offset,
1874                length: u32::try_from(directory.len())
1875                    .map_err(|_| invalid("directory length overflow"))?,
1876                hash: checksum(&directory),
1877            },
1878        })
1879    }
1880
1881    /// Commits every table this writer has written and syncs the file before publishing its header
1882    /// slot.
1883    ///
1884    /// The table handed back is the one the writer was on, which is the last of them. Callers that
1885    /// wrote several already know the others, since they named them.
1886    ///
1887    /// # Errors
1888    ///
1889    /// If directory encoding, writing, or syncing fails.
1890    pub fn finish(mut self) -> Result<Table> {
1891        let entry = self.close()?;
1892        let mut tables = std::mem::take(&mut self.closed);
1893        tables.push(entry);
1894        let catalog = encode_catalog(&tables, &self.views)?;
1895        if catalog.len() > MAX_DIRECTORY {
1896            return Err(invalid("catalog exceeds the configured bound"));
1897        }
1898        let offset = self.at;
1899        self.put(&catalog)?;
1900        // Every page and every table directory is on the disk before anything points at them. The
1901        // slot write below is what makes this generation the one a reader picks, so the order of
1902        // these two syncs is the whole of the commit.
1903        self.file.sync_all().map_err(io)?;
1904        let slot = Slot {
1905            offset,
1906            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1907            generation: self.generation,
1908            hash: checksum(&catalog),
1909        };
1910        // The one write that is not an append, and the last one. It goes back over the slot in the
1911        // header, so it names its offset rather than going through `put`, and `at` does not move.
1912        // Which of the two slots it is alternates with the generation, so the one naming the
1913        // generation before this is still intact and still valid until this write lands.
1914        write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
1915        self.file.sync_all().map_err(io)?;
1916        Ok(self.table)
1917    }
1918
1919    /// Commits a generation that changes the views and leaves every table exactly where it is.
1920    ///
1921    /// There was no way to do this before views existed, because everything that could change the
1922    /// catalog also wrote a table, so the only way to say something new about a file was to go
1923    /// through a table. A view is the first thing that can change on its own. Without this, adding
1924    /// a view to a database with eight tables in it would rewrite all eight, since the append path
1925    /// needs a table to append and the fallback is the whole file.
1926    ///
1927    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
1928    /// entries are carried forward by directory pointer the way an append carries them, the new
1929    /// catalog goes on the end, and the slot write at the end is what publishes it.
1930    ///
1931    /// # Errors
1932    ///
1933    /// If the file has no valid committed directory, is not this build's format, or cannot be
1934    /// written.
1935    pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1936        let path = path.as_ref();
1937        let (_, size, slot, bytes, _) = slot_bytes(path)?;
1938        let (closed, _) = decode_catalog(&bytes, size)?;
1939        let generation = slot
1940            .generation
1941            .checked_add(1)
1942            .ok_or_else(|| invalid("native file generation overflow"))?;
1943        let catalog = encode_catalog(&closed, views)?;
1944        if catalog.len() > MAX_DIRECTORY {
1945            return Err(invalid("catalog exceeds the configured bound"));
1946        }
1947        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1948        write_at(&file, size, &catalog)?;
1949        file.sync_all().map_err(io)?;
1950        let slot = Slot {
1951            offset: size,
1952            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1953            generation,
1954            hash: checksum(&catalog),
1955        };
1956        write_at(&file, slot_offset(generation), &slot.bytes())?;
1957        file.sync_all().map_err(io)?;
1958        Ok(())
1959    }
1960}
1961
1962/// Appends one run of bytes at `at` and moves it past them, answering where they went.
1963///
1964/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
1965/// table. Every byte a section costs goes through here, so the offsets in an extent table come
1966/// from one place.
1967fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
1968    let offset = *at;
1969    write_at(file, offset, bytes)?;
1970    *at =
1971        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
1972    Ok(offset)
1973}
1974
1975/// Writes one attachment's payload as extents and returns the entry that names it.
1976///
1977/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
1978/// whose extents should break on a row boundary instead will want to hand its extents over already
1979/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
1980fn write_section(
1981    file: &File,
1982    at: &mut u64,
1983    one: &section::Attachment<'_>,
1984    generation: u64,
1985) -> Result<Section> {
1986    if one.header_bytes as usize > one.bytes.len() {
1987        return Err(invalid("a section's header is longer than its payload"));
1988    }
1989    let mut extents = Vec::new();
1990    let mut first = 0_u64;
1991    for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
1992        let offset = append(file, at, chunk)?;
1993        extents.push(section::Extent {
1994            offset,
1995            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
1996            hash: checksum(chunk),
1997            first,
1998        });
1999        first += chunk.len() as u64;
2000    }
2001    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
2002    section::encode_extents(&extents, &mut table)?;
2003    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
2004    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
2005    // relationship that did not fit the budget is recorded as not built rather than forgotten.
2006    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
2007    Ok(Section {
2008        kind: one.kind,
2009        id: one.id,
2010        generation,
2011        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
2012        extent_page,
2013        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
2014        hash: checksum(&table),
2015        flags: one.flags,
2016        header_bytes: one.header_bytes,
2017    })
2018}
2019
2020/// Attaches graph sections to a table already committed in a file, without rewriting a page.
2021///
2022/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
2023/// exist before the link that uses it can be built, and it is built by reading the key column back,
2024/// so the structures of a table cannot be written during the load that wrote the table. They are
2025/// written afterwards, by this, and the file in between the two is a correct file that answers
2026/// every query more slowly.
2027///
2028/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
2029/// the new catalog all go on the end of the file past the committed generation, and the last write
2030/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
2031/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
2032/// writes past.
2033///
2034/// An attachment replaces any section of the same kind and id, and every other section is carried
2035/// through untouched, including one whose kind this build does not know. The table's own generation
2036/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
2037///
2038/// # Errors
2039///
2040/// If the file has no valid committed directory, is an older format than this build writes, holds
2041/// no table of that name, names a section whose payload cannot be written, or would end up naming
2042/// more sections than the format allows.
2043pub fn attach(
2044    path: impl AsRef<Path>,
2045    table: &str,
2046    attachments: &[section::Attachment<'_>],
2047) -> Result<Table> {
2048    let path = path.as_ref();
2049    let (_, size, slot, bytes, _) = slot_bytes(path)?;
2050    let (mut entries, views) = decode_catalog(&bytes, size)?;
2051    let at = entries
2052        .iter()
2053        .position(|entry| entry.name == table)
2054        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
2055    let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
2056    let mut version = [0; 4];
2057    read_at(&file, 8, &mut version)?;
2058    let version = u32::from_le_bytes(version);
2059    // Readable is not the same as writable. A format 22 file has no section table, and giving its
2060    // directory one without moving the number in its header would leave a file that claims to be
2061    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
2062    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
2063    // just make.
2064    if version != FORMAT {
2065        return Err(invalid(&format!(
2066            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
2067             to be written again"
2068        )));
2069    }
2070    let mut directory = vec![0; entries[at].directory.length as usize];
2071    read_at(&file, entries[at].directory.offset, &mut directory)?;
2072    if checksum(&directory) != entries[at].directory.hash {
2073        return Err(invalid(&format!("the directory of table {table} does not checksum")));
2074    }
2075    let mut held = decode_directory(&directory, size)?;
2076    let mut cursor = size;
2077    for one in attachments {
2078        let written = write_section(&file, &mut cursor, one, held.generation)?;
2079        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
2080        held.sections.push(written);
2081    }
2082    if held.sections.len() > MAX_SECTIONS {
2083        return Err(invalid("the table would name more sections than the bound allows"));
2084    }
2085    let encoded = encode_directory(&held)?;
2086    if encoded.len() > MAX_DIRECTORY {
2087        return Err(invalid("directory exceeds the configured bound"));
2088    }
2089    let offset = append(&file, &mut cursor, &encoded)?;
2090    entries[at].directory = Page {
2091        offset,
2092        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
2093        hash: checksum(&encoded),
2094    };
2095    // The views the file already had, written back unchanged. Attaching a section to a table says
2096    // nothing about a view and must not drop one.
2097    let catalog = encode_catalog(&entries, &views)?;
2098    if catalog.len() > MAX_DIRECTORY {
2099        return Err(invalid("catalog exceeds the configured bound"));
2100    }
2101    let offset = append(&file, &mut cursor, &catalog)?;
2102    file.sync_all().map_err(io)?;
2103    let generation =
2104        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
2105    let committed = Slot {
2106        offset,
2107        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2108        generation,
2109        hash: checksum(&catalog),
2110    };
2111    write_at(&file, slot_offset(generation), &committed.bytes())?;
2112    file.sync_all().map_err(io)?;
2113    Ok(held)
2114}
2115
2116/// Reads committed native column pages without holding the table in memory.
2117#[derive(Debug, Clone)]
2118pub struct Reader {
2119    file: Arc<File>,
2120    table: Arc<Table>,
2121    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
2122    /// Held while a global dictionary is being opened, one per column.
2123    ///
2124    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
2125    /// already has it needs answered and is free. It does not say whether one is being opened, and
2126    /// the difference matters because every worker of a scan wants the same dictionary at the same
2127    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
2128    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
2129    /// entries, and was paying for it twice.
2130    loading: Arc<Vec<Mutex<()>>>,
2131    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
2132    /// dictionary once however many workers it has, and the test that says so is the only thing
2133    /// keeping it that way.
2134    opened: Arc<AtomicUsize>,
2135    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
2136    /// first time a probe asks about them. A query filters on one or two columns and never looks at
2137    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
2138    sieves: Arc<Vec<Vec<SieveSlot>>>,
2139    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
2140    /// first time something compares that column and kept after that.
2141    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
2142    /// Which stripe and which part of it every part of the table is, by table wide part number.
2143    places: Arc<Vec<Place>>,
2144    cache: Arc<Vec<Mutex<Cached>>>,
2145    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
2146    /// scan of a column should read each of its stripes once however many workers it has.
2147    pages: Arc<AtomicUsize>,
2148    /// How many index sections have been read. A scan of a column should read each of its stripes
2149    /// once here too, and the test that says so is the only thing keeping it that way.
2150    indexes: Arc<AtomicUsize>,
2151    /// How many stripes of one column the page cache keeps. See [`CACHED_STRIPES_PER_COLUMN`] for
2152    /// what sets it and [`Reader::keep_stripes`] for who raises it.
2153    kept: Arc<AtomicUsize>,
2154    /// The file's size when it was opened, for [`Reader::layout`].
2155    size: u64,
2156    /// The committed directory's size, for [`Reader::layout`].
2157    directory: u64,
2158    /// What opening the file cost, which is a number rather than a claim.
2159    opening: Opening,
2160}
2161
2162/// What [`Reader::open`] read before it returned.
2163///
2164/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
2165/// and nothing else, and once that document's statistics are in the file the tempting change is to
2166/// load a column summary or two on the way past, because they are small and the next query will
2167/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
2168/// embedded database is opened by processes that are about to run one trivial query.
2169///
2170/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
2171/// independent of how many rows the file holds, and the test that says so is what stops the
2172/// tempting change from landing quietly.
2173#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2174pub struct Opening {
2175    /// How many times the file was read. The header, then each directory slot that looked valid
2176    /// enough to check, so three at the most.
2177    pub reads: u32,
2178    /// How many bytes those reads asked for.
2179    pub bytes: u64,
2180}
2181
2182/// What a reader has read, while it was being opened and since.
2183#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2184pub struct Reads {
2185    /// What opening cost, before any query had been planned.
2186    pub opening: Opening,
2187    /// Whole stripe pages read since.
2188    pub pages: usize,
2189    /// Index sections read since.
2190    pub indexes: usize,
2191    /// Global dictionaries opened since. One per dictionary column that a query touched, however
2192    /// many workers touched it, which is a claim only a test can keep true.
2193    pub dictionaries: usize,
2194}
2195
2196/// Where one table wide part number lands.
2197#[derive(Debug, Clone, Copy)]
2198struct Place {
2199    stripe: u32,
2200    part: u32,
2201    rows: u32,
2202}
2203
2204/// One part's bytes inside one column page.
2205#[derive(Debug, Clone, Copy)]
2206struct PartSpan {
2207    start: usize,
2208    length: usize,
2209    hash: u64,
2210}
2211
2212/// What a reader holds for one stripe of one column.
2213///
2214/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
2215/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
2216/// four thousand would be reading sixty four times what it uses.
2217#[derive(Debug, Clone)]
2218struct CachedColumn {
2219    stripe: usize,
2220    index: Arc<Vec<PartSpan>>,
2221    page: Option<Arc<Vec<u8>>>,
2222}
2223
2224/// One column's stripes a reader holds, and which of them somebody is reading right now.
2225///
2226/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
2227/// finding a page is an index and not a walk. That matters because the walk happened under the
2228/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
2229/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
2230/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
2231/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
2232/// first, because that is the one thing the slots cannot say by themselves.
2233///
2234/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
2235/// a set because it holds at most one stripe per worker on the column and is walked far less often
2236/// than a hash of it would be built.
2237///
2238/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
2239/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
2240/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
2241/// stripe after its page had been evicted read the index again with it, which on the full
2242/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
2243#[derive(Debug, Default)]
2244struct Cached {
2245    pages: Vec<Option<Arc<Vec<u8>>>>,
2246    order: VecDeque<usize>,
2247    loading: Vec<usize>,
2248    index: Vec<Option<Arc<Vec<PartSpan>>>>,
2249}
2250
2251/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
2252///
2253/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
2254/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
2255/// needs, because then every worker is within a few parts of every other and at most a couple of
2256/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
2257/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
2258/// than paying for sixteen slots on every table that is read one part at a time.
2259///
2260/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
2261/// the number of columns a query touches.
2262const CACHED_STRIPES_PER_COLUMN: usize = 4;
2263
2264/// The sieves of one stripe of one column, once somebody has asked for them.
2265type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
2266
2267type RangeSlot = OnceLock<Arc<Vec<Range>>>;
2268
2269#[derive(Debug)]
2270struct NativeText {
2271    file: Arc<File>,
2272    /// How many values the dictionary holds.
2273    values: usize,
2274    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
2275    /// [`TEXT_OFFSET_RUN`].
2276    ///
2277    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
2278    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
2279    /// starts at zero by construction. Relative to the block rather than to the payload, because a
2280    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
2281    /// would have to subtract a base from anyway.
2282    offsets: Vec<u8>,
2283    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
2284    /// same for every block of it.
2285    offset_bits: usize,
2286    /// How many entries the sorted order has, which is the value count.
2287    ranks: usize,
2288    /// Where the sorted order starts in the file. It is read a block at a time and only when
2289    /// something searches it, so a query that never compares this column against a literal never
2290    /// touches it at all.
2291    rank_at: u64,
2292    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
2293    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
2294    /// arithmetic on the block number.
2295    rank_ends: Vec<u64>,
2296    rank_hashes: Vec<u64>,
2297    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2298    /// Bits one code is packed at, which is what the value count needs and is the same for every
2299    /// block of the column.
2300    code_bits: usize,
2301    /// The sorted order turned round, built the first time a reader asks for it.
2302    ///
2303    /// Four bytes per value against the four the offsets already hold, so a column that has this is
2304    /// carrying half again what it carried before rather than something of a new order. It is built
2305    /// only when something asks, which is a grouped min or max over this column and nothing else,
2306    /// and that reader was going to read the payload of this column once per row otherwise.
2307    code_ranks: OnceLock<Option<Vec<u32>>>,
2308    payload: u64,
2309    /// Where each block of the payload ends in the file, as a byte offset from `payload`. The
2310    /// blocks are stored back to back, so a block starts where the one before it ended.
2311    ends: Vec<u64>,
2312    hashes: Vec<u64>,
2313    /// The payload, read and decoded a block at a time and kept after that.
2314    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2315    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
2316    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
2317    keep_budget: usize,
2318    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
2319    /// is measured against.
2320    ///
2321    /// Roughly, because two threads that keep the same block at the same time both add its length
2322    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
2323    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
2324    /// than a lock on the path every scan of a string column goes through.
2325    payload_kept: AtomicUsize,
2326    /// The boundaries this dictionary has already been searched for, by the value searched for.
2327    ///
2328    /// A search is the expensive thing this type does. It settles a probe on the stored head where
2329    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
2330    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
2331    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
2332    /// worst candidate, and the worst candidate settles long before the chunks run out.
2333    ///
2334    /// Shared across the instances of a scan rather than kept per instance, because each of them has
2335    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
2336    /// is nothing next to a probe of a file.
2337    ///
2338    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
2339    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
2340    /// bound is there for the filter that searches for a different literal every chunk rather than
2341    /// for anything this is meant to help.
2342    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
2343}
2344
2345/// How many searched for values a column's dictionary remembers the boundary of.
2346///
2347/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
2348/// larger one would be wrong.
2349const TEXT_SEARCH_MEMO: usize = 64;
2350
2351/// How many values of a dictionary go in one block of the payload.
2352///
2353/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
2354/// reader has to decode to get at a single value, so it is the one number the payload format turns
2355/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
2356/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
2357///
2358/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
2359/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
2360/// better all the way up, because front coding and the LZ matcher have more to look back at and
2361/// because the per chunk setup is spread over more values. What stops it is the point read: a query
2362/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
2363/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
2364/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
2365/// Going down to 512 gives up five to nine percent.
2366const TEXT_PAYLOAD_VALUES: usize = 1024;
2367
2368/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
2369///
2370/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
2371/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
2372/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
2373/// asking the same thing decodes all of it again, and on the same column at a million rows that
2374/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
2375/// is now paid by every statement in it. Neither end is the answer. A bound is.
2376///
2377/// So a sweep keeps what it decodes until the column is holding this much and decodes without
2378/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
2379/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
2380/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
2381/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
2382///
2383/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
2384/// what should replace it: this wants to be a buffer pool over the whole database, sized against
2385/// the memory limit the session was given, with the blocks of every column competing for it and the
2386/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
2387/// without an eviction order, which is a ceiling.
2388const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
2389
2390/// How many offsets go in one packed run.
2391///
2392/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
2393/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
2394/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
2395/// a run starts where a multiply says it does and nothing is padded.
2396const TEXT_OFFSET_RUN: usize = 512;
2397
2398/// Bytes at the front of a global dictionary index: the value count, the values a payload block
2399/// holds, the block count and the bits an offset is packed at.
2400const DICTIONARY_HEADER: usize = 16;
2401
2402/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
2403/// unit.
2404///
2405/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
2406/// columns, which is well under a page. A binary search over half a million entries makes nineteen
2407/// probes, and the first ten land in ten different blocks while the last nine land in the one block
2408/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
2409/// smaller block would save a little on the early probes, cost a checksum and an end list four times
2410/// as long, and give the heads less to share a base with. A larger one would read more than it uses
2411/// on every probe.
2412const TEXT_RANK_BLOCK: usize = 512;
2413
2414/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
2415/// at.
2416///
2417/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
2418/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
2419/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
2420/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
2421/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
2422/// dictionary of eighteen million, which is twenty five bits and not thirty two.
2423///
2424/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
2425/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
2426/// and the codes.
2427const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
2428
2429impl NativeText {
2430    /// One block of the payload, read and decoded the first time anything asks for a value in it.
2431    ///
2432    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
2433    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
2434    /// file is the only thing the caller cannot work out for itself, because the stored form is
2435    /// shorter than the decoded one and by a different amount in every block.
2436    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
2437        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
2438        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
2439        Ok(Some(bytes.as_slice()))
2440    }
2441
2442    /// Reads and decodes one block of the payload, without deciding who keeps it.
2443    ///
2444    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
2445    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
2446    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
2447        let start = if block == 0 { 0 } else { self.ends[block - 1] };
2448        let end = self.ends[block];
2449        let len = end
2450            .checked_sub(start)
2451            .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
2452        let mut stored = vec![
2453            0;
2454            usize::try_from(len).map_err(|_| invalid(
2455                "global dictionary block does not fit in memory"
2456            ))?
2457        ];
2458        read_at(&self.file, self.payload + start, &mut stored)?;
2459        if checksum(&stored) != self.hashes[block] {
2460            return Err(invalid("global dictionary payload checksum differs"));
2461        }
2462        let first = block * TEXT_PAYLOAD_VALUES;
2463        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
2464        let want = self.end_within(last - 1)? as usize;
2465        let values = string::decode_flat(&stored)?;
2466        if values.len() != last - first {
2467            return Err(invalid("global dictionary block holds the wrong value count"));
2468        }
2469        let bytes = values.into_bytes();
2470        if bytes.len() != want {
2471            return Err(invalid("global dictionary block decodes to the wrong length"));
2472        }
2473        Ok(bytes)
2474    }
2475
2476    /// Where the value at `index` ends inside its payload block.
2477    fn end_within(&self, index: usize) -> Result<u32> {
2478        let run = index / TEXT_OFFSET_RUN;
2479        let bytes = self
2480            .offsets
2481            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2482            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2483        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
2484            .map_err(|_| invalid("global dictionary offsets are short"))?;
2485        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
2486    }
2487
2488    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
2489    ///
2490    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
2491    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
2492    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
2493    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
2494    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
2495    ///
2496    /// [`bitpack::unpack_tail`] walks the run instead, which makes the window a fixed width and so
2497    /// an unaligned load, and reads the bit position off a counter. A run is five hundred and twelve
2498    /// values and a block is two of them, so a block of a thousand and twenty four values costs two
2499    /// calls here and nothing per value.
2500    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
2501        let mut ends = Vec::with_capacity(last.saturating_sub(first));
2502        let mut at = first;
2503        while at < last {
2504            let run = at / TEXT_OFFSET_RUN;
2505            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
2506            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
2507            let bytes = self
2508                .offsets
2509                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2510                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2511            let run_ends = bitpack::unpack_tail(bytes, self.offset_bits, held)
2512                .map_err(|_| invalid("global dictionary offsets are short"))?;
2513            let within = run_ends
2514                .get(at % TEXT_OFFSET_RUN..stop - run * TEXT_OFFSET_RUN)
2515                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2516            ends.extend_from_slice(within);
2517            at = stop;
2518        }
2519        Ok(ends)
2520    }
2521
2522    /// Where the value at `index` starts inside its payload block, which is where the value before
2523    /// it ended unless it is the first of the block.
2524    fn start_within(&self, index: usize) -> Result<u32> {
2525        if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
2526    }
2527
2528    /// Where the value at `index` starts and ends inside its payload block.
2529    ///
2530    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
2531    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
2532    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
2533    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
2534    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
2535    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
2536        let within = index % TEXT_OFFSET_RUN;
2537        let (start, end) = if within == 0 {
2538            (self.start_within(index)?, self.end_within(index)?)
2539        } else {
2540            let run = index / TEXT_OFFSET_RUN;
2541            let bytes = self
2542                .offsets
2543                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2544                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2545            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
2546                .map_err(|_| invalid("global dictionary offsets are short"))?;
2547            let ends = u32::try_from(end)
2548                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2549            let starts = u32::try_from(start)
2550                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2551            (starts, ends)
2552        };
2553        if start > end {
2554            return Err(invalid("global dictionary value ends before it starts"));
2555        }
2556        Ok((start, end))
2557    }
2558
2559    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
2560    ///
2561    /// The block is read from the file and checked against the hash the index carries for it the
2562    /// first time anything asks, and kept after that, the same way a payload block is. A search
2563    /// makes about as many probes as the order has bits, so the whole search reads a handful of
2564    /// these and never the rest.
2565    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
2566        let slot = self
2567            .rank_blocks
2568            .get(rank / TEXT_RANK_BLOCK)
2569            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
2570        let block = slot
2571            .get_or_init(|| {
2572                let which = rank / TEXT_RANK_BLOCK;
2573                let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
2574                let end = self.rank_ends[which];
2575                let mut bytes = vec![0; (end - start) as usize];
2576                read_at(&self.file, self.rank_at + start, &mut bytes)?;
2577                if checksum(&bytes)
2578                    != *self
2579                        .rank_hashes
2580                        .get(rank / TEXT_RANK_BLOCK)
2581                        .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
2582                {
2583                    return Err(invalid("global dictionary rank checksum differs"));
2584                }
2585                Ok(bytes)
2586            })
2587            .as_ref()
2588            .map_err(Clone::clone)?;
2589        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
2590    }
2591
2592    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
2593    fn head_at(&self, rank: usize) -> Result<u64> {
2594        let (block, within) = self.rank_parts(rank)?;
2595        let (base, width, packed) = rank_heads(block)?;
2596        let above = bitpack::tail_at(packed, width, within)
2597            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
2598        Ok(base.wrapping_add(above))
2599    }
2600
2601    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
2602    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
2603        let (_, width, packed) = rank_heads(block)?;
2604        packed
2605            .get(bitpack::tail_len(count, width)..)
2606            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
2607    }
2608
2609    /// How many entries the block holding `rank` has, which is a full block except at the end.
2610    fn rank_block_len(&self, rank: usize) -> usize {
2611        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
2612        TEXT_RANK_BLOCK.min(self.ranks - first)
2613    }
2614}
2615
2616/// The base, the width and the packed bytes of one rank block's heads.
2617fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
2618    let header = block
2619        .get(..RANK_BLOCK_HEADER)
2620        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
2621    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
2622    let width = header[8] as usize;
2623    if width > 64 {
2624        return Err(invalid("global dictionary rank block packs heads past a word"));
2625    }
2626    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
2627}
2628
2629/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
2630///
2631/// One width for the whole column rather than one a block. A block is 1,024 values of the same
2632/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
2633/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
2634/// the arithmetic that finds where a block starts.
2635fn offset_width(offsets: &[u32]) -> usize {
2636    let values = offsets.len() - 1;
2637    let mut span = 0;
2638    for first in (0..values).step_by(TEXT_PAYLOAD_VALUES) {
2639        let last = (first + TEXT_PAYLOAD_VALUES).min(values);
2640        span = span.max(offsets[last] - offsets[first]);
2641    }
2642    (u32::BITS - span.leading_zeros()) as usize
2643}
2644
2645/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
2646/// has read any of them.
2647fn offset_bytes(values: usize, bits: usize) -> usize {
2648    let full = values / TEXT_OFFSET_RUN;
2649    let rest = values % TEXT_OFFSET_RUN;
2650    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
2651}
2652
2653/// The end of every value within its payload block, packed a run at a time.
2654fn encode_offsets(offsets: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
2655    let values = offsets.len() - 1;
2656    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
2657    for first in (0..values).step_by(TEXT_OFFSET_RUN) {
2658        let last = (first + TEXT_OFFSET_RUN).min(values);
2659        let base = offsets[first / TEXT_PAYLOAD_VALUES * TEXT_PAYLOAD_VALUES];
2660        run.clear();
2661        run.extend((first..last).map(|value| u64::from(offsets[value + 1] - base)));
2662        bitpack::pack_tail(&run, bits, out)
2663            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
2664    }
2665    Ok(())
2666}
2667
2668/// How many bits a code of a dictionary of `values` entries takes.
2669fn code_width(values: usize) -> usize {
2670    match u64::try_from(values).unwrap_or(u64::MAX) {
2671        0 | 1 => 0,
2672        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
2673    }
2674}
2675
2676impl TextSource for NativeText {
2677    fn len(&self) -> usize {
2678        self.values
2679    }
2680
2681    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
2682        if index >= self.values {
2683            return Ok(None);
2684        }
2685        let (start, end) = self.span_within(index)?;
2686        if start == end {
2687            return Ok(Some(&[]));
2688        }
2689        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
2690        // is in one block and the offsets already say where in it.
2691        let block = index / TEXT_PAYLOAD_VALUES;
2692        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
2693        Ok(bytes.get(start as usize..end as usize))
2694    }
2695
2696    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
2697        if index >= self.values {
2698            return Ok(None);
2699        }
2700        let (start, end) = self.span_within(index)?;
2701        Ok(Some((end - start) as usize))
2702    }
2703
2704    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
2705    ///
2706    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
2707    /// every block whatever it does. The question is whether it keeps them, and both answers are
2708    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
2709    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
2710    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
2711    /// the same question decode all of it again, which on the same column at a million rows is a
2712    /// `LIKE` going from 2.7 ms to 16.2 ms.
2713    ///
2714    /// So a sweep keeps what it decodes while the column is under [`TEXT_KEEP_BUDGET`] and drops it
2715    /// after that. A block already in hand is used where it is there and costs nothing either way.
2716    fn sweep(
2717        &self,
2718        first: usize,
2719        limit: usize,
2720        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
2721    ) -> Result<usize> {
2722        let limit = limit.min(self.values);
2723        if first >= limit {
2724            return Ok(first);
2725        }
2726        let block = first / TEXT_PAYLOAD_VALUES;
2727        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
2728        let decoded;
2729        let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
2730            Some(Ok(kept)) => kept,
2731            _ if self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
2732                let kept = self
2733                    .payload_block(block)?
2734                    .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
2735                self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
2736                kept
2737            }
2738            _ => {
2739                decoded = self.decode_block(block)?;
2740                &decoded
2741            }
2742        };
2743        let ends = self.ends_within(first, last)?;
2744        if ends.len() != last - first {
2745            return Err(invalid("global dictionary offsets are short"));
2746        }
2747        let mut start = u64::from(self.start_within(first)?);
2748        // row at a time: the caller is handed one value after another, and what it does with one is
2749        // its own business, so there is no shape here for anything but a walk.
2750        for (index, &end) in (first..last).zip(&ends) {
2751            let value = usize::try_from(start)
2752                .ok()
2753                .zip(usize::try_from(end).ok())
2754                .and_then(|(from, to)| bytes.get(from..to))
2755                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
2756            body(index, value)?;
2757            start = end;
2758        }
2759        Ok(last)
2760    }
2761
2762    fn ranks(&self) -> Option<usize> {
2763        (self.ranks > 0).then_some(self.ranks)
2764    }
2765
2766    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
2767    /// it is not.
2768    ///
2769    /// The lock is held over the search rather than dropped and taken again, so that two threads
2770    /// asking for the same value at the same time do the work once between them. That is the shape
2771    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
2772    /// improving their bound over the same early chunks.
2773    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
2774        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
2775        if let Some(&answer) = memo.get(wanted) {
2776            return Ok(answer);
2777        }
2778        let answer = search_below(self, ranks, wanted)?;
2779        if memo.len() >= TEXT_SEARCH_MEMO {
2780            memo.clear();
2781        }
2782        memo.insert(wanted.to_vec(), answer);
2783        Ok(answer)
2784    }
2785
2786    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
2787        // The head settles the probe unless the two values start with the same eight bytes, and
2788        // only then is a value read. On a column of URLs that is the difference between a search
2789        // that touches one block of the payload and a search that touches nineteen of them.
2790        let settled = self.head_at(rank)?.cmp(&head(wanted));
2791        if settled != Ordering::Equal {
2792            return Ok(settled);
2793        }
2794        let code = self.code_at_rank(rank)?;
2795        let bytes = self
2796            .bytes_at(code as usize)?
2797            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
2798        Ok(bytes.cmp(wanted))
2799    }
2800
2801    fn code_at_rank(&self, rank: usize) -> Result<u32> {
2802        let (block, within) = self.rank_parts(rank)?;
2803        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
2804        let code = bitpack::tail_at(codes, self.code_bits, within)
2805            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
2806        let code = u32::try_from(code)
2807            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
2808        if code as usize >= self.len() {
2809            return Err(invalid("global dictionary order names a code it does not have"));
2810        }
2811        Ok(code)
2812    }
2813
2814    fn code_ranks(&self) -> Option<&[u32]> {
2815        // The order is a permutation of the positions, so inverting it needs every position to be
2816        // named exactly once. Anything else and the slice would have holes, and a caller indexing
2817        // it by a code would read a rank that belongs to nothing.
2818        if self.ranks == 0 || self.ranks != self.len() {
2819            return None;
2820        }
2821        self.code_ranks
2822            .get_or_init(|| {
2823                let mut ranks = vec![u32::MAX; self.ranks];
2824                // A block at a time rather than a rank at a time, because reading it per rank pays
2825                // for the bounds check, the division and the lock on every one of them.
2826                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
2827                    let (block, _) = self.rank_parts(first).ok()?;
2828                    let count = self.rank_block_len(first);
2829                    let codes = self.rank_codes(block, count).ok()?;
2830                    for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
2831                        .ok()?
2832                        .into_iter()
2833                        .enumerate()
2834                    {
2835                        let code = usize::try_from(code).ok()?;
2836                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
2837                    }
2838                }
2839                if ranks.contains(&u32::MAX) {
2840                    return None;
2841                }
2842                Some(ranks)
2843            })
2844            .as_deref()
2845    }
2846
2847    fn footprint(&self) -> usize {
2848        self.offsets.capacity()
2849            + self
2850                .code_ranks
2851                .get()
2852                .and_then(Option::as_ref)
2853                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
2854            + self.rank_hashes.capacity() * size_of::<u64>()
2855            + self.rank_ends.capacity() * size_of::<u64>()
2856            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2857            + self
2858                .rank_blocks
2859                .iter()
2860                .filter_map(OnceLock::get)
2861                .filter_map(|result| result.as_ref().ok())
2862                .map(Vec::capacity)
2863                .sum::<usize>()
2864            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2865            + self.hashes.capacity() * size_of::<u64>()
2866            + self.ends.capacity() * size_of::<u64>()
2867            + self
2868                .blocks
2869                .iter()
2870                .filter_map(OnceLock::get)
2871                .filter_map(|result| result.as_ref().ok())
2872                .map(Vec::capacity)
2873                .sum::<usize>()
2874    }
2875}
2876
2877/// Every table wide part number in order, with the stripe it belongs to.
2878fn places(table: &Table) -> Result<Vec<Place>> {
2879    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
2880    for (at, stripe) in table.stripes.iter().enumerate() {
2881        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
2882        for (part, &rows) in stripe.parts.iter().enumerate() {
2883            places.push(Place {
2884                stripe: index,
2885                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
2886                rows,
2887            });
2888        }
2889    }
2890    Ok(places)
2891}
2892
2893/// Reads one column's section of a stripe's index page.
2894///
2895/// The section carries its own checksum, so a reader that wants one column out of a hundred and
2896/// five preads a few hundred bytes and still knows that what it got is what was written.
2897fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
2898    let parts = stripe.parts.len();
2899    let section = index_section(parts)?;
2900    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
2901    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
2902    if end > stripe.index.length as usize {
2903        return Err(invalid("index page is shorter than its columns"));
2904    }
2905    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
2906    let mut bytes = vec![0; section];
2907    let offset = stripe
2908        .index
2909        .offset
2910        .checked_add(at as u64)
2911        .ok_or_else(|| invalid("index page offset overflow"))?;
2912    read_at(file, offset, &mut bytes)?;
2913    let entries = section - size_of::<u64>();
2914    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
2915    if checksum(&bytes[..entries]) != stored {
2916        // With where it was read from, because the two ways this fires look identical from the
2917        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
2918        return Err(invalid(&format!(
2919            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
2920             wanted {stored:016x} and got {:016x}",
2921            checksum(&bytes[..entries]),
2922        )));
2923    }
2924    let mut spans = Vec::with_capacity(parts);
2925    let mut start = 0_usize;
2926    for part in 0..parts {
2927        let at = part * INDEX_ENTRY;
2928        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
2929        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
2930        spans.push(PartSpan { start, length, hash });
2931        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
2932    }
2933    if start != page.length as usize {
2934        return Err(invalid("column page length differs from its index"));
2935    }
2936    Ok(spans)
2937}
2938
2939/// One part's bytes out of a whole column page.
2940fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
2941    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
2942    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
2943}
2944
2945/// Puts one stripe of one column in the cache, dropping the stripe that has been there longest.
2946///
2947/// The index goes in its own slot and stays. Only the page is under the budget, and `kept` is how
2948/// many pages that budget is.
2949fn remember(cached: &mut Cached, held: &CachedColumn, kept: usize) {
2950    if let Some(slot) = cached.index.get_mut(held.stripe) {
2951        if slot.is_none() {
2952            *slot = Some(Arc::clone(&held.index));
2953        }
2954    }
2955    let Some(page) = held.page.clone() else { return };
2956    let Some(slot) = cached.pages.get_mut(held.stripe) else { return };
2957    if slot.is_none() {
2958        cached.order.push_back(held.stripe);
2959    }
2960    *slot = Some(page);
2961    while cached.order.len() > kept.max(1) {
2962        let Some(oldest) = cached.order.pop_front() else { break };
2963        if let Some(slot) = cached.pages.get_mut(oldest) {
2964            *slot = None;
2965        }
2966    }
2967}
2968
2969/// Every table a native file holds, without the directory of any of them.
2970///
2971/// This is what opening a database reads. It is the small level of the directory, so the cost is
2972/// proportional to how many tables there are rather than to how much data they hold, and a session
2973/// that touches two tables of eight decodes two table directories.
2974///
2975/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
2976/// file descriptor, not eight, which is the other thing one file buys over a file per table.
2977#[derive(Debug, Clone)]
2978pub struct Catalog {
2979    file: Arc<File>,
2980    size: u64,
2981    entries: Arc<Vec<Entry>>,
2982    /// The views the file holds, whole, since a view has no second level to read later.
2983    views: Arc<Vec<ViewEntry>>,
2984    opening: Opening,
2985}
2986
2987impl Catalog {
2988    /// Reads the highest valid catalog slot and nothing under it.
2989    ///
2990    /// # Errors
2991    ///
2992    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
2993    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
2994        let (file, size, _, bytes, opening) = slot_bytes(path)?;
2995        let (entries, views) = decode_catalog(&bytes, size)?;
2996        Ok(Self {
2997            file: Arc::new(file),
2998            size,
2999            entries: Arc::new(entries),
3000            views: Arc::new(views),
3001            opening,
3002        })
3003    }
3004
3005    /// The tables in the file, in the order they were written.
3006    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
3007        self.entries.iter().map(|entry| entry.name.as_str())
3008    }
3009
3010    /// The same tables with how many rows each of them holds.
3011    ///
3012    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
3013    /// A load asks a second question: whether a table already in the file is really in the way of
3014    /// the one it wants to write. A table with no rows is not, because it has no pages the next
3015    /// generation would have to carry, so the count has to come out of the catalog beside the name.
3016    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
3017        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
3018    }
3019
3020    /// The views in the file, in the order they were written.
3021    ///
3022    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
3023    /// by one. A view is a few strings and a column list and it was all read at open, so there is
3024    /// nothing left to go and fetch and no reason to make the caller ask twice.
3025    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
3026        self.views.iter()
3027    }
3028
3029    /// How many tables the file holds.
3030    #[must_use]
3031    pub fn len(&self) -> usize {
3032        self.entries.len()
3033    }
3034
3035    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
3036    /// database somebody dropped the last table out of comes back as.
3037    #[must_use]
3038    pub fn is_empty(&self) -> bool {
3039        self.entries.is_empty()
3040    }
3041
3042    /// Opens one table by name, decoding its directory now.
3043    ///
3044    /// # Errors
3045    ///
3046    /// If there is no table by that name, or its directory is torn or points outside the file.
3047    pub fn table(&self, name: &str) -> Result<Reader> {
3048        let entry = self
3049            .entries
3050            .iter()
3051            .find(|entry| entry.name == name)
3052            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
3053        let mut bytes = vec![0; entry.directory.length as usize];
3054        read_at(&self.file, entry.directory.offset, &mut bytes)?;
3055        if checksum(&bytes) != entry.directory.hash {
3056            return Err(invalid(&format!("the directory of table {name} does not checksum")));
3057        }
3058        let mut opening = self.opening;
3059        opening.reads += 1;
3060        opening.bytes += u64::from(entry.directory.length);
3061        Reader::build(
3062            Arc::clone(&self.file),
3063            self.size,
3064            decode_directory(&bytes, self.size)?,
3065            u64::from(entry.directory.length),
3066            opening,
3067        )
3068    }
3069}
3070
3071/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
3072///
3073/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
3074/// before there was a second generation to write.
3075fn slot_offset(generation: u64) -> u64 {
3076    16 + (generation - 1) % 2 * SLOT_BYTES as u64
3077}
3078
3079/// The header and the bytes the highest valid slot points at.
3080///
3081/// Both levels of the directory are reached this way, so the magic check, the version check and the
3082/// choice between the two slots live here rather than being written out twice.
3083fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
3084    let mut file = File::open(path).map_err(io)?;
3085    let size = file.metadata().map_err(io)?.len();
3086    if size < HEADER {
3087        return Err(invalid("file is shorter than its header"));
3088    }
3089    let mut header = [0; HEADER as usize];
3090    file.read_exact(&mut header).map_err(io)?;
3091    let mut opening = Opening { reads: 1, bytes: HEADER };
3092    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
3093    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
3094    // the answer is to look at the path. A wrong version is our own file from another build,
3095    // and the number this build wants is the only thing that tells the reader whether to
3096    // rebuild the file or to go back to the binary that wrote it.
3097    if &header[..8] != MAGIC {
3098        return Err(invalid("the header does not begin with a rudb native magic"));
3099    }
3100    if !READABLE.contains(&version) {
3101        return Err(invalid(&format!(
3102            "the file is format {version} and this build reads format {FORMAT}, so it has to \
3103                 be written again"
3104        )));
3105    }
3106    let mut selected = None;
3107    for start in [16, 16 + SLOT_BYTES] {
3108        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
3109        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
3110            continue;
3111        }
3112        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
3113        if slot.offset < HEADER || end > size {
3114            continue;
3115        }
3116        let mut bytes = vec![0; slot.length as usize];
3117        file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
3118        file.read_exact(&mut bytes).map_err(io)?;
3119        opening.reads += 1;
3120        opening.bytes += u64::from(slot.length);
3121        if checksum(&bytes) == slot.hash
3122            && selected
3123                .as_ref()
3124                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
3125        {
3126            selected = Some((slot, bytes));
3127        }
3128    }
3129    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
3130    Ok((file, size, slot, bytes, opening))
3131}
3132
3133impl Reader {
3134    /// Opens a file that holds exactly one table.
3135    ///
3136    /// # Errors
3137    ///
3138    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
3139    /// file holds more than one table, which is a file that has to be opened by name.
3140    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
3141        let catalog = Catalog::open(path)?;
3142        let mut names = catalog.names();
3143        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
3144        if names.next().is_some() {
3145            return Err(invalid(
3146                "the file holds more than one table, so it has to be opened by name",
3147            ));
3148        }
3149        catalog.table(&name)
3150    }
3151
3152    /// Builds a reader over one decoded table directory.
3153    fn build(
3154        file: Arc<File>,
3155        size: u64,
3156        table: Table,
3157        directory: u64,
3158        opening: Opening,
3159    ) -> Result<Self> {
3160        let places = places(&table)?;
3161        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
3162        let table_fields = table.fields.len();
3163        let stripes = table.stripes.len();
3164        let cache = (0..table.fields.len())
3165            .map(|_| {
3166                Mutex::new(Cached {
3167                    pages: (0..stripes).map(|_| None).collect(),
3168                    index: (0..stripes).map(|_| None).collect(),
3169                    ..Cached::default()
3170                })
3171            })
3172            .collect::<Vec<_>>();
3173        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
3174            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3175            .collect();
3176        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
3177            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3178            .collect();
3179        Ok(Self {
3180            file,
3181            table: Arc::new(table),
3182            dictionaries: Arc::new(dictionaries),
3183            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
3184            opened: Arc::new(AtomicUsize::new(0)),
3185            sieves: Arc::new(sieves),
3186            part_ranges: Arc::new(part_ranges),
3187            places: Arc::new(places),
3188            cache: Arc::new(cache),
3189            pages: Arc::new(AtomicUsize::new(0)),
3190            indexes: Arc::new(AtomicUsize::new(0)),
3191            kept: Arc::new(AtomicUsize::new(CACHED_STRIPES_PER_COLUMN)),
3192            size,
3193            directory,
3194            opening,
3195        })
3196    }
3197
3198    /// What this reader has read so far, and what opening it cost.
3199    ///
3200    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
3201    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
3202    /// file touched the data asks here, and gets an answer that does not depend on what the page
3203    /// cache happened to hold.
3204    #[must_use]
3205    pub fn reads(&self) -> Reads {
3206        Reads {
3207            opening: self.opening,
3208            pages: self.pages.load(Atomic::Relaxed),
3209            indexes: self.indexes.load(Atomic::Relaxed),
3210            dictionaries: self.opened.load(Atomic::Relaxed),
3211        }
3212    }
3213
3214    /// Where the file's bytes went, from the directory alone.
3215    ///
3216    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
3217    /// for what is charged where and for why the three things that are not columns stay separate.
3218    #[must_use]
3219    pub fn layout(&self) -> Layout {
3220        let table = &self.table;
3221        let stripes = table.stripes.as_slice();
3222        let columns = table
3223            .fields
3224            .iter()
3225            .enumerate()
3226            .map(|(at, field)| ColumnLayout {
3227                name: field.name.clone(),
3228                kind: field.ty.to_string(),
3229                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
3230                memberships: sum(stripes.iter().map(|stripe| page_bytes(&stripe.memberships, at))),
3231                sieves: sum(stripes.iter().map(|stripe| page_bytes(&stripe.sieves, at))),
3232                part_ranges: sum(stripes.iter().map(|stripe| page_bytes(&stripe.part_ranges, at))),
3233                dictionary: page_bytes(&table.dictionaries, at),
3234            })
3235            .collect();
3236        Layout {
3237            file: self.size,
3238            rows: table.rows,
3239            stripes: stripes.len(),
3240            parts: self.places.len(),
3241            columns,
3242            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
3243            directory: self.directory,
3244            header: HEADER,
3245        }
3246    }
3247
3248    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
3249    ///
3250    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
3251    /// nowhere else. The directory says how many bytes a column took and says nothing about what
3252    /// shape they are in, and the shape is the question worth asking: the same rows in a different
3253    /// order come back bit packed on one file and plain on another, and that is the difference a
3254    /// clustered load makes to a scan.
3255    ///
3256    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
3257    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
3258    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
3259    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
3260    ///
3261    /// # Errors
3262    ///
3263    /// If the column is outside the schema, or a page, index section or checksum is invalid.
3264    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
3265        let field = self
3266            .table
3267            .fields
3268            .get(column)
3269            .ok_or_else(|| invalid("stored column index out of range"))?;
3270        let mut stored = Vec::with_capacity(self.places.len());
3271        let mut row = 0;
3272        for (at, stripe) in self.table.stripes.iter().enumerate() {
3273            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3274            let index = read_index(&self.file, stripe, column)?;
3275            let mut bytes = vec![0; page.length as usize];
3276            read_at(&self.file, page.offset, &mut bytes)?;
3277            let ranges = self.stripe_part_ranges(at, column);
3278            for (part, &rows) in stripe.parts.iter().enumerate() {
3279                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
3280                let held = part_bytes(&bytes, span)?;
3281                let range = ranges.and_then(|held| held.get(part));
3282                stored.push(StoredPart {
3283                    stripe: at,
3284                    part,
3285                    row,
3286                    rows: rows as usize,
3287                    encoding: page_encoding(&field.ty, rows as usize, held),
3288                    bytes: span.length as u64,
3289                    page: page.offset,
3290                    offset: span.start as u64,
3291                    low: range
3292                        .and_then(|range| range.low.clone())
3293                        .and_then(|bound| bound.into_value(&field.ty)),
3294                    high: range
3295                        .and_then(|range| range.high.clone())
3296                        .and_then(|bound| bound.into_value(&field.ty)),
3297                    nulls: range.map(|range| range.nulls),
3298                });
3299                row += rows as usize;
3300            }
3301        }
3302        Ok(stored)
3303    }
3304
3305    /// How many parts the table has, which is how many chunks a scan of it reads.
3306    #[must_use]
3307    pub fn parts(&self) -> usize {
3308        self.places.len()
3309    }
3310
3311    /// The parts of each stripe, in table wide part numbers.
3312    ///
3313    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
3314    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
3315    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
3316    /// directory rather than worked out from a constant.
3317    #[must_use]
3318    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
3319        let mut runs = Vec::with_capacity(self.table.stripes.len());
3320        let mut start = 0;
3321        for stripe in &self.table.stripes {
3322            let end = start + stripe.parts.len();
3323            runs.push(start..end);
3324            start = end;
3325        }
3326        runs
3327    }
3328
3329    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
3330    ///
3331    /// Off the directory, which is already in memory, rather than by the caller asking for each
3332    /// part in turn through the catalog. Nothing past the end holds any rows.
3333    #[must_use]
3334    pub fn stripe_rows(&self, stripe: usize) -> usize {
3335        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
3336    }
3337
3338    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
3339    ///
3340    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
3341    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
3342    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
3343    /// reads a quarter of a megabyte for every part it takes out of it.
3344    pub fn keep_stripes(&self, stripes: usize) {
3345        self.kept.fetch_max(stripes, Atomic::Relaxed);
3346    }
3347
3348    /// Rows in one part, or zero when the part number is past the table.
3349    #[must_use]
3350    pub fn part_rows(&self, at: usize) -> usize {
3351        self.places.get(at).map_or(0, |place| place.rows as usize)
3352    }
3353
3354    /// The committed table directory.
3355    #[must_use]
3356    pub fn table(&self) -> &Table {
3357        &self.table
3358    }
3359
3360    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
3361    ///
3362    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
3363    /// additional ordering keys without losing a value tied with the requested boundary.
3364    ///
3365    /// # Errors
3366    ///
3367    /// If the column is outside the schema or a stored value does not fit its declared type.
3368    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
3369        let field = self
3370            .table
3371            .fields
3372            .get(column)
3373            .ok_or_else(|| invalid("frequency column index out of range"))?;
3374        let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3375            return Ok(None);
3376        };
3377        if top == 0 || summary.entries.len() < top {
3378            return Ok(None);
3379        }
3380        let boundary = summary.entries[top - 1].count;
3381        if boundary <= summary.omitted_max {
3382            return Ok(None);
3383        }
3384        self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
3385    }
3386
3387    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
3388    ///
3389    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
3390    /// out of room, so what it usually ends with is the leading values and a bound on everything it
3391    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
3392    /// the entries did not overflow the stored budget, so the list is every distinct value of the
3393    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
3394    ///
3395    /// That makes a whole class of question answerable without reading a row. How many rows hold a
3396    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
3397    /// all in here. It is only ever true of a column with few enough distinct values, which is the
3398    /// case worth having, because that is exactly the column a grouping or an equality filter would
3399    /// otherwise walk every row to answer.
3400    ///
3401    /// `None` when the column has no synopsis, or has one that dropped anything.
3402    ///
3403    /// # Errors
3404    ///
3405    /// If the column is outside the schema or a stored value does not fit its declared type.
3406    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
3407        let Some(prefix) = self.frequency_prefix(column)? else {
3408            return Ok(None);
3409        };
3410        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
3411    }
3412
3413    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
3414    ///
3415    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
3416    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
3417    /// made it into the list carries the number of rows that really hold it rather than whatever the
3418    /// pass had left over. What the pass loses is values, not counts.
3419    ///
3420    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
3421    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
3422    /// leading values of the column and everything else is somewhere between no rows and that bound.
3423    ///
3424    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
3425    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
3426    /// the rows by the distinct count is furthest from the truth.
3427    ///
3428    /// `None` when the column has no synopsis.
3429    ///
3430    /// # Errors
3431    ///
3432    /// If the column is outside the schema or a stored value does not fit its declared type.
3433    ///
3434    /// [`exact_frequencies`]: Self::exact_frequencies
3435    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
3436        let field = self
3437            .table
3438            .fields
3439            .get(column)
3440            .ok_or_else(|| invalid("frequency column index out of range"))?;
3441        let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3442            return Ok(None);
3443        };
3444        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
3445        Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
3446    }
3447
3448    /// Turns stored frequency entries into values of the column's own type.
3449    fn decode_frequencies(
3450        &self,
3451        column: usize,
3452        ty: &LogicalType,
3453        entries: &[FrequencyEntry],
3454    ) -> Result<Vec<(Value, u64)>> {
3455        let dictionary = if *ty == LogicalType::Varchar { self.dictionary(column)? } else { None };
3456        let mut out = Vec::with_capacity(entries.len());
3457        for entry in entries {
3458            let value = match entry.value {
3459                FrequencyValue::Null => Value::Null,
3460                FrequencyValue::Integer(value) => match *ty {
3461                    LogicalType::TinyInt => Value::TinyInt(
3462                        i8::try_from(value)
3463                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
3464                    ),
3465                    LogicalType::UTinyInt => Value::UTinyInt(
3466                        u8::try_from(value)
3467                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
3468                    ),
3469                    LogicalType::USmallInt => Value::USmallInt(
3470                        u16::try_from(value)
3471                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
3472                    ),
3473                    LogicalType::UInteger => Value::UInteger(
3474                        u32::try_from(value)
3475                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
3476                    ),
3477                    LogicalType::UBigInt => Value::UBigInt(
3478                        u64::try_from(value)
3479                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
3480                    ),
3481                    LogicalType::SmallInt => Value::SmallInt(
3482                        i16::try_from(value)
3483                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
3484                    ),
3485                    LogicalType::Integer => Value::Integer(
3486                        i32::try_from(value)
3487                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
3488                    ),
3489                    LogicalType::BigInt => Value::BigInt(
3490                        i64::try_from(value)
3491                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
3492                    ),
3493                    LogicalType::Date => Value::Date(
3494                        i32::try_from(value)
3495                            .map_err(|_| invalid("frequency DATE is out of range"))?,
3496                    ),
3497                    LogicalType::Timestamp => Value::Timestamp(
3498                        i64::try_from(value)
3499                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
3500                    ),
3501                    _ => return Err(invalid("integer frequency belongs to another type")),
3502                },
3503                FrequencyValue::Code(code) => dictionary
3504                    .as_ref()
3505                    .ok_or_else(|| invalid("frequency code has no dictionary"))?
3506                    .try_value_at(code as usize)?,
3507            };
3508            out.push((value, entry.count));
3509        }
3510        Ok(out)
3511    }
3512
3513    /// Sparse rows belonging to the bounded numeric frequency candidate set.
3514    ///
3515    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
3516    /// aggregate may accept a result over these rows only when its requested boundary is strictly
3517    /// greater than `omitted_max`.
3518    ///
3519    /// # Errors
3520    ///
3521    /// If the column is outside the schema.
3522    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
3523        self.table
3524            .fields
3525            .get(column)
3526            .ok_or_else(|| invalid("frequency column index out of range"))?;
3527        let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3528            return Ok(None);
3529        };
3530        if summary.ordinals.is_empty() {
3531            return Ok(None);
3532        }
3533        Ok(Some(FrequencyOccurrences {
3534            omitted_max: summary.omitted_max,
3535            ordinals: summary.ordinals.clone(),
3536        }))
3537    }
3538
3539    /// How many distinct values one column holds, counting a null as no value.
3540    ///
3541    /// A string column of this format is written against one dictionary that covers the whole table.
3542    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
3543    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
3544    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
3545    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
3546    /// every row.
3547    ///
3548    /// A null in the column used to make this `None` and no longer does. A null row is written as
3549    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
3550    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
3551    /// The writer does know, because it counts the non-null rows that use each code on its way to
3552    /// the frequency summary, so it records how many codes any row holds and the directory carries
3553    /// that number. This reads it rather than the size of the dictionary, which also means the
3554    /// dictionary page is not opened to answer.
3555    ///
3556    /// `None` for a column the file has no dictionary for, which is every column that is not a
3557    /// string. A sketch would answer that approximately and SQL asked for the exact number.
3558    ///
3559    /// # Errors
3560    ///
3561    /// If the column is outside the schema.
3562    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
3563        self.table
3564            .distincts
3565            .get(column)
3566            .copied()
3567            .ok_or_else(|| invalid("distinct column index out of range"))
3568    }
3569
3570    /// How many rows of one column are null, added up over the stripes.
3571    ///
3572    /// Every stripe records this exactly when it is written, because a null count is not a bound
3573    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
3574    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
3575    /// already in memory is what makes `COUNT(column)` over a whole table free.
3576    ///
3577    /// # Errors
3578    ///
3579    /// If the column is outside the schema.
3580    pub fn null_count(&self, column: usize) -> Result<u64> {
3581        if column >= self.table.fields.len() {
3582            return Err(invalid("null count column index out of range"));
3583        }
3584        let mut nulls = 0_u64;
3585        for stripe in &self.table.stripes {
3586            let range = stripe
3587                .zone
3588                .column(column)
3589                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3590            nulls = nulls
3591                .checked_add(range.nulls as u64)
3592                .ok_or_else(|| invalid("null count overflow"))?;
3593        }
3594        Ok(nulls)
3595    }
3596
3597    /// The smallest and the largest value of one string column, from the order beside its values.
3598    ///
3599    /// The dictionary holds exactly the values the column holds, so the first and the last of them
3600    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
3601    /// otherwise walks a million rows.
3602    ///
3603    /// `None` when the column is not a string, when the file was written before version 9 and so has
3604    /// no order, when the column has no values at all, or when it has a null in it, which is the
3605    /// placeholder again: the empty string a null is written as would sort ahead of every real
3606    /// value and be reported as the minimum.
3607    ///
3608    /// # Errors
3609    ///
3610    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
3611    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
3612        if self.null_count(column)? > 0 {
3613            return Ok(None);
3614        }
3615        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
3616        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
3617        if ranks == 0 {
3618            return Ok(None);
3619        }
3620        let low = text_at_rank(&dictionary, 0)?;
3621        let high = text_at_rank(&dictionary, ranks - 1)?;
3622        Ok(Some((low, high)))
3623    }
3624
3625    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
3626    ///
3627    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
3628    /// chunk that could not match is still correct when it rules out nothing. That is what makes
3629    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
3630    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
3631    /// all of them walked their rows.
3632    ///
3633    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
3634    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
3635    ///
3636    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
3637    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
3638    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
3639    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
3640    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
3641    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
3642    /// and the fix is a row count per part rather than anything here.
3643    ///
3644    /// # Errors
3645    ///
3646    /// If the column is outside the schema.
3647    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
3648        if column >= self.table.fields.len() {
3649            return Err(invalid("extremes column index out of range"));
3650        }
3651        let mut low: Option<Bound> = None;
3652        let mut high: Option<Bound> = None;
3653        for stripe in &self.table.stripes {
3654            let range = stripe
3655                .zone
3656                .column(column)
3657                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3658            if !range.exact {
3659                return Ok(None);
3660            }
3661            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
3662            // is why this skips it rather than giving up on the whole column. A stripe that has
3663            // rows and still has no end is a layout whose values this cannot see, and skipping that
3664            // one would answer with an end taken from the other stripes, so it gives up instead.
3665            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
3666                if stripe.rows > range.nulls {
3667                    return Ok(None);
3668                }
3669                continue;
3670            };
3671            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
3672            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
3673        }
3674        Ok(low.zip(high))
3675    }
3676
3677    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
3678    ///
3679    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
3680    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
3681    /// count would be doing the same walk twice.
3682    ///
3683    /// `None` for anything that is not an integer column, for a file written by something that did
3684    /// not record it, and when adding the stripes together would overflow.
3685    ///
3686    /// # Errors
3687    ///
3688    /// If the column is outside the schema.
3689    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
3690        if column >= self.table.fields.len() {
3691            return Err(invalid("sum column index out of range"));
3692        }
3693        let mut total = 0_i128;
3694        let mut rows = 0_u64;
3695        for stripe in &self.table.stripes {
3696            let range = stripe
3697                .zone
3698                .column(column)
3699                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3700            let Some(part) = range.sum else { return Ok(None) };
3701            let Some(sum) = total.checked_add(part) else { return Ok(None) };
3702            total = sum;
3703            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
3704        }
3705        Ok(Some((total, rows)))
3706    }
3707
3708    /// The global dictionary of a column, opened once however many workers ask for it at once.
3709    ///
3710    /// The unlocked look is first because it is the answer every time after the first and it costs a
3711    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
3712    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
3713    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
3714    /// dictionary that can hold half a million entries, and the alternative is every worker of the
3715    /// scan doing all of it and all but one dropping the result on the floor.
3716    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
3717        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
3718        if let Some(dictionary) = self.dictionaries[column].get() {
3719            return Ok(Some(Arc::clone(dictionary)));
3720        }
3721        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
3722        if let Some(dictionary) = self.dictionaries[column].get() {
3723            return Ok(Some(Arc::clone(dictionary)));
3724        }
3725        self.opened.fetch_add(1, Atomic::Relaxed);
3726        let dictionary = Arc::new(open_global_dictionary(
3727            Arc::clone(&self.file),
3728            page,
3729            &self.table.fields[column].ty,
3730            TEXT_KEEP_BUDGET,
3731        )?);
3732        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
3733        Ok(Some(dictionary))
3734    }
3735
3736    /// Reads one section's extent table and checks it against the entry that names it.
3737    ///
3738    /// # Errors
3739    ///
3740    /// If the entry points outside the file, the table does not checksum, or it does not decode as
3741    /// a run of extents in element order.
3742    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
3743        if of.extent_bytes == 0 {
3744            return Ok(Vec::new());
3745        }
3746        let mut bytes = vec![0; of.extent_bytes as usize];
3747        read_at(&self.file, of.extent_page, &mut bytes)?;
3748        if checksum(&bytes) != of.hash {
3749            return Err(invalid("a section's extent table does not checksum"));
3750        }
3751        let extents = section::decode_extents(&bytes)?;
3752        if extents.len() != of.extents as usize {
3753            return Err(invalid("a section's extent table is not the length the entry says"));
3754        }
3755        Ok(extents)
3756    }
3757
3758    /// Reads and verifies one extent of a section.
3759    ///
3760    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
3761    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
3762    /// difference between a structure that works at SF100 and issue #745.
3763    ///
3764    /// # Errors
3765    ///
3766    /// If the extent points outside the file, or its bytes do not checksum.
3767    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
3768        let end = of
3769            .offset
3770            .checked_add(u64::from(of.length))
3771            .ok_or_else(|| invalid("an extent overflows the file"))?;
3772        if of.offset < HEADER || end > self.size {
3773            return Err(invalid("an extent is outside the file"));
3774        }
3775        let mut bytes = vec![0; of.length as usize];
3776        read_at(&self.file, of.offset, &mut bytes)?;
3777        if checksum(&bytes) != of.hash {
3778            return Err(invalid("an extent does not checksum"));
3779        }
3780        Ok(bytes)
3781    }
3782
3783    /// Reads a whole section's payload, every extent of it, in order.
3784    ///
3785    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
3786    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
3787    ///
3788    /// # Errors
3789    ///
3790    /// If the extent table or any extent fails its check.
3791    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
3792        let extents = self.extents(of)?;
3793        let mut bytes =
3794            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
3795        for one in &extents {
3796            if one.first != bytes.len() as u64 {
3797                return Err(invalid("a section's extents do not join up"));
3798            }
3799            bytes.extend_from_slice(&self.extent(one)?);
3800        }
3801        if of.header_bytes as usize > bytes.len() {
3802            return Err(invalid("a section's header is longer than its payload"));
3803        }
3804        Ok(bytes)
3805    }
3806
3807    /// Reads only the named columns from one part.
3808    ///
3809    /// The whole stripe page each column lives in is read and kept, because a scan asks for the
3810    /// parts of a stripe one after another and this is what turns sixty four reads into one.
3811    ///
3812    /// # Errors
3813    ///
3814    /// If a part, column, page, or checksum is invalid.
3815    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3816        self.read_impl(part, columns, true)
3817    }
3818
3819    /// Reads named columns from one part without keeping the stripe page it came out of.
3820    ///
3821    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
3822    /// a stripe rather than all of them. A caller that will read most of a stripe should use
3823    /// [`Self::read`] instead, because this reads and discards the page index every time.
3824    ///
3825    /// # Errors
3826    ///
3827    /// If a part, column, page, or checksum is invalid.
3828    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3829        self.read_impl(part, columns, false)
3830    }
3831
3832    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
3833    /// contain any of the sorted candidate codes.
3834    ///
3835    /// # Errors
3836    ///
3837    /// If the part, column, index page, checksum, or delta stream is invalid.
3838    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
3839        if candidates.is_empty() {
3840            return Ok(true);
3841        }
3842        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
3843            return Err(Error::internal("native code candidates are not sorted and unique"));
3844        }
3845        let stripe = self.stripe_of(part)?;
3846        let Some(page) = stripe.memberships.get(column).copied().flatten() else {
3847            return Ok(false);
3848        };
3849        let mut bytes = vec![0; page.length as usize];
3850        read_at(&self.file, page.offset, &mut bytes)?;
3851        if checksum(&bytes) != page.hash {
3852            return Err(invalid("membership page checksum differs"));
3853        }
3854        let codes = decode_membership(&bytes)?;
3855        let mut left = 0;
3856        let mut right = 0;
3857        while left < codes.len() && right < candidates.len() {
3858            match codes[left].cmp(&candidates[right]) {
3859                Ordering::Less => left += 1,
3860                Ordering::Greater => right += 1,
3861                Ordering::Equal => return Ok(false),
3862            }
3863        }
3864        Ok(true)
3865    }
3866
3867    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
3868        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
3869        self.table
3870            .stripes
3871            .get(place.stripe as usize)
3872            .ok_or_else(|| invalid("stripe index out of range"))
3873    }
3874
3875    /// The page index of one column of one stripe, and its page when the caller wants all of it.
3876    ///
3877    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
3878    /// a few parts of the others and they all want the same page at the same moment. This used to
3879    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
3880    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
3881    /// look at 400 MB of column.
3882    ///
3883    /// A worker that finds the page it wants already being read neither waits for it nor reads it
3884    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
3885    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
3886    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
3887    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
3888    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
3889    ///
3890    /// The file is never read under the lock.
3891    fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
3892        let cache = self.cache.get(column).ok_or_else(|| invalid("column index out of range"))?;
3893        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3894        let known = cached.index.get(at).and_then(Clone::clone);
3895        let page = cached.pages.get(at).and_then(Clone::clone);
3896        if let Some(index) = known.clone() {
3897            if !whole || page.is_some() {
3898                return Ok(CachedColumn { stripe: at, index, page });
3899            }
3900        }
3901        if cached.loading.contains(&at) {
3902            drop(cached);
3903            // The index is almost always already here, because somebody read this stripe to get
3904            // into the loading list in the first place, so this branch usually costs no read at
3905            // all and the one part read in `read_impl` is all the losing worker pays for.
3906            if let Some(index) = known {
3907                return Ok(CachedColumn { stripe: at, index, page: None });
3908            }
3909            let held = self.page_of(stripe, column, at, false, None)?;
3910            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3911            remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3912            return Ok(held);
3913        }
3914        cached.loading.push(at);
3915        drop(cached);
3916
3917        let read = self.page_of(stripe, column, at, whole, known);
3918
3919        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
3920        // them separately would leave a moment where another worker sees neither and reads the
3921        // page a second time, which is the whole thing this is here to stop.
3922        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3923        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
3924            cached.loading.remove(position);
3925        }
3926        let held = read?;
3927        remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3928        Ok(held)
3929    }
3930
3931    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
3932    ///
3933    /// `known` is the index when the reader has already read it, which after the first worker
3934    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
3935    /// reader. Without that a scan reads the index again on every part that misses the page cache.
3936    fn page_of(
3937        &self,
3938        stripe: &Stripe,
3939        column: usize,
3940        at: usize,
3941        whole: bool,
3942        known: Option<Arc<Vec<PartSpan>>>,
3943    ) -> Result<CachedColumn> {
3944        let index = match known {
3945            Some(index) => index,
3946            None => {
3947                self.indexes.fetch_add(1, Atomic::Relaxed);
3948                Arc::new(read_index(&self.file, stripe, column)?)
3949            }
3950        };
3951        let page = if whole {
3952            self.pages.fetch_add(1, Atomic::Relaxed);
3953            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3954            let mut bytes = vec![0; span.length as usize];
3955            read_at(&self.file, span.offset, &mut bytes)?;
3956            Some(Arc::new(bytes))
3957        } else {
3958            None
3959        };
3960        Ok(CachedColumn { stripe: at, index, page })
3961    }
3962
3963    fn read_impl(&self, at: usize, columns: &[usize], whole: bool) -> Result<Chunk> {
3964        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
3965        let index = place.stripe as usize;
3966        let stripe =
3967            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
3968        let rows = place.rows as usize;
3969        let mut picked = Vec::with_capacity(columns.len());
3970        for &column in columns {
3971            let field = self
3972                .table
3973                .fields
3974                .get(column)
3975                .ok_or_else(|| invalid("column index out of range"))?;
3976            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3977            let held = self.held(index, stripe, column, whole)?;
3978            let span = *held
3979                .index
3980                .get(place.part as usize)
3981                .ok_or_else(|| invalid("part index out of range"))?;
3982            let owned;
3983            let bytes = match &held.page {
3984                Some(held) => part_bytes(held, span)?,
3985                None => {
3986                    let offset = page
3987                        .offset
3988                        .checked_add(span.start as u64)
3989                        .ok_or_else(|| invalid("part range overflow"))?;
3990                    let mut bytes = vec![0; span.length];
3991                    read_at(&self.file, offset, &mut bytes)?;
3992                    owned = bytes;
3993                    &owned
3994                }
3995            };
3996            if checksum(bytes) != span.hash {
3997                return Err(invalid(&format!(
3998                    "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
3999                     wanted {:016x} and got {:016x}",
4000                    place.part,
4001                    page.offset,
4002                    span.start,
4003                    span.length,
4004                    span.hash,
4005                    checksum(bytes),
4006                )));
4007            }
4008            let dictionary = self.dictionary(column)?;
4009            // Held as a page, because a column that came out of a file is handed out more than
4010            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
4011            // projection of a bare column name does the same, and a cut of a flat run copies unless
4012            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
4013            // run into the `Arc` without touching a value.
4014            picked.push(decode(&field.ty, rows, bytes, dictionary)?.into_pages());
4015        }
4016        Chunk::with_rows(picked, rows)
4017    }
4018
4019    /// Whether persisted statistics prove that a part cannot match the predicates.
4020    ///
4021    /// Three of them, asked cheapest first.
4022    ///
4023    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
4024    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
4025    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
4026    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
4027    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
4028    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
4029    /// really hold the value.
4030    ///
4031    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
4032    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
4033    /// and the part bounds leave thirty parts of nine hundred and seventy four.
4034    #[must_use]
4035    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
4036        let Some(place) = self.places.get(part).copied() else { return false };
4037        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
4038        if stripe.zone.skips(probes) {
4039            return true;
4040        }
4041        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
4042    }
4043
4044    /// Whether the bounds of one part rule out one probe.
4045    ///
4046    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
4047    /// time this is asked about a column. A column with no page here answers `false`, which is the
4048    /// answer a caller got before there were any.
4049    fn outside(&self, place: Place, probe: &Probe) -> bool {
4050        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
4051            Some(ranges) => ranges
4052                .get(place.part as usize)
4053                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
4054            None => false,
4055        }
4056    }
4057
4058    /// The per part ranges of one stripe of one column, read once and kept.
4059    ///
4060    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
4061    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
4062    /// cannot read one reads the rows and gets the right answer slowly.
4063    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
4064        let slot = self.part_ranges.get(column)?.get(stripe)?;
4065        if let Some(held) = slot.get() {
4066            return Some(held);
4067        }
4068        let page = self.table.stripes.get(stripe)?.part_ranges.get(column).copied().flatten()?;
4069        let mut bytes = vec![0; page.length as usize];
4070        read_at(&self.file, page.offset, &mut bytes).ok()?;
4071        if checksum(&bytes) != page.hash {
4072            return None;
4073        }
4074        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
4075        let _ = slot.set(ranges);
4076        slot.get().map(|held| held.as_slice())
4077    }
4078
4079    /// Whether persisted statistics prove that every row of a part matches the predicates.
4080    ///
4081    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
4082    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
4083    /// through.
4084    ///
4085    /// The stripe first and the part after it, the same two steps and in the same order as
4086    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
4087    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
4088    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
4089    /// stretch where everything passes contains no narrower stretch where something fails, and a
4090    /// stripe with no nulls has no nulls in any of its parts.
4091    ///
4092    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
4093    /// wider than its rows really are as well. That is the same safe direction for the same reason,
4094    /// and it is why this asks the two ends rather than anything `exact` says.
4095    #[must_use]
4096    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
4097        let Some(place) = self.places.get(part).copied() else { return false };
4098        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
4099        if stripe.zone.certain(probes) {
4100            return true;
4101        }
4102        probes
4103            .iter()
4104            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
4105    }
4106
4107    /// Whether one part's own two ends prove that every row of it passes `probe`.
4108    ///
4109    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
4110    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
4111    /// part's and the caller has already asked them.
4112    fn inside(&self, place: Place, probe: &Probe) -> bool {
4113        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
4114            Some(ranges) => ranges
4115                .get(place.part as usize)
4116                .is_some_and(|range| range.certain(probe.op, &probe.value)),
4117            None => false,
4118        }
4119    }
4120
4121    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
4122    ///
4123    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
4124    /// directory and are already in memory, so this answers without touching the file, and that is
4125    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
4126    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
4127    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
4128    ///
4129    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
4130    /// it to be wrong: the parts are still checked when they are read.
4131    #[must_use]
4132    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
4133        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
4134    }
4135
4136    /// Whether the sieve of one part rules out one probe.
4137    ///
4138    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
4139    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
4140    /// sieve gets anyway.
4141    fn sifted(&self, place: Place, probe: &Probe) -> bool {
4142        if probe.op != Op::Equal {
4143            return false;
4144        }
4145        match self.stripe_sieves(place.stripe as usize, probe.column) {
4146            Some(sieves) => sieves
4147                .get(place.part as usize)
4148                .and_then(Option::as_ref)
4149                .is_some_and(|sieve| sieve.excludes(&probe.value)),
4150            None => false,
4151        }
4152    }
4153
4154    /// The sieves of one stripe of one column, read once and kept.
4155    ///
4156    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
4157    /// bytes are not a page this version can read. A sieve is an index over data that is still there
4158    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
4159    /// a bad checksum is a slow query rather than an error.
4160    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
4161        let slot = self.sieves.get(column)?.get(stripe)?;
4162        if let Some(held) = slot.get() {
4163            return Some(held);
4164        }
4165        let page = self.table.stripes.get(stripe)?.sieves.get(column).copied().flatten()?;
4166        let mut bytes = vec![0; page.length as usize];
4167        read_at(&self.file, page.offset, &mut bytes).ok()?;
4168        if checksum(&bytes) != page.hash {
4169            return None;
4170        }
4171        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
4172        let _ = slot.set(sieves);
4173        slot.get().map(|held| held.as_slice())
4174    }
4175}
4176
4177/// The value sitting at one position of a dictionary's sorted order.
4178fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
4179    let code = dictionary.code_at_rank(rank)? as usize;
4180    let text = dictionary
4181        .try_text_at(code)?
4182        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4183    Ok(Value::Varchar(text.into()))
4184}
4185
4186/// Writes one span of a file at an offset, without depending on where the cursor is.
4187///
4188/// The writer owns an offset of its own and passes it in here, so that nothing it writes depends on
4189/// a cursor that a read is entitled to move. Both of these can come back short and both loop.
4190#[cfg(unix)]
4191fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4192    use std::os::unix::fs::FileExt;
4193    while !bytes.is_empty() {
4194        let written = file.write_at(bytes, offset).map_err(io)?;
4195        if written == 0 {
4196            return Err(invalid("a write to the native file wrote nothing"));
4197        }
4198        offset += written as u64;
4199        bytes = &bytes[written..];
4200    }
4201    Ok(())
4202}
4203
4204/// The same write, on the call Windows spells differently.
4205#[cfg(windows)]
4206fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4207    use std::os::windows::fs::FileExt;
4208    while !bytes.is_empty() {
4209        let written = file.seek_write(bytes, offset).map_err(io)?;
4210        if written == 0 {
4211            return Err(invalid("a write to the native file wrote nothing"));
4212        }
4213        offset += written as u64;
4214        bytes = &bytes[written..];
4215    }
4216    Ok(())
4217}
4218
4219/// Somewhere that is neither, where the cursor is all there is.
4220#[cfg(not(any(unix, windows)))]
4221fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
4222    use std::io::Write;
4223    let mut file = file.try_clone().map_err(io)?;
4224    file.seek(SeekFrom::Start(offset)).map_err(io)?;
4225    file.write_all(bytes).map_err(io)
4226}
4227
4228/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
4229///
4230/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
4231/// pages from several threads at once, so this has to be positional. Seeking and then reading is
4232/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
4233/// comes back with somebody else's bytes.
4234///
4235/// Both of these can come back short, so both loop. A read of zero bytes before the span is filled
4236/// means the file stops earlier than the directory said it does.
4237#[cfg(unix)]
4238fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4239    use std::os::unix::fs::FileExt;
4240    while !bytes.is_empty() {
4241        let read = file.read_at(bytes, offset).map_err(io)?;
4242        if read == 0 {
4243            return Err(invalid("column page ends before its declared length"));
4244        }
4245        offset += read as u64;
4246        bytes = &mut bytes[read..];
4247    }
4248    Ok(())
4249}
4250
4251/// The same read, on the call Windows spells differently.
4252///
4253/// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave the
4254/// way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is why
4255/// nothing in this file may read that cursor.
4256#[cfg(windows)]
4257fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4258    use std::os::windows::fs::FileExt;
4259    while !bytes.is_empty() {
4260        let read = file.seek_read(bytes, offset).map_err(io)?;
4261        if read == 0 {
4262            return Err(invalid("column page ends before its declared length"));
4263        }
4264        offset += read as u64;
4265        bytes = &mut bytes[read..];
4266    }
4267    Ok(())
4268}
4269
4270/// Somewhere that is neither, where the cursor is all there is.
4271///
4272/// This one does race, and there is no way to write it so it does not. Nothing we build for runs
4273/// here, so it exists to keep the crate compiling rather than to be correct under threads.
4274#[cfg(not(any(unix, windows)))]
4275fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
4276    let mut file = file.try_clone().map_err(io)?;
4277    file.seek(SeekFrom::Start(offset)).map_err(io)?;
4278    file.read_exact(bytes).map_err(io)
4279}
4280
4281/// What a column type is called in the directory.
4282///
4283/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
4284/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
4285/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
4286/// rather than in an order that means anything.
4287fn type_tag(ty: &LogicalType) -> Result<u8> {
4288    match ty {
4289        LogicalType::SmallInt => Ok(1),
4290        LogicalType::Integer => Ok(2),
4291        LogicalType::BigInt => Ok(3),
4292        LogicalType::Varchar => Ok(4),
4293        LogicalType::Date => Ok(5),
4294        LogicalType::Timestamp => Ok(6),
4295        LogicalType::Boolean => Ok(7),
4296        LogicalType::TinyInt => Ok(8),
4297        LogicalType::UTinyInt => Ok(9),
4298        LogicalType::USmallInt => Ok(10),
4299        LogicalType::UInteger => Ok(11),
4300        LogicalType::UBigInt => Ok(12),
4301        LogicalType::Decimal { .. } => Ok(13),
4302        LogicalType::Float => Ok(14),
4303        LogicalType::Double => Ok(15),
4304        LogicalType::HugeInt => Ok(16),
4305        LogicalType::UHugeInt => Ok(17),
4306        LogicalType::Time => Ok(18),
4307        LogicalType::TimeTz => Ok(19),
4308        LogicalType::TimestampTz => Ok(20),
4309        LogicalType::Interval => Ok(21),
4310        LogicalType::Uuid => Ok(22),
4311        LogicalType::Blob => Ok(23),
4312        LogicalType::Bit => Ok(24),
4313        LogicalType::TimestampS => Ok(25),
4314        LogicalType::TimestampMs => Ok(26),
4315        LogicalType::TimestampNs => Ok(27),
4316        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
4317    }
4318}
4319
4320/// The tag of a column type, and the parameters of the ones that have any.
4321///
4322/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
4323/// because they are what says how wide a value is on disk, and a reader that guessed would read the
4324/// wrong number of bytes per row rather than the wrong number of digits.
4325fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
4326    out.push(type_tag(ty)?);
4327    if let LogicalType::Decimal { width, scale } = ty {
4328        out.push(*width);
4329        out.push(*scale);
4330    }
4331    Ok(())
4332}
4333
4334/// The other half of [`put_type`], reading the parameters the tag says are there.
4335fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
4336    let tag = cur.u8()?;
4337    if tag == 13 {
4338        let width = cur.u8()?;
4339        let scale = cur.u8()?;
4340        return LogicalType::decimal(width, scale)
4341            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
4342    }
4343    tag_type(tag)
4344}
4345
4346fn tag_type(tag: u8) -> Result<LogicalType> {
4347    match tag {
4348        1 => Ok(LogicalType::SmallInt),
4349        2 => Ok(LogicalType::Integer),
4350        3 => Ok(LogicalType::BigInt),
4351        4 => Ok(LogicalType::Varchar),
4352        5 => Ok(LogicalType::Date),
4353        6 => Ok(LogicalType::Timestamp),
4354        7 => Ok(LogicalType::Boolean),
4355        8 => Ok(LogicalType::TinyInt),
4356        9 => Ok(LogicalType::UTinyInt),
4357        10 => Ok(LogicalType::USmallInt),
4358        11 => Ok(LogicalType::UInteger),
4359        12 => Ok(LogicalType::UBigInt),
4360        14 => Ok(LogicalType::Float),
4361        15 => Ok(LogicalType::Double),
4362        16 => Ok(LogicalType::HugeInt),
4363        17 => Ok(LogicalType::UHugeInt),
4364        18 => Ok(LogicalType::Time),
4365        19 => Ok(LogicalType::TimeTz),
4366        20 => Ok(LogicalType::TimestampTz),
4367        21 => Ok(LogicalType::Interval),
4368        22 => Ok(LogicalType::Uuid),
4369        23 => Ok(LogicalType::Blob),
4370        24 => Ok(LogicalType::Bit),
4371        25 => Ok(LogicalType::TimestampS),
4372        26 => Ok(LogicalType::TimestampMs),
4373        27 => Ok(LogicalType::TimestampNs),
4374        _ => Err(invalid("column type tag is unknown")),
4375    }
4376}
4377
4378fn put_u16(out: &mut Vec<u8>, value: u16) {
4379    out.extend_from_slice(&value.to_le_bytes());
4380}
4381fn put_u32(out: &mut Vec<u8>, value: u32) {
4382    out.extend_from_slice(&value.to_le_bytes());
4383}
4384fn put_u64(out: &mut Vec<u8>, value: u64) {
4385    out.extend_from_slice(&value.to_le_bytes());
4386}
4387fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
4388    while value >= 0x80 {
4389        out.push((value as u8 & 0x7f) | 0x80);
4390        value >>= 7;
4391    }
4392    out.push(value as u8);
4393}
4394
4395fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
4396    match (left, right) {
4397        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
4398        (FrequencyValue::Null, _) => Ordering::Less,
4399        (_, FrequencyValue::Null) => Ordering::Greater,
4400        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
4401        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
4402        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
4403        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
4404    }
4405}
4406
4407fn code_frequency(dictionary: &GlobalDictionary) -> FrequencySummary {
4408    let mut entries = dictionary
4409        .counts
4410        .iter()
4411        .enumerate()
4412        .filter(|(_, count)| **count != 0)
4413        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
4414        .collect::<Vec<_>>();
4415    if dictionary.nulls != 0 {
4416        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
4417    }
4418    entries.sort_unstable_by(|left, right| {
4419        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
4420    });
4421    let omitted_max = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
4422    entries.truncate(FREQUENCY_ENTRIES);
4423    FrequencySummary { entries, omitted_max, ordinals: Vec::new() }
4424}
4425
4426fn encode_directory(table: &Table) -> Result<Vec<u8>> {
4427    let mut out = DIRECTORY.to_vec();
4428    let name = table.name.as_bytes();
4429    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4430    out.extend_from_slice(name);
4431    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
4432    for field in &table.fields {
4433        let name = field.name.as_bytes();
4434        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
4435        out.extend_from_slice(name);
4436        put_type(&mut out, &field.ty)?;
4437        out.push(u8::from(field.not_null));
4438    }
4439    for dictionary in &table.dictionaries {
4440        match dictionary {
4441            None => out.push(0),
4442            Some(page) => {
4443                out.push(1);
4444                put_u64(&mut out, page.offset);
4445                put_u32(&mut out, page.length);
4446                put_u64(&mut out, page.hash);
4447            }
4448        }
4449    }
4450    for distinct in &table.distincts {
4451        match distinct {
4452            None => out.push(0),
4453            Some(count) => {
4454                out.push(1);
4455                put_u64(&mut out, *count);
4456            }
4457        }
4458    }
4459    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
4460    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
4461    for stripe in &table.stripes {
4462        put_u32(
4463            &mut out,
4464            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
4465        );
4466        for &rows in &stripe.parts {
4467            put_u32(&mut out, rows);
4468        }
4469        put_u64(&mut out, stripe.index.offset);
4470        put_u32(&mut out, stripe.index.length);
4471        for page in &stripe.pages {
4472            put_u64(&mut out, page.offset);
4473            put_u32(&mut out, page.length);
4474        }
4475        // A membership index says which of a dictionary's codes a part holds, so a column the writer
4476        // decided against giving a dictionary has nothing for it to be about and writes none. Every
4477        // file written before that decision existed has a dictionary on every varchar column, so
4478        // this reads those files byte for byte the way it always did.
4479        for ((field, dictionary), membership) in
4480            table.fields.iter().zip(&table.dictionaries).zip(&stripe.memberships)
4481        {
4482            if field.ty != LogicalType::Varchar || dictionary.is_none() {
4483                continue;
4484            }
4485            let page =
4486                membership.ok_or_else(|| invalid("string page has no code membership index"))?;
4487            put_u64(&mut out, page.offset);
4488            put_u32(&mut out, page.length);
4489            put_u64(&mut out, page.hash);
4490        }
4491        for sieve in &stripe.sieves {
4492            match sieve {
4493                None => out.push(0),
4494                Some(page) => {
4495                    out.push(1);
4496                    put_u64(&mut out, page.offset);
4497                    put_u32(&mut out, page.length);
4498                    put_u64(&mut out, page.hash);
4499                }
4500            }
4501        }
4502        for held in &stripe.part_ranges {
4503            match held {
4504                None => out.push(0),
4505                Some(page) => {
4506                    out.push(1);
4507                    put_u64(&mut out, page.offset);
4508                    put_u32(&mut out, page.length);
4509                    put_u64(&mut out, page.hash);
4510                }
4511            }
4512        }
4513        for range in stripe.zone.columns() {
4514            put_bound(&mut out, range.low.as_ref())?;
4515            put_bound(&mut out, range.high.as_ref())?;
4516            put_u32(
4517                &mut out,
4518                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
4519            );
4520            out.push(u8::from(range.exact));
4521            match range.sum {
4522                None => out.push(0),
4523                Some(total) => {
4524                    out.push(1);
4525                    out.extend_from_slice(&total.to_le_bytes());
4526                }
4527            }
4528        }
4529    }
4530    out.extend_from_slice(FREQUENCIES);
4531    put_u16(
4532        &mut out,
4533        u16::try_from(table.frequencies.len())
4534            .map_err(|_| invalid("too many frequency columns"))?,
4535    );
4536    for summary in &table.frequencies {
4537        let Some(summary) = summary else {
4538            out.push(0);
4539            continue;
4540        };
4541        out.push(1);
4542        put_u64(&mut out, summary.omitted_max);
4543        put_u32(
4544            &mut out,
4545            u32::try_from(summary.entries.len())
4546                .map_err(|_| invalid("too many frequency entries"))?,
4547        );
4548        for entry in &summary.entries {
4549            match entry.value {
4550                FrequencyValue::Null => out.push(0),
4551                FrequencyValue::Integer(value) => {
4552                    out.push(1);
4553                    out.extend_from_slice(&value.to_le_bytes());
4554                }
4555                FrequencyValue::Code(value) => {
4556                    out.push(2);
4557                    put_u32(&mut out, value);
4558                }
4559            }
4560            put_u64(&mut out, entry.count);
4561        }
4562        put_u32(
4563            &mut out,
4564            u32::try_from(summary.ordinals.len())
4565                .map_err(|_| invalid("too many frequency ordinals"))?,
4566        );
4567        let mut previous = 0_u64;
4568        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
4569            let delta = if at == 0 {
4570                ordinal
4571            } else {
4572                ordinal
4573                    .checked_sub(previous)
4574                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
4575            };
4576            if at != 0 && delta == 0 {
4577                return Err(invalid("frequency ordinals are not unique"));
4578            }
4579            put_var_u64(&mut out, delta);
4580            previous = ordinal;
4581        }
4582    }
4583    // Written only when there is a declaration, so that the common file is the same bytes it was
4584    // and the section is not a byte of zero on every table in the world that never asked for one.
4585    if let Some(clustering) = &table.clustering {
4586        out.extend_from_slice(CLUSTERING);
4587        out.push(clustering.width().tag());
4588        put_u16(
4589            &mut out,
4590            u16::try_from(clustering.columns().len())
4591                .map_err(|_| invalid("too many clustering columns"))?,
4592        );
4593        for &column in clustering.columns() {
4594            put_u16(
4595                &mut out,
4596                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
4597            );
4598        }
4599    }
4600    // The section table, last, behind its own magic, for the same reason the frequency block is
4601    // behind its own: a reader that stops before it gets a table with no sections, and a table with
4602    // no sections is a correct table. The one difference from the blocks before it is that this one
4603    // is written even when it is empty, so that a file written by this build always says which
4604    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
4605    out.extend_from_slice(SECTIONS);
4606    put_u64(&mut out, table.generation);
4607    put_u16(
4608        &mut out,
4609        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
4610    );
4611    for held in &table.sections {
4612        held.encode(&mut out)?;
4613    }
4614    Ok(out)
4615}
4616
4617/// The small level of the directory, naming every table in the file.
4618///
4619/// This is what a footer slot points at. Each entry carries its own checksum over its table
4620/// directory, so a table whose directory is torn is found when that table is first touched rather
4621/// than being trusted because the catalog around it checksummed.
4622///
4623/// The views go after the tables and are whole here, since a view is text and a column list and has
4624/// no pages for a second level to point at.
4625fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
4626    let mut out = CATALOG.to_vec();
4627    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
4628    for entry in entries {
4629        let name = entry.name.as_bytes();
4630        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4631        out.extend_from_slice(name);
4632        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
4633        put_u16(
4634            &mut out,
4635            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
4636        );
4637        for field in &entry.fields {
4638            let name = field.name.as_bytes();
4639            put_u16(
4640                &mut out,
4641                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
4642            );
4643            out.extend_from_slice(name);
4644            put_type(&mut out, &field.ty)?;
4645            out.push(u8::from(field.not_null));
4646        }
4647        put_u64(&mut out, entry.directory.offset);
4648        put_u32(&mut out, entry.directory.length);
4649        put_u64(&mut out, entry.directory.hash);
4650    }
4651    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
4652    for view in views {
4653        let name = view.name.as_bytes();
4654        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
4655        out.extend_from_slice(name);
4656        put_long_text(&mut out, &view.sql, "view body")?;
4657        put_long_text(&mut out, &view.statement, "view statement")?;
4658        put_u16(
4659            &mut out,
4660            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
4661        );
4662        for alias in &view.aliases {
4663            let alias = alias.as_bytes();
4664            put_u16(
4665                &mut out,
4666                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
4667            );
4668            out.extend_from_slice(alias);
4669        }
4670        put_u16(
4671            &mut out,
4672            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
4673        );
4674        for field in &view.columns {
4675            let name = field.name.as_bytes();
4676            put_u16(
4677                &mut out,
4678                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
4679            );
4680            out.extend_from_slice(name);
4681            put_type(&mut out, &field.ty)?;
4682            out.push(u8::from(field.not_null));
4683        }
4684    }
4685    Ok(out)
4686}
4687
4688/// A length and that many bytes, for text that is allowed to be longer than a name.
4689fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
4690    let bytes = text.as_bytes();
4691    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
4692    out.extend_from_slice(bytes);
4693    Ok(())
4694}
4695
4696/// Reads the catalog directory back, checking every span against the file before anything is
4697/// allocated for it.
4698fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
4699    let mut cur = Cursor { bytes, at: 0 };
4700    if cur.take(8)? != CATALOG {
4701        return Err(invalid("catalog magic differs"));
4702    }
4703    let count = cur.u32()? as usize;
4704    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
4705    for _ in 0..count {
4706        let name = cur.text()?;
4707        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4708        let width = cur.u16()? as usize;
4709        let mut fields = Vec::with_capacity(width);
4710        for _ in 0..width {
4711            let name = cur.text()?;
4712            let ty = read_type(&mut cur)?;
4713            let not_null = match cur.u8()? {
4714                0 => false,
4715                1 => true,
4716                _ => return Err(invalid("nullability flag differs")),
4717            };
4718            fields.push(Field { name, ty, not_null });
4719        }
4720        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4721        let end = directory
4722            .offset
4723            .checked_add(u64::from(directory.length))
4724            .ok_or_else(|| invalid("table directory offset overflow"))?;
4725        if directory.offset < HEADER
4726            || end > size
4727            || directory.length as usize > MAX_DIRECTORY
4728            || directory.length == 0
4729        {
4730            return Err(invalid("table directory range is outside the file"));
4731        }
4732        if entries.iter().any(|held| held.name == name) {
4733            return Err(invalid("two tables in the catalog have the same name"));
4734        }
4735        entries.push(Entry { name, fields, rows, directory });
4736    }
4737    // A catalog that ends where the tables end is a catalog with no views in it, which is every
4738    // file written before format 25. That is why the count is allowed to be missing rather than
4739    // read as a zero that has to be there: an older file has nothing after the last table entry at
4740    // all, and [`READABLE`] says those files still open.
4741    let count = if cur.done() { 0 } else { cur.u32()? as usize };
4742    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
4743    for _ in 0..count {
4744        let name = cur.text()?;
4745        let sql = cur.long_text()?;
4746        let statement = cur.long_text()?;
4747        let width = cur.u16()? as usize;
4748        let mut aliases = Vec::with_capacity(width);
4749        for _ in 0..width {
4750            aliases.push(cur.text()?);
4751        }
4752        let width = cur.u16()? as usize;
4753        let mut columns = Vec::with_capacity(width);
4754        for _ in 0..width {
4755            let name = cur.text()?;
4756            let ty = read_type(&mut cur)?;
4757            let not_null = match cur.u8()? {
4758                0 => false,
4759                1 => true,
4760                _ => return Err(invalid("nullability flag differs")),
4761            };
4762            columns.push(Field { name, ty, not_null });
4763        }
4764        // The same rule the tables above get, and for the same reason. Two entries under one name
4765        // is a catalog nothing can answer a lookup from, and finding that out here is better than
4766        // finding it out from whichever of the two a search happened to reach first.
4767        if views.iter().any(|held| held.name == name) {
4768            return Err(invalid("two views in the catalog have the same name"));
4769        }
4770        if entries.iter().any(|held| held.name == name) {
4771            return Err(invalid("a table and a view in the catalog have the same name"));
4772        }
4773        views.push(ViewEntry { name, sql, statement, aliases, columns });
4774    }
4775    Ok((entries, views))
4776}
4777
4778struct Cursor<'a> {
4779    bytes: &'a [u8],
4780    at: usize,
4781}
4782impl<'a> Cursor<'a> {
4783    fn take(&mut self, len: usize) -> Result<&'a [u8]> {
4784        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
4785        let bytes =
4786            self.bytes.get(self.at..end).ok_or_else(|| invalid("directory is truncated"))?;
4787        self.at = end;
4788        Ok(bytes)
4789    }
4790    fn u8(&mut self) -> Result<u8> {
4791        Ok(self.take(1)?[0])
4792    }
4793    fn u16(&mut self) -> Result<u16> {
4794        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
4795    }
4796    fn u32(&mut self) -> Result<u32> {
4797        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
4798    }
4799    fn u64(&mut self) -> Result<u64> {
4800        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
4801    }
4802    fn var_u64(&mut self) -> Result<u64> {
4803        let mut value = 0_u64;
4804        for shift in (0..=63).step_by(7) {
4805            let byte = self.u8()?;
4806            let part = u64::from(byte & 0x7f);
4807            if shift == 63 && part > 1 {
4808                return Err(invalid("frequency ordinal varint overflows"));
4809            }
4810            value |= part << shift;
4811            if byte & 0x80 == 0 {
4812                return Ok(value);
4813            }
4814        }
4815        Err(invalid("frequency ordinal varint is too long"))
4816    }
4817    /// A zone map's end, in the layout `rudb_common::bounds` defines.
4818    ///
4819    /// The bytes are the ones this directory has written since format 10 and the codec moved to
4820    /// rank zero rather than being copied, because a column summary now writes the same two ends
4821    /// and two encodings of one type is how the two quietly stop agreeing.
4822    fn bound(&mut self) -> Result<Option<Bound>> {
4823        bounds::get(self.bytes, &mut self.at)
4824    }
4825    fn text(&mut self) -> Result<String> {
4826        let len = self.u16()? as usize;
4827        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
4828    }
4829    /// Whether everything has been read, which is how a section that an older file does not have at
4830    /// all is told from one that is there and empty.
4831    fn done(&self) -> bool {
4832        self.at >= self.bytes.len()
4833    }
4834    /// The same, for text that is a query rather than a name.
4835    ///
4836    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
4837    /// kilobyte identifier by accident and people do write generated queries that long, and a view
4838    /// that could not be written down because its body was too big would be a limit invented here
4839    /// rather than one anything else in the engine has.
4840    fn long_text(&mut self) -> Result<String> {
4841        let len = self.u32()? as usize;
4842        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
4843    }
4844}
4845
4846fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
4847    let mut cur = Cursor { bytes, at: 0 };
4848    if cur.take(8)? != DIRECTORY {
4849        return Err(invalid("directory magic differs"));
4850    }
4851    let name = cur.text()?;
4852    let width = cur.u16()? as usize;
4853    let mut fields = Vec::with_capacity(width);
4854    for _ in 0..width {
4855        let name = cur.text()?;
4856        let ty = read_type(&mut cur)?;
4857        let not_null = match cur.u8()? {
4858            0 => false,
4859            1 => true,
4860            _ => return Err(invalid("nullability flag differs")),
4861        };
4862        fields.push(Field { name, ty, not_null });
4863    }
4864    let mut dictionaries = Vec::with_capacity(width);
4865    for _ in 0..width {
4866        dictionaries.push(match cur.u8()? {
4867            0 => None,
4868            1 => {
4869                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4870                let end = page
4871                    .offset
4872                    .checked_add(u64::from(page.length))
4873                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
4874                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
4875                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
4876                // pages are capped there. `Writer::finish` has already bounded this length by the
4877                // on-disk `u32`, and the range check below keeps it inside the file.
4878                if page.offset < HEADER || end > size {
4879                    return Err(invalid("dictionary page range is outside the file"));
4880                }
4881                Some(page)
4882            }
4883            _ => return Err(invalid("dictionary page tag differs")),
4884        });
4885    }
4886    let mut distincts = Vec::with_capacity(width);
4887    for _ in 0..width {
4888        distincts.push(match cur.u8()? {
4889            0 => None,
4890            1 => Some(cur.u64()?),
4891            _ => return Err(invalid("distinct count tag differs")),
4892        });
4893    }
4894    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4895    let count = cur.u32()? as usize;
4896    let mut stripes = Vec::with_capacity(count);
4897    let mut total = 0_usize;
4898    for _ in 0..count {
4899        let count = cur.u32()? as usize;
4900        if count == 0 || count > STRIPE_PARTS {
4901            return Err(invalid("stripe part count is outside its bound"));
4902        }
4903        let mut parts = Vec::with_capacity(count);
4904        let mut stripe_rows = 0_usize;
4905        for _ in 0..count {
4906            let rows = cur.u32()?;
4907            if rows == 0 {
4908                return Err(invalid("empty part"));
4909            }
4910            parts.push(rows);
4911            stripe_rows = stripe_rows
4912                .checked_add(rows as usize)
4913                .ok_or_else(|| invalid("stripe row count overflow"))?;
4914        }
4915        total =
4916            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
4917        let index = Span { offset: cur.u64()?, length: cur.u32()? };
4918        let section = index_section(count)?;
4919        let wanted = section
4920            .checked_mul(width)
4921            .and_then(|bytes| u32::try_from(bytes).ok())
4922            .ok_or_else(|| invalid("index page length overflow"))?;
4923        let end = index
4924            .offset
4925            .checked_add(u64::from(index.length))
4926            .ok_or_else(|| invalid("index page offset overflow"))?;
4927        if index.offset < HEADER || end > size || index.length != wanted {
4928            return Err(invalid("index page range is outside the file"));
4929        }
4930        let mut pages = Vec::with_capacity(width);
4931        for _ in 0..width {
4932            let offset = cur.u64()?;
4933            let length = cur.u32()?;
4934            let end = offset
4935                .checked_add(u64::from(length))
4936                .ok_or_else(|| invalid("page offset overflow"))?;
4937            if offset < HEADER || end > size || length as usize > MAX_PAGE {
4938                return Err(invalid("page range is outside the file"));
4939            }
4940            pages.push(Span { offset, length });
4941        }
4942        let mut memberships = vec![None; width];
4943        for (column, field) in fields.iter().enumerate() {
4944            if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
4945                continue;
4946            }
4947            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4948            let end = page
4949                .offset
4950                .checked_add(u64::from(page.length))
4951                .ok_or_else(|| invalid("membership page offset overflow"))?;
4952            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4953                return Err(invalid("membership page range is outside the file"));
4954            }
4955            memberships[column] = Some(page);
4956        }
4957        let mut sieves = vec![None; width];
4958        for sieve in sieves.iter_mut().take(width) {
4959            match cur.u8()? {
4960                0 => continue,
4961                1 => {}
4962                _ => return Err(invalid("a sieve page has an unknown tag")),
4963            }
4964            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4965            let end = page
4966                .offset
4967                .checked_add(u64::from(page.length))
4968                .ok_or_else(|| invalid("sieve page offset overflow"))?;
4969            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4970                return Err(invalid("sieve page range is outside the file"));
4971            }
4972            *sieve = Some(page);
4973        }
4974        let mut part_ranges = vec![None; width];
4975        for held in part_ranges.iter_mut().take(width) {
4976            match cur.u8()? {
4977                0 => continue,
4978                1 => {}
4979                _ => return Err(invalid("a part range page has an unknown tag")),
4980            }
4981            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4982            let end = page
4983                .offset
4984                .checked_add(u64::from(page.length))
4985                .ok_or_else(|| invalid("part range page offset overflow"))?;
4986            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4987                return Err(invalid("part range page range is outside the file"));
4988            }
4989            *held = Some(page);
4990        }
4991        let mut ranges = Vec::with_capacity(width);
4992        for column in 0..width {
4993            let low = cur.bound()?;
4994            let high = cur.bound()?;
4995            let nulls = cur.u32()? as usize;
4996            if nulls > stripe_rows {
4997                return Err(invalid("null count exceeds stripe rows"));
4998            }
4999            let exact = cur.u8()? != 0;
5000            let sum = match cur.u8()? {
5001                0 => None,
5002                1 => Some(i128::from_le_bytes(
5003                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
5004                )),
5005                _ => return Err(invalid("a stripe sum has an unknown tag")),
5006            };
5007            // Files written before the ends of a decimal or a timestamp column carried their power
5008            // of ten hold a bare integer here, and that integer is the one the column holds, which
5009            // is what the power is over. So the type puts it back on the way in and an old file
5010            // prunes as well as a new one. A file that already wrote the power keeps it, because
5011            // this leaves anything that is not an integer alone.
5012            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
5013            let low = low.map(|bound| scaled_as(bound, ty));
5014            let high = high.map(|bound| scaled_as(bound, ty));
5015            ranges.push(Range { low, high, nulls, exact, sum });
5016        }
5017        stripes.push(Stripe {
5018            rows: stripe_rows,
5019            parts,
5020            index,
5021            pages,
5022            memberships,
5023            sieves,
5024            part_ranges,
5025            zone: Zone::from_ranges(ranges),
5026        });
5027    }
5028    if total != rows {
5029        return Err(invalid("table row count differs from stripes"));
5030    }
5031    let frequencies = if cur.at == bytes.len() {
5032        vec![None; width]
5033    } else {
5034        if cur.take(8)? != FREQUENCIES {
5035            return Err(invalid("directory extension magic differs"));
5036        }
5037        if cur.u16()? as usize != width {
5038            return Err(invalid("frequency column count differs"));
5039        }
5040        let mut frequencies = Vec::with_capacity(width);
5041        for field in &fields {
5042            let summary = match cur.u8()? {
5043                0 => None,
5044                1 => {
5045                    let omitted_max = cur.u64()?;
5046                    let count = cur.u32()? as usize;
5047                    if count > FREQUENCY_ENTRIES {
5048                        return Err(invalid("frequency entry count exceeds its bound"));
5049                    }
5050                    let mut entries = Vec::with_capacity(count);
5051                    // row at a time: directory decoding validates each persisted bounded frequency entry.
5052                    for _ in 0..count {
5053                        let value = match cur.u8()? {
5054                            0 => FrequencyValue::Null,
5055                            1 => FrequencyValue::Integer(i128::from_le_bytes(
5056                                cur.take(16)?.try_into().expect("sixteen bytes"),
5057                            )),
5058                            2 => FrequencyValue::Code(cur.u32()?),
5059                            _ => return Err(invalid("frequency value tag differs")),
5060                        };
5061                        let valid = matches!(
5062                            (&field.ty, value),
5063                            (_, FrequencyValue::Null)
5064                                | (LogicalType::Varchar, FrequencyValue::Code(_))
5065                                | (
5066                                    LogicalType::TinyInt
5067                                        | LogicalType::SmallInt
5068                                        | LogicalType::Integer
5069                                        | LogicalType::BigInt
5070                                        | LogicalType::UTinyInt
5071                                        | LogicalType::USmallInt
5072                                        | LogicalType::UInteger
5073                                        | LogicalType::UBigInt
5074                                        | LogicalType::Date
5075                                        | LogicalType::Timestamp,
5076                                    FrequencyValue::Integer(_),
5077                                )
5078                        );
5079                        if !valid {
5080                            return Err(invalid("frequency value does not match its column"));
5081                        }
5082                        let count = cur.u64()?;
5083                        if count == 0 || count > rows as u64 {
5084                            return Err(invalid("frequency count is outside the table"));
5085                        }
5086                        entries.push(FrequencyEntry { value, count });
5087                    }
5088                    if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
5089                        return Err(invalid("frequency entries are not descending"));
5090                    }
5091                    let ordinals = {
5092                        let ordinal_count = cur.u32()? as usize;
5093                        if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
5094                            return Err(invalid("frequency ordinal count exceeds its bound"));
5095                        }
5096                        let mut ordinals = Vec::with_capacity(ordinal_count);
5097                        let mut previous = 0_u64;
5098                        for at in 0..ordinal_count {
5099                            let delta = cur.var_u64()?;
5100                            if at != 0 && delta == 0 {
5101                                return Err(invalid("frequency ordinals are not increasing"));
5102                            }
5103                            let ordinal = if at == 0 {
5104                                delta
5105                            } else {
5106                                previous
5107                                    .checked_add(delta)
5108                                    .ok_or_else(|| invalid("frequency ordinal overflows"))?
5109                            };
5110                            if ordinal >= rows as u64 {
5111                                return Err(invalid("frequency ordinal is outside the table"));
5112                            }
5113                            ordinals.push(ordinal);
5114                            previous = ordinal;
5115                        }
5116                        ordinals
5117                    };
5118                    Some(FrequencySummary { entries, omitted_max, ordinals })
5119                }
5120                _ => return Err(invalid("frequency summary tag differs")),
5121            };
5122            frequencies.push(summary);
5123        }
5124        frequencies
5125    };
5126    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
5127    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
5128    // independently: a format 22 directory ends here and has neither, a directory written before
5129    // the section table has only the clustering declaration, and each one still opens without a
5130    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
5131    // a file that predates them and answers every query, only without the graph path.
5132    //
5133    // A repeated block is refused rather than allowed to win, because two clustering declarations
5134    // in one directory is a torn directory and the only question is which of them is the lie.
5135    let mut clustering = None;
5136    let mut sections = Vec::new();
5137    let mut seen_sections = false;
5138    // Zero until a section table says otherwise, which is what a format 22 table gets and what
5139    // makes every section stamp fail to match on one, because real generations start at one.
5140    let mut generation = 0;
5141    while cur.at != bytes.len() {
5142        let mut tag = [0u8; 8];
5143        tag.copy_from_slice(cur.take(8)?);
5144        if &tag == CLUSTERING {
5145            if clustering.is_some() {
5146                return Err(invalid("directory names two clustering declarations"));
5147            }
5148            let bucket = Width::from_tag(cur.u8()?)
5149                .ok_or_else(|| invalid("clustering width tag differs"))?;
5150            let count = cur.u16()? as usize;
5151            let mut columns = Vec::with_capacity(count.min(fields.len()));
5152            for _ in 0..count {
5153                columns.push(u32::from(cur.u16()?));
5154            }
5155            // Through the constructor and not built by hand, so that a file claiming a column the
5156            // table does not have is caught at open rather than at the first scan that trusted it.
5157            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
5158                invalid("stored clustering declaration does not match the table it is on")
5159            })?);
5160        } else if &tag == SECTIONS {
5161            if seen_sections {
5162                return Err(invalid("directory names two section tables"));
5163            }
5164            seen_sections = true;
5165            generation = cur.u64()?;
5166            let count = cur.u16()? as usize;
5167            if count > MAX_SECTIONS {
5168                return Err(invalid("section count exceeds its bound"));
5169            }
5170            sections = Vec::with_capacity(count);
5171            // entry at a time: a malformed section entry is refused rather than turned into an
5172            // offset.
5173            for _ in 0..count {
5174                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
5175            }
5176            for held in &sections {
5177                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
5178                    return Err(invalid("a section's extent table overflows the file"));
5179                };
5180                // The bound check is here and not in `section`, because only the caller knows how
5181                // big the file is. A section pointing past the end is a torn directory, and reading
5182                // the payload it names would be reading whatever else is at that offset.
5183                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
5184                    return Err(invalid("a section's extent table is outside the file"));
5185                }
5186                if held.extents == 0 && held.extent_bytes != 0 {
5187                    return Err(invalid("a section with no extents names an extent table"));
5188                }
5189            }
5190        } else {
5191            return Err(invalid("directory extension magic differs"));
5192        }
5193    }
5194    if cur.at != bytes.len() {
5195        return Err(invalid("directory has trailing bytes"));
5196    }
5197    Ok(Table {
5198        name,
5199        fields,
5200        stripes,
5201        rows,
5202        dictionaries,
5203        distincts,
5204        frequencies,
5205        clustering,
5206        generation,
5207        sections,
5208    })
5209}
5210
5211/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
5212fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
5213    bounds::put(out, bound)
5214}
5215
5216/// Which cascades are worth trying on a run of dictionary codes.
5217///
5218/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
5219/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
5220/// three candidates were always going to win. It is the right default for a crate that does not
5221/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
5222/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
5223/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
5224///
5225/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
5226/// already the dictionary, and it is also the most expensive one to try. Below the top level the
5227/// streams are an RLE's run values and run lengths, which are integers in their own right with no
5228/// runs left in them, so only the two flat candidates go down there.
5229///
5230/// This is size given up for time on purpose, and the ablation is this chooser against
5231/// [`chooser::EXHAUSTIVE`] on the same file.
5232#[derive(Debug)]
5233struct Codes;
5234
5235impl chooser::Chooser for Codes {
5236    fn name(&self) -> &'static str {
5237        "codes"
5238    }
5239
5240    fn narrow_strings(
5241        &self,
5242        _values: &[&[u8]],
5243        offered: &[string::Kind],
5244        _depth: u8,
5245    ) -> Vec<string::Kind> {
5246        // Never reached, because nothing here encodes strings through the cascade. The trait asks
5247        // for it and the honest answer to a question we have no opinion on is the whole list.
5248        offered.to_vec()
5249    }
5250
5251    fn narrow_integers(
5252        &self,
5253        _values: &[i64],
5254        offered: &[integer::Kind],
5255        depth: u8,
5256    ) -> Vec<integer::Kind> {
5257        let keep: &[integer::Kind] = if depth == 0 {
5258            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
5259        } else {
5260            &[integer::Kind::Constant, integer::Kind::Packed]
5261        };
5262        let narrowed: Vec<integer::Kind> =
5263            offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5264        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
5265        // this has no opinion about rather than one that cannot be written.
5266        if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5267    }
5268}
5269
5270/// Which cascades are worth trying on a part of plain integers.
5271///
5272/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
5273/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
5274/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
5275/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
5276/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
5277/// every value. A column that is one value with a handful of exceptions is sparse. What is still
5278/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
5279/// expensive candidate to try and this file already puts the columns that want one through a
5280/// dictionary of their own before they ever reach here.
5281#[derive(Debug)]
5282struct Fixed;
5283
5284impl chooser::Chooser for Fixed {
5285    fn name(&self) -> &'static str {
5286        "fixed"
5287    }
5288
5289    fn narrow_strings(
5290        &self,
5291        _values: &[&[u8]],
5292        offered: &[string::Kind],
5293        _depth: u8,
5294    ) -> Vec<string::Kind> {
5295        offered.to_vec()
5296    }
5297
5298    fn narrow_integers(
5299        &self,
5300        _values: &[i64],
5301        offered: &[integer::Kind],
5302        depth: u8,
5303    ) -> Vec<integer::Kind> {
5304        let keep: &[integer::Kind] = if depth == 0 {
5305            &[
5306                integer::Kind::Constant,
5307                integer::Kind::Packed,
5308                integer::Kind::Delta,
5309                integer::Kind::Rle,
5310                integer::Kind::Sparse,
5311                integer::Kind::Strided,
5312            ]
5313        } else {
5314            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
5315        };
5316        let narrowed: Vec<integer::Kind> =
5317            offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5318        if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5319    }
5320}
5321
5322/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
5323/// losing one.
5324///
5325/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
5326/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
5327/// integers and have their own ways of being small.
5328fn widened(data: &Data) -> Option<Vec<i64>> {
5329    match data {
5330        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5331        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5332        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5333        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5334        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5335        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5336        Data::Int64(values) => Some(values.to_vec()),
5337        _ => None,
5338    }
5339}
5340
5341/// An integer type a cascaded page can be read back into, and how to tell whether a value fits.
5342///
5343/// This exists so that the check and the conversion can be two loops instead of one. `TryFrom` puts
5344/// them together, which is the right shape for one value and the wrong one for a page: a fallible
5345/// conversion a value at a time is a branch a value at a time, the branch decides whether the loop
5346/// keeps going, and a loop like that is one no compiler will widen.
5347trait Narrow: Copy {
5348    /// How wide this type is, and what to add to a value to put its range at the bottom of a `u64`.
5349    ///
5350    /// Half the width for a signed type, which is what moves its smallest value to zero, and nothing
5351    /// for an unsigned one, whose smallest value is already there.
5352    const BIASED: (u32, u64);
5353
5354    /// The value narrowed, which the caller has already shown fits.
5355    fn narrow(value: i64) -> Self;
5356}
5357
5358/// The bits of `value` a `T` cannot hold, and zero when the value fits.
5359///
5360/// The question is asked this way round because the answers or together. A page fits when every
5361/// residue in it is zero, so the loop is an or into an accumulator and the decision is one test
5362/// after it, where asking whether each value is between a floor and a ceiling gives an answer that
5363/// does not combine and turns into a running minimum and maximum.
5364///
5365/// Biasing and shifting is what the answer is made of, rather than anything that reads more like the
5366/// question, because those are the operations a machine has four of. A 64 bit integer minimum is
5367/// AVX-512. So is a 64 bit arithmetic shift right, which is how the sign extension this could be
5368/// written as would have to be done. An add and a logical shift right are AVX2 and are on every
5369/// machine this runs on, so this is the form that gets four values a cycle instead of one.
5370///
5371/// Adding the bias moves the type's range to `0..=2^bits`, wrapping, so everything in range shifts
5372/// away to nothing and everything outside it leaves something behind. A negative value under an
5373/// unsigned type is caught by the same shift, because a negative `i64` read as a `u64` is enormous.
5374#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
5375fn residue<T: Narrow>(value: i64) -> u64 {
5376    let (bits, bias) = T::BIASED;
5377    (value as u64).wrapping_add(bias) >> bits
5378}
5379
5380/// Says a primitive integer narrows with `as`, and where the bottom of its range is.
5381///
5382/// `as` is a truncation and is the right operation here only because [`fit`] has already found every
5383/// residue zero, and it is what makes the second loop a narrowing store with no branch in it.
5384macro_rules! narrows {
5385    ($($ty:ty => $bias:expr),* $(,)?) => {$(
5386        impl Narrow for $ty {
5387            const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
5388
5389            #[allow(
5390                clippy::cast_possible_truncation,
5391                clippy::cast_sign_loss,
5392                reason = "the caller has checked the bits this truncates away"
5393            )]
5394            fn narrow(value: i64) -> Self {
5395                value as Self
5396            }
5397        }
5398    )*};
5399}
5400
5401narrows! {
5402    i8 => 1 << 7,
5403    u8 => 0,
5404    i16 => 1 << 15,
5405    u16 => 0,
5406    i32 => 1 << 31,
5407    u32 => 0,
5408}
5409
5410/// Narrows a page's values, refusing the page if any of them does not fit.
5411///
5412/// The check first and the conversion second, rather than a fallible conversion a value at a time.
5413/// Both loops here are ones a compiler widens: [`residue`] is three instructions a lane and a
5414/// narrowing store is one. The version before this was a `TryFrom` and a `collect` into a `Result`,
5415/// which is a compare, a branch and a short circuit a value at a time, and on ClickBench 39 it was
5416/// seven percent of the query. The version after that kept a running minimum and maximum, which is
5417/// the obvious way to ask and needs a 64 bit integer minimum that AVX2 does not have, so it stayed
5418/// a value at a time and was still ten percent of the same query.
5419///
5420/// An empty page has nothing to refuse, which falls out of the accumulator starting at zero rather
5421/// than needing a case of its own.
5422fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
5423    let mut spilled = 0u64;
5424    for value in values {
5425        spilled |= residue::<T>(*value);
5426    }
5427    if spilled != 0 {
5428        return Err(invalid("page value is not of its type"));
5429    }
5430    Ok(values.iter().map(|value| T::narrow(*value)).collect())
5431}
5432
5433/// The same values back in the width the column is declared at.
5434///
5435/// A value that does not fit is a page that disagrees with the directory about what the column is,
5436/// which is a damaged file rather than a caller error, so it is refused rather than truncated.
5437fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
5438    Ok(match ty {
5439        LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
5440        LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
5441        LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
5442        LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
5443        LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
5444        LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
5445        LogicalType::BigInt
5446        | LogicalType::Timestamp
5447        | LogicalType::Time
5448        | LogicalType::TimeTz
5449        | LogicalType::TimestampTz
5450        | LogicalType::TimestampS
5451        | LogicalType::TimestampMs
5452        | LogicalType::TimestampNs => Data::Int64(values.into()),
5453        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
5454        // integer the declared width says the column is stored as.
5455        LogicalType::Decimal { .. } => match ty.physical() {
5456            PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
5457            PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
5458            PhysicalType::Int64 => Data::Int64(values.into()),
5459            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
5460        },
5461        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
5462    })
5463}
5464
5465/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
5466/// beat before it is worth the decode.
5467fn plain_width(ty: &LogicalType) -> Option<usize> {
5468    Some(match ty {
5469        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
5470        LogicalType::SmallInt | LogicalType::USmallInt => 2,
5471        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
5472        LogicalType::BigInt
5473        | LogicalType::Timestamp
5474        | LogicalType::Time
5475        | LogicalType::TimeTz
5476        | LogicalType::TimestampTz
5477        | LogicalType::TimestampS
5478        | LogicalType::TimestampMs
5479        | LogicalType::TimestampNs => 8,
5480        LogicalType::Decimal { .. } => match ty.physical() {
5481            PhysicalType::Int16 => 2,
5482            PhysicalType::Int32 => 4,
5483            PhysicalType::Int64 => 8,
5484            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
5485            // they take the plain path and there is nothing here to compare against.
5486            _ => return None,
5487        },
5488        _ => return None,
5489    })
5490}
5491
5492/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
5493///
5494/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
5495/// where there is one and the plain width where there is not. Both are cheaper to decode than a
5496/// cascade, so a tie goes to them.
5497fn cascaded(
5498    flat: &Vector,
5499    ty: &LogicalType,
5500    packed: Option<&Packed<'_>>,
5501) -> Result<Option<Vec<u8>>> {
5502    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
5503    let Some(values) = widened(data) else { return Ok(None) };
5504    let plain = values.len().saturating_mul(width);
5505    let best = match packed {
5506        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
5507        Some(packed) => plain.min(21 + size_of_val(packed.words())),
5508        None => plain,
5509    };
5510    let out = integer::encode_with(&values, &Fixed)?;
5511    Ok((out.len() < best).then_some(out))
5512}
5513
5514/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
5515///
5516/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
5517/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
5518/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
5519/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
5520/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
5521///
5522/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
5523/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
5524/// values, and there is no reason to pay for the decode when it does.
5525/// A varchar page as one FSST layer, or `None` when it did not pay.
5526///
5527/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
5528/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
5529/// a page of values with nothing in common and the wrong one for a page of English, and a column of
5530/// comments is the case this exists for.
5531///
5532/// One layer and not the full string cascade, which is what the payload blocks of a global
5533/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
5534/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
5535/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
5536/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
5537/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
5538/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
5539/// what the page has to be put back together from.
5540///
5541/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
5542/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
5543/// already lays them out, and what the reader hands a chunk is views over that buffer.
5544///
5545/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
5546/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
5547/// page that was being written raw.
5548///
5549/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
5550/// nothing at read time for having been offered.
5551fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
5552    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
5553    let mut payload = 0_usize;
5554    for row in 0..flat.len() {
5555        let text = flat.text_at(row).unwrap_or("").as_bytes();
5556        payload = payload.saturating_add(text.len());
5557        values.push(text);
5558    }
5559    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
5560    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
5561    let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
5562        return Ok(None);
5563    };
5564    Ok((out.len() < plain).then_some(out))
5565}
5566
5567fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
5568    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
5569    let coded = integer::encode_with(&wide, &Codes)?;
5570    let plain = codes.len().saturating_mul(size_of::<u32>());
5571    Ok((coded.len() < plain).then_some(coded))
5572}
5573
5574fn encode(
5575    vector: &Vector,
5576    global: Option<&mut GlobalDictionary>,
5577) -> Result<(Vec<u8>, Option<Vec<u32>>)> {
5578    let ty = vector.logical_type();
5579    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
5580    let flat = vector.flatten()?;
5581    let mut out = Vec::new();
5582    let mut global_codes = None;
5583    if let Some(global) = global {
5584        let mut codes = Vec::with_capacity(flat.len());
5585        for row in 0..flat.len() {
5586            let text = flat.text_at(row).unwrap_or("");
5587            let code = global.code(text)?;
5588            global.observe(code, flat.is_null_at(row))?;
5589            codes.push(code);
5590        }
5591        global_codes = Some(codes);
5592    }
5593    let membership = global_codes.as_deref().map(unique_codes);
5594    let dictionary = if global_codes.is_none() && ty == &LogicalType::Varchar {
5595        string_dictionary(&flat)?
5596    } else {
5597        None
5598    };
5599    let compressed_text =
5600        if global_codes.is_none() && dictionary.is_none() && ty == &LogicalType::Varchar {
5601            text_compressed(&flat)?
5602        } else {
5603            None
5604        };
5605    let packed_vector = if dictionary.is_none() && global_codes.is_none() {
5606        Some(flat.bit_packed()?)
5607    } else {
5608        None
5609    };
5610    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
5611    let coded = match global_codes.as_deref() {
5612        Some(codes) => encoded_codes(codes)?,
5613        None => None,
5614    };
5615    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
5616    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
5617    // when it halves it, so a column that shrinks by a third was coming out whole.
5618    let cascade = if dictionary.is_none() && global_codes.is_none() {
5619        cascaded(&flat, ty, packed.as_ref())?
5620    } else {
5621        None
5622    };
5623    out.push(if coded.is_some() {
5624        4
5625    } else if cascade.is_some() {
5626        5
5627    } else if global_codes.is_some() {
5628        3
5629    } else if dictionary.is_some() {
5630        1
5631    } else if compressed_text.is_some() {
5632        6
5633    } else if packed.is_some() {
5634        2
5635    } else {
5636        0
5637    });
5638    let nulls = flat.validity();
5639    let flag = match nulls {
5640        Validity::AllValid => 0,
5641        Validity::AllInvalid => 1,
5642        Validity::Mask(_) => 2,
5643    };
5644    out.push(flag);
5645    if flag == 2 {
5646        for group in (0..vector.len()).step_by(8) {
5647            let mut bits = 0_u8;
5648            for bit in 0..8 {
5649                if group + bit < vector.len() && !flat.is_null_at(group + bit) {
5650                    bits |= 1 << bit;
5651                }
5652            }
5653            out.push(bits);
5654        }
5655    }
5656    if let Some(coded) = coded {
5657        out.extend_from_slice(&coded);
5658        return Ok((out, membership));
5659    }
5660    if let Some(cascade) = cascade {
5661        out.extend_from_slice(&cascade);
5662        return Ok((out, membership));
5663    }
5664    if let Some(codes) = global_codes {
5665        for code in codes {
5666            put_u32(&mut out, code);
5667        }
5668        return Ok((out, membership));
5669    }
5670    if let Some(dictionary) = dictionary {
5671        out.extend_from_slice(&dictionary);
5672        return Ok((out, membership));
5673    }
5674    if let Some(compressed_text) = compressed_text {
5675        out.extend_from_slice(&compressed_text);
5676        return Ok((out, membership));
5677    }
5678    if let Some(packed) = packed {
5679        if packed.offset() != 0 {
5680            return Err(invalid("writer received a sliced packed vector"));
5681        }
5682        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
5683        out.extend_from_slice(&packed.base().to_le_bytes());
5684        put_u32(
5685            &mut out,
5686            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
5687        );
5688        for word in packed.words() {
5689            put_u64(&mut out, *word);
5690        }
5691        return Ok((out, membership));
5692    }
5693    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
5694    match (ty, data) {
5695        (LogicalType::TinyInt, Data::Int8(values)) => {
5696            for value in &**values {
5697                out.extend_from_slice(&value.to_le_bytes());
5698            }
5699        }
5700        (LogicalType::UTinyInt, Data::UInt8(values)) => {
5701            for value in &**values {
5702                out.extend_from_slice(&value.to_le_bytes());
5703            }
5704        }
5705        (LogicalType::SmallInt, Data::Int16(values)) => {
5706            for value in &**values {
5707                out.extend_from_slice(&value.to_le_bytes());
5708            }
5709        }
5710        (LogicalType::USmallInt, Data::UInt16(values)) => {
5711            for value in &**values {
5712                out.extend_from_slice(&value.to_le_bytes());
5713            }
5714        }
5715        (LogicalType::UInteger, Data::UInt32(values)) => {
5716            for value in &**values {
5717                out.extend_from_slice(&value.to_le_bytes());
5718            }
5719        }
5720        (LogicalType::UBigInt, Data::UInt64(values)) => {
5721            for value in &**values {
5722                out.extend_from_slice(&value.to_le_bytes());
5723            }
5724        }
5725        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
5726            for value in &**values {
5727                out.extend_from_slice(&value.to_le_bytes());
5728            }
5729        }
5730        (
5731            LogicalType::BigInt
5732            | LogicalType::Timestamp
5733            | LogicalType::Time
5734            | LogicalType::TimeTz
5735            | LogicalType::TimestampTz
5736            | LogicalType::TimestampS
5737            | LogicalType::TimestampMs
5738            | LogicalType::TimestampNs,
5739            Data::Int64(values),
5740        ) => {
5741            for value in &**values {
5742                out.extend_from_slice(&value.to_le_bytes());
5743            }
5744        }
5745        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
5746        // the engine already carries it in, so nothing about the value changes on the way down.
5747        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
5748            for value in &**values {
5749                out.extend_from_slice(&value.to_le_bytes());
5750            }
5751        }
5752        (LogicalType::UHugeInt, Data::UInt128(values)) => {
5753            for value in &**values {
5754                out.extend_from_slice(&value.to_le_bytes());
5755            }
5756        }
5757        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
5758        // float codecs is worth having before somebody has measured a corpus of them.
5759        (LogicalType::Float, Data::Float32(values)) => {
5760            for value in &**values {
5761                out.extend_from_slice(&value.to_le_bytes());
5762            }
5763        }
5764        (LogicalType::Double, Data::Float64(values)) => {
5765            for value in &**values {
5766                out.extend_from_slice(&value.to_le_bytes());
5767            }
5768        }
5769        // Three counts and not one number. Months, days and microseconds stay apart on disk because
5770        // they are apart in the value: a month is not a fixed number of days and a day is not a
5771        // fixed number of microseconds, which is the whole reason the type has three fields.
5772        (LogicalType::Interval, Data::Interval(values)) => {
5773            for (months, days, micros) in &**values {
5774                out.extend_from_slice(&months.to_le_bytes());
5775                out.extend_from_slice(&days.to_le_bytes());
5776                out.extend_from_slice(&micros.to_le_bytes());
5777            }
5778        }
5779        (LogicalType::Boolean, Data::Bool(values)) => {
5780            for value in &**values {
5781                out.push(u8::from(*value));
5782            }
5783        }
5784        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
5785        // directory already, so writing it a value at a time would be paying for it twice.
5786        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
5787            for value in &**values {
5788                out.extend_from_slice(&value.to_le_bytes());
5789            }
5790        }
5791        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
5792            for value in &**values {
5793                out.extend_from_slice(&value.to_le_bytes());
5794            }
5795        }
5796        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
5797            for value in &**values {
5798                out.extend_from_slice(&value.to_le_bytes());
5799            }
5800        }
5801        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
5802            for value in &**values {
5803                out.extend_from_slice(&value.to_le_bytes());
5804            }
5805        }
5806        // A blob and a bit string go down the way a varchar does, because the layout is the same
5807        // one: an offset a value and then the bytes. What is not the same is that nothing here may
5808        // read the payload as text, which is why this arm asks the column for bytes rather than for
5809        // a string, and why the codecs above that do read text are all asked of a varchar by name.
5810        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
5811            let mut bytes = Vec::new();
5812            put_u32(&mut out, 0);
5813            for row in 0..vector.len() {
5814                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
5815                bytes.extend_from_slice(value);
5816                put_u32(
5817                    &mut out,
5818                    u32::try_from(bytes.len())
5819                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
5820                );
5821            }
5822            out.extend_from_slice(&bytes);
5823        }
5824        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
5825    }
5826    Ok((out, membership))
5827}
5828
5829fn put_varint(out: &mut Vec<u8>, mut value: u32) {
5830    while value >= 0x80 {
5831        out.push((value as u8 & 0x7f) | 0x80);
5832        value >>= 7;
5833    }
5834    out.push(value as u8);
5835}
5836
5837/// The distinct codes of one part, which is what a stripe's membership index is merged from.
5838fn unique_codes(codes: &[u32]) -> Vec<u32> {
5839    let mut unique = codes.to_vec();
5840    unique.sort_unstable();
5841    unique.dedup();
5842    unique
5843}
5844
5845/// The union of the sorted distinct codes of every part in a stripe.
5846///
5847/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
5848/// work on paper and the tree is the one that does not sort what is already in order: sixty four
5849/// sorted lists become one in six passes over the values.
5850fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
5851    let mut lists = lists;
5852    while lists.len() > 1 {
5853        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
5854        for pair in lists.chunks(2) {
5855            match pair {
5856                [left, right] => next.push(merged_pair(left, right)),
5857                [only] => next.push(only.clone()),
5858                _ => {}
5859            }
5860        }
5861        lists = next;
5862    }
5863    lists.pop().unwrap_or_default()
5864}
5865
5866fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
5867    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
5868    let mut at = 0;
5869    let mut to = 0;
5870    while at < left.len() && to < right.len() {
5871        match left[at].cmp(&right[to]) {
5872            Ordering::Less => {
5873                out.push(left[at]);
5874                at += 1;
5875            }
5876            Ordering::Greater => {
5877                out.push(right[to]);
5878                to += 1;
5879            }
5880            Ordering::Equal => {
5881                out.push(left[at]);
5882                at += 1;
5883                to += 1;
5884            }
5885        }
5886    }
5887    out.extend_from_slice(&left[at..]);
5888    out.extend_from_slice(&right[to..]);
5889    out
5890}
5891
5892/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
5893///
5894/// A bound that is missing from any part is missing from the stripe, because a missing bound means
5895/// nothing is known and a stripe that holds an unknown cannot claim one.
5896fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
5897    let mut merged = Range::default();
5898    let mut first = true;
5899    for range in ranges {
5900        merged.nulls = merged.nulls.saturating_add(range.nulls);
5901        // Both of these have to survive every part, so one part that could not say anything makes
5902        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
5903        // which leaves the stripe with exact ends and no total, which is a true thing to say.
5904        merged.sum = match (merged.sum.take(), range.sum) {
5905            (Some(held), Some(next)) if !first => held.checked_add(next),
5906            (_, next) if first => next,
5907            _ => None,
5908        };
5909        merged.exact = if first { range.exact } else { merged.exact && range.exact };
5910        if first {
5911            merged.low = range.low;
5912            merged.high = range.high;
5913            first = false;
5914            continue;
5915        }
5916        merged.low = match (merged.low.take(), range.low) {
5917            (Some(held), Some(next)) => Some(held.smaller(next)),
5918            _ => None,
5919        };
5920        merged.high = match (merged.high.take(), range.high) {
5921            (Some(held), Some(next)) => Some(held.larger(next)),
5922            _ => None,
5923        };
5924    }
5925    merged
5926}
5927
5928/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
5929///
5930/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
5931/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
5932/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
5933/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
5934///
5935/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
5936/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
5937/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
5938/// bound rather than claiming one that is too small. Anything that is not a string is already a
5939/// fixed width and is left alone.
5940fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
5941    match bound {
5942        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
5943            value.truncate(PART_BOUND_BYTES);
5944            if !high {
5945                return Some(Bound::Bytes(value));
5946            }
5947            while let Some(last) = value.pop() {
5948                if last < u8::MAX {
5949                    value.push(last + 1);
5950                    return Some(Bound::Bytes(value));
5951                }
5952            }
5953            None
5954        }
5955        other => other,
5956    }
5957}
5958
5959/// The ranges of one column's parts of one stripe, as a page.
5960///
5961/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
5962/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
5963/// number costs sixty times less to keep. What a part range is for is skipping the part, and
5964/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
5965/// string end that was cut down anyway.
5966fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
5967    let mut out = Vec::new();
5968    put_u32(
5969        &mut out,
5970        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
5971    );
5972    for range in ranges {
5973        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
5974        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
5975        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
5976    }
5977    Ok(out)
5978}
5979
5980/// The ranges one encoded page holds, one entry per part of the stripe.
5981fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
5982    let mut cur = Cursor { bytes, at: 0 };
5983    let parts = cur.u32()? as usize;
5984    let mut out = Vec::new();
5985    for _ in 0..parts {
5986        let low = cur.bound()?;
5987        let high = cur.bound()?;
5988        let nulls = cur.u32()? as usize;
5989        out.push(Range { low, high, nulls, exact: false, sum: None });
5990    }
5991    Ok(out)
5992}
5993
5994fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
5995    let held: Vec<&Option<Sieve>> = sieves.collect();
5996    let mut out = Vec::new();
5997    put_u32(
5998        &mut out,
5999        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
6000    );
6001    for sieve in &held {
6002        let length = sieve.as_ref().map_or(0, Sieve::len);
6003        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
6004    }
6005    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
6006    for sieve in held.into_iter().flatten() {
6007        out.extend_from_slice(&sieve.to_bytes());
6008    }
6009    Ok(out)
6010}
6011
6012/// The sieves one encoded page holds, one entry per part of the stripe.
6013///
6014/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
6015/// that gets read. That is how a file written by a later version of the sieve stays readable rather
6016/// than being a corrupt page.
6017fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
6018    let parts = u32::from_le_bytes(
6019        bytes
6020            .get(..4)
6021            .ok_or_else(|| invalid("sieve page is truncated"))?
6022            .try_into()
6023            .map_err(|_| invalid("sieve page is truncated"))?,
6024    ) as usize;
6025    let mut lengths = Vec::with_capacity(parts);
6026    for part in 0..parts {
6027        let at = 4 + part * 4;
6028        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
6029        lengths.push(u32::from_le_bytes(
6030            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
6031        ) as usize);
6032    }
6033    let mut at = 4 + parts * 4;
6034    let mut out = Vec::with_capacity(parts);
6035    for length in lengths {
6036        if length == 0 {
6037            out.push(None);
6038            continue;
6039        }
6040        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
6041        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
6042        out.push(Sieve::from_bytes(field));
6043        at = end;
6044    }
6045    if at != bytes.len() {
6046        return Err(invalid("sieve page has trailing bytes"));
6047    }
6048    Ok(out)
6049}
6050
6051/// One stripe's membership index: the code count and then the codes as ascending deltas.
6052///
6053/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
6054/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
6055/// a step a caller can skip.
6056fn encode_membership(unique: &[u32]) -> Vec<u8> {
6057    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
6058    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
6059    let mut previous = 0;
6060    for (at, &code) in unique.iter().enumerate() {
6061        put_varint(&mut out, if at == 0 { code } else { code - previous });
6062        previous = code;
6063    }
6064    out
6065}
6066
6067fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
6068    let mut value = 0_u32;
6069    for shift in (0..35).step_by(7) {
6070        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
6071        *at += 1;
6072        let part = u32::from(byte & 0x7f);
6073        if shift == 28 && part > 0x0f {
6074            return Err(invalid("membership varint overflow"));
6075        }
6076        value = value
6077            .checked_add(
6078                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
6079            )
6080            .ok_or_else(|| invalid("membership varint overflow"))?;
6081        if byte & 0x80 == 0 {
6082            return Ok(value);
6083        }
6084    }
6085    Err(invalid("membership varint is too long"))
6086}
6087
6088fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
6089    let mut at = 0;
6090    let count = take_varint(bytes, &mut at)? as usize;
6091    let mut codes = Vec::with_capacity(count);
6092    let mut previous = 0_u32;
6093    for index in 0..count {
6094        let delta = take_varint(bytes, &mut at)?;
6095        let code = if index == 0 {
6096            delta
6097        } else {
6098            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
6099        };
6100        if index > 0 && code <= previous {
6101            return Err(invalid("membership codes are not increasing"));
6102        }
6103        codes.push(code);
6104        previous = code;
6105    }
6106    if at != bytes.len() {
6107        return Err(invalid("membership page has trailing bytes"));
6108    }
6109    Ok(codes)
6110}
6111
6112fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
6113    let mut by_text = HashMap::new();
6114    let mut values = Vec::new();
6115    let mut codes = Vec::with_capacity(vector.len());
6116    let mut plain_bytes = 0_usize;
6117    for row in 0..vector.len() {
6118        let text = vector.text_at(row).unwrap_or("");
6119        plain_bytes = plain_bytes.saturating_add(text.len());
6120        let code = match by_text.get(text) {
6121            Some(&code) => code,
6122            None => {
6123                let code = u32::try_from(values.len())
6124                    .map_err(|_| invalid("too many dictionary values"))?;
6125                by_text.insert(text, code);
6126                values.push(text);
6127                code
6128            }
6129        };
6130        codes.push(code);
6131    }
6132    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
6133    let encoded = 8_usize
6134        .saturating_add((values.len() + 1).saturating_mul(4))
6135        .saturating_add(dictionary_bytes)
6136        .saturating_add(codes.len().saturating_mul(4));
6137    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
6138    if encoded >= plain {
6139        return Ok(None);
6140    }
6141    let mut out = Vec::with_capacity(encoded);
6142    put_u32(
6143        &mut out,
6144        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
6145    );
6146    put_u32(
6147        &mut out,
6148        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
6149    );
6150    let mut offset = 0_u32;
6151    put_u32(&mut out, offset);
6152    for value in &values {
6153        offset = offset
6154            .checked_add(
6155                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
6156            )
6157            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
6158        put_u32(&mut out, offset);
6159    }
6160    for value in values {
6161        out.extend_from_slice(value.as_bytes());
6162    }
6163    for code in codes {
6164        put_u32(&mut out, code);
6165    }
6166    Ok(Some(out))
6167}
6168
6169struct EncodedDictionary {
6170    index: Vec<u8>,
6171    ranks: Vec<u8>,
6172    /// The payload as the blocks it is written as, kept apart rather than joined because joining
6173    /// them is a second copy of a thing that is already gigabytes on the columns that matter.
6174    payload: Vec<Vec<u8>>,
6175}
6176
6177/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
6178fn head(bytes: &[u8]) -> u64 {
6179    let mut word = [0; 8];
6180    let take = bytes.len().min(8);
6181    word[..take].copy_from_slice(&bytes[..take]);
6182    u64::from_be_bytes(word)
6183}
6184
6185/// The sorted order of every global dictionary, one entry per column and empty where there is no
6186/// dictionary.
6187///
6188/// One column's sort has nothing to do with another's, and a table like `hits` has fifteen string
6189/// columns, so this runs across threads the way the numeric synopses above do. It is the only part
6190/// of committing a file that is more than bookkeeping, and doing it serially would show up as a
6191/// pause at the end of a load that thirty two threads had been busy with until then.
6192fn rankings(dictionaries: &[Option<GlobalDictionary>]) -> Result<Vec<Vec<(u64, u32)>>> {
6193    let present =
6194        dictionaries.iter().enumerate().filter(|(_, held)| held.is_some()).map(|(at, _)| at);
6195    let present = present.collect::<Vec<_>>();
6196    let mut orders = vec![Vec::new(); dictionaries.len()];
6197    let workers = std::thread::available_parallelism()
6198        .map_or(1, usize::from)
6199        .min(MAX_FREQUENCY_WORKERS)
6200        .min(present.len());
6201    if workers <= 1 {
6202        for at in present {
6203            if let Some(dictionary) = &dictionaries[at] {
6204                orders[at] = dictionary.ranked();
6205            }
6206        }
6207        return Ok(orders);
6208    }
6209    let width = present.len().div_ceil(workers);
6210    let pieces = std::thread::scope(|scope| {
6211        present
6212            .chunks(width)
6213            .map(|columns| {
6214                scope.spawn(|| {
6215                    columns
6216                        .iter()
6217                        .filter_map(|&at| dictionaries[at].as_ref().map(|held| (at, held.ranked())))
6218                        .collect::<Vec<_>>()
6219                })
6220            })
6221            .collect::<Vec<_>>()
6222            .into_iter()
6223            .map(|handle| {
6224                handle.join().map_err(|_| Error::internal("a dictionary sort worker panicked"))
6225            })
6226            .collect::<Result<Vec<_>>>()
6227    })?;
6228    for piece in pieces {
6229        for (at, order) in piece {
6230            orders[at] = order;
6231        }
6232    }
6233    Ok(orders)
6234}
6235
6236fn encode_global_dictionary(
6237    dictionary: GlobalDictionary,
6238    order: &[(u64, u32)],
6239) -> Result<EncodedDictionary> {
6240    let values = dictionary.offsets.len() - 1;
6241    if order.len() != values {
6242        return Err(invalid("global dictionary order does not cover its values"));
6243    }
6244    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6245    let payload = encode_payload(&dictionary)?;
6246    if payload.len() != blocks {
6247        return Err(invalid("global dictionary payload is not the blocks it says it is"));
6248    }
6249    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
6250    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
6251    let offset_bits = offset_width(&dictionary.offsets);
6252    let mut index = Vec::with_capacity(
6253        DICTIONARY_HEADER + offset_bytes(values, offset_bits) + (blocks + rank_blocks) * 16,
6254    );
6255    put_u32(
6256        &mut index,
6257        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
6258    );
6259    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
6260    put_u32(
6261        &mut index,
6262        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
6263    );
6264    put_u32(&mut index, offset_bits as u32);
6265    encode_offsets(&dictionary.offsets, offset_bits, &mut index)?;
6266    // Where each block ends, so a reader can find one. The stored blocks are shorter than the
6267    // decoded ones and by a different amount each, so this is the one thing the offsets above no
6268    // longer say.
6269    let mut at = 0_u64;
6270    for block in &payload {
6271        at = at
6272            .checked_add(block.len() as u64)
6273            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
6274        put_u64(&mut index, at);
6275    }
6276    for block in &payload {
6277        put_u64(&mut index, checksum(block));
6278    }
6279    // The same two lists for the sorted order. A rank block is packed at whatever width its own
6280    // heads need, so where one ends is no longer arithmetic on the block number.
6281    if rank_ends.len() != rank_blocks {
6282        return Err(invalid("global dictionary order is not the blocks it says it is"));
6283    }
6284    for end in &rank_ends {
6285        put_u64(&mut index, *end);
6286    }
6287    let mut at = 0_usize;
6288    for end in &rank_ends {
6289        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
6290        put_u64(&mut index, checksum(&ranks[at..end]));
6291        at = end;
6292    }
6293    Ok(EncodedDictionary { index, ranks, payload })
6294}
6295
6296/// How many blocks of the payload the shape is settled on.
6297///
6298/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
6299/// the same reason. They are spread across the dictionary rather than taken off the front, because
6300/// a dictionary is in the order values were first seen and the front of it is the first morsel of
6301/// the load.
6302const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
6303
6304/// The shapes the payload encoder picks between.
6305///
6306/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
6307/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
6308/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
6309/// settles the outer level and the one below it, which is where almost all of that hour goes, and
6310/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
6311/// to cost nothing.
6312///
6313/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
6314/// block, against the exhaustive search over the same blocks:
6315///
6316/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
6317/// |---|---|---|---|---|
6318/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
6319/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
6320/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
6321/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
6322/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
6323///
6324/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
6325/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
6326/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
6327/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
6328/// rather than searched for an answer that does not exist.
6329fn payload_shapes() -> Vec<chooser::Settled> {
6330    let integers = vec![integer::Kind::Packed];
6331    [
6332        vec![string::Kind::Front, string::Kind::Lz],
6333        vec![string::Kind::Lz, string::Kind::Fsst],
6334        vec![string::Kind::Lz, string::Kind::Plain],
6335        vec![string::Kind::Fsst],
6336        vec![string::Kind::Plain],
6337    ]
6338    .into_iter()
6339    .map(|strings| chooser::Settled::new(strings, integers.clone()))
6340    .collect()
6341}
6342
6343/// The payload as encoded blocks of [`TEXT_PAYLOAD_VALUES`] values each.
6344///
6345/// Across threads because this is the only part of committing a file that is real work rather than
6346/// bookkeeping. The blocks are the same size and cost about the same, so an index each is enough of
6347/// a queue and there is nothing to weight the way the numeric synopses are weighted.
6348fn encode_payload(dictionary: &GlobalDictionary) -> Result<Vec<Vec<u8>>> {
6349    let values = dictionary.offsets.len() - 1;
6350    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6351    let run = |block: usize| {
6352        let first = block * TEXT_PAYLOAD_VALUES;
6353        let last = (first + TEXT_PAYLOAD_VALUES).min(values);
6354        (first..last)
6355            .map(|value| {
6356                let from = dictionary.offsets[value] as usize;
6357                let to = dictionary.offsets[value + 1] as usize;
6358                &dictionary.payload[from..to]
6359            })
6360            .collect::<Vec<_>>()
6361    };
6362    // A dictionary small enough to be the sample is small enough to search in full, and searching
6363    // it costs less than deciding not to.
6364    let shape = (blocks > PAYLOAD_SAMPLE_BLOCKS).then(|| settle_shape(&run, blocks)).transpose()?;
6365    let one = |block: usize| match &shape {
6366        Some(shape) => string::encode_with(&run(block), shape),
6367        None => string::encode(&run(block)),
6368    };
6369    let workers = std::thread::available_parallelism()
6370        .map_or(1, usize::from)
6371        .min(MAX_FREQUENCY_WORKERS)
6372        .min(blocks);
6373    if workers <= 1 {
6374        return (0..blocks).map(one).collect();
6375    }
6376    let next = AtomicUsize::new(0);
6377    let pieces = std::thread::scope(|scope| {
6378        (0..workers)
6379            .map(|_| {
6380                scope.spawn(|| {
6381                    let mut mine = Vec::new();
6382                    loop {
6383                        let block = next.fetch_add(1, Atomic::Relaxed);
6384                        if block >= blocks {
6385                            break;
6386                        }
6387                        mine.push((block, one(block)?));
6388                    }
6389                    Ok(mine)
6390                })
6391            })
6392            .collect::<Vec<_>>()
6393            .into_iter()
6394            .map(|handle| {
6395                handle.join().map_err(|_| Error::internal("a dictionary encode worker panicked"))?
6396            })
6397            .collect::<Result<Vec<_>>>()
6398    })?;
6399    let mut payload = vec![Vec::new(); blocks];
6400    for piece in pieces {
6401        for (block, bytes) in piece {
6402            payload[block] = bytes;
6403        }
6404    }
6405    Ok(payload)
6406}
6407
6408/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
6409///
6410/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
6411/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
6412/// sample is spread across the dictionary so that the first and last blocks are both in it, because
6413/// a dictionary written in first seen order has its common values at the front and its long tail at
6414/// the back, and those do not compress alike.
6415fn settle_shape<'a>(
6416    run: &dyn Fn(usize) -> Vec<&'a [u8]>,
6417    blocks: usize,
6418) -> Result<chooser::Settled> {
6419    let last = blocks - 1;
6420    let sample = (0..PAYLOAD_SAMPLE_BLOCKS)
6421        .map(|region| run(region * last / (PAYLOAD_SAMPLE_BLOCKS - 1)))
6422        .collect::<Vec<_>>();
6423    let mut best: Option<(chooser::Settled, usize)> = None;
6424    for shape in payload_shapes() {
6425        let mut size = 0;
6426        for block in &sample {
6427            size += string::encode_with(block, &shape)?.len();
6428        }
6429        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
6430            best = Some((shape, size));
6431        }
6432    }
6433    best.map(|(shape, _)| shape)
6434        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
6435}
6436
6437/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
6438///
6439/// Each block holds its heads first and then its codes, rather than pairing them, because a search
6440/// asks for a head at every probe and for a code about once a search. Keeping the heads together
6441/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
6442/// probes of a search, which are the ones that land in the same block, touch the same cache line.
6443fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
6444    let mut out = Vec::with_capacity(order.len() * 4);
6445    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
6446    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
6447    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
6448    for block in order.chunks(TEXT_RANK_BLOCK) {
6449        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
6450        // rise, the smallest is the first and the largest is the last.
6451        let base = block.first().map_or(0, |&(head, _)| head);
6452        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
6453        let width = (u64::BITS - span.leading_zeros()) as usize;
6454        heads.clear();
6455        codes.clear();
6456        for &(head, code) in block {
6457            heads.push(head.wrapping_sub(base));
6458            codes.push(u64::from(code));
6459        }
6460        put_u64(&mut out, base);
6461        out.push(width as u8);
6462        bitpack::pack_tail(&heads, width, &mut out)
6463            .map_err(|_| invalid("global dictionary heads do not pack"))?;
6464        bitpack::pack_tail(&codes, code_bits, &mut out)
6465            .map_err(|_| invalid("global dictionary codes do not pack"))?;
6466        ends.push(out.len() as u64);
6467    }
6468    Ok((out, ends))
6469}
6470
6471/// Opens a column's global dictionary, which reads its index and none of its payload.
6472///
6473/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
6474/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
6475/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
6476/// a quarter of a gigabyte of dictionary to reach it.
6477fn open_global_dictionary(
6478    file: Arc<File>,
6479    page: Page,
6480    ty: &LogicalType,
6481    keep_budget: usize,
6482) -> Result<Vector> {
6483    if ty != &LogicalType::Varchar {
6484        return Err(invalid("global dictionary belongs to a non-string column"));
6485    }
6486    let mut header = [0; DICTIONARY_HEADER];
6487    read_at(&file, page.offset, &mut header)?;
6488    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
6489    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
6490    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
6491    let offset_bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
6492    if per_block != TEXT_PAYLOAD_VALUES {
6493        return Err(invalid("global dictionary block width differs"));
6494    }
6495    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
6496        return Err(invalid("global dictionary block count differs from its value count"));
6497    }
6498    if offset_bits > u32::BITS as usize {
6499        return Err(invalid("global dictionary packs offsets past a payload"));
6500    }
6501    let offset_len = offset_bytes(count, offset_bits);
6502    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
6503    // full the moment the column is first touched, and the order is half again the size of the
6504    // offsets, so putting it there would make every query that reads a string column pay for a
6505    // search that most of them never make.
6506    let ranks = count;
6507    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
6508    // Two words a payload block, one for where it ends in the file and one for its checksum, and the
6509    // same two a rank block.
6510    let hash_len = blocks
6511        .checked_add(rank_blocks)
6512        .and_then(|words| words.checked_mul(16))
6513        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
6514    let index_len = DICTIONARY_HEADER
6515        .checked_add(offset_len)
6516        .and_then(|len| len.checked_add(hash_len))
6517        .ok_or_else(|| invalid("global dictionary header overflow"))?;
6518    if index_len > page.length as usize {
6519        return Err(invalid("global dictionary offset index exceeds its page"));
6520    }
6521    let mut index = vec![0; index_len];
6522    index[..DICTIONARY_HEADER].copy_from_slice(&header);
6523    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
6524    if checksum(&index) != page.hash {
6525        return Err(invalid("global dictionary index checksum differs"));
6526    }
6527    let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
6528    let mut words = index[DICTIONARY_HEADER + offset_len..]
6529        .chunks_exact(8)
6530        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
6531        .collect::<Vec<_>>();
6532    let mut hashes = words.split_off(blocks);
6533    let mut rank_ends = hashes.split_off(blocks);
6534    let rank_hashes = rank_ends.split_off(rank_blocks);
6535    let ends = words;
6536    // A rank block packs its heads at whatever width its own values need, so its length is no longer
6537    // arithmetic on the block number and the reader has to be told where each one ends.
6538    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
6539        return Err(invalid("global dictionary order blocks do not rise"));
6540    }
6541    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
6542        .map_err(|_| invalid("global dictionary rank overflow"))?;
6543    let body_len = index_len
6544        .checked_add(rank_len)
6545        .ok_or_else(|| invalid("global dictionary header overflow"))?;
6546    if body_len > page.length as usize {
6547        return Err(invalid("global dictionary order exceeds its page"));
6548    }
6549    // What the offsets bound is the decoded payload, and what the page holds is the stored one, so
6550    // the last block end is the only thing that ties the index to the length of the page.
6551    let stored_len = page.length as usize - body_len;
6552    if ends.last().copied().unwrap_or_default() as usize != stored_len
6553        || ends.windows(2).any(|pair| pair[0] > pair[1])
6554    {
6555        return Err(invalid("global dictionary blocks do not bound the payload"));
6556    }
6557    Vector::external_text(
6558        LogicalType::Varchar,
6559        Arc::new(NativeText {
6560            file,
6561            values: count,
6562            offsets,
6563            offset_bits,
6564            ranks,
6565            rank_at: page.offset + index_len as u64,
6566            rank_ends,
6567            rank_hashes,
6568            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
6569            code_bits: code_width(count),
6570            code_ranks: OnceLock::new(),
6571            payload: page.offset + body_len as u64,
6572            ends,
6573            hashes,
6574            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
6575            keep_budget,
6576            payload_kept: AtomicUsize::new(0),
6577            searched: Mutex::new(HashMap::new()),
6578        }),
6579    )
6580}
6581
6582/// What a stored page is, without decoding a value out of it.
6583///
6584/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
6585/// the format's own choice, and it is what says whether the column came back as codes into a table
6586/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
6587/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
6588/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
6589///
6590/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
6591/// cannot walk comes back as text rather than as an error, because a caller asking what a file
6592/// looks like is usually asking because something is wrong with it, and a report that stops at the
6593/// first bad page is a report that says nothing about the other nine hundred.
6594fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
6595    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
6596    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
6597        let mut cur = Cursor { bytes, at: 0 };
6598        let codec = cur.u8()?;
6599        if cur.u8()? == 2 {
6600            cur.take(rows.div_ceil(8))?;
6601        }
6602        Ok((codec, cur.at))
6603    }
6604    let Ok((codec, at)) = cascade_at(rows, bytes) else {
6605        return "UNREADABLE".to_string();
6606    };
6607    let tail = &bytes[at..];
6608    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
6609    match codec {
6610        0 => match ty {
6611            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
6612            _ => "FIXED".to_string(),
6613        },
6614        1 => "DICT(PLAIN)".to_string(),
6615        2 => "FOR+BITPACK".to_string(),
6616        3 => "TABLE DICT".to_string(),
6617        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
6618        5 => described(integer::describe(tail)),
6619        6 => described(string::describe(tail)),
6620        other => format!("CODEC {other}"),
6621    }
6622}
6623
6624fn decode(
6625    ty: &LogicalType,
6626    rows: usize,
6627    bytes: &[u8],
6628    global: Option<Arc<Vector>>,
6629) -> Result<Vector> {
6630    let mut cur = Cursor { bytes, at: 0 };
6631    let codec = cur.u8()?;
6632    let flag = cur.u8()?;
6633    let validity = match flag {
6634        0 => Validity::AllValid,
6635        1 => Validity::AllInvalid,
6636        2 => {
6637            let mask = cur.take(rows.div_ceil(8))?;
6638            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
6639        }
6640        _ => return Err(invalid("page validity tag differs")),
6641    };
6642    if codec == 1 {
6643        if ty != &LogicalType::Varchar {
6644            return Err(invalid("dictionary codec belongs to a non-string page"));
6645        }
6646        let count = cur.u32()? as usize;
6647        let payload_len = cur.u32()? as usize;
6648        let offset_bytes = cur.take(
6649            (count + 1)
6650                .checked_mul(4)
6651                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
6652        )?;
6653        let offsets = offset_bytes
6654            .chunks_exact(4)
6655            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6656            .collect::<Vec<_>>();
6657        let payload = cur.take(payload_len)?.to_vec();
6658        if offsets.first() != Some(&0)
6659            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6660            || offsets.windows(2).any(|pair| pair[0] > pair[1])
6661        {
6662            return Err(invalid("dictionary offsets do not bound the payload"));
6663        }
6664        // A page, because every chunk cut out of this dictionary points at the same payload and a
6665        // page is what lets a cut be the views and nothing else.
6666        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
6667        for pair in offsets.windows(2) {
6668            strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
6669        }
6670        let mut codes = Vec::with_capacity(rows);
6671        for _ in 0..rows {
6672            codes.push(cur.u32()?);
6673        }
6674        if codes.iter().any(|code| *code as usize >= count) {
6675            return Err(invalid("dictionary code is out of range"));
6676        }
6677        if cur.at != bytes.len() {
6678            return Err(invalid("dictionary page has trailing bytes"));
6679        }
6680        let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
6681        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
6682    }
6683    if codec == 3 || codec == 4 {
6684        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
6685        let codes = if codec == 4 {
6686            // The cascade holds the whole tail of the page and says how long it is itself, so the
6687            // check that nothing is left over is the one the decoder already makes.
6688            let wide = integer::decode(&bytes[cur.at..])?;
6689            if wide.len() != rows {
6690                return Err(invalid("encoded code page holds the wrong number of rows"));
6691            }
6692            // Converted in one pass and checked in the same one, rather than a fallible conversion
6693            // per code. A `Result` an element is a short circuit the loop cannot be vectorized past,
6694            // and it was costing about twelve instructions a row to narrow a number that already
6695            // fits. Every code a file holds is inside a `u32` or the file is corrupt, so the check
6696            // belongs once at the end: or the codes together and the answer has a bit set above the
6697            // low thirty two, or the sign bit, exactly when one of them did.
6698            let mut codes = Vec::with_capacity(wide.len());
6699            let mut seen = 0_i64;
6700            for &code in &wide {
6701                seen |= code;
6702                codes.push(code as u32);
6703            }
6704            if seen < 0 || seen > i64::from(u32::MAX) {
6705                return Err(invalid("code is not a code"));
6706            }
6707            codes
6708        } else {
6709            let mut codes = Vec::with_capacity(rows);
6710            for _ in 0..rows {
6711                codes.push(cur.u32()?);
6712            }
6713            if cur.at != bytes.len() {
6714                return Err(invalid("global code page has trailing bytes"));
6715            }
6716            codes
6717        };
6718        let highest = codes.iter().copied().max();
6719        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
6720            .with_validity(validity));
6721    }
6722    if codec == 6 {
6723        if ty != &LogicalType::Varchar {
6724            return Err(invalid("compressed text codec belongs to a non-string page"));
6725        }
6726        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
6727        // It comes back as one buffer with the values laid end to end and where each one ends, which
6728        // is the raw form's layout, so what is left to do here is what codec 0 does.
6729        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
6730        if ends.len() != rows {
6731            return Err(invalid("compressed text page holds the wrong number of rows"));
6732        }
6733        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
6734        // payload moves views rather than bytes.
6735        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6736        let mut start = 0;
6737        for end in ends {
6738            let len = end
6739                .checked_sub(start)
6740                .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
6741            values.push_in_place(start, len)?;
6742            start = end;
6743        }
6744        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
6745    }
6746    if codec == 5 {
6747        // The cascade holds the whole tail of the page and says how long it is itself.
6748        let values = integer::decode(&bytes[cur.at..])?;
6749        if values.len() != rows {
6750            return Err(invalid("cascade page holds the wrong number of rows"));
6751        }
6752        let data = narrowed(ty, values)?;
6753        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
6754    }
6755    if codec == 2 {
6756        let width = u32::from(cur.u8()?);
6757        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
6758        let count = cur.u32()? as usize;
6759        let mut words = Vec::with_capacity(count);
6760        for _ in 0..count {
6761            words.push(cur.u64()?);
6762        }
6763        if cur.at != bytes.len() {
6764            return Err(invalid("packed page has trailing bytes"));
6765        }
6766        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
6767    }
6768    if codec != 0 {
6769        return Err(invalid("page codec is unknown"));
6770    }
6771    let data = match ty {
6772        LogicalType::TinyInt => {
6773            let values = cur.take(rows)?;
6774            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
6775        }
6776        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
6777        LogicalType::SmallInt => {
6778            let values =
6779                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6780            Data::Int16(
6781                values
6782                    .chunks_exact(2)
6783                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6784                    .collect::<Vec<_>>()
6785                    .into(),
6786            )
6787        }
6788        LogicalType::USmallInt => {
6789            let values =
6790                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6791            Data::UInt16(
6792                values
6793                    .chunks_exact(2)
6794                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
6795                    .collect::<Vec<_>>()
6796                    .into(),
6797            )
6798        }
6799        LogicalType::UInteger => {
6800            let values =
6801                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6802            Data::UInt32(
6803                values
6804                    .chunks_exact(4)
6805                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
6806                    .collect::<Vec<_>>()
6807                    .into(),
6808            )
6809        }
6810        LogicalType::UBigInt => {
6811            let values =
6812                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6813            Data::UInt64(
6814                values
6815                    .chunks_exact(8)
6816                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
6817                    .collect::<Vec<_>>()
6818                    .into(),
6819            )
6820        }
6821        LogicalType::Integer | LogicalType::Date => {
6822            let values =
6823                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6824            Data::Int32(
6825                values
6826                    .chunks_exact(4)
6827                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6828                    .collect::<Vec<_>>()
6829                    .into(),
6830            )
6831        }
6832        LogicalType::BigInt
6833        | LogicalType::Timestamp
6834        | LogicalType::Time
6835        | LogicalType::TimeTz
6836        | LogicalType::TimestampTz
6837        | LogicalType::TimestampS
6838        | LogicalType::TimestampMs
6839        | LogicalType::TimestampNs => {
6840            let values =
6841                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6842            Data::Int64(
6843                values
6844                    .chunks_exact(8)
6845                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6846                    .collect::<Vec<_>>()
6847                    .into(),
6848            )
6849        }
6850        LogicalType::HugeInt | LogicalType::Uuid => {
6851            let values =
6852                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6853            Data::Int128(
6854                values
6855                    .chunks_exact(16)
6856                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6857                    .collect::<Vec<_>>()
6858                    .into(),
6859            )
6860        }
6861        LogicalType::UHugeInt => {
6862            let values =
6863                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6864            Data::UInt128(
6865                values
6866                    .chunks_exact(16)
6867                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6868                    .collect::<Vec<_>>()
6869                    .into(),
6870            )
6871        }
6872        LogicalType::Float => {
6873            let values =
6874                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6875            Data::Float32(
6876                values
6877                    .chunks_exact(4)
6878                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
6879                    .collect::<Vec<_>>()
6880                    .into(),
6881            )
6882        }
6883        LogicalType::Double => {
6884            let values =
6885                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6886            Data::Float64(
6887                values
6888                    .chunks_exact(8)
6889                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
6890                    .collect::<Vec<_>>()
6891                    .into(),
6892            )
6893        }
6894        LogicalType::Interval => {
6895            let values =
6896                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6897            Data::Interval(
6898                values
6899                    .chunks_exact(16)
6900                    .map(|item| {
6901                        (
6902                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
6903                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
6904                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
6905                        )
6906                    })
6907                    .collect::<Vec<_>>()
6908                    .into(),
6909            )
6910        }
6911        LogicalType::Boolean => {
6912            let values = cur.take(rows)?;
6913            if values.iter().any(|value| *value > 1) {
6914                return Err(invalid("boolean page has another value"));
6915            }
6916            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
6917        }
6918        // Whichever integer the declared width says, which is the mapping the rest of the engine
6919        // already uses for a decimal in memory.
6920        LogicalType::Decimal { .. } => match ty.physical() {
6921            PhysicalType::Int16 => {
6922                let values =
6923                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6924                Data::Int16(
6925                    values
6926                        .chunks_exact(2)
6927                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6928                        .collect::<Vec<_>>()
6929                        .into(),
6930                )
6931            }
6932            PhysicalType::Int32 => {
6933                let values =
6934                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6935                Data::Int32(
6936                    values
6937                        .chunks_exact(4)
6938                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6939                        .collect::<Vec<_>>()
6940                        .into(),
6941                )
6942            }
6943            PhysicalType::Int64 => {
6944                let values =
6945                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6946                Data::Int64(
6947                    values
6948                        .chunks_exact(8)
6949                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6950                        .collect::<Vec<_>>()
6951                        .into(),
6952                )
6953            }
6954            _ => {
6955                let values =
6956                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6957                Data::Int128(
6958                    values
6959                        .chunks_exact(16)
6960                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6961                        .collect::<Vec<_>>()
6962                        .into(),
6963                )
6964            }
6965        },
6966        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
6967            let offset_bytes = cur
6968                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
6969            let offsets = offset_bytes
6970                .chunks_exact(4)
6971                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6972                .collect::<Vec<_>>();
6973            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
6974            if offsets.first() != Some(&0)
6975                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6976                || offsets.windows(2).any(|pair| pair[0] > pair[1])
6977            {
6978                return Err(invalid("string offsets do not bound the payload"));
6979            }
6980            // A page for the reason the dictionary payload above is one: the page is read once and
6981            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
6982            // bytes.
6983            //
6984            // A varchar is checked for text on the way in and a blob and a bit string are not,
6985            // because the second pair never claimed to hold any. Reading them through the checking
6986            // seam would refuse a column for holding exactly what it was told to hold.
6987            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6988            let text = ty == &LogicalType::Varchar;
6989            for pair in offsets.windows(2) {
6990                let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
6991                if text {
6992                    values.push_in_place(at, len)?;
6993                } else {
6994                    values.push_bytes_in_place(at, len)?;
6995                }
6996            }
6997            Data::Varlen(values)
6998        }
6999        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
7000    };
7001    if cur.at != bytes.len() {
7002        return Err(invalid("page has trailing bytes"));
7003    }
7004    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
7005}
7006
7007#[cfg(test)]
7008mod tests {
7009    use std::fs;
7010    use std::io::{Seek, SeekFrom, Write};
7011    use std::path::PathBuf;
7012    use std::time::{SystemTime, UNIX_EPOCH};
7013
7014    use rudb_common::Stat;
7015    use rudb_common::Value;
7016    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
7017    use rudb_common::stat::Provenance;
7018
7019    use super::*;
7020
7021    #[test]
7022    fn checksum_matches_fixed_vectors() {
7023        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
7024        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
7025        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
7026    }
7027
7028    fn path(label: &str) -> PathBuf {
7029        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
7030        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
7031    }
7032
7033    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
7034    #[test]
7035    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
7036        const SPANS: usize = 64;
7037        const SPAN: usize = 512;
7038        let path = path("positional");
7039        let content: Vec<u8> =
7040            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
7041        fs::write(&path, &content).expect("the file is written");
7042        let file = Arc::new(File::open(&path).expect("the file opens"));
7043        std::thread::scope(|scope| {
7044            for _ in 0..8 {
7045                let file = Arc::clone(&file);
7046                scope.spawn(move || {
7047                    for _ in 0..64 {
7048                        for span in 0..SPANS {
7049                            let mut bytes = [0_u8; SPAN];
7050                            read_at(&file, (span * SPAN) as u64, &mut bytes)
7051                                .expect("the span reads");
7052                            assert!(
7053                                bytes.iter().all(|byte| *byte == span as u8),
7054                                "span {span} came back as {}",
7055                                bytes[0],
7056                            );
7057                        }
7058                    }
7059                });
7060            }
7061        });
7062        let mut past = [0_u8; SPAN];
7063        let end = (SPANS * SPAN) as u64;
7064        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
7065        assert!(error.message().contains("ends before its declared length"), "{error}");
7066        drop(file);
7067        let _ = fs::remove_file(&path);
7068    }
7069
7070    /// The writer records where it put a page and puts it there, whatever the cursor is doing.
7071    ///
7072    /// The cursor is moved between the steps that record an offset, which is what reading the pages
7073    /// back to build the frequencies does on a platform with no `pread`. Without the fix the
7074    /// directory lands on top of a page and the file fails to reopen.
7075    #[test]
7076    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
7077        let path = path("cursor");
7078        let mut writer = Writer::create(
7079            &path,
7080            "items",
7081            vec![
7082                Field::required("id", LogicalType::Integer),
7083                Field::new("text", LogicalType::Varchar),
7084            ],
7085        )
7086        .expect("new file");
7087        writer.append(&sample()).expect("first part");
7088        writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
7089        writer.append(&sample()).expect("second part");
7090        writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
7091        writer.finish().expect("commit");
7092        let reader = Reader::open(&path).expect("reopen from disk");
7093        assert_eq!(reader.table().rows(), 6);
7094        let ids = reader.read(0, &[0]).expect("the integer page reads back");
7095        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
7096        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
7097        let text = reader.read(1, &[1]).expect("the text page reads back");
7098        assert_eq!(text.value_at(1, 0), Value::Null);
7099        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7100        // Nothing the directory points at may run past the end of the file, which is the shape the
7101        // failure took: a page recorded at an offset the directory had already been written over.
7102        let end = reader.table().stripes().iter().flat_map(|stripe| {
7103            stripe
7104                .pages
7105                .iter()
7106                .map(|page| page.offset + u64::from(page.length))
7107                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
7108        });
7109        let last = end.fold(HEADER, u64::max);
7110        let directory = fs::metadata(&path).expect("the file is there").len();
7111        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
7112        fs::remove_file(path).expect("remove scratch file");
7113    }
7114
7115    /// How long a global dictionary index is, read out of the page's own header.
7116    ///
7117    /// The tests below damage a byte of the order or of the payload, so they need to know where each
7118    /// one starts, and working it out here rather than writing a number down means adding something
7119    /// to the index does not quietly turn one of them into a test that damages the index instead.
7120    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
7121        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
7122        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
7123        let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
7124        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
7125        DICTIONARY_HEADER as u64
7126            + offset_bytes(count as usize, bits) as u64
7127            + (blocks + rank_blocks) * 16
7128    }
7129
7130    /// How long the sorted order is, which is where its last block ends.
7131    fn last_rank_end(file: &File, offset: u64, header: &[u8; DICTIONARY_HEADER]) -> u64 {
7132        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
7133        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
7134        let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
7135        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
7136        let at = offset
7137            + DICTIONARY_HEADER as u64
7138            + offset_bytes(count as usize, bits) as u64
7139            + blocks * 16
7140            + (rank_blocks - 1) * 8;
7141        let mut end = [0; 8];
7142        read_at(file, at, &mut end).expect("the last rank block end");
7143        u64::from_le_bytes(end)
7144    }
7145
7146    fn sample() -> Chunk {
7147        Chunk::new(vec![
7148            Vector::from_values(
7149                LogicalType::Integer,
7150                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
7151            )
7152            .expect("integers"),
7153            Vector::from_values(
7154                LogicalType::Varchar,
7155                &[
7156                    Value::Varchar("alpha".into()),
7157                    Value::Null,
7158                    Value::Varchar("long text after a slash".into()),
7159                ],
7160            )
7161            .expect("strings"),
7162        ])
7163        .expect("matching rows")
7164    }
7165
7166    fn sample_ids() -> Chunk {
7167        Chunk::new(vec![
7168            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
7169                .expect("integers"),
7170        ])
7171        .expect("one column")
7172    }
7173
7174    #[test]
7175    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
7176        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
7177        // condition gets, and the number was in the stripe entry next to the bounds all along.
7178        let path = path("nulls_for_the_planner");
7179        let mut writer =
7180            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
7181                .expect("new file");
7182        let rows = Chunk::new(vec![
7183            Vector::from_values(
7184                LogicalType::Integer,
7185                &[
7186                    Value::Integer(4),
7187                    Value::Null,
7188                    Value::Integer(9),
7189                    Value::Null,
7190                    Value::Integer(1),
7191                    Value::Integer(2),
7192                ],
7193            )
7194            .expect("integers"),
7195        ])
7196        .expect("one column");
7197        writer.append(&rows).expect("the only part");
7198        writer.finish().expect("commit");
7199        let reader = Reader::open(&path).expect("reopen from disk");
7200        let stripes = Stripes::new(reader);
7201        let column = stripes.column("a").expect("the file has that column");
7202        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
7203        // A column the file does not have. Zero here would be a fact about a column that is not
7204        // there, which the planner would then divide by.
7205        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
7206        fs::remove_file(&path).expect("clean up");
7207    }
7208
7209    #[test]
7210    fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
7211        // The whole of the frequency half of #1106, end to end over a real file. Six rows, three
7212        // of one value and two of another, and a complete synopsis because six rows is well inside
7213        // what the writer can account for. The estimate for `id = 4` is three rows rather than a
7214        // sixth of the table, and for a value the file does not hold it is none.
7215        let path = path("frequencies_for_the_planner");
7216        let mut writer =
7217            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7218                .expect("new file");
7219        let rows = Chunk::new(vec![
7220            Vector::from_values(
7221                LogicalType::Integer,
7222                &[
7223                    Value::Integer(4),
7224                    Value::Integer(4),
7225                    Value::Integer(4),
7226                    Value::Integer(9),
7227                    Value::Integer(9),
7228                    Value::Integer(1),
7229                ],
7230            )
7231            .expect("integers"),
7232        ])
7233        .expect("one column");
7234        writer.append(&rows).expect("the only part");
7235        writer.finish().expect("commit");
7236        let reader = Reader::open(&path).expect("reopen from disk");
7237        let common = Common::new(reader);
7238        assert_eq!(common.rows(), 6);
7239        let column = common.column("id").expect("the file has that column");
7240        assert_eq!(common.column("nothing"), None);
7241        assert_eq!(
7242            common.rows_with(column, &Bound::Int(4)),
7243            Stat::exact(3, Provenance::FrequencySynopsis)
7244        );
7245        // Not in the file, and a synopsis that accounts for all six rows proves it.
7246        assert_eq!(
7247            common.rows_with(column, &Bound::Int(7)),
7248            Stat::exact(0, Provenance::FrequencySynopsis)
7249        );
7250        // A constant of another domain against an integer column. Nothing in the list compares
7251        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
7252        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
7253        // A complete list has no remainder. Answering one of no rows over no values would hand the
7254        // caller a division to special case, and the counts above already answer this column.
7255        assert_eq!(common.remainder(column), None);
7256        fs::remove_file(&path).expect("clean up");
7257    }
7258
7259    /// A table directory with nothing in it but a name and one column, for the section tests.
7260    ///
7261    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
7262    /// say so by starting from the emptiest table that encodes.
7263    fn bare_table(sections: Vec<Section>) -> Table {
7264        Table {
7265            name: "linked".to_owned(),
7266            fields: vec![Field::required("id", LogicalType::Integer)],
7267            stripes: Vec::new(),
7268            rows: 0,
7269            dictionaries: vec![None],
7270            distincts: vec![None],
7271            frequencies: vec![None],
7272            clustering: None,
7273            generation: 1,
7274            sections,
7275        }
7276    }
7277
7278    fn a_key_map_section() -> Section {
7279        Section {
7280            kind: *section::KEY_MAP,
7281            id: 1,
7282            generation: 3,
7283            extents: 1,
7284            extent_page: HEADER,
7285            extent_bytes: section::EXTENT_BYTES as u32,
7286            hash: 0x1234_5678_9abc_def0,
7287            flags: 0,
7288            header_bytes: 24,
7289        }
7290    }
7291
7292    #[test]
7293    fn a_section_table_round_trips_through_a_directory() {
7294        let mut later = a_key_map_section();
7295        later.kind = *b"RUDBZZ9\0";
7296        later.id = 2;
7297        let table = bare_table(vec![a_key_map_section(), later]);
7298        let directory = encode_directory(&table).expect("directory");
7299        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7300        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
7301        // The second is a kind this build has no name for, and it survived the round trip anyway.
7302        // That is what keeps an old build from silently discarding a newer build's work when it
7303        // rewrites a directory.
7304        assert!(decoded.sections()[0].known());
7305        assert!(!decoded.sections()[1].known());
7306    }
7307
7308    #[test]
7309    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
7310        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
7311        // build's directory with the trailing section block cut off, so cutting it off is the
7312        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
7313        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7314        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
7315        let older = &directory[..directory.len() - block];
7316        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
7317        assert!(decoded.sections().is_empty());
7318        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
7319        assert_eq!(decoded.name(), "linked");
7320        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
7321    }
7322
7323    #[test]
7324    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
7325        // The same criterion end to end, which is the one the milestone actually asks for: a build
7326        // that knows about sections opens a file written by a build that did not, with no rewrite
7327        // and no repair, and answers from it. The version field is patched rather than a file
7328        // committed by an old binary because the bytes either side of it are identical: format 22
7329        // and format 23 differ only in a trailing directory block, and a reader that stops before
7330        // that block gets a table with no sections.
7331        let path = path("format_twenty_two");
7332        let mut writer =
7333            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7334                .expect("new file");
7335        let rows = Chunk::new(vec![
7336            Vector::from_values(
7337                LogicalType::Integer,
7338                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
7339            )
7340            .expect("integers"),
7341        ])
7342        .expect("one column");
7343        writer.append(&rows).expect("the only part");
7344        writer.finish().expect("commit");
7345
7346        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7347        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7348        drop(file);
7349
7350        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
7351        assert_eq!(reader.table().rows(), 3);
7352        assert!(reader.table().sections().is_empty());
7353
7354        // And a format this build has never written is still refused, so the accept set is a list
7355        // and not an absence of a check.
7356        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7357        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
7358        drop(file);
7359        let error = Reader::open(&path).expect_err("format 21 is not readable");
7360        assert!(error.to_string().contains("format 21"), "{error}");
7361
7362        fs::remove_file(&path).expect("clean up");
7363    }
7364
7365    #[test]
7366    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
7367        // The bound the format has to check and `section` cannot, because only the reader knows how
7368        // big the file is. Reading the payload a section like this names would be reading whatever
7369        // else happens to be at that offset, which is the one way a graph section could turn into a
7370        // wrong answer rather than a slow one.
7371        let mut past = a_key_map_section();
7372        past.extent_page = 1 << 30;
7373        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
7374        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
7375        assert!(error.to_string().contains("outside the file"), "{error}");
7376
7377        let mut inside_the_header = a_key_map_section();
7378        inside_the_header.extent_page = 8;
7379        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
7380        assert!(
7381            decode_directory(&directory, 1 << 20).is_err(),
7382            "a section may not overlap a header"
7383        );
7384    }
7385
7386    #[test]
7387    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
7388        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
7389        // that `rudb_links()` can report what a larger budget would buy. That record is a section
7390        // entry with no extents, so it has to survive a round trip while naming nothing.
7391        let not_built = Section {
7392            kind: *section::FORWARD_LINK,
7393            id: 9,
7394            generation: 3,
7395            extents: 0,
7396            extent_page: 0,
7397            extent_bytes: 0,
7398            hash: 0,
7399            flags: 0,
7400            header_bytes: 0,
7401        };
7402        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
7403        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7404        assert_eq!(decoded.sections(), &[not_built]);
7405
7406        // But a section with no extents that still names an extent table is incoherent, and an
7407        // incoherent entry is a torn directory rather than a relationship that was skipped.
7408        let mut incoherent = not_built;
7409        incoherent.extent_bytes = 28;
7410        incoherent.extent_page = HEADER;
7411        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
7412        assert!(decode_directory(&directory, 1 << 20).is_err());
7413    }
7414
7415    #[test]
7416    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
7417        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7418        let mut torn = directory.clone();
7419        let count_at = torn.len() - size_of::<u16>();
7420        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
7421        // Not an allocation of sixty five thousand entries off a torn count: either the bound
7422        // refuses it or the bytes run out, and both are errors rather than a read past the end.
7423        assert!(decode_directory(&torn, 1 << 20).is_err());
7424    }
7425
7426    /// A committed one column file of `rows` integers, for the attach tests.
7427    fn linked_file(label: &str, rows: i32) -> PathBuf {
7428        let path = path(label);
7429        let mut writer =
7430            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7431                .expect("new file");
7432        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
7433        let chunk =
7434            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
7435                .expect("one column");
7436        writer.append(&chunk).expect("the only part");
7437        writer.finish().expect("commit");
7438        path
7439    }
7440
7441    fn a_key_map_payload() -> Vec<u8> {
7442        // Shaped like one without being one: this crate never reads a payload, so what matters here
7443        // is that every byte comes back and that the header the entry measures is at the front.
7444        (0..512_u32).flat_map(u32::to_le_bytes).collect()
7445    }
7446
7447    #[test]
7448    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
7449        let path = linked_file("attach", 64);
7450        let payload = a_key_map_payload();
7451        let table = attach(
7452            &path,
7453            "items",
7454            &[section::Attachment {
7455                kind: *section::KEY_MAP,
7456                id: 0,
7457                flags: 2,
7458                header_bytes: 40,
7459                bytes: &payload,
7460            }],
7461        )
7462        .expect("attach a key map");
7463        assert_eq!(table.sections().len(), 1);
7464
7465        let reader = Reader::open(&path).expect("reopen after the attach");
7466        let held = reader.table().sections();
7467        assert_eq!(held.len(), 1);
7468        assert_eq!(held[0].kind, *section::KEY_MAP);
7469        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
7470        assert_eq!(held[0].header_bytes, 40);
7471        // The generation is the one the pages were written at, not the one the attach committed at.
7472        // Attaching a section moved no row, so a section written by it is current, and a second
7473        // table added to this file later would not make it stale.
7474        assert_eq!(held[0].generation, 1);
7475        assert!(held[0].usable(reader.table().generation()));
7476        assert_eq!(reader.payload(&held[0]).expect("read the payload"), payload);
7477        assert_eq!(reader.extents(&held[0]).expect("extent table").len(), 1);
7478
7479        fs::remove_file(&path).expect("clean up");
7480    }
7481
7482    #[test]
7483    fn attaching_a_section_answers_every_row_exactly_as_before() {
7484        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
7485        // file with a section in it and the same file without one have to agree row for row, so the
7486        // comparison is made against the answers taken before the attach rather than against a
7487        // constant somebody typed.
7488        let path = linked_file("attach_changes_nothing", 300);
7489        let before = Reader::open(&path).expect("open before");
7490        let rows = before.table().rows();
7491        let first = before.read(0, &[0]).expect("read before");
7492        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
7493        let layout = before.layout().columns_total();
7494        drop(before);
7495
7496        let payload = a_key_map_payload();
7497        attach(
7498            &path,
7499            "items",
7500            &[section::Attachment {
7501                kind: *section::KEY_MAP,
7502                id: 0,
7503                flags: 0,
7504                header_bytes: 0,
7505                bytes: &payload,
7506            }],
7507        )
7508        .expect("attach");
7509
7510        let after = Reader::open(&path).expect("open after");
7511        assert_eq!(after.table().rows(), rows);
7512        let read = after.read(0, &[0]).expect("read after");
7513        for (at, value) in values.iter().enumerate() {
7514            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
7515        }
7516        assert_eq!(
7517            after.layout().columns_total(),
7518            layout,
7519            "an attach appends and does not rewrite a column page"
7520        );
7521
7522        fs::remove_file(&path).expect("clean up");
7523    }
7524
7525    #[test]
7526    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
7527        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
7528        // replaced, a table rebuilt a few times would name several maps for one column and a reader
7529        // would have to pick, which is a decision with no right answer in it.
7530        let path = linked_file("attach_twice", 32);
7531        let one = a_key_map_payload();
7532        let two = vec![7_u8; 1024];
7533        let entry = |bytes| section::Attachment {
7534            kind: *section::KEY_MAP,
7535            id: 4,
7536            flags: 1,
7537            header_bytes: 0,
7538            bytes,
7539        };
7540        attach(&path, "items", &[entry(&one)]).expect("first build");
7541        attach(&path, "items", &[entry(&two)]).expect("rebuild");
7542
7543        let reader = Reader::open(&path).expect("reopen");
7544        let held = reader.table().sections();
7545        assert_eq!(held.len(), 1, "one map per column and not one per build");
7546        assert_eq!(reader.payload(&held[0]).expect("payload"), two);
7547
7548        fs::remove_file(&path).expect("clean up");
7549    }
7550
7551    #[test]
7552    fn an_attach_carries_through_a_kind_it_does_not_know() {
7553        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
7554        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
7555        // an older binary and attaching one section quietly deletes the work of a newer one.
7556        let path = linked_file("attach_unknown", 16);
7557        let payload = vec![3_u8; 96];
7558        attach(
7559            &path,
7560            "items",
7561            &[section::Attachment {
7562                kind: *b"RUDBZZ9\0",
7563                id: 1,
7564                flags: 0,
7565                header_bytes: 0,
7566                bytes: &payload,
7567            }],
7568        )
7569        .expect("a kind this build does not know still writes");
7570        let key_map = a_key_map_payload();
7571        attach(
7572            &path,
7573            "items",
7574            &[section::Attachment {
7575                kind: *section::KEY_MAP,
7576                id: 0,
7577                flags: 0,
7578                header_bytes: 0,
7579                bytes: &key_map,
7580            }],
7581        )
7582        .expect("attach beside it");
7583
7584        let reader = Reader::open(&path).expect("reopen");
7585        let held = reader.table().sections();
7586        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
7587        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
7588        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
7589
7590        fs::remove_file(&path).expect("clean up");
7591    }
7592
7593    #[test]
7594    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
7595        let path = linked_file("attach_not_built", 8);
7596        attach(
7597            &path,
7598            "items",
7599            &[section::Attachment {
7600                kind: *section::FORWARD_LINK,
7601                id: 2,
7602                flags: 0,
7603                header_bytes: 0,
7604                bytes: &[],
7605            }],
7606        )
7607        .expect("record a link that did not fit the budget");
7608
7609        let reader = Reader::open(&path).expect("reopen");
7610        let held = reader.table().sections();
7611        assert_eq!(held.len(), 1);
7612        assert_eq!(held[0].extents, 0);
7613        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
7614        assert!(reader.extents(&held[0]).expect("no extent table").is_empty());
7615        assert!(reader.payload(&held[0]).expect("no payload").is_empty());
7616
7617        fs::remove_file(&path).expect("clean up");
7618    }
7619
7620    #[test]
7621    fn a_payload_past_one_extent_is_split_and_joined_back() {
7622        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
7623        // payload that has to be two extents, and it is the case a split written for the common
7624        // size gets wrong.
7625        let path = linked_file("attach_two_extents", 8);
7626        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
7627        attach(
7628            &path,
7629            "items",
7630            &[section::Attachment {
7631                kind: *section::KEY_MAP,
7632                id: 0,
7633                flags: 0,
7634                header_bytes: 0,
7635                bytes: &payload,
7636            }],
7637        )
7638        .expect("attach a payload past the bound");
7639
7640        let reader = Reader::open(&path).expect("reopen");
7641        let held = reader.table().sections();
7642        let extents = reader.extents(&held[0]).expect("extent table");
7643        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
7644        assert_eq!(extents[0].length, section::MAX_EXTENT);
7645        assert_eq!(extents[1].length, 1);
7646        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
7647        // And the extent the caller wants is readable on its own, which is the point of the split.
7648        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
7649        assert_eq!(reader.payload(&held[0]).expect("the whole payload").len(), payload.len());
7650
7651        fs::remove_file(&path).expect("clean up");
7652    }
7653
7654    #[test]
7655    fn a_torn_extent_is_refused_rather_than_decoded() {
7656        let path = linked_file("attach_torn", 8);
7657        let payload = a_key_map_payload();
7658        attach(
7659            &path,
7660            "items",
7661            &[section::Attachment {
7662                kind: *section::KEY_MAP,
7663                id: 0,
7664                flags: 0,
7665                header_bytes: 0,
7666                bytes: &payload,
7667            }],
7668        )
7669        .expect("attach");
7670
7671        let reader = Reader::open(&path).expect("reopen");
7672        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
7673        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
7674        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
7675        drop(file);
7676
7677        let reader = Reader::open(&path).expect("the table still opens");
7678        let error = reader
7679            .payload(&reader.table().sections()[0])
7680            .expect_err("a corrupt payload is not handed out");
7681        assert!(error.to_string().contains("checksum"), "{error}");
7682        // And the table is still readable, which is section 3.1: a section that cannot be trusted
7683        // costs the query its shortcut and nothing else.
7684        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
7685
7686        fs::remove_file(&path).expect("clean up");
7687    }
7688
7689    #[test]
7690    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
7691        // Readable is not writable. A format 22 directory has no section block, and adding one
7692        // without moving the number in the header would leave a file claiming a format it is not.
7693        let path = linked_file("attach_old_format", 8);
7694        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7695        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7696        drop(file);
7697
7698        let payload = a_key_map_payload();
7699        let error = attach(
7700            &path,
7701            "items",
7702            &[section::Attachment {
7703                kind: *section::KEY_MAP,
7704                id: 0,
7705                flags: 0,
7706                header_bytes: 0,
7707                bytes: &payload,
7708            }],
7709        )
7710        .expect_err("format 22 cannot gain a section");
7711        assert!(error.to_string().contains("format 22"), "{error}");
7712        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
7713
7714        fs::remove_file(&path).expect("clean up");
7715    }
7716
7717    #[test]
7718    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
7719        let path = linked_file("attach_bad_header", 8);
7720        let error = attach(
7721            &path,
7722            "items",
7723            &[section::Attachment {
7724                kind: *section::KEY_MAP,
7725                id: 0,
7726                flags: 0,
7727                header_bytes: 40,
7728                bytes: &[1, 2, 3],
7729            }],
7730        )
7731        .expect_err("a writer's bug stops at the write");
7732        assert!(error.to_string().contains("header is longer"), "{error}");
7733
7734        fs::remove_file(&path).expect("clean up");
7735    }
7736
7737    #[test]
7738    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
7739        let path = linked_file("attach_wrong_name", 8);
7740        let error = attach(&path, "orders", &[]).expect_err("no such table");
7741        assert!(error.to_string().contains("orders"), "{error}");
7742        fs::remove_file(&path).expect("clean up");
7743    }
7744
7745    #[test]
7746    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
7747        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
7748        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
7749        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
7750        // the tail is outside it. The counts inside it are still exact, because the pass recounts
7751        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
7752        // twenty six a distinct count of 601 would divide its way to.
7753        let path = path("frequency_prefix_for_the_planner");
7754        let mut writer =
7755            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7756                .expect("new file");
7757        let mut values = vec![Value::Integer(1); 10_000];
7758        for _ in 0..10 {
7759            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
7760        }
7761        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
7762        // synopsis walks the whole column rather than a part, so the counts are the same either way.
7763        for part in values.chunks(8_000) {
7764            let rows = Chunk::new(vec![
7765                Vector::from_values(LogicalType::Integer, part).expect("integers"),
7766            ])
7767            .expect("one column");
7768            writer.append(&rows).expect("a part");
7769        }
7770        writer.finish().expect("commit");
7771        let reader = Reader::open(&path).expect("reopen from disk");
7772        let prefix =
7773            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
7774        // A prefix and not the whole column, and the writer said how many rows anything left out of
7775        // it can hold.
7776        assert_eq!(prefix.entries.len(), 512);
7777        assert_eq!(prefix.omitted_max, 10);
7778        let common = Common::new(reader);
7779        assert_eq!(common.rows(), 16_000);
7780        let column = common.column("id").expect("the file has that column");
7781        assert_eq!(
7782            common.rows_with(column, &Bound::Int(1)),
7783            Stat::exact(10_000, Provenance::FrequencySynopsis)
7784        );
7785        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
7786        assert_eq!(
7787            common.rows_with(column, &Bound::Int(1_100)),
7788            Stat::exact(10, Provenance::FrequencySynopsis)
7789        );
7790        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
7791        // what a complete list would say, and the file holds ten rows of this one.
7792        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
7793        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
7794        // two apart, which is the whole of what it gives up.
7795        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
7796        // What the prefix left out, which is what turns the unknown above into a number. The 512
7797        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
7798        // and 890 over 89 is the ten rows each of them really holds.
7799        let remainder = common.remainder(column).expect("the list is a prefix");
7800        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
7801        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
7802        fs::remove_file(&path).expect("clean up");
7803    }
7804
7805    /// A file with no table in it is a file, and opening it says so rather than failing.
7806    #[test]
7807    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
7808        let path = path("empty");
7809        Writer::empty(&path, &[]).expect("a file with nothing in it");
7810        let catalog = Catalog::open(&path).expect("the empty file opens");
7811        assert_eq!(catalog.len(), 0);
7812        assert!(catalog.is_empty());
7813        assert_eq!(catalog.names().count(), 0);
7814        // The next generation goes over the top of it the way it goes over any other, which is what
7815        // says this is a committed file and not a special case somebody has to know about.
7816        let mut writer =
7817            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7818                .expect("a table goes into the empty file");
7819        writer.append(&sample_ids()).expect("rows");
7820        writer.finish().expect("commit");
7821        let catalog = Catalog::open(&path).expect("the file opens again");
7822        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7823        fs::remove_file(&path).expect("clean up");
7824    }
7825
7826    /// A committed table with no rows is a name the next generation takes over, and one with rows
7827    /// is a name it refuses.
7828    ///
7829    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
7830    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
7831    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
7832    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
7833    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
7834    /// instead of through memory.
7835    #[test]
7836    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
7837        let path = path("empty-name");
7838        let field = || vec![Field::required("id", LogicalType::Integer)];
7839        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
7840        let catalog = Catalog::open(&path).expect("the file opens");
7841        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
7842
7843        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
7844        writer.append(&sample_ids()).expect("rows");
7845        writer.finish().expect("commit");
7846        let catalog = Catalog::open(&path).expect("the file opens again");
7847        // One entry and not two. The generation replaced the empty table rather than joining it.
7848        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7849        let held = catalog.rows().collect::<Vec<_>>();
7850        assert_eq!(held.len(), 1);
7851        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
7852
7853        // The same call against the same name now that it holds rows, which is still refused.
7854        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
7855        assert!(error.to_string().contains("same name"), "{error}");
7856        fs::remove_file(&path).expect("clean up");
7857    }
7858
7859    /// A view, with everything about it that a reopened catalog has to be able to answer from.
7860    fn sample_view(name: &str) -> ViewEntry {
7861        ViewEntry {
7862            name: name.to_string(),
7863            sql: "SELECT id FROM items WHERE id > 0".to_string(),
7864            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
7865            aliases: vec!["n".to_string()],
7866            columns: vec![Field::new("n", LogicalType::Integer)],
7867        }
7868    }
7869
7870    #[test]
7871    fn a_view_written_into_the_catalog_comes_back_whole() {
7872        let path = path("views");
7873        let mut writer =
7874            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7875                .expect("new file");
7876        writer.append(&sample_ids()).expect("rows");
7877        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
7878        let catalog = Catalog::open(&path).expect("reopen");
7879        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
7880        // The tables are still there and are still read the same way, so the section on the end did
7881        // not move anything in front of it.
7882        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7883        fs::remove_file(&path).expect("clean up");
7884    }
7885
7886    /// A writer opened to append a table says nothing about views and must not lose them.
7887    #[test]
7888    fn appending_a_table_carries_the_views_forward() {
7889        let path = path("viewscarry");
7890        let mut writer =
7891            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7892                .expect("new file");
7893        writer.append(&sample_ids()).expect("rows");
7894        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
7895        let mut writer =
7896            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
7897                .expect("a second table");
7898        writer.append(&sample_ids()).expect("rows");
7899        writer.finish().expect("commit");
7900        let catalog = Catalog::open(&path).expect("reopen");
7901        assert_eq!(catalog.views().count(), 1);
7902        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
7903        fs::remove_file(&path).expect("clean up");
7904    }
7905
7906    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
7907    #[test]
7908    fn restating_the_views_leaves_every_table_where_it_was() {
7909        let path = path("restate");
7910        let mut writer =
7911            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7912                .expect("new file");
7913        writer.append(&sample_ids()).expect("rows");
7914        writer.finish().expect("commit");
7915        let before = fs::metadata(&path).expect("the file is there").len();
7916        Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
7917        let catalog = Catalog::open(&path).expect("reopen");
7918        assert_eq!(catalog.views().count(), 2);
7919        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7920        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
7921        // than the size of the table.
7922        let after = fs::metadata(&path).expect("the file is there").len();
7923        assert!(after > before, "a generation was written");
7924        assert!(after - before < before, "the table was not written again");
7925        // The rows are still readable through the new generation, which is the part that would go
7926        // wrong if the catalog carried the wrong directory pointers forward.
7927        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
7928        assert_eq!(reader.table().rows, 3);
7929        // And a restate over a restate keeps working, because each one reads the slot that
7930        // checksummed rather than the highest number in the header.
7931        Writer::restate(&path, &[]).expect("no views at all");
7932        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
7933        fs::remove_file(&path).expect("clean up");
7934    }
7935
7936    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
7937    #[test]
7938    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
7939        let bytes = encode_catalog(
7940            &[Entry {
7941                name: "items".to_string(),
7942                fields: vec![Field::required("id", LogicalType::Integer)],
7943                rows: 1,
7944                directory: Page { offset: HEADER, length: 8, hash: 0 },
7945            }],
7946            &[sample_view("items")],
7947        )
7948        .expect("it encodes, because encoding does not look");
7949        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
7950        assert!(error.to_string().contains("same name"), "{error}");
7951    }
7952
7953    #[test]
7954    fn committed_file_reopens_and_reads_only_requested_columns() {
7955        let path = path("reopen");
7956        let mut writer = Writer::create(
7957            &path,
7958            "items",
7959            vec![
7960                Field::required("id", LogicalType::Integer),
7961                Field::new("text", LogicalType::Varchar),
7962            ],
7963        )
7964        .expect("new file");
7965        writer.append(&sample()).expect("first part");
7966        writer.append(&sample()).expect("second part");
7967        writer.finish().expect("commit");
7968        let reader = Reader::open(&path).expect("reopen from disk");
7969        assert_eq!(reader.table().rows(), 6);
7970        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
7971        // of the split: the directory describes the stripe and the scan still reads a part.
7972        assert_eq!(reader.table().stripes().len(), 1);
7973        assert_eq!(reader.parts(), 2);
7974        assert_eq!(reader.part_rows(0), 3);
7975        assert_eq!(reader.part_rows(1), 3);
7976        let text = reader.read(1, &[1]).expect("only text page");
7977        assert_eq!(text.width(), 1);
7978        assert_eq!(text.value_at(1, 0), Value::Null);
7979        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7980        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
7981        assert_eq!(sparse.width(), 1);
7982        assert_eq!(sparse.value_at(1, 0), Value::Null);
7983        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7984        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
7985        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
7986        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
7987        let count = reader.read(0, &[]).expect("no page is needed for count");
7988        assert_eq!(count.len(), 3);
7989        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
7990        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
7991        let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
7992        assert_eq!(
7993            integers,
7994            vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
7995        );
7996        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
7997        assert_eq!(strings.len(), 3);
7998        assert!(strings.contains(&(Value::Null, 2)));
7999        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
8000        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
8001        fs::remove_file(path).expect("remove scratch file");
8002    }
8003
8004    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
8005    /// instance.
8006    ///
8007    /// The runs arrive in the order the instances finished reading them rather than in source
8008    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
8009    /// a stripe of its own and the table still reads back in source order, which is the whole of
8010    /// what the writer promises about ordering.
8011    #[test]
8012    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
8013        let path = path("interleaved-runs");
8014        let mut writer =
8015            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
8016                .expect("new file");
8017        for morsel in [2_u64, 0, 3, 1] {
8018            let parts = (0..4_u64)
8019                .map(|chunk| {
8020                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
8021                    let values =
8022                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
8023                    let column =
8024                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
8025                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
8026                })
8027                .collect::<Vec<_>>();
8028            writer.append_stripe(parts).expect("a stripe");
8029        }
8030        writer.finish().expect("commit");
8031
8032        let reader = Reader::open(&path).expect("valid directory");
8033        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
8034        assert_eq!(reader.table().rows(), 128);
8035        for part in 0..16_usize {
8036            let read = reader.read(part, &[0]).expect("a part back");
8037            for row in 0..8_usize {
8038                let want = i64::try_from(part * 8 + row).expect("small");
8039                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
8040            }
8041        }
8042        fs::remove_file(path).expect("remove scratch file");
8043    }
8044
8045    /// Runs from different callers may interleave and may not overlap, and the commit is what
8046    /// catches an overlap.
8047    #[test]
8048    fn runs_that_overlap_each_other_are_refused_at_commit() {
8049        let path = path("overlapping-runs");
8050        let mut writer =
8051            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
8052                .expect("new file");
8053        let one = |order: (u64, u64)| {
8054            let column =
8055                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
8056            (order, Chunk::new(vec![column]).expect("one column"))
8057        };
8058        // The second run sits inside the first rather than after it, which is a thing no instance
8059        // holding its own contiguous run can produce and a thing the file cannot represent.
8060        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
8061        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
8062        let error = writer.finish().expect_err("the runs overlap");
8063        assert!(error.message().contains("source order"), "{error}");
8064        fs::remove_file(path).expect("remove scratch file");
8065    }
8066
8067    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
8068    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
8069    #[test]
8070    fn a_run_longer_than_a_stripe_is_refused() {
8071        let path = path("overlong-run");
8072        let mut writer =
8073            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
8074                .expect("new file");
8075        let parts = (0..=STRIPE_PARTS)
8076            .map(|at| {
8077                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
8078                    .expect("a column");
8079                let chunk = Chunk::new(vec![column]).expect("one column");
8080                ((0, u64::try_from(at).expect("small")), chunk)
8081            })
8082            .collect::<Vec<_>>();
8083        let error = writer.append_stripe(parts).expect_err("one part too many");
8084        assert!(error.message().contains("more parts than it holds"), "{error}");
8085        fs::remove_file(path).expect("remove scratch file");
8086    }
8087
8088    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
8089    ///
8090    /// This is the shape the format exists for, so both ends of the split are checked here. The
8091    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
8092    /// part still answers with that part's rows rather than with its whole stripe's.
8093    #[test]
8094    fn parts_past_the_stripe_bound_start_a_new_stripe() {
8095        let path = path("stripe-bound");
8096        let mut writer = Writer::create(
8097            &path,
8098            "items",
8099            vec![
8100                Field::required("id", LogicalType::Integer),
8101                Field::new("text", LogicalType::Varchar),
8102            ],
8103        )
8104        .expect("new file");
8105        let parts = STRIPE_PARTS * 2 + 3;
8106        for part in 0..parts {
8107            let id = part as i32;
8108            let chunk = Chunk::new(vec![
8109                Vector::from_values(
8110                    LogicalType::Integer,
8111                    &[Value::Integer(id), Value::Integer(-id)],
8112                )
8113                .expect("integers"),
8114                Vector::from_values(
8115                    LogicalType::Varchar,
8116                    &[Value::Varchar(format!("value {part}")), Value::Null],
8117                )
8118                .expect("strings"),
8119            ])
8120            .expect("matching rows");
8121            writer.append(&chunk).expect("one part");
8122        }
8123        writer.finish().expect("commit");
8124
8125        let reader = Reader::open(&path).expect("reopen from disk");
8126        assert_eq!(reader.parts(), parts);
8127        assert_eq!(reader.table().rows(), parts * 2);
8128        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
8129        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
8130        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
8131        assert_eq!(reader.table().stripes()[2].parts(), 3);
8132        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
8133        // table the other way is what catches a cache that only ever holds what it just read.
8134        for part in (0..parts).rev() {
8135            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
8136            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
8137            for chunk in [&dense, &sparse] {
8138                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
8139                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8140                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
8141                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
8142                assert_eq!(chunk.value_at(1, 1), Value::Null);
8143            }
8144        }
8145        // The bounds are merged over the stripe, so they answer for the range the whole stripe
8146        // covers and not for the part that was asked about.
8147        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
8148        assert!(reader.skips(0, &above), "the first stripe stops at 63");
8149        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
8150        fs::remove_file(path).expect("remove scratch file");
8151    }
8152
8153    /// A scattered value in the column that decides `WHERE UserID = ?`.
8154    fn scattered(n: i64) -> i64 {
8155        n.wrapping_mul(-7_046_029_254_386_353_131)
8156    }
8157
8158    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
8159    ///
8160    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
8161    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
8162    /// holds the value is the only one a scan has to read.
8163    #[test]
8164    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
8165        let path = path("sieve-skip");
8166        let mut writer =
8167            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
8168                .expect("new file");
8169        let parts = STRIPE_PARTS + 3;
8170        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
8171        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
8172        // that small costs about as much to read as the rows do and is no longer written.
8173        let per_part = 128;
8174        for part in 0..parts {
8175            let held: Vec<Value> = (0..per_part)
8176                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
8177                .collect();
8178            let chunk =
8179                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8180                    .expect("one column");
8181            writer.append(&chunk).expect("one part");
8182        }
8183        writer.finish().expect("commit");
8184
8185        let reader = Reader::open(&path).expect("reopen from disk");
8186        let probe = |value: i64| Probe {
8187            column: 0,
8188            op: Op::Equal,
8189            value: Bound::Int(i128::from(scattered(value))),
8190        };
8191        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
8192            let tests = [probe(wanted)];
8193            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
8194            let home = wanted as usize / per_part;
8195            assert!(kept.contains(&home), "the part holding {wanted} is read");
8196            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
8197            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
8198            // stray part across the whole file and that is what this leaves room for.
8199            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
8200        }
8201        let absent = [probe((parts * per_part) as i64 + 1)];
8202        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
8203        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
8204        // The same probes against the bounds alone, which is what this replaces. A column of
8205        // scattered numbers has a range per stripe that covers nearly the whole type.
8206        let tests = [probe(0)];
8207        assert!(
8208            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
8209            "the bounds rule out no stripe at all"
8210        );
8211        fs::remove_file(path).expect("remove scratch file");
8212    }
8213
8214    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
8215    ///
8216    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
8217    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
8218    /// rules out none of it and rules out all but a few parts.
8219    #[test]
8220    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
8221        let path = path("part-range-skip");
8222        let mut writer =
8223            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8224                .expect("new file");
8225        let parts = STRIPE_PARTS + 3;
8226        let per_part = 128;
8227        for part in 0..parts {
8228            // Scattered inside the part's own band rather than a run, because a run of
8229            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
8230            // costs more than reading the column it indexes, which is the case the writer declines.
8231            let held: Vec<Value> = (0..per_part)
8232                .map(|row| {
8233                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
8234                })
8235                .collect();
8236            let chunk =
8237                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8238                    .expect("one column");
8239            writer.append(&chunk).expect("one part");
8240        }
8241        writer.finish().expect("commit");
8242
8243        let reader = Reader::open(&path).expect("reopen from disk");
8244        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
8245        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
8246        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
8247        // The same question asked of the stripe alone, which is what this replaces.
8248        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
8249        fs::remove_file(path).expect("remove scratch file");
8250    }
8251
8252    /// The other half of the same page. A part whose own bounds put every row of it inside the
8253    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
8254    /// across every part and can prove nothing.
8255    #[test]
8256    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
8257        let path = path("part-range-certain");
8258        let mut writer =
8259            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8260                .expect("new file");
8261        let parts = STRIPE_PARTS + 3;
8262        let per_part = 128;
8263        for part in 0..parts {
8264            let held: Vec<Value> = (0..per_part)
8265                .map(|row| {
8266                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
8267                })
8268                .collect();
8269            let chunk =
8270                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8271                    .expect("one column");
8272            writer.append(&chunk).expect("one part");
8273        }
8274        writer.finish().expect("commit");
8275
8276        let reader = Reader::open(&path).expect("reopen from disk");
8277        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
8278        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
8279        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
8280        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
8281        // and settles nothing either way. The three yeses above are the parts' own ends talking.
8282        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
8283        fs::remove_file(path).expect("remove scratch file");
8284    }
8285
8286    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
8287    /// that has a single part, where the stripe bounds already are the part's.
8288    #[test]
8289    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
8290        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
8291            let path = path("part-range-page");
8292            let mut writer =
8293                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8294                    .expect("new file");
8295            for part in 0..parts {
8296                let held: Vec<Value> = (0..128)
8297                    .map(|row| {
8298                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
8299                    })
8300                    .collect();
8301                let chunk = Chunk::new(vec![
8302                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
8303                ])
8304                .expect("one column");
8305                writer.append(&chunk).expect("one part");
8306            }
8307            writer.finish().expect("commit");
8308            let reader = Reader::open(&path).expect("reopen from disk");
8309            let bytes = reader.layout().columns[0].part_ranges;
8310            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
8311            fs::remove_file(path).expect("remove scratch file");
8312        }
8313    }
8314
8315    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
8316    /// a shortened bound from turning a skip into a wrong answer.
8317    #[test]
8318    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
8319        let long = vec![b'a'; PART_BOUND_BYTES * 2];
8320        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
8321        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
8322        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
8323        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
8324        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
8325        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
8326        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
8327    }
8328
8329    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
8330    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
8331    #[test]
8332    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
8333        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
8334        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
8335        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
8336        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
8337    }
8338
8339    /// What a column is stored as, asked of two files holding the same rows in a different order.
8340    ///
8341    /// This is the question the report exists to answer and it is the one the directory cannot. The
8342    /// two files have the same rows, the same schema and the same number of parts, and the column
8343    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
8344    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
8345    /// says so, and reading it is what this does.
8346    ///
8347    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
8348    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
8349    /// pays for the wider ones.
8350    #[test]
8351    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
8352        let parts = 4;
8353        let per_part = 1024;
8354        let rows = parts * per_part;
8355        let written = |name: &str, keys: &[i64]| {
8356            let path = path(name);
8357            let fields = vec![Field::required("key", LogicalType::BigInt)];
8358            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
8359            for part in 0..parts {
8360                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
8361                    .iter()
8362                    .map(|key| Value::BigInt(*key))
8363                    .collect();
8364                let chunk = Chunk::new(vec![
8365                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
8366                ])
8367                .expect("one column");
8368                writer.append(&chunk).expect("one part");
8369            }
8370            writer.finish().expect("commit");
8371            path
8372        };
8373        // Ascending with a small irregular step, which is what a key column in arrival order looks
8374        // like: an order has one to seven line items, so the key repeats and then moves on by one.
8375        let climbing = |step: &dyn Fn(usize) -> i64| {
8376            let mut key = 0;
8377            (0..rows)
8378                .map(|row| {
8379                    key += step(row);
8380                    key
8381                })
8382                .collect::<Vec<i64>>()
8383        };
8384        let ascending = climbing(&|row| (row % 3) as i64);
8385        // The same rows in the same direction over a range a thousand times wider, which is what a
8386        // partition of a clustered table holds: still ascending, and far enough apart that the
8387        // deltas no longer fit in a handful of bits.
8388        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
8389        let near_path = written("stored-near", &ascending);
8390        let far_path = written("stored-far", &sparse);
8391
8392        let one = Reader::open(&near_path).expect("reopen from disk");
8393        let other = Reader::open(&far_path).expect("reopen from disk");
8394        let near = one.stored(0).expect("the column is stored");
8395        let far = other.stored(0).expect("the column is stored");
8396        assert_eq!(near.len(), parts, "one row per part");
8397        assert_eq!(far.len(), parts);
8398        // The bytes are the same bytes the directory totals, which is the check that this is
8399        // reading the pages the file really holds rather than some other pages.
8400        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
8401        assert_eq!(total(&near), one.layout().columns[0].pages);
8402        assert_eq!(total(&far), other.layout().columns[0].pages);
8403        assert!(
8404            total(&near) * 2 < total(&far),
8405            "the sparse keys cost more, {} against {}",
8406            total(&far),
8407            total(&near)
8408        );
8409        // Every part accounted for, in order, with the row it starts at following the one before.
8410        for (at, part) in near.iter().enumerate() {
8411            assert_eq!(part.part, at);
8412            assert_eq!(part.row, at * per_part);
8413            assert_eq!(part.rows, per_part);
8414            let held = &ascending[at * per_part..(at + 1) * per_part];
8415            assert_eq!(part.low, Some(Value::BigInt(held[0])));
8416            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
8417            assert_eq!(part.nulls, Some(0));
8418        }
8419        // And the encoding is a line of text that names what the encoder chose, which is the whole
8420        // point. Both are a cascade over deltas and the widths inside them are what differ.
8421        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
8422        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
8423        assert_ne!(near[0].encoding, far[0].encoding);
8424        fs::remove_file(near_path).expect("remove scratch file");
8425        fs::remove_file(far_path).expect("remove scratch file");
8426    }
8427
8428    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
8429    ///
8430    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
8431    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
8432    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
8433    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
8434    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
8435    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
8436    /// the part, every time, and that is the case this drops.
8437    #[test]
8438    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
8439        let path = path("sieve-pays");
8440        let fields = vec![
8441            Field::required("spread", LogicalType::BigInt),
8442            Field::required("repeated", LogicalType::BigInt),
8443        ];
8444        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
8445        let parts = 3;
8446        let per_part = 1024;
8447        for part in 0..parts {
8448            let base = (part * per_part) as i64;
8449            let spread: Vec<Value> =
8450                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
8451            let repeated: Vec<Value> =
8452                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
8453            let chunk = Chunk::new(vec![
8454                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
8455                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
8456            ])
8457            .expect("two columns");
8458            writer.append(&chunk).expect("one part");
8459        }
8460        writer.finish().expect("commit");
8461
8462        let reader = Reader::open(&path).expect("reopen from disk");
8463        let layout = reader.layout();
8464        let spread = &layout.columns[0];
8465        let repeated = &layout.columns[1];
8466        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
8467        assert_eq!(
8468            repeated.sieves, 0,
8469            "a column whose filter costs more than its parts keeps none"
8470        );
8471        // Per part this is the rule itself, so it holds over the column as well: a part without a
8472        // sieve adds to one side of this and to nothing on the other.
8473        for column in &layout.columns {
8474            assert!(
8475                column.sieves < column.pages,
8476                "{} spends {} on sieves over {} of data",
8477                column.name,
8478                column.sieves,
8479                column.pages
8480            );
8481        }
8482        // The filter that was kept still does what it is for.
8483        let absent = [Probe {
8484            column: 0,
8485            op: Op::Equal,
8486            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
8487        }];
8488        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
8489        fs::remove_file(path).expect("remove scratch file");
8490    }
8491
8492    /// A damaged sieve page is a part that gets read, not a query that fails.
8493    ///
8494    /// A sieve is an index over rows that are still there and still correct, so losing one costs
8495    /// time and costs no answers. That is the opposite of the membership index beside it, which is
8496    /// the only thing standing between a string page and a wrong answer.
8497    #[test]
8498    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
8499        let path = path("sieve-damaged");
8500        let mut writer =
8501            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
8502                .expect("new file");
8503        let rows = 128;
8504        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
8505        let chunk =
8506            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8507                .expect("one column");
8508        writer.append(&chunk).expect("one part");
8509        writer.finish().expect("commit");
8510
8511        let page =
8512            Reader::open(&path).expect("reopen").table.stripes[0].sieves[0].expect("a sieve page");
8513        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
8514        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
8515        file.write_all(&[0xff]).expect("damage one byte");
8516        drop(file);
8517
8518        let reader = Reader::open(&path).expect("reopen the damaged file");
8519        let absent =
8520            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
8521        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
8522        assert_eq!(
8523            reader.read(0, &[0]).expect("the rows are untouched").len(),
8524            usize::try_from(rows).expect("a small count")
8525        );
8526        fs::remove_file(path).expect("remove scratch file");
8527    }
8528
8529    /// Eight workers over one stripe read it once between them.
8530    ///
8531    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
8532    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
8533    /// started sharing the read every one of them read the whole page. On the full ClickBench file
8534    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
8535    /// column, which is most of what a first touch costs.
8536    ///
8537    /// The workers that lose the race still answer, out of the part reads they do instead, which is
8538    /// what the values below are checking.
8539    #[test]
8540    fn workers_that_want_the_same_stripe_read_it_once() {
8541        let path = path("single-flight");
8542        let mut writer =
8543            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8544                .expect("new file");
8545        for part in 0..STRIPE_PARTS {
8546            let id = part as i32;
8547            let chunk = Chunk::new(vec![
8548                Vector::from_values(
8549                    LogicalType::Integer,
8550                    &[Value::Integer(id), Value::Integer(-id)],
8551                )
8552                .expect("integers"),
8553            ])
8554            .expect("matching rows");
8555            writer.append(&chunk).expect("one part");
8556        }
8557        writer.finish().expect("commit");
8558
8559        let reader = Reader::open(&path).expect("reopen from disk");
8560        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
8561        let barrier = std::sync::Barrier::new(8);
8562        std::thread::scope(|scope| {
8563            for worker in 0..8 {
8564                let reader = &reader;
8565                let barrier = &barrier;
8566                scope.spawn(move || {
8567                    barrier.wait();
8568                    for part in (worker..STRIPE_PARTS).step_by(8) {
8569                        let chunk = reader.read(part, &[0]).expect("a whole page read");
8570                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8571                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
8572                    }
8573                });
8574            }
8575        });
8576        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
8577        fs::remove_file(path).expect("remove scratch file");
8578    }
8579
8580    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
8581    ///
8582    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
8583    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
8584    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
8585    /// the next query will want them, so read them on the way past. A process that opened the
8586    /// database to run one trivial query pays for all of it and gets nothing.
8587    ///
8588    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
8589    /// two openings cost the same. The stripe count is held equal so that the directory is the same
8590    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
8591    /// data would show up here.
8592    #[test]
8593    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
8594        let opened = |label: &str, rows_per_part: i32| {
8595            let path = path(label);
8596            let mut writer =
8597                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8598                    .expect("new file");
8599            for part in 0..STRIPE_PARTS * 3 {
8600                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
8601                // of consecutive integers encodes to almost nothing and would leave the two files
8602                // the same size, which would make this test pass for the wrong reason.
8603                let values = (0..rows_per_part)
8604                    .map(|row| {
8605                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
8606                    })
8607                    .collect::<Vec<_>>();
8608                let chunk = Chunk::new(vec![
8609                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
8610                ])
8611                .expect("matching rows");
8612                writer.append(&chunk).expect("one part");
8613            }
8614            writer.finish().expect("commit");
8615            let reader = Reader::open(&path).expect("reopen from disk");
8616            let size = fs::metadata(&path).expect("the file is there").len();
8617            let out = (reader.reads(), reader.table().stripes().len(), size);
8618            fs::remove_file(path).expect("remove scratch file");
8619            out
8620        };
8621
8622        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
8623        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
8624        assert_eq!(
8625            thin_stripes, fat_stripes,
8626            "the same stripe count is what makes this a fair ask"
8627        );
8628        assert!(
8629            fat_size > thin_size * 50,
8630            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
8631        );
8632
8633        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
8634        assert_eq!(thin.pages, 0, "opening read a page");
8635        assert_eq!(fat.pages, 0, "opening read a page");
8636        assert_eq!(thin.indexes, 0, "opening read an index");
8637        assert_eq!(fat.indexes, 0, "opening read an index");
8638        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
8639        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
8640        assert!(
8641            fat.opening.bytes < thin.opening.bytes * 2,
8642            "opening the thin file read {} bytes and the fat one read {}",
8643            thin.opening.bytes,
8644            fat.opening.bytes
8645        );
8646    }
8647
8648    /// The reads a file costs to open are fixed by its shape and not by what ran before.
8649    ///
8650    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
8651    /// the plan is a function of the data, the generation and the settings, and never of what
8652    /// happened to be in cache. Opening the same file twice in the same process has to cost the
8653    /// same, because a second open that read less would be an open that was about to plan
8654    /// differently.
8655    #[test]
8656    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
8657        let path = path("open-twice");
8658        let mut writer =
8659            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8660                .expect("new file");
8661        for part in 0..STRIPE_PARTS * 3 {
8662            let chunk = Chunk::new(vec![
8663                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8664                    .expect("integers"),
8665            ])
8666            .expect("matching rows");
8667            writer.append(&chunk).expect("one part");
8668        }
8669        writer.finish().expect("commit");
8670
8671        let first = Reader::open(&path).expect("open");
8672        // A whole scan in between, so the operating system's page cache is as warm as it gets and
8673        // anything that consulted it would show up in the second open.
8674        for part in 0..first.parts() {
8675            first.read(part, &[0]).expect("a part");
8676        }
8677        assert!(first.reads().pages > 0, "the scan has to have read something");
8678        let second = Reader::open(&path).expect("open again");
8679
8680        assert_eq!(first.reads().opening, second.reads().opening);
8681        assert_eq!(
8682            second.reads().pages,
8683            0,
8684            "the second open read a page off the back of the first"
8685        );
8686        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
8687        fs::remove_file(path).expect("remove scratch file");
8688    }
8689
8690    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
8691    ///
8692    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
8693    /// stripes than that read the index again every time a stripe came back around. The index is a
8694    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
8695    /// different budgets. This is the test that keeps them there, since the saving is small enough
8696    /// that nothing in a benchmark would notice it going away again.
8697    #[test]
8698    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
8699        let path = path("index-cache");
8700        let mut writer =
8701            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8702                .expect("new file");
8703        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
8704        for part in 0..parts {
8705            let id = part as i32;
8706            let chunk = Chunk::new(vec![
8707                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
8708            ])
8709            .expect("matching rows");
8710            writer.append(&chunk).expect("one part");
8711        }
8712        writer.finish().expect("commit");
8713
8714        let reader = Reader::open(&path).expect("reopen from disk");
8715        let stripes = reader.table().stripes().len();
8716        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
8717        // Twice over, so that the second pass finds every page evicted and every index kept.
8718        for _ in 0..2 {
8719            for part in 0..parts {
8720                let chunk = reader.read(part, &[0]).expect("a part");
8721                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8722            }
8723        }
8724        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
8725        assert!(
8726            reader.pages.load(Atomic::Relaxed) > stripes,
8727            "the pages are the ones that get read again, which is what makes the index count mean \
8728             something"
8729        );
8730        fs::remove_file(path).expect("remove scratch file");
8731    }
8732
8733    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
8734    ///
8735    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
8736    /// Nobody races for a page any more, but every worker holds a different one for the length of a
8737    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
8738    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
8739    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
8740    /// without it a worker can run a whole stripe before the next one starts and never collide.
8741    #[test]
8742    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
8743        let workers = CACHED_STRIPES_PER_COLUMN + 4;
8744        let path = path("stripe-per-worker");
8745        let mut writer =
8746            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8747                .expect("new file");
8748        for part in 0..STRIPE_PARTS * workers {
8749            let chunk = Chunk::new(vec![
8750                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8751                    .expect("integers"),
8752            ])
8753            .expect("matching rows");
8754            writer.append(&chunk).expect("one part");
8755        }
8756        writer.finish().expect("commit");
8757
8758        let read = |told: bool| {
8759            let reader = Reader::open(&path).expect("reopen from disk");
8760            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
8761            if told {
8762                reader.keep_stripes(workers);
8763            }
8764            let barrier = std::sync::Barrier::new(workers);
8765            std::thread::scope(|scope| {
8766                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
8767                    let reader = &reader;
8768                    let barrier = &barrier;
8769                    scope.spawn(move || {
8770                        for part in run {
8771                            barrier.wait();
8772                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
8773                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8774                        }
8775                        assert!(worker < workers);
8776                    });
8777                }
8778            });
8779            reader.pages.load(Atomic::Relaxed)
8780        };
8781
8782        assert_eq!(read(true), workers, "one page read per stripe and no more");
8783        assert!(read(false) > workers, "a cache that small is read again on every part");
8784        fs::remove_file(path).expect("remove scratch file");
8785    }
8786
8787    /// A damaged index page is caught before anything decodes a part out of it.
8788    ///
8789    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
8790    /// per column section rather than one for the page, and this is what says that check runs.
8791    #[test]
8792    fn a_damaged_index_page_is_an_error() {
8793        let path = path("damaged-index");
8794        let mut writer =
8795            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8796                .expect("new file");
8797        writer.append(&sample_ids()).expect("first part");
8798        writer.append(&sample_ids()).expect("second part");
8799        writer.finish().expect("commit");
8800
8801        let reader = Reader::open(&path).expect("valid directory");
8802        let index = reader.table.stripes[0].index;
8803        let mut byte = [0; 1];
8804        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
8805        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
8806        file.seek(SeekFrom::Start(index.offset)).expect("index start");
8807        file.write_all(&[!byte[0]]).expect("damage the first part length");
8808        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
8809        assert!(error.message().contains("index page section checksum differs"), "{error}");
8810        fs::remove_file(path).expect("remove scratch file");
8811    }
8812
8813    /// Every integer width the format knows about, written and read back.
8814    ///
8815    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
8816    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
8817    /// are in here on purpose, because a width that round trips through the wrong signedness only
8818    /// goes wrong at the end of its range.
8819    #[test]
8820    fn every_integer_width_round_trips_through_a_page() {
8821        let path = path("integer-widths");
8822        let columns = [
8823            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
8824            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
8825            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
8826            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
8827            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
8828            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
8829            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
8830            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
8831        ];
8832        let fields = columns
8833            .iter()
8834            .enumerate()
8835            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8836            .collect::<Vec<_>>();
8837        let vectors = columns
8838            .iter()
8839            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8840            .collect::<Vec<_>>();
8841        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
8842        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8843        writer.finish().expect("commit");
8844
8845        let reader = Reader::open(&path).expect("reopen from disk");
8846        let wanted = (0..columns.len()).collect::<Vec<_>>();
8847        let read = reader.read(0, &wanted).expect("every column");
8848        assert_eq!(read.len(), 2);
8849        // row at a time: each column has its own type and its own pair of extremes.
8850        for (at, (ty, values)) in columns.iter().enumerate() {
8851            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8852            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8853        }
8854        fs::remove_file(path).expect("remove scratch file");
8855    }
8856
8857    /// The rest of the fixed width types, and the byte strings, written and read back.
8858    ///
8859    /// The extremes again, and for a float that means more than the ends of the range. Negative
8860    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
8861    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
8862    /// `==`, which a NaN fails against itself.
8863    ///
8864    /// A blob is here beside them because it is the same round trip asked of bytes that are not
8865    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
8866    /// past turns this test red rather than turning a user's column into nulls.
8867    #[test]
8868    fn every_other_type_the_format_knows_round_trips_through_a_page() {
8869        let path = path("other-types");
8870        let columns = [
8871            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
8872            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
8873            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
8874            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
8875            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
8876            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
8877            (
8878                LogicalType::TimestampTz,
8879                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
8880            ),
8881            (
8882                LogicalType::Interval,
8883                vec![
8884                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
8885                    Value::Interval { months: 13, days: -1, micros: 1 },
8886                ],
8887            ),
8888            (
8889                LogicalType::Blob,
8890                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
8891            ),
8892        ];
8893        let fields = columns
8894            .iter()
8895            .enumerate()
8896            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8897            .collect::<Vec<_>>();
8898        let vectors = columns
8899            .iter()
8900            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8901            .collect::<Vec<_>>();
8902        let mut writer = Writer::create(&path, "others", fields).expect("new file");
8903        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8904        writer.finish().expect("commit");
8905
8906        let reader = Reader::open(&path).expect("reopen from disk");
8907        let wanted = (0..columns.len()).collect::<Vec<_>>();
8908        let read = reader.read(0, &wanted).expect("every column");
8909        assert_eq!(read.len(), 2);
8910        for (at, (ty, values)) in columns.iter().enumerate() {
8911            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8912            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8913        }
8914        // A float keeps its sign through a zero, which `==` says nothing about because negative
8915        // zero and zero compare equal.
8916        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
8917        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
8918
8919        fs::remove_file(path).expect("remove scratch file");
8920    }
8921
8922    /// A NaN is still a NaN after a trip through a page.
8923    ///
8924    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
8925    /// to itself, so a comparison against the value that was written passes for every NaN and for
8926    /// nothing else, which is the one assertion that would not catch a page that lost it.
8927    #[test]
8928    fn a_nan_survives_being_written_down() {
8929        let path = path("nan");
8930        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
8931            .expect("a NaN vector");
8932        let mut writer =
8933            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
8934                .expect("new file");
8935        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
8936        writer.finish().expect("commit");
8937        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
8938        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
8939        assert!(back.is_nan(), "a NaN came back as {back}");
8940        fs::remove_file(path).expect("remove scratch file");
8941    }
8942
8943    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
8944    ///
8945    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
8946    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
8947    /// whatever the file held. The data underneath is what the storage promise is about, so that is
8948    /// what this reads.
8949    #[test]
8950    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
8951        let path = path("uuid-and-bit");
8952        let uuids = vec![0_i128, i128::MIN, -1];
8953        let mut bits = StringColumn::new();
8954        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
8955            bits.push_bytes(value);
8956        }
8957        let expected = bits.clone();
8958        let fields =
8959            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
8960        let vectors = vec![
8961            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
8962            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
8963        ];
8964        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
8965        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8966        writer.finish().expect("commit");
8967
8968        let reader = Reader::open(&path).expect("reopen from disk");
8969        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
8970        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
8971            panic!("a uuid column is the 128 bit lane")
8972        };
8973        assert_eq!(back.as_slice(), uuids.as_slice());
8974        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
8975            panic!("a bit column is bytes")
8976        };
8977        for row in 0..expected.len() {
8978            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
8979        }
8980        fs::remove_file(path).expect("remove scratch file");
8981    }
8982
8983    #[test]
8984    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
8985        let path = path("frequency-ordinals");
8986        let mut writer =
8987            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
8988                .expect("new file");
8989        let mut values = Vec::new();
8990        for leader in 0..10_i64 {
8991            values.extend(std::iter::repeat_n(leader, 100));
8992        }
8993        values.extend(1_000_i64..41_000);
8994        for part in values.chunks(1_024) {
8995            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
8996                .expect("big integers");
8997            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
8998        }
8999        writer.finish().expect("commit");
9000
9001        let reader = Reader::open(&path).expect("reopen from disk");
9002        let occurrences =
9003            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
9004        assert!(occurrences.omitted_max < 100);
9005        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
9006        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
9007        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
9008        fs::remove_file(path).expect("remove scratch file");
9009    }
9010
9011    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
9012    /// format went from 11 to 12, every binary built after that said "magic or major version is
9013    /// unsupported" about the file, and there was no way to tell from the message whether the path
9014    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
9015    /// wants is the whole answer and it was the one thing the message did not carry.
9016    #[test]
9017    fn a_file_from_another_format_says_which_format_it_is() {
9018        let older = path("older-format");
9019        let mut writer =
9020            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
9021                .expect("new file");
9022        let chunk = Chunk::new(vec![
9023            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
9024                .expect("integers"),
9025        ])
9026        .expect("chunk");
9027        writer.append(&chunk).expect("page written");
9028        writer.finish().expect("commit");
9029
9030        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
9031        // more than one member now: format 22 is deliberately still readable, so the version that
9032        // has to be refused is the one under the oldest one accepted.
9033        let unreadable =
9034            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
9035        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
9036        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
9037        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
9038        drop(file);
9039        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
9040        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
9041        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
9042
9043        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
9044        file.seek(SeekFrom::Start(0)).expect("the magic is first");
9045        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
9046        drop(file);
9047        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
9048        assert!(complaint.contains("magic"), "{complaint}");
9049        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
9050        fs::remove_file(older).expect("remove scratch file");
9051    }
9052
9053    #[test]
9054    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
9055        let unfinished = path("unfinished");
9056        let mut writer =
9057            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
9058                .expect("new file");
9059        let chunk = Chunk::new(vec![
9060            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
9061                .expect("integers"),
9062        ])
9063        .expect("chunk");
9064        writer.append(&chunk).expect("page written");
9065        drop(writer);
9066        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
9067        fs::remove_file(unfinished).expect("remove scratch file");
9068
9069        let damaged = path("damaged");
9070        let mut writer =
9071            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
9072                .expect("new file");
9073        writer.append(&chunk).expect("page written");
9074        writer.finish().expect("commit");
9075        let reader = Reader::open(&damaged).expect("valid directory");
9076        let mut file =
9077            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
9078        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
9079        file.write_all(&[255]).expect("damage one byte");
9080        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
9081        fs::remove_file(damaged).expect("remove scratch file");
9082    }
9083
9084    #[test]
9085    fn damaged_lazy_dictionary_payload_is_an_error() {
9086        let path = path("damaged-dictionary");
9087        let mut writer = Writer::create(
9088            &path,
9089            "items",
9090            vec![
9091                Field::required("id", LogicalType::Integer),
9092                Field::new("text", LogicalType::Varchar),
9093            ],
9094        )
9095        .expect("new file");
9096        writer.append(&sample()).expect("stripe written");
9097        writer.finish().expect("commit");
9098
9099        let reader = Reader::open(&path).expect("valid directory");
9100        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
9101        // Read the count out of the page rather than writing it here, so that adding something
9102        // else to the index does not silently turn this into a test that damages the index.
9103        let mut header = [0; DICTIONARY_HEADER];
9104        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
9105        let index_len = dictionary_index_len(&header);
9106        let rank_len = last_rank_end(&reader.file, dictionary.offset, &header);
9107        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9108        file.seek(SeekFrom::Start(dictionary.offset + index_len + rank_len))
9109            .expect("inside dictionary payload");
9110        file.write_all(&[255]).expect("damage dictionary payload");
9111
9112        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
9113        let error =
9114            chunk.validate_external().expect_err("payload corruption must reach the caller");
9115        assert!(error.message().contains("payload checksum differs"), "{error}");
9116        fs::remove_file(path).expect("remove scratch file");
9117    }
9118
9119    /// A column whose values are all different is written without a dictionary, and one whose
9120    /// values repeat keeps it.
9121    ///
9122    /// The two columns go in the same table and hold the same number of rows, so the only thing
9123    /// separating them is how much of the first stripe was a value it had not seen before. Both have
9124    /// to read back the values that were written, because the decision is about cost and nothing
9125    /// else. The file size is the other half of it: a column written without a dictionary goes
9126    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
9127    /// column raw.
9128    #[test]
9129    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
9130        let path = path("dictionary-decide");
9131        let rows = 20_000;
9132        // Long enough that storing it raw would show, and different in every row.
9133        let unique =
9134            |row: usize| format!("{row:09} a value that appears exactly once in the table");
9135        // The same values in the same shape, each one used forty times over.
9136        let repeated = |row: usize| unique(row / 40);
9137        let mut writer = Writer::create(
9138            &path,
9139            "items",
9140            vec![
9141                Field::required("unique", LogicalType::Varchar),
9142                Field::required("repeated", LogicalType::Varchar),
9143            ],
9144        )
9145        .expect("new file");
9146        for part in (0..rows).step_by(1_000) {
9147            let span = part..(part + 1_000).min(rows);
9148            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
9149            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
9150            writer
9151                .append(
9152                    &Chunk::new(vec![
9153                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
9154                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
9155                    ])
9156                    .expect("two columns"),
9157                )
9158                .expect("a part");
9159        }
9160        writer.finish().expect("commit");
9161
9162        let reader = Reader::open(&path).expect("reopen from disk");
9163        assert!(
9164            reader.table.dictionaries[0].is_none(),
9165            "a column with no repeats has nothing to say twice"
9166        );
9167        assert!(
9168            reader.table.dictionaries[1].is_some(),
9169            "a column whose values come round again keeps its dictionary"
9170        );
9171        let mut first = 0;
9172        for part in 0..reader.parts() {
9173            let chunk = reader.read(part, &[0, 1]).expect("a part");
9174            for row in 0..chunk.len() {
9175                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
9176                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
9177            }
9178            first += chunk.len();
9179        }
9180        assert_eq!(first, rows, "every row was read back");
9181        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
9182        let size = fs::metadata(&path).expect("the file is there").len() as usize;
9183        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
9184        fs::remove_file(path).expect("remove scratch file");
9185    }
9186
9187    /// A payload of many blocks reads and checks every block of it.
9188    ///
9189    /// The test above has a dictionary of three values, which is one block, so it says nothing
9190    /// about a reader finding the right block among many. This one has thirty two thousand values,
9191    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
9192    /// the last and then damages the last and asks for it again.
9193    ///
9194    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
9195    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
9196    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
9197    /// The repeats are put at the front so that the values still arrive in order after them, which
9198    /// is what keeps the last part of the table on the last block of the payload.
9199    #[test]
9200    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
9201        let path = path("dictionary-blocks");
9202        let value = |row: usize| {
9203            let row = row.saturating_sub(8_000);
9204            format!("{row:07} a value long enough to be worth a payload block")
9205        };
9206        let parts = 40;
9207        let per_part = 1000;
9208        let mut writer =
9209            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9210                .expect("new file");
9211        for part in 0..parts {
9212            let values = (0..per_part)
9213                .map(|row| Value::Varchar(value(part * per_part + row)))
9214                .collect::<Vec<_>>();
9215            let chunk = Chunk::new(vec![
9216                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
9217            ])
9218            .expect("matching rows");
9219            writer.append(&chunk).expect("a part");
9220        }
9221        writer.finish().expect("commit");
9222
9223        let reader = Reader::open(&path).expect("reopen from disk");
9224        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
9225        assert!(
9226            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
9227            "the dictionary has to be several blocks for this to be testing anything"
9228        );
9229        for part in [0, parts - 1] {
9230            let chunk = reader.read(part, &[0]).expect("a part");
9231            chunk.validate_external().expect("every payload block checks out");
9232            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
9233        }
9234
9235        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9236        file.seek(SeekFrom::Start(dictionary.offset + u64::from(dictionary.length) - 4))
9237            .expect("the last bytes of the page are payload");
9238        file.write_all(&[255]).expect("damage the last payload block");
9239        let reader = Reader::open(&path).expect("the directory and the index are untouched");
9240        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
9241        let error = chunk.validate_external().expect_err("the damage must reach the caller");
9242        assert!(error.message().contains("payload checksum differs"), "{error}");
9243        fs::remove_file(path).expect("remove scratch file");
9244    }
9245
9246    /// Values of different lengths read back where the offsets say they do.
9247    ///
9248    /// The offsets are packed at one width for the column, they are relative to the payload block a
9249    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
9250    /// arithmetic could be off by one and neither shows up on values that are all the same length.
9251    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
9252    /// so the first value of a block, the last value of a run and the last value of a block are all
9253    /// covered several times over. An empty value is in the cycle because a zero length span is the
9254    /// case the reader short circuits.
9255    ///
9256    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
9257    /// distinct is written without a dictionary and then there are no packed offsets to be off by
9258    /// one in.
9259    #[test]
9260    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
9261        let path = path("dictionary-offsets");
9262        let value = |row: usize| {
9263            let row = row % 5_000;
9264            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
9265        };
9266        let rows = 6_000;
9267        let mut writer =
9268            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9269                .expect("new file");
9270        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
9271        for part in values.chunks(1_000) {
9272            let chunk =
9273                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
9274                    .expect("matching rows");
9275            writer.append(&chunk).expect("a part");
9276        }
9277        writer.finish().expect("commit");
9278
9279        let reader = Reader::open(&path).expect("reopen from disk");
9280        assert!(
9281            rows > TEXT_PAYLOAD_VALUES * 4,
9282            "the dictionary has to be several blocks for this to be testing anything"
9283        );
9284        for part in 0..rows / 1_000 {
9285            let chunk = reader.read(part, &[0]).expect("a part");
9286            for row in 0..1_000 {
9287                let row = part * 1_000 + row;
9288                assert_eq!(
9289                    chunk.value_at(row % 1_000, 0),
9290                    Value::Varchar(value(row)),
9291                    "value {row}"
9292                );
9293            }
9294        }
9295        fs::remove_file(path).expect("remove scratch file");
9296    }
9297
9298    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
9299    ///
9300    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
9301    /// the dictionary is asking and not the one a worker without it is asking, which is whether
9302    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
9303    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
9304    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
9305    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
9306    ///
9307    /// The barrier is what makes the test about that rather than about luck. Without it the first
9308    /// thread is usually finished before the last one starts and the count is one either way.
9309    #[test]
9310    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
9311        let path = path("dictionary-once");
9312        let parts = 8;
9313        let per_part = 500;
9314        let value =
9315            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
9316        let mut writer =
9317            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9318                .expect("new file");
9319        for part in 0..parts {
9320            let values = (0..per_part)
9321                .map(|row| Value::Varchar(value(part * per_part + row)))
9322                .collect::<Vec<_>>();
9323            let chunk = Chunk::new(vec![
9324                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
9325            ])
9326            .expect("matching rows");
9327            writer.append(&chunk).expect("a part");
9328        }
9329        writer.finish().expect("commit");
9330
9331        let reader = Reader::open(&path).expect("reopen from disk");
9332        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
9333        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
9334
9335        let workers = 16;
9336        let gate = std::sync::Barrier::new(workers);
9337        std::thread::scope(|scope| {
9338            for worker in 0..workers {
9339                let reader = reader.clone();
9340                let gate = &gate;
9341                scope.spawn(move || {
9342                    gate.wait();
9343                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
9344                    assert_eq!(
9345                        chunk.value_at(0, 0),
9346                        Value::Varchar(value((worker % parts) * per_part))
9347                    );
9348                });
9349            }
9350        });
9351
9352        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
9353        fs::remove_file(path).expect("remove scratch file");
9354    }
9355
9356    /// The sorted order sits outside the index the page checksum covers, because a query that
9357    /// never searches a dictionary should not read it, so it carries its own checksums and this is
9358    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
9359    /// rather than a slow one.
9360    #[test]
9361    fn a_damaged_sorted_order_is_an_error() {
9362        let path = path("damaged-order");
9363        let mut writer = Writer::create(
9364            &path,
9365            "items",
9366            vec![
9367                Field::required("id", LogicalType::Integer),
9368                Field::new("text", LogicalType::Varchar),
9369            ],
9370        )
9371        .expect("new file");
9372        writer.append(&sample()).expect("stripe written");
9373        writer.finish().expect("commit");
9374
9375        let reader = Reader::open(&path).expect("valid directory");
9376        let page = reader.table.dictionaries[1].expect("string dictionary page");
9377        let mut header = [0; DICTIONARY_HEADER];
9378        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
9379        let index_len = dictionary_index_len(&header);
9380        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9381        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
9382        file.write_all(&[255]).expect("damage the order");
9383
9384        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
9385        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
9386        assert!(error.message().contains("rank checksum differs"), "{error}");
9387        fs::remove_file(path).expect("remove scratch file");
9388    }
9389
9390    /// Codes stay in first appearance order and the sorted order is written beside them, so a
9391    /// reader can put the values back in order without the writer having had to know them all
9392    /// before it handed out the first code.
9393    #[test]
9394    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
9395        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
9396        // a nine byte prefix, one is a prefix of another, and one is empty.
9397        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
9398        let path = path("dictionary-order");
9399        let mut writer =
9400            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9401                .expect("new file");
9402        writer
9403            .append(
9404                &Chunk::new(vec![
9405                    Vector::from_values(
9406                        LogicalType::Varchar,
9407                        &spellings.map(|text| Value::Varchar(text.into())),
9408                    )
9409                    .expect("strings"),
9410                ])
9411                .expect("one column"),
9412            )
9413            .expect("stripe written");
9414        writer.finish().expect("commit");
9415
9416        let reader = Reader::open(&path).expect("valid directory");
9417        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9418        let count = dictionary.ranks().expect("a v10 file stores one");
9419        assert_eq!(count, spellings.len(), "every distinct value has a rank");
9420        let order = (0..count)
9421            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
9422            .collect::<Vec<_>>();
9423        let mut seen = order.clone();
9424        seen.sort_unstable();
9425        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
9426
9427        let ranked = order
9428            .iter()
9429            .map(|&code| {
9430                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
9431            })
9432            .collect::<Vec<_>>();
9433        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
9434        expected.sort();
9435        assert_eq!(ranked, expected, "rank order is value order");
9436
9437        // What a search asks, on the values themselves rather than through a kernel, so that a
9438        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
9439        for (rank, value) in expected.iter().enumerate() {
9440            assert_eq!(
9441                dictionary.compare_rank(rank, value).expect("compare"),
9442                Ordering::Equal,
9443                "rank {rank} is its own value"
9444            );
9445            if rank > 0 {
9446                assert_eq!(
9447                    dictionary.compare_rank(rank - 1, value).expect("compare"),
9448                    Ordering::Less,
9449                    "rank {rank} follows the one before it"
9450                );
9451            }
9452        }
9453        fs::remove_file(path).expect("remove scratch file");
9454    }
9455
9456    /// A sweep of the dictionary reads every value and keeps what it read, up to the budget.
9457    ///
9458    /// The point of the sweep is the resident size rather than the answer, so both are checked
9459    /// here. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so it keeps everything and
9460    /// a second sweep decodes nothing, which is what makes the second statement of a session asking
9461    /// the same question cost what it should. The ceiling is the other half of it and it has its own
9462    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
9463    #[test]
9464    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
9465        let path = path("dictionary-sweep");
9466        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
9467        // third, so the sweep has to be called more than once and the last call has to stop short.
9468        let spellings = (0..2_500)
9469            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9470            .collect::<Vec<_>>();
9471        let mut writer =
9472            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9473                .expect("new file");
9474        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
9475        // The dictionary is table wide and does not care where a value was written.
9476        for part in spellings.chunks(1_024) {
9477            writer
9478                .append(
9479                    &Chunk::new(vec![
9480                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9481                    ])
9482                    .expect("one column"),
9483                )
9484                .expect("stripe written");
9485        }
9486        writer.finish().expect("commit");
9487
9488        let reader = Reader::open(&path).expect("valid directory");
9489        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9490        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9491
9492        let resting = dictionary.footprint();
9493        let mut swept: Vec<Vec<u8>> = Vec::new();
9494        let mut at = 0;
9495        let mut calls = 0;
9496        while at < dictionary.len() {
9497            let stopped = dictionary
9498                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9499                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9500                    swept.push(text.to_vec());
9501                    Ok(())
9502                })
9503                .expect("a sweep reads");
9504            assert!(stopped > at, "a sweep moves");
9505            at = stopped;
9506            calls += 1;
9507        }
9508        assert_eq!(calls, 3, "a sweep hands over one block at a time");
9509        let after = dictionary.footprint();
9510        assert!(after > resting, "a sweep under the budget keeps what it decoded");
9511
9512        let read = (0..dictionary.len())
9513            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9514            .collect::<Vec<_>>();
9515        assert_eq!(swept, read, "a sweep answers what a point read answers");
9516        assert_eq!(dictionary.footprint(), after, "a point read of a kept block decodes nothing");
9517        fs::remove_file(path).expect("remove scratch file");
9518    }
9519
9520    /// A sweep over a block whose second run of offsets is short reads the same values as a point
9521    /// read does.
9522    ///
9523    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
9524    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
9525    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
9526    /// never puts a short run second in its block: the last block there begins on a run boundary and
9527    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
9528    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
9529    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
9530    #[test]
9531    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
9532        let path = path("dictionary-sweep-short-run");
9533        let spellings = (0..2_800)
9534            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9535            .collect::<Vec<_>>();
9536        let mut writer =
9537            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9538                .expect("new file");
9539        for part in spellings.chunks(1_024) {
9540            writer
9541                .append(
9542                    &Chunk::new(vec![
9543                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9544                    ])
9545                    .expect("one column"),
9546                )
9547                .expect("stripe written");
9548        }
9549        writer.finish().expect("commit");
9550
9551        let reader = Reader::open(&path).expect("valid directory");
9552        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9553        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9554        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
9555        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
9556        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
9557
9558        let mut swept: Vec<Vec<u8>> = Vec::new();
9559        let mut at = 0;
9560        while at < dictionary.len() {
9561            let stopped = dictionary
9562                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9563                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9564                    swept.push(text.to_vec());
9565                    Ok(())
9566                })
9567                .expect("a sweep reads");
9568            assert!(stopped > at, "a sweep moves");
9569            at = stopped;
9570        }
9571        let read = (0..dictionary.len())
9572            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9573            .collect::<Vec<_>>();
9574        assert_eq!(swept, read, "a sweep answers what a point read answers");
9575        fs::remove_file(path).expect("remove scratch file");
9576    }
9577
9578    /// Narrowing a page takes what fits and refuses the page for anything that does not.
9579    ///
9580    /// The edges of the range on both sides and one step past each of them, for every type, because
9581    /// checking a page separately from converting it is only right if the check refuses exactly what
9582    /// `TryFrom` would have refused, and off by one there is a file that reads back a different
9583    /// number than it was given. The check is a bit pattern rather than a comparison, so it is not
9584    /// the shape a reader would guess from the bounds, which is why all six are here. The empty page
9585    /// is here because a check written the obvious way starts with the extremes the wrong way round
9586    /// and refuses it.
9587    #[test]
9588    fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
9589        assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
9590        assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
9591        fit::<i8>(&[128]).expect_err("one past the top does not fit");
9592        fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
9593        assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
9594        fit::<u8>(&[256]).expect_err("one past the top does not fit");
9595        fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
9596        assert_eq!(
9597            fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
9598            vec![-32_768_i16, 0, 32_767]
9599        );
9600        fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
9601        fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
9602        assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
9603        fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
9604        fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
9605        assert_eq!(
9606            fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
9607            vec![i32::MIN, 0, i32::MAX]
9608        );
9609        fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
9610        fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
9611        assert_eq!(
9612            fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
9613            vec![0_u32, 4_294_967_295]
9614        );
9615        fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
9616        fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
9617
9618        // One value in a page that fits is still a page that does not, which is the thing an or
9619        // into an accumulator could get wrong in a way a page of one value would never show.
9620        fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
9621    }
9622
9623    /// The residue says yes to exactly what `TryFrom` says yes to.
9624    ///
9625    /// The edges above are the cases anyone would think to write down. This is the argument that
9626    /// there are no others, made by asking both questions about every value either narrow type could
9627    /// have an opinion about, and then about the values around the wide edges and the ends of an
9628    /// `i64`, which a range that size cannot reach.
9629    #[test]
9630    fn the_residue_agrees_with_a_checked_conversion_everywhere() {
9631        for value in -70_000_i64..70_000 {
9632            assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
9633            assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
9634            assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
9635            assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
9636        }
9637        let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
9638        for edge in wide {
9639            for step in -2_i64..=2 {
9640                let value = edge.saturating_add(step);
9641                assert_eq!(
9642                    fit::<i32>(&[value]).is_ok(),
9643                    i32::try_from(value).is_ok(),
9644                    "{value} as i32"
9645                );
9646                assert_eq!(
9647                    fit::<u32>(&[value]).is_ok(),
9648                    u32::try_from(value).is_ok(),
9649                    "{value} as u32"
9650                );
9651            }
9652        }
9653    }
9654
9655    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
9656    ///
9657    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
9658    /// column and no size at all for a test, so this opens the same dictionary a second time with a
9659    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
9660    /// somewhere in the middle of itself and everything past that point is read and dropped, which
9661    /// costs the decode again and holds none of it.
9662    #[test]
9663    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
9664        let path = path("dictionary-budget");
9665        let spellings = (0..2_500)
9666            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
9667            .collect::<Vec<_>>();
9668        let mut writer =
9669            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9670                .expect("new file");
9671        for part in spellings.chunks(1_024) {
9672            writer
9673                .append(
9674                    &Chunk::new(vec![
9675                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9676                    ])
9677                    .expect("one column"),
9678                )
9679                .expect("stripe written");
9680        }
9681        writer.finish().expect("commit");
9682
9683        let reader = Reader::open(&path).expect("valid directory");
9684        let page = reader.table.dictionaries[0].expect("a string column has one");
9685        let file = Arc::clone(&reader.file);
9686        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
9687            .expect("a dictionary opens whatever it may keep");
9688
9689        let resting = starved.footprint();
9690        let mut swept: Vec<Vec<u8>> = Vec::new();
9691        let mut at = 0;
9692        while at < starved.len() {
9693            at = starved
9694                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
9695                    swept.push(text.to_vec());
9696                    Ok(())
9697                })
9698                .expect("a sweep reads");
9699        }
9700        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
9701        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
9702
9703        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
9704        let read = (0..generous.len())
9705            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
9706            .collect::<Vec<_>>();
9707        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
9708        fs::remove_file(path).expect("remove scratch file");
9709    }
9710
9711    #[test]
9712    fn damaged_membership_cannot_skip_a_string_page() {
9713        let path = path("damaged-membership");
9714        let mut writer = Writer::create(
9715            &path,
9716            "items",
9717            vec![
9718                Field::required("id", LogicalType::Integer),
9719                Field::new("text", LogicalType::Varchar),
9720            ],
9721        )
9722        .expect("new file");
9723        writer.append(&sample()).expect("stripe written");
9724        writer.finish().expect("commit");
9725
9726        let reader = Reader::open(&path).expect("valid directory");
9727        let membership = reader.table.stripes[0].memberships[1].expect("string membership");
9728        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
9729        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
9730        file.write_all(&[255]).expect("damage membership");
9731        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
9732        assert!(error.message().contains("membership page checksum differs"), "{error}");
9733        fs::remove_file(path).expect("remove scratch file");
9734    }
9735
9736    #[test]
9737    fn membership_delta_stream_is_sorted_exact_and_bounded() {
9738        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
9739        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
9740        let encoded = encode_membership(&unique);
9741        assert_eq!(
9742            decode_membership(&encoded).expect("valid membership"),
9743            [4, 9, 72, 900, u32::MAX]
9744        );
9745        // A stripe's index is the union of its parts', so a code in two of them is in it once and
9746        // the result is still one ascending run of deltas.
9747        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
9748        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
9749        assert_eq!(
9750            decode_membership(&encode_membership(&merged)).expect("valid membership"),
9751            unique
9752        );
9753        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
9754        assert!(
9755            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
9756            "a value past u32 is invalid"
9757        );
9758    }
9759
9760    #[test]
9761    fn a_global_dictionary_may_be_larger_than_one_column_page() {
9762        let dictionary = Page {
9763            offset: HEADER,
9764            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
9765            hash: 0,
9766        };
9767        let table = Table {
9768            name: "items".to_owned(),
9769            fields: vec![Field::new("text", LogicalType::Varchar)],
9770            stripes: Vec::new(),
9771            rows: 0,
9772            dictionaries: vec![Some(dictionary)],
9773            distincts: vec![None],
9774            frequencies: vec![None],
9775            clustering: None,
9776            generation: 1,
9777            sections: Vec::new(),
9778        };
9779        let directory = encode_directory(&table).expect("directory");
9780        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
9781
9782        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
9783        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
9784    }
9785
9786    #[test]
9787    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
9788        let path = path("constant-codes");
9789        let mut writer =
9790            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9791                .expect("new file");
9792        let empty = vec![Value::Varchar(String::new()); 1024];
9793        for _ in 0..4 {
9794            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
9795            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
9796        }
9797        writer.finish().expect("commit");
9798
9799        let reader = Reader::open(&path).expect("valid directory");
9800        let pages = reader.layout().columns.first().expect("one column").pages;
9801        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
9802        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
9803        // a tag, a count and the value, and the row count stops being what drives the number.
9804        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
9805        let read = reader.read(3, &[0]).expect("the last part back");
9806        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
9807        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
9808        fs::remove_file(path).expect("remove scratch file");
9809    }
9810
9811    #[test]
9812    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
9813        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
9814        // truncated, but the values do not belong to the column the directory says they do.
9815        let over = vec![i64::from(i32::MAX) + 1];
9816        let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
9817        assert!(format!("{error}").contains("not of its type"), "{error}");
9818        assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
9819        assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
9820    }
9821
9822    #[test]
9823    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
9824        // A shift register rather than a run, because an arithmetic run is the one wide shape the
9825        // cascade does shrink. This is what a column with tens of millions of distinct values hands
9826        // over: full width codes with no order to them.
9827        let mut state: u32 = 0x9e37_79b9;
9828        let spread: Vec<u32> = (0..1024)
9829            .map(|_| {
9830                state ^= state << 13;
9831                state ^= state >> 17;
9832                state ^= state << 5;
9833                state
9834            })
9835            .collect();
9836        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
9837        let near: Vec<u32> = (0..1024).collect();
9838        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
9839        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
9840    }
9841
9842    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
9843    /// must not depend on which thread that was is the file. Two writes of the same rows are
9844    /// compared byte for byte rather than value for value, because a dictionary that two columns
9845    /// somehow shared would still read back correctly and would hand out its codes in the order the
9846    /// threads happened to run in, which is exactly what this is here to catch.
9847    #[test]
9848    fn two_writes_of_the_same_rows_give_the_same_bytes() {
9849        fn written(path: &PathBuf) {
9850            let fields = (0..40)
9851                .map(|column| {
9852                    let ty =
9853                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
9854                    Field::new(format!("c{column}"), ty)
9855                })
9856                .collect::<Vec<_>>();
9857            let mut writer = Writer::create(path, "wide", fields).expect("new file");
9858            for part in 0..70_u64 {
9859                let columns = (0..40)
9860                    .map(|column| {
9861                        let values = (0..64_u64)
9862                            .map(|row| {
9863                                let seed = part.wrapping_mul(31).wrapping_add(row);
9864                                if column % 4 == 0 {
9865                                    Value::Varchar(format!("v{}", seed % 17))
9866                                } else {
9867                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
9868                                }
9869                            })
9870                            .collect::<Vec<_>>();
9871                        let ty = if column % 4 == 0 {
9872                            LogicalType::Varchar
9873                        } else {
9874                            LogicalType::BigInt
9875                        };
9876                        Vector::from_values(ty, &values).expect("a column")
9877                    })
9878                    .collect::<Vec<_>>();
9879                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
9880            }
9881            writer.finish().expect("commit");
9882        }
9883
9884        let first = path("repeatable-one");
9885        let second = path("repeatable-two");
9886        written(&first);
9887        written(&second);
9888        let left = fs::read(&first).expect("the first file");
9889        let right = fs::read(&second).expect("the second file");
9890        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
9891        assert!(left == right, "two writes of the same rows differ in their bytes");
9892
9893        // And the rows are still there, since a pair of identically wrong files would pass the
9894        // comparison above on its own.
9895        let reader = Reader::open(&first).expect("valid directory");
9896        assert_eq!(reader.table().rows(), 70 * 64);
9897        let read = reader.read(0, &[0, 1]).expect("the first part back");
9898        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
9899        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
9900        fs::remove_file(first).expect("remove scratch file");
9901        fs::remove_file(second).expect("remove scratch file");
9902    }
9903
9904    /// Three tables of different shapes in one file, read back by name.
9905    fn three_tables(path: &PathBuf) {
9906        let writer = Writer::create(
9907            path,
9908            "region",
9909            vec![
9910                Field::new("r_key", LogicalType::Integer),
9911                Field::new("r_name", LogicalType::Varchar),
9912            ],
9913        )
9914        .expect("new file");
9915        let mut writer = writer;
9916        writer
9917            .append(
9918                &Chunk::new(vec![
9919                    Vector::from_values(
9920                        LogicalType::Integer,
9921                        &[Value::Integer(0), Value::Integer(1)],
9922                    )
9923                    .expect("keys"),
9924                    Vector::from_values(
9925                        LogicalType::Varchar,
9926                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
9927                    )
9928                    .expect("names"),
9929                ])
9930                .expect("two columns"),
9931            )
9932            .expect("a part");
9933        let mut writer = writer
9934            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
9935            .expect("a second table");
9936        writer
9937            .append(
9938                &Chunk::new(vec![
9939                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
9940                ])
9941                .expect("one column"),
9942            )
9943            .expect("a part");
9944        let mut writer =
9945            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
9946        for part in 0..70_i64 {
9947            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
9948            writer
9949                .append(
9950                    &Chunk::new(vec![
9951                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
9952                    ])
9953                    .expect("one column"),
9954                )
9955                .expect("a part");
9956        }
9957        writer.finish().expect("commit");
9958    }
9959
9960    #[test]
9961    fn three_tables_in_one_file_read_back_by_name() {
9962        let file = path("three-tables");
9963        three_tables(&file);
9964        let catalog = Catalog::open(&file).expect("a committed catalog");
9965        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
9966
9967        let region = catalog.table("region").expect("the first table");
9968        assert_eq!(region.table().rows(), 2);
9969        assert_eq!(
9970            region.read(0, &[1]).expect("names").value_at(1, 0),
9971            Value::Varchar("ASIA".to_owned())
9972        );
9973
9974        let wide = catalog.table("wide").expect("the third table");
9975        assert_eq!(wide.table().rows(), 70 * 64);
9976        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
9977
9978        // The middle table is reached without the one after it having been touched, which is what
9979        // a directory per table buys over one directory of everything.
9980        let empty = catalog.table("empty").expect("the second table");
9981        assert_eq!(empty.table().rows(), 1);
9982        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
9983
9984        fs::remove_file(file).expect("remove scratch file");
9985    }
9986
9987    #[test]
9988    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
9989        let file = path("three-tables-missing");
9990        three_tables(&file);
9991        let catalog = Catalog::open(&file).expect("a committed catalog");
9992        let error = catalog.table("nation").expect_err("no such table");
9993        assert!(error.message().contains("nation"), "{}", error.message());
9994        fs::remove_file(file).expect("remove scratch file");
9995    }
9996
9997    #[test]
9998    fn a_file_of_three_tables_will_not_open_as_one() {
9999        let file = path("three-tables-unnamed");
10000        three_tables(&file);
10001        let error = Reader::open(&file).expect_err("more than one table");
10002        assert!(error.message().contains("more than one table"), "{}", error.message());
10003        fs::remove_file(file).expect("remove scratch file");
10004    }
10005
10006    /// One column per storage width, because the width is what decides how many bytes a row costs.
10007    #[test]
10008    fn decimals_of_every_storage_width_round_trip() {
10009        let file = path("decimals");
10010        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
10011        let fields = widths
10012            .iter()
10013            .enumerate()
10014            .map(|(index, (width, scale))| {
10015                Field::new(
10016                    format!("d{index}"),
10017                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
10018                )
10019            })
10020            .collect::<Vec<_>>();
10021        let mut writer = Writer::create(&file, "money", fields).expect("new file");
10022        let rows: [i128; 3] = [-1234, 0, 999];
10023        let columns = widths
10024            .iter()
10025            .map(|(width, scale)| {
10026                let values = rows
10027                    .iter()
10028                    .map(|unscaled| Value::Decimal {
10029                        unscaled: *unscaled,
10030                        width: *width,
10031                        scale: *scale,
10032                    })
10033                    .collect::<Vec<_>>();
10034                Vector::from_values(
10035                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
10036                    &values,
10037                )
10038                .expect("a decimal column")
10039            })
10040            .collect::<Vec<_>>();
10041        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
10042        writer.finish().expect("commit");
10043
10044        let reader = Reader::open(&file).expect("a committed file");
10045        for (index, (width, scale)) in widths.iter().enumerate() {
10046            assert_eq!(
10047                reader.table().fields()[index].ty,
10048                LogicalType::decimal(*width, *scale).expect("a decimal type"),
10049                "column {index} came back as another type"
10050            );
10051            let column = reader.read(0, &[index]).expect("the column");
10052            for (row, unscaled) in rows.iter().enumerate() {
10053                assert_eq!(
10054                    column.value_at(row, 0),
10055                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
10056                    "column {index} row {row}"
10057                );
10058            }
10059        }
10060        fs::remove_file(file).expect("remove scratch file");
10061    }
10062
10063    #[test]
10064    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
10065        let file = path("two-of-a-name");
10066        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
10067            .expect("new file");
10068        let error = writer
10069            .next("t", vec![Field::new("a", LogicalType::BigInt)])
10070            .expect_err("the same name twice");
10071        assert!(error.message().contains("same name"), "{}", error.message());
10072        fs::remove_file(file).expect("remove scratch file");
10073    }
10074
10075    #[test]
10076    fn opening_the_catalog_reads_no_table_directory() {
10077        let file = path("catalog-only");
10078        three_tables(&file);
10079        let catalog = Catalog::open(&file).expect("a committed catalog");
10080        // The header and one slot, and nothing under it. The third table's directory covers seventy
10081        // stripes and reading it here would be the whole point of the two levels thrown away.
10082        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
10083        assert_eq!(catalog.names().len(), 3);
10084        fs::remove_file(file).expect("remove scratch file");
10085    }
10086
10087    /// The checksum answers what it has always answered, at every length its branches split on.
10088    ///
10089    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
10090    /// any particular function, but a file already on disk carries the answers the version that
10091    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
10092    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
10093    /// a block and a word, a word and a half word, and a half word and a byte.
10094    ///
10095    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
10096    /// also a check that this is the function it says it is.
10097    #[test]
10098    fn the_checksum_answers_what_it_has_always_answered() {
10099        let bytes: Vec<u8> =
10100            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
10101        for (length, expected) in [
10102            (0, 0xef46_db37_51d8_e999),
10103            (1, 0xa96c_7f0c_e858_bbb7),
10104            (3, 0x56e6_9576_32a4_87f9),
10105            (4, 0xc60d_15b1_e3ff_8f04),
10106            (5, 0x8088_1585_8624_dd4e),
10107            (7, 0xafbe_fc3d_6c6f_9a8e),
10108            (8, 0x3da5_c7aa_2696_83e0),
10109            (9, 0x465e_c429_b13c_3892),
10110            (15, 0xdee8_9d8a_065a_6233),
10111            (16, 0x1330_489a_7767_9c80),
10112            (31, 0x3391_303d_485e_846e),
10113            (32, 0x40b7_aff7_5d45_bbc8),
10114            (33, 0x4997_cae4_951c_17a5),
10115            (39, 0x5807_28fd_5c14_5739),
10116            (40, 0xf95c_f6f5_c08a_3d3b),
10117            (63, 0x2944_b4da_fc69_b206),
10118            (64, 0xbb76_f6ef_19bd_5a1b),
10119            (65, 0x814e_0c65_4a9f_d640),
10120            (127, 0x00de_aab1_31cf_f89b),
10121            (1000, 0x9e33_00c1_cde3_c58d),
10122        ] {
10123            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
10124        }
10125        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
10126    }
10127    /// A declared order survives the file, and a table that declared none stays as it was.
10128    ///
10129    /// The second half is the one worth a test. The clustering section is written only when there
10130    /// is a declaration, so a file of two tables where one is clustered exercises both the present
10131    /// and the absent branch of the decoder in one directory, which is where a length bug would
10132    /// show up as one table reading the other's bytes.
10133    #[test]
10134    fn a_declared_order_comes_back_out_of_the_file() {
10135        let path = path("clustered");
10136        let shipped = vec![
10137            Field::new("key", LogicalType::BigInt),
10138            Field::new("line", LogicalType::Integer),
10139            Field::new("shipdate", LogicalType::Date),
10140        ];
10141        let plain = vec![Field::new("a", LogicalType::Integer)];
10142        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
10143
10144        let mut writer = Writer::create(&path, "lineitem", shipped)
10145            .expect("new file")
10146            .declare(stage_zero.clone())
10147            .expect("the columns are the table's");
10148        let column = |ty: LogicalType, values: &[Value]| {
10149            Vector::from_values(ty, values).expect("the values match the type")
10150        };
10151        writer
10152            .append(
10153                &Chunk::new(vec![
10154                    column(
10155                        LogicalType::BigInt,
10156                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
10157                    ),
10158                    column(
10159                        LogicalType::Integer,
10160                        &[
10161                            Value::Integer(1),
10162                            Value::Integer(1),
10163                            Value::Integer(1),
10164                            Value::Integer(1),
10165                        ],
10166                    ),
10167                    column(
10168                        LogicalType::Date,
10169                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
10170                    ),
10171                ])
10172                .expect("three columns"),
10173            )
10174            .expect("four rows");
10175        let mut writer = writer.next("nation", plain).expect("a second table");
10176        writer
10177            .append(
10178                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
10179                    .expect("one column"),
10180            )
10181            .expect("one row");
10182        writer.finish().expect("commit");
10183
10184        let catalog = Catalog::open(&path).expect("reopen");
10185        let lineitem = catalog.table("lineitem").expect("the clustered table");
10186        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
10187        let nation = catalog.table("nation").expect("the plain table");
10188        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
10189
10190        // And the rows are still the rows, because the section goes on the end of the directory
10191        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
10192        assert_eq!(lineitem.table().rows(), 4);
10193        assert_eq!(nation.table().rows(), 1);
10194        fs::remove_file(&path).ok();
10195    }
10196
10197    /// A declaration naming a column the table does not have is refused where it is made.
10198    #[test]
10199    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
10200        let path = path("clustered-bad");
10201        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
10202            .expect("new file");
10203        let four =
10204            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
10205        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
10206        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
10207        fs::remove_file(&path).ok();
10208    }
10209}