Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::cmp::Ordering;
36use std::collections::{HashMap, VecDeque};
37use std::fs::{File, OpenOptions};
38use std::io::{Read, Seek, SeekFrom};
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Mutex, OnceLock};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::{bitpack, chooser, integer, string};
48use rudb_storage::sieve::Sieve;
49use rudb_storage::{Probe, Range, Zone};
50use rudb_vector::string::StringColumn;
51use rudb_vector::validity::Validity;
52use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
53
54pub mod graph;
55pub mod section;
56pub mod stats;
57mod zones;
58
59pub use section::Section;
60pub use zones::{Common, Stripes, distincts};
61
62const MAGIC: &[u8; 8] = b"RUDBNV10";
63const DIRECTORY: &[u8; 8] = b"RUDBDI10";
64const CATALOG: &[u8; 8] = b"RUDBCA10";
65const FORMAT: u32 = 25;
66
67/// Formats this build can open.
68///
69/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
70/// criterion: a build with the section table in it has to open a file written before the section
71/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
72/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
73/// graph sections is.
74///
75/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
76/// was tags for fourteen more column types, and a file written before that has none of them in it,
77/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
78/// section table, which a file written before it simply does not have. What takes it from 24 to 25
79/// is the view section on the end of the catalog, which an older file does not have either, and a
80/// catalog that ends where the tables end reads as a catalog with no views in it.
81///
82/// This is not a general compatibility promise. Four formats are readable because there was a
83/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
84/// carrying.
85const READABLE: &[u32] = &[22, 23, 24, FORMAT];
86
87const HEADER: u64 = 80;
88const SLOT_BYTES: usize = 28;
89const MAX_PAGE: usize = 256 * 1024 * 1024;
90const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
91const FREQUENCIES: &[u8; 8] = b"RUDBFQ2\0";
92/// The clustering declaration, written after the frequencies and only when there is one.
93///
94/// No format bump for this, which is the convention the frequency section set in #728: a new
95/// optional trailing section with its own magic leaves every file that does not use it byte for
96/// byte what it was, and the version is bumped for a change to a layout that already exists, as
97/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
98const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
99/// The graph section table, written after the clustering declaration and written even when empty.
100///
101/// Same convention and the same reason as the block above it, with one difference: this one is
102/// always there, so a file written by this build says which sections it has rather than leaving a
103/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
104/// that safe to add without a format bump, because a table with no sections answers every query
105/// the way it did before, only without the graph path.
106const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
107
108/// The most sections one table's directory may name.
109///
110/// A relationship contributes at most three sections, so this bounds a table at a few thousand
111/// relationships, which is far past anything a schema has. The bound is here so that a torn
112/// directory naming four billion of them is refused at decode rather than turned into an
113/// allocation, the same reason the extent count has one.
114const MAX_SECTIONS: usize = 4096;
115const FREQUENCY_CANDIDATES: usize = 32_768;
116const FREQUENCY_ENTRIES: usize = 512;
117const FREQUENCY_BUILD_RANK: usize = 10;
118const FREQUENCY_ORDINALS: usize = 65_536;
119/// The most threads the two per column passes at the end of a commit are spread over.
120///
121/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
122/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
123/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
124/// on a narrow machine would be worse than waiting.
125const MAX_FREQUENCY_WORKERS: usize = 32;
126
127/// The most threads one stripe's encode is spread over.
128///
129/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
130/// it, and the work is one column of sixty four parts, which is large enough that a thread that
131/// takes one is not a thread that was started for nothing. A machine with more cores than this has
132/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
133const MAX_ENCODE_WORKERS: usize = 32;
134
135/// The most bytes one column of one part may spend on a membership sieve.
136///
137/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
138/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
139/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
140/// rule in `encode_column` that a sieve may not be as large as the part it indexes, which is a cap
141/// per column rather than one number for the whole file.
142const SIEVE_BUDGET: usize = 8 * 1024;
143
144/// The most bytes one end of a per part range may spend on a string.
145///
146/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
147/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
148/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
149/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
150/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
151/// where two URLs of the same site still look alike.
152const PART_BOUND_BYTES: usize = 24;
153
154fn io(error: std::io::Error) -> Error {
155    Error::io(error.to_string())
156}
157
158fn invalid(message: &str) -> Error {
159    Error::invalid_input(format!("invalid rudb native file: {message}"))
160}
161
162/// Adds a sequence of byte counts without an overflow the caller has to think about.
163fn sum(counts: impl Iterator<Item = u64>) -> u64 {
164    counts.fold(0, u64::saturating_add)
165}
166
167/// One column's span out of a per column list, or zero when the list is shorter than the column.
168fn span_bytes(spans: &[Span], at: usize) -> u64 {
169    spans.get(at).map_or(0, |span| u64::from(span.length))
170}
171
172/// One column's page out of a per column list, or zero when that column has no page at all.
173fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
174    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
175}
176
177/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
178///
179/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
180/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
181/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
182/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
183/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
184/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
185/// 8 is about five percent of the query.
186fn checksum(bytes: &[u8]) -> u64 {
187    const P1: u64 = 11_400_714_785_074_694_791;
188    const P2: u64 = 14_029_467_366_897_019_727;
189    const P3: u64 = 1_609_587_929_392_839_161;
190    const P4: u64 = 9_650_029_242_287_828_579;
191    const P5: u64 = 2_870_177_450_012_600_261;
192    let round = |state: u64, word: u64| {
193        state.wrapping_add(word.wrapping_mul(P2)).rotate_left(31).wrapping_mul(P1)
194    };
195    let merge = |state: u64, lane: u64| (state ^ round(0, lane)).wrapping_mul(P1).wrapping_add(P4);
196    let word = |chunk: &[u8]| u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"));
197
198    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
199    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
200    let mut blocks = bytes.chunks_exact(32);
201    let mut rest = blocks.remainder();
202    let mut hash = if bytes.len() >= 32 {
203        let mut one = P1.wrapping_add(P2);
204        let mut two = P2;
205        let mut three = 0;
206        let mut four = 0_u64.wrapping_sub(P1);
207        for block in blocks.by_ref() {
208            one = round(one, word(&block[..8]));
209            two = round(two, word(&block[8..16]));
210            three = round(three, word(&block[16..24]));
211            four = round(four, word(&block[24..]));
212        }
213        let combined = one
214            .rotate_left(1)
215            .wrapping_add(two.rotate_left(7))
216            .wrapping_add(three.rotate_left(12))
217            .wrapping_add(four.rotate_left(18));
218        merge(merge(merge(merge(combined, one), two), three), four)
219    } else {
220        P5
221    };
222    hash = hash.wrapping_add(bytes.len() as u64);
223    let mut words = rest.chunks_exact(8);
224    for chunk in words.by_ref() {
225        hash ^= round(0, word(chunk));
226        hash = hash.rotate_left(27).wrapping_mul(P1).wrapping_add(P4);
227    }
228    rest = words.remainder();
229    if rest.len() >= 4 {
230        let (head, tail) = rest.split_at(4);
231        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
232        hash ^= u64::from(quarter).wrapping_mul(P1);
233        hash = hash.rotate_left(23).wrapping_mul(P2).wrapping_add(P3);
234        rest = tail;
235    }
236    for &byte in rest {
237        hash ^= u64::from(byte).wrapping_mul(P5);
238        hash = hash.rotate_left(11).wrapping_mul(P1);
239    }
240    hash ^= hash >> 33;
241    hash = hash.wrapping_mul(P2);
242    hash ^= hash >> 29;
243    hash = hash.wrapping_mul(P3);
244    hash ^ (hash >> 32)
245}
246
247#[derive(Debug, Clone, Copy)]
248struct Slot {
249    offset: u64,
250    length: u32,
251    generation: u64,
252    hash: u64,
253}
254
255impl Slot {
256    fn bytes(self) -> [u8; SLOT_BYTES] {
257        let mut result = [0; SLOT_BYTES];
258        result[..8].copy_from_slice(&self.offset.to_le_bytes());
259        result[8..12].copy_from_slice(&self.length.to_le_bytes());
260        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
261        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
262        result
263    }
264
265    fn read(bytes: &[u8]) -> Self {
266        Self {
267            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
268            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
269            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
270            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
271        }
272    }
273}
274
275#[derive(Debug, Clone, Copy)]
276struct Page {
277    offset: u64,
278    length: u32,
279    hash: u64,
280}
281
282impl Page {
283    /// How much of the file this page takes, for [`Reader::layout`].
284    fn bytes(&self) -> u64 {
285        u64::from(self.length)
286    }
287}
288
289#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
290enum FrequencyValue {
291    Null,
292    Integer(i128),
293    Code(u32),
294}
295
296#[derive(Debug, Clone)]
297struct FrequencyEntry {
298    value: FrequencyValue,
299    count: u64,
300}
301
302/// Exact leading frequencies for one column.
303///
304/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
305/// use the synopsis only when its last winner is strictly above every omitted value.
306#[derive(Debug, Clone)]
307struct FrequencySummary {
308    entries: Vec<FrequencyEntry>,
309    omitted_max: u64,
310    ordinals: Vec<u64>,
311}
312
313/// The values one column's frequency synopsis lists, with a bound on everything it left out.
314///
315/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
316/// rows any value not in the list can hold, which is zero when nothing was left out at all.
317#[derive(Debug, Clone)]
318pub struct FrequencyPrefix {
319    /// Every value the synopsis lists, with the number of rows holding it, count descending.
320    pub entries: Vec<(Value, u64)>,
321    /// How many rows the most common value outside the list holds, and zero for a complete list.
322    pub omitted_max: u64,
323}
324
325/// Sparse row ordinals covered by a numeric frequency candidate set.
326#[derive(Debug, Clone, PartialEq, Eq)]
327pub struct FrequencyOccurrences {
328    /// Upper bound for the frequency of every value absent from the fetched rows.
329    pub omitted_max: u64,
330    /// Table-wide row ordinals in ascending order.
331    pub ordinals: Vec<u64>,
332}
333
334/// Where one column's page for one stripe sits in the file.
335///
336/// A column page has no checksum of its own because every part inside it carries one, and the
337/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
338/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
339/// or pulled one part out of the middle of it.
340#[derive(Debug, Clone, Copy, Default)]
341struct Span {
342    offset: u64,
343    length: u32,
344}
345
346/// One independently readable stripe of a table.
347#[derive(Debug, Clone)]
348pub struct Stripe {
349    rows: usize,
350    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
351    /// part, which every sparse fetch does, never reads the file.
352    parts: Vec<u32>,
353    /// The index page: one section per column, holding a length and a checksum for every part and
354    /// then a checksum of the section itself, so that a reader can pread one column's section and
355    /// still know it is intact.
356    index: Span,
357    pages: Vec<Span>,
358    memberships: Vec<Option<Page>>,
359    /// One page per column holding the membership sieve of every part of the stripe, for the
360    /// columns that have one. A column whose parts all declined a sieve has no page at all.
361    sieves: Vec<Option<Page>>,
362    /// One page per column holding the two ends and the null count of every part of the stripe.
363    ///
364    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
365    /// not the one the rows are ordered by that is the difference between skipping half the file and
366    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
367    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
368    ///
369    /// A page per column rather than one page for the stripe, so that a query that compares one
370    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
371    /// for the same reason, like the sieves.
372    part_ranges: Vec<Option<Page>>,
373    zone: Zone,
374}
375
376impl Stripe {
377    /// Number of rows in this stripe.
378    #[must_use]
379    pub fn rows(&self) -> usize {
380        self.rows
381    }
382
383    /// Number of parts in this stripe.
384    #[must_use]
385    pub fn parts(&self) -> usize {
386        self.parts.len()
387    }
388
389    /// The two ends and the null count of every column over the whole stripe.
390    ///
391    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
392    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
393    /// scan wants to know which parts to open.
394    #[must_use]
395    pub fn zone(&self) -> &Zone {
396        &self.zone
397    }
398}
399
400/// The committed table directory.
401#[derive(Debug, Clone)]
402pub struct Table {
403    name: String,
404    fields: Vec<Field>,
405    stripes: Vec<Stripe>,
406    rows: usize,
407    dictionaries: Vec<Option<Page>>,
408    frequencies: Vec<Option<FrequencySummary>>,
409    /// How many distinct values each column holds, for the columns that know.
410    ///
411    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
412    /// the size of the dictionary is the number of distinct values in the column. That is the whole
413    /// story for a column with no null in it, and the wrong number by one for a column with a null
414    /// in it, because a null row is written as the code for the empty string and makes an entry the
415    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
416    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
417    /// work it out from the dictionary alone. So the writer settles it here.
418    distincts: Vec<Option<u64>>,
419    /// The order the rows of this table are meant to be stored in, if anybody declared one.
420    ///
421    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
422    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
423    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
424    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
425    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
426    clustering: Option<Clustering>,
427    /// The file generation of the commit that last wrote this table's column pages.
428    ///
429    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
430    /// the definition is deliberately about the pages rather than about the directory. A graph
431    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
432    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
433    /// section to this one, commits a new file generation without touching a single row of this
434    /// table, and a definition that moved with those would declare every section in the file stale
435    /// for no reason.
436    ///
437    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
438    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
439    /// sections for it to match anyway.
440    generation: u64,
441    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
442    ///
443    /// Empty for every table written before the section table existed, and empty is not a
444    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
445    /// only the time, so a table with none here answers every query the same way and slower. That
446    /// is what lets this field arrive without a migration.
447    sections: Vec<Section>,
448}
449
450impl Table {
451    /// The SQL table name held by this snapshot.
452    #[must_use]
453    pub fn name(&self) -> &str {
454        &self.name
455    }
456
457    /// Columns in their SQL order.
458    #[must_use]
459    pub fn fields(&self) -> &[Field] {
460        &self.fields
461    }
462
463    /// Committed row count.
464    #[must_use]
465    pub fn rows(&self) -> usize {
466        self.rows
467    }
468
469    /// Independently readable stripes.
470    #[must_use]
471    pub fn stripes(&self) -> &[Stripe] {
472        &self.stripes
473    }
474
475    /// The order the rows are meant to be stored in, if this table was declared with one.
476    #[must_use]
477    pub fn clustering(&self) -> Option<&Clustering> {
478        self.clustering.as_ref()
479    }
480
481    /// The generation every section of this table is judged against.
482    ///
483    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
484    /// this.
485    #[must_use]
486    pub fn generation(&self) -> u64 {
487        self.generation
488    }
489
490    /// Every graph section this table names, including the kinds this build does not know.
491    ///
492    /// Including them is the point. A caller that wants only the ones it can use asks
493    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
494    /// file opened by an older build and written again does not silently lose a section that build
495    /// had no name for.
496    #[must_use]
497    pub fn sections(&self) -> &[Section] {
498        &self.sections
499    }
500}
501
502/// One table's line in the catalog directory.
503///
504/// The small level of the two. It holds what opening a database needs and nothing else: the name to
505/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
506/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
507/// thousand rows or a billion.
508///
509/// The name, the fields and the row count are repeated here rather than pointed at inside the table
510/// directory, which is the entire point of having two levels. A catalog that pointed at them would
511/// have to read every table directory at open to answer what tables there are, which is the cost
512/// this level exists to avoid.
513#[derive(Debug, Clone)]
514struct Entry {
515    name: String,
516    fields: Vec<Field>,
517    rows: usize,
518    /// Where this table's own directory sits, with the checksum it was committed under.
519    directory: Page,
520}
521
522/// One view's line in the catalog directory.
523///
524/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
525/// What it is made of is text: the body the binder binds again at every reference, and the whole
526/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
527///
528/// The columns are a cache and they are written down anyway, which is worth saying out loud because
529/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
530/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
531/// true without anything having bound the body, so the list survived the write. Not writing it
532/// would answer null and false there, and the only way back would be to bind every view at open,
533/// which is the thing the cache exists to avoid.
534#[derive(Debug, Clone, PartialEq, Eq)]
535pub struct ViewEntry {
536    /// The view's own name, without the schema, the way a table entry holds its name.
537    pub name: String,
538    /// The query the view stands for, as the text that was written.
539    pub sql: String,
540    /// The whole `CREATE VIEW` written back out.
541    pub statement: String,
542    /// The column names the statement gave, which rename a prefix of what the body produces.
543    pub aliases: Vec<String>,
544    /// The columns the last bind of the body produced.
545    pub columns: Vec<Field>,
546}
547
548/// Where one column's bytes went, taken from the directory rather than by reading pages.
549#[derive(Debug, Clone)]
550pub struct ColumnLayout {
551    /// The column's name, so a report does not have to carry the field list beside this.
552    pub name: String,
553    /// The type, spelled the way the catalog spells it.
554    pub kind: String,
555    /// Every stripe's page of this column added up, which is the encoded data itself.
556    pub pages: u64,
557    /// Every stripe's exact code membership page for this column.
558    pub memberships: u64,
559    /// Every stripe's membership sieve page for this column.
560    pub sieves: u64,
561    /// Every stripe's per part range page for this column.
562    pub part_ranges: u64,
563    /// The table wide dictionary of this column, if it has one.
564    pub dictionary: u64,
565}
566
567impl ColumnLayout {
568    /// Everything this column costs, which is what the file would lose if the column went.
569    #[must_use]
570    pub fn total(&self) -> u64 {
571        self.pages
572            .saturating_add(self.memberships)
573            .saturating_add(self.sieves)
574            .saturating_add(self.part_ranges)
575            .saturating_add(self.dictionary)
576    }
577}
578
579/// Where a whole file's bytes went.
580///
581/// Every number here comes out of the committed directory, so taking it costs one directory read
582/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
583/// without being read, or nobody will ask.
584///
585/// The parts that are not a column are kept apart rather than shared out over the columns. The
586/// stripe index page holds a section per column and could be split, and the directory and the
587/// header cannot be, so splitting one of the three and not the others would read as if the columns
588/// accounted for everything. They do not, and the gap is the thing worth looking at.
589#[derive(Debug, Clone)]
590pub struct Layout {
591    /// The size of the file on disk.
592    pub file: u64,
593    /// Committed rows.
594    pub rows: usize,
595    /// Committed stripes.
596    pub stripes: usize,
597    /// Committed parts, which is how many chunks a scan reads.
598    pub parts: usize,
599    /// One entry per column, in the table's column order.
600    pub columns: Vec<ColumnLayout>,
601    /// Every stripe's index page, which carries a length and a checksum for every part of every
602    /// column and is charged per stripe rather than per column.
603    pub indexes: u64,
604    /// The committed directory itself, the one that was read to build this.
605    pub directory: u64,
606    /// The fixed header, which holds the magic, the format and the two directory slots.
607    pub header: u64,
608}
609
610impl Layout {
611    /// Everything the columns cost together.
612    #[must_use]
613    pub fn columns_total(&self) -> u64 {
614        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
615    }
616
617    /// What the file holds that this does not account for.
618    ///
619    /// A committed file is written once and never rewritten in place, so an earlier directory and
620    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
621    /// are bytes on disk that no column owns.
622    #[must_use]
623    pub fn unaccounted(&self) -> u64 {
624        self.file
625            .saturating_sub(self.columns_total())
626            .saturating_sub(self.indexes)
627            .saturating_sub(self.directory)
628            .saturating_sub(self.header)
629    }
630}
631
632/// How one part of one column is stored, which is one row of `pragma_storage_info`.
633///
634/// Everything here is read off the file rather than worked out from the schema, because the whole
635/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
636/// holding the same rows in a different order give different answers and that difference is the
637/// reason to ask.
638///
639/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
640/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
641/// of a page that is a quarter of a megabyte.
642#[derive(Debug, Clone)]
643pub struct StoredPart {
644    /// Which stripe the part belongs to.
645    pub stripe: usize,
646    /// Which part of that stripe it is, counting from zero inside the stripe.
647    pub part: usize,
648    /// The table wide row number the part starts at.
649    pub row: usize,
650    /// How many rows it holds.
651    pub rows: usize,
652    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
653    pub encoding: String,
654    /// The stored bytes of the part, which is what it costs in the file.
655    pub bytes: u64,
656    /// Where in the file the column page holding this part starts.
657    pub page: u64,
658    /// Where in that page the part starts.
659    pub offset: u64,
660    /// The smallest value the part holds, when the stored ranges say.
661    pub low: Option<Value>,
662    /// The largest, same.
663    pub high: Option<Value>,
664    /// How many of its rows are null, when the stored ranges say.
665    pub nulls: Option<usize>,
666}
667
668/// Appends pages and commits a new directory for one table.
669#[derive(Debug)]
670struct GlobalDictionary {
671    primary: HashMap<u64, u32>,
672    collisions: HashMap<u64, Vec<u32>>,
673    offsets: Vec<u32>,
674    payload: Vec<u8>,
675    counts: Vec<u64>,
676    nulls: u64,
677}
678
679impl GlobalDictionary {
680    fn new() -> Self {
681        Self {
682            primary: HashMap::new(),
683            collisions: HashMap::new(),
684            offsets: vec![0],
685            payload: Vec::new(),
686            counts: Vec::new(),
687            nulls: 0,
688        }
689    }
690
691    fn bytes(&self, code: u32) -> Option<&[u8]> {
692        let start = *self.offsets.get(code as usize)? as usize;
693        let end = *self.offsets.get(code as usize + 1)? as usize;
694        self.payload.get(start..end)
695    }
696
697    fn code(&mut self, text: &str) -> Result<u32> {
698        let hash = checksum(text.as_bytes());
699        if let Some(&code) = self.primary.get(&hash) {
700            if self.bytes(code) == Some(text.as_bytes()) {
701                return Ok(code);
702            }
703            if let Some(codes) = self.collisions.get(&hash) {
704                if let Some(code) =
705                    codes.iter().copied().find(|&code| self.bytes(code) == Some(text.as_bytes()))
706                {
707                    return Ok(code);
708                }
709            }
710            let code = self.insert(text)?;
711            self.collisions.entry(hash).or_default().push(code);
712            return Ok(code);
713        }
714        let code = self.insert(text)?;
715        self.primary.insert(hash, code);
716        Ok(code)
717    }
718
719    fn insert(&mut self, text: &str) -> Result<u32> {
720        let code = u32::try_from(self.offsets.len() - 1)
721            .map_err(|_| invalid("global dictionary has too many values"))?;
722        self.payload.extend_from_slice(text.as_bytes());
723        self.offsets.push(
724            u32::try_from(self.payload.len())
725                .map_err(|_| invalid("global dictionary payload exceeds 4 GiB"))?,
726        );
727        self.counts.push(0);
728        Ok(code)
729    }
730
731    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
732    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
733    /// are sorted by their bytes.
734    ///
735    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
736    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
737    /// stripe's codes close together because the data is clustered. This is what puts the values
738    /// back in order for anything that needs it, and it is separate from the codes so that getting
739    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
740    ///
741    /// The sort compares the first eight bytes as one integer before it compares the values, which
742    /// settles almost every pair without touching the payload. Padding with zero on the right is
743    /// order preserving for byte strings, because a shorter value differs from a longer one that
744    /// starts the same way at a position where the shorter one has run out, and zero is below every
745    /// byte that could be there. A pair the head cannot settle falls through to the bytes.
746    ///
747    /// The heads are kept rather than thrown away once the sort is over, because a reader searching
748    /// this order wants exactly the same comparison and for exactly the same reason. Eight bytes an
749    /// entry of file is what buys a binary search that reads no values at all in the ordinary case.
750    fn ranked(&self) -> Vec<(u64, u32)> {
751        let count = self.offsets.len() - 1;
752        let mut ranked = (0..count)
753            .map(|code| {
754                let code = code as u32;
755                (head(self.bytes(code).unwrap_or_default()), code)
756            })
757            .collect::<Vec<_>>();
758        ranked.sort_unstable_by(|left, right| {
759            left.0.cmp(&right.0).then_with(|| self.bytes(left.1).cmp(&self.bytes(right.1)))
760        });
761        ranked
762    }
763
764    fn observe(&mut self, code: u32, null: bool) -> Result<()> {
765        if null {
766            self.nulls = self.nulls.saturating_add(1);
767            return Ok(());
768        }
769        let count = self
770            .counts
771            .get_mut(code as usize)
772            .ok_or_else(|| invalid("global dictionary count code is out of range"))?;
773        *count = count.saturating_add(1);
774        Ok(())
775    }
776}
777
778/// Appends pages and commits a new directory.
779///
780/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
781/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
782/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
783/// the end of it and a reader sees every table at the generation before it or every table at the
784/// generation after it.
785#[derive(Debug)]
786pub struct Writer {
787    file: File,
788    /// Where the next write goes, counted here rather than asked of the file.
789    ///
790    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
791    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
792    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
793    /// it read. A writer that asked the file where it was would then write the directory over a
794    /// page it had already written, which is what it did.
795    at: u64,
796    table: Table,
797    generation: u64,
798    /// The first and the last source position in every stripe, in the order the stripes were
799    /// written.
800    order: Vec<((u64, u64), (u64, u64))>,
801    next_order: u64,
802    dictionaries: Vec<Option<GlobalDictionary>>,
803    pending: Vec<PendingChunk>,
804    /// The tables already closed in this generation, in the order they were written.
805    closed: Vec<Entry>,
806    /// The views the next commit writes down, which [`Writer::with_views`] sets.
807    ///
808    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
809    /// opened to append a table does not have to know about views to avoid dropping them.
810    views: Vec<ViewEntry>,
811}
812
813/// A chunk that has arrived and is waiting for the rest of its stripe.
814///
815/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
816/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
817/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
818/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
819/// that share nothing.
820#[derive(Debug)]
821struct PendingChunk {
822    order: (u64, u64),
823    chunk: Chunk,
824}
825
826/// One column's share of a stripe, which is what one encode worker produces.
827///
828/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
829/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
830/// parts next to each other, and it used to reach across a row of parts to do it.
831#[derive(Debug)]
832struct ColumnStripe {
833    pages: Vec<Vec<u8>>,
834    codes: Vec<Option<Vec<u32>>>,
835    sieves: Vec<Option<Sieve>>,
836    ranges: Vec<Range>,
837}
838
839/// Roughly what encoding a column of this type costs, for ordering the encode queue.
840///
841/// Only the order matters and only roughly. A string column hashes and copies every value into a
842/// dictionary and is in a different class from everything else, and among the fixed widths the wide
843/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
844/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
845/// a column nobody else can help with.
846fn weight(ty: &LogicalType) -> usize {
847    match ty {
848        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
849        LogicalType::HugeInt
850        | LogicalType::UHugeInt
851        | LogicalType::Uuid
852        | LogicalType::Interval => 16,
853        LogicalType::BigInt
854        | LogicalType::UBigInt
855        | LogicalType::Timestamp
856        | LogicalType::Time
857        | LogicalType::TimeTz
858        | LogicalType::TimestampTz
859        | LogicalType::TimestampS
860        | LogicalType::TimestampMs
861        | LogicalType::TimestampNs
862        | LogicalType::Double
863        | LogicalType::Decimal { .. } => 8,
864        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
865        LogicalType::SmallInt | LogicalType::USmallInt => 2,
866        _ => 1,
867    }
868}
869
870/// Parts in one stripe.
871///
872/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
873/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
874/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
875/// and cost a sparse fetch, which has to read a page index before it can reach one part.
876pub const STRIPE_PARTS: usize = 64;
877
878/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
879/// its global dictionary.
880///
881/// See [`Writer::encode_column`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
882/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
883/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
884/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
885const DICTIONARY_DECIDE_ROWS: usize = 4_096;
886
887/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
888/// first stripe held a value that stripe had not seen before.
889///
890/// See [`Writer::encode_column`]. Nine and not five, because the properties a dictionary buys are
891/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
892/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
893/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
894/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
895///
896/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
897/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
898/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
899/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
900/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
901/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
902const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
903
904/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
905const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
906
907/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
908fn index_section(parts: usize) -> Result<usize> {
909    parts
910        .checked_mul(INDEX_ENTRY)
911        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
912        .ok_or_else(|| invalid("index page length overflow"))
913}
914
915impl Writer {
916    /// Opens a committed file and starts a table in the generation after the one it holds.
917    ///
918    /// The tables already in the file are carried forward by name and by directory pointer, and
919    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
920    /// new catalog go on the end, past the catalog the committed generation points at, and the one
921    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
922    ///
923    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
924    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
925    /// still reads as the generation before it, and a slot torn across a write fails its checksum
926    /// and the reader falls back to the one beside it. This is what the second slot has always been
927    /// for.
928    ///
929    /// # Errors
930    ///
931    /// If the file has no valid committed directory, is not this build's format, repeats the name
932    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
933    /// written.
934    pub fn open(
935        path: impl AsRef<Path>,
936        name: impl Into<String>,
937        fields: Vec<Field>,
938    ) -> Result<Self> {
939        for field in &fields {
940            type_tag(&field.ty)?;
941        }
942        let name = name.into();
943        let path = path.as_ref();
944        let (_, size, slot, bytes, _) = slot_bytes(path)?;
945        let (mut closed, views) = decode_catalog(&bytes, size)?;
946        // A table already in the file under this name is only in the way if it holds rows. One that
947        // holds none has no pages for this generation to carry and no reader that could lose
948        // anything, so the table being started here takes its place in the catalog rather than
949        // colliding with it, and `finish` writes the new entry where the old one was.
950        //
951        // That is not a corner. It is the shape every loading script writes: the schema goes in one
952        // statement and the rows go in the next, and a checkpoint between them commits the empty
953        // table. Before this, the second statement had to build the whole table in memory because
954        // the first had already put the name in the file, which is how a load of a table larger
955        // than memory became a load that needed memory the size of the table.
956        if let Some(at) = closed.iter().position(|held| held.name == name) {
957            if closed[at].rows > 0 {
958                return Err(invalid("two tables in one native file have the same name"));
959            }
960            closed.remove(at);
961        }
962        // The generation of the slot whose bytes checksummed, and not the highest number in the
963        // header. A slot torn across a write can hold any number at all, and taking that one would
964        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
965        // half written commit gets to destroy the one good copy beside it.
966        let generation = slot
967            .generation
968            .checked_add(1)
969            .ok_or_else(|| invalid("native file generation overflow"))?;
970        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
971        Ok(Self {
972            file,
973            // The end of the file, so that the committed generation's catalog stays where its slot
974            // says it is and keeps naming a file a reader can still open.
975            at: size,
976            dictionaries: fields
977                .iter()
978                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
979                .collect(),
980            table: Table {
981                name,
982                dictionaries: vec![None; fields.len()],
983                distincts: vec![None; fields.len()],
984                fields,
985                stripes: Vec::new(),
986                rows: 0,
987                frequencies: Vec::new(),
988                clustering: None,
989                generation,
990                sections: Vec::new(),
991            },
992            generation,
993            order: Vec::new(),
994            next_order: 0,
995            pending: Vec::with_capacity(STRIPE_PARTS),
996            closed,
997            views,
998        })
999    }
1000
1001    /// Creates a new v10 file and its first table.
1002    ///
1003    /// # Errors
1004    ///
1005    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
1006    pub fn create(
1007        path: impl AsRef<Path>,
1008        name: impl Into<String>,
1009        fields: Vec<Field>,
1010    ) -> Result<Self> {
1011        for field in &fields {
1012            type_tag(&field.ty)?;
1013        }
1014        let file =
1015            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1016        let mut header = [0; HEADER as usize];
1017        header[..8].copy_from_slice(MAGIC);
1018        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1019        write_at(&file, 0, &header)?;
1020        Ok(Self {
1021            file,
1022            at: HEADER,
1023            dictionaries: fields
1024                .iter()
1025                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1026                .collect(),
1027            table: Table {
1028                name: name.into(),
1029                dictionaries: vec![None; fields.len()],
1030                distincts: vec![None; fields.len()],
1031                fields,
1032                stripes: Vec::new(),
1033                rows: 0,
1034                frequencies: Vec::new(),
1035                clustering: None,
1036                generation: 1,
1037                sections: Vec::new(),
1038            },
1039            generation: 1,
1040            order: Vec::new(),
1041            next_order: 0,
1042            pending: Vec::with_capacity(STRIPE_PARTS),
1043            closed: Vec::new(),
1044            views: Vec::new(),
1045        })
1046    }
1047
1048    /// Creates a new file that holds no table at all, committed and ready to open.
1049    ///
1050    /// A database somebody dropped the last table out of is still a database, and until this there
1051    /// was no way to write one down. Every other way into this file goes through a table, because
1052    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
1053    /// catalog with nothing in it could be read and not written. The format already allowed it: the
1054    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
1055    /// way every other count does, which is why nothing here is a version change.
1056    ///
1057    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
1058    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
1059    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
1060    /// wrote the same way it reads any other generation.
1061    ///
1062    /// It takes the views anyway, because a database with no table can still have views in it. A
1063    /// view over `range` or over another view names no table, so dropping the last table out of a
1064    /// database does not have to leave the catalog with nothing worth writing down.
1065    ///
1066    /// # Errors
1067    ///
1068    /// If the file exists or the path cannot be written.
1069    pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1070        let file =
1071            OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1072        let mut header = [0; HEADER as usize];
1073        header[..8].copy_from_slice(MAGIC);
1074        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1075        write_at(&file, 0, &header)?;
1076        let catalog = encode_catalog(&[], views)?;
1077        write_at(&file, HEADER, &catalog)?;
1078        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
1079        // catalog is on the disk before the slot names it, so a file this is interrupted in the
1080        // middle of is a header with no valid slot rather than a slot pointing at nothing.
1081        file.sync_all().map_err(io)?;
1082        let slot = Slot {
1083            offset: HEADER,
1084            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1085            generation: 1,
1086            hash: checksum(&catalog),
1087        };
1088        write_at(&file, slot_offset(1), &slot.bytes())?;
1089        file.sync_all().map_err(io)?;
1090        Ok(())
1091    }
1092
1093    /// Closes the table this writer is on and starts another one in the same file.
1094    ///
1095    /// Nothing is published here. The closed table's directory is written so that the bytes are on
1096    /// disk and its span is known, and the catalog that names it is only written by
1097    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
1098    ///
1099    /// # Errors
1100    ///
1101    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
1102    /// being closed cannot be written.
1103    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
1104        for field in &fields {
1105            type_tag(&field.ty)?;
1106        }
1107        let name = name.into();
1108        let entry = self.close()?;
1109        if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
1110            return Err(invalid("two tables in one native file have the same name"));
1111        }
1112        let Self { file, at, generation, mut closed, views, .. } = self;
1113        closed.push(entry);
1114        Ok(Self {
1115            file,
1116            at,
1117            generation,
1118            closed,
1119            views,
1120            dictionaries: fields
1121                .iter()
1122                .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1123                .collect(),
1124            table: Table {
1125                name,
1126                dictionaries: vec![None; fields.len()],
1127                distincts: vec![None; fields.len()],
1128                fields,
1129                stripes: Vec::new(),
1130                rows: 0,
1131                frequencies: Vec::new(),
1132                clustering: None,
1133                generation,
1134                sections: Vec::new(),
1135            },
1136            order: Vec::new(),
1137            next_order: 0,
1138            pending: Vec::with_capacity(STRIPE_PARTS),
1139        })
1140    }
1141
1142    /// Sets the views the next commit writes down, replacing whatever was carried forward.
1143    ///
1144    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
1145    /// writer does not. A view that was dropped is a view that is not in the list any more, and
1146    /// there is no other way for the writer to hear about that, since nothing else it is told about
1147    /// mentions views at all.
1148    ///
1149    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
1150    /// checkpoint that only had a table to append does not quietly drop them.
1151    #[must_use]
1152    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
1153        self.views = views;
1154        self
1155    }
1156
1157    /// Records the order this table's rows are meant to be stored in.
1158    ///
1159    /// The declaration goes in the table directory and comes back out of
1160    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
1161    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
1162    /// the thing that was missing was a place to write the order down, and a loader that honours
1163    /// the declaration is the next piece rather than this one.
1164    ///
1165    /// The declaration applies to the table the writer is currently on, so it is set after
1166    /// [`Writer::next`] rather than once for the file.
1167    ///
1168    /// # Errors
1169    ///
1170    /// If the declaration names a column this table does not have.
1171    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
1172        // Rebuilt against this table's own column count rather than trusted, because the caller
1173        // built it against a catalog entry and the two could have drifted.
1174        self.table.clustering = Some(Clustering::new(
1175            clustering.columns().to_vec(),
1176            clustering.width(),
1177            &self.table.fields,
1178        )?);
1179        Ok(self)
1180    }
1181
1182    /// Appends bytes at the end of the file and moves the writer's own offset past them.
1183    ///
1184    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
1185    /// anything is and the file's cursor is never consulted for it.
1186    fn put(&mut self, bytes: &[u8]) -> Result<()> {
1187        write_at(&self.file, self.at, bytes)?;
1188        self.at = self
1189            .at
1190            .checked_add(bytes.len() as u64)
1191            .ok_or_else(|| invalid("native file length overflow"))?;
1192        Ok(())
1193    }
1194
1195    /// Writes one chunk as independently readable column pages.
1196    ///
1197    /// # Errors
1198    ///
1199    /// If its width or types differ from the declared table, or a page exceeds its bound.
1200    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
1201        let order = (self.next_order, 0);
1202        self.next_order = self.next_order.saturating_add(1);
1203        self.append_at(order, chunk)
1204    }
1205
1206    /// Writes one chunk and records its source position for directory ordering.
1207    ///
1208    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
1209    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
1210    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
1211    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
1212    ///
1213    /// # Errors
1214    ///
1215    /// The same as [`Self::append`].
1216    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
1217        if chunk.is_empty() {
1218            return Ok(());
1219        }
1220        self.admit(chunk)?;
1221        if self.pending.last().is_some_and(|last| last.order > order) {
1222            self.flush_pending()?;
1223        }
1224        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
1225        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
1226        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
1227        // against the hundreds of seconds of encode this is what lets off one thread.
1228        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
1229        if self.pending.len() == STRIPE_PARTS {
1230            self.flush_pending()?;
1231        }
1232        Ok(())
1233    }
1234
1235    /// Writes a run of chunks as one stripe of its own.
1236    ///
1237    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
1238    /// when one caller hands over every chunk in source order and does not when several do. A
1239    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
1240    /// that ends every time two of them cross is a stripe of one or two parts.
1241    ///
1242    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
1243    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
1244    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
1245    /// so the runs from different callers may interleave with each other but may not overlap.
1246    ///
1247    /// # Errors
1248    ///
1249    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
1250    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
1251        if parts.len() > STRIPE_PARTS {
1252            return Err(invalid("a stripe was handed more parts than it holds"));
1253        }
1254        // Whatever an earlier caller left behind is its own stripe rather than the front of this
1255        // one, because the two runs are from different places in the source and a stripe is a run.
1256        self.flush_pending()?;
1257        for (order, chunk) in parts {
1258            if chunk.is_empty() {
1259                continue;
1260            }
1261            self.admit(&chunk)?;
1262            self.pending.push(PendingChunk { order, chunk });
1263        }
1264        self.flush_pending()
1265    }
1266
1267    /// Checks a chunk against the declared table and counts its rows in.
1268    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
1269        if chunk.width() != self.table.fields.len() {
1270            return Err(invalid("chunk width differs from table schema"));
1271        }
1272        for (index, field) in self.table.fields.iter().enumerate() {
1273            if chunk.column(index)?.logical_type() != &field.ty {
1274                return Err(invalid("chunk type differs from table schema"));
1275            }
1276        }
1277        self.table.rows = self
1278            .table
1279            .rows
1280            .checked_add(chunk.len())
1281            .ok_or_else(|| invalid("row count overflow"))?;
1282        Ok(())
1283    }
1284
1285    /// Encodes one column's parts of a stripe, and on the first stripe decides whether the column
1286    /// should have a dictionary at all.
1287    ///
1288    /// Every varchar column starts with one, because the writer cannot know what is in a column
1289    /// before it has seen some of it. A global dictionary is the right shape for a column of a few
1290    /// dozen values repeated down the table: the pages become small integers, a filter against a
1291    /// literal is one search of the sorted order rather than a comparison a row, and a group by is
1292    /// on the codes. It is the wrong shape for a column whose values are nearly all different.
1293    /// There the codes are as wide as row numbers, nothing is saved on the pages, and the
1294    /// membership index of a stripe is a list of very nearly every code in the column. On TPC-H the
1295    /// orders table written on its own goes from 52.3 MB to 41.4 MB, the load from 6.9 s to 5.8 s,
1296    /// and `select o_comment from orders` from 1.810 G instructions to 1.213 G, which is what the
1297    /// rudb parquet reader takes over the same values.
1298    ///
1299    /// So the first stripe of a column is the sample and the decision is made once on it. Once,
1300    /// rather than per stripe, because the codes of one column have to mean the same thing in every
1301    /// page of it, and a column that changed its mind halfway would need its earlier stripes
1302    /// rewritten. The first stripe is re-encoded when the answer comes out against the dictionary,
1303    /// which is the one stripe that pays for the decision.
1304    ///
1305    /// The threshold is deliberately near the top. [`DICTIONARY_DISTINCT_IN_TEN`] of the sample has
1306    /// to be values never seen before, which is a column with essentially no repeats. Everything
1307    /// with real repetition keeps its dictionary and keeps every property that hangs off it, and
1308    /// nothing is claimed here about where between the two the crossover really sits.
1309    ///
1310    /// Nothing here is shared with another column. The dictionary belongs to this one, the sieve
1311    /// reads only this one, and the page bytes go in a vector of this one's own. That is why the
1312    /// fan out below can hand a whole column to a thread and take a plain `&mut` on the dictionary
1313    /// rather than making it something several threads can grow at once, which is the harder half
1314    /// of #808 and is still open.
1315    fn encode_column(
1316        index: usize,
1317        held: &[PendingChunk],
1318        dictionary: &mut Option<GlobalDictionary>,
1319    ) -> Result<ColumnStripe> {
1320        // Empty means nothing has been written through it yet, so this is the column's first stripe
1321        // and the only stripe the decision below is allowed to be made on.
1322        let deciding = dictionary.as_ref().is_some_and(|held| held.offsets.len() == 1);
1323        let stripe = Self::encode_pages(index, held, dictionary.as_mut())?;
1324        if !deciding {
1325            return Ok(stripe);
1326        }
1327        let rows: usize = held.iter().map(|pending| pending.chunk.len()).sum();
1328        let distinct = dictionary.as_ref().map_or(0, |held| held.offsets.len() - 1);
1329        if rows < DICTIONARY_DECIDE_ROWS
1330            || distinct.saturating_mul(10) <= rows.saturating_mul(DICTIONARY_DISTINCT_IN_TEN)
1331        {
1332            return Ok(stripe);
1333        }
1334        *dictionary = None;
1335        Self::encode_pages(index, held, None)
1336    }
1337
1338    /// One column's parts of a stripe, with whatever dictionary it was given.
1339    fn encode_pages(
1340        index: usize,
1341        held: &[PendingChunk],
1342        mut dictionary: Option<&mut GlobalDictionary>,
1343    ) -> Result<ColumnStripe> {
1344        let mut stripe = ColumnStripe {
1345            pages: Vec::with_capacity(held.len()),
1346            codes: Vec::with_capacity(held.len()),
1347            sieves: Vec::with_capacity(held.len()),
1348            ranges: Vec::with_capacity(held.len()),
1349        };
1350        for pending in held {
1351            let column = pending.chunk.column(index)?;
1352            let (bytes, unique) = encode(column, dictionary.as_deref_mut())?;
1353            if bytes.len() > MAX_PAGE {
1354                return Err(invalid("column page exceeds the configured bound"));
1355            }
1356            // The range is built first because the sieve reads it rather than walking the column a
1357            // second time to find out how wide it is.
1358            let range = Range::of(column);
1359            // A column with a global dictionary already has an exact membership index per stripe,
1360            // so an approximate one beside it would cost a hash of every string in the table to
1361            // answer a question that is already answered. What it would buy is the finer grain, a
1362            // part rather than a stripe, and that is worth coming back for on its own.
1363            //
1364            // A sieve at least as large as the part it indexes is not written. A reader reads the
1365            // sieve to decide whether to read the part, so when the sieve is the larger of the two
1366            // it has already spent more than the read it is trying to avoid, and that holds even if
1367            // it rejects every time. It is a necessary condition rather than the whole rule, which
1368            // is that a sieve pays when its bytes are under the rejection rate times the part's,
1369            // but the rejection rate depends on what a query probes for and the writer does not
1370            // know that. The necessary half needs two numbers that are both in hand here.
1371            let sieve = match dictionary {
1372                Some(_) => None,
1373                None => Sieve::of(column, &range, SIEVE_BUDGET)
1374                    .filter(|sieve| sieve.len() < bytes.len()),
1375            };
1376            stripe.pages.push(bytes);
1377            stripe.codes.push(unique);
1378            stripe.sieves.push(sieve);
1379            stripe.ranges.push(range);
1380        }
1381        Ok(stripe)
1382    }
1383
1384    /// Encodes a whole stripe, one column to a worker.
1385    ///
1386    /// The columns are handed out through a queue rather than dealt in equal piles, because they
1387    /// are nothing like equal: `URL` on ClickBench is a global dictionary of sixty one million
1388    /// strings and `IsMobile` is a byte. A pile that happened to hold the four large string columns
1389    /// would be the whole stripe and the other workers would be waiting on it. The queue is sorted
1390    /// so the expensive ones are taken first, which is the classic answer to a last job that runs
1391    /// longer than everything after it.
1392    fn encode_columns(&mut self, held: &[PendingChunk]) -> Result<Vec<ColumnStripe>> {
1393        let width = self.table.fields.len();
1394        let workers = std::thread::available_parallelism()
1395            .map_or(1, usize::from)
1396            .min(MAX_ENCODE_WORKERS)
1397            .min(width);
1398        if workers <= 1 || held.len() <= 1 {
1399            return self
1400                .dictionaries
1401                .iter_mut()
1402                .enumerate()
1403                .map(|(index, dictionary)| Self::encode_column(index, held, dictionary))
1404                .collect();
1405        }
1406        // The dictionaries are moved out and back rather than borrowed, because a worker that takes
1407        // the next column off a queue cannot be holding a borrow of the vector the queue came from.
1408        let mut jobs: Vec<(usize, Option<GlobalDictionary>)> =
1409            std::mem::take(&mut self.dictionaries).into_iter().enumerate().collect();
1410        // Popped from the back, so the expensive columns go last in the vector.
1411        jobs.sort_by_key(|(index, _)| weight(&self.table.fields[*index].ty));
1412        let queue = Mutex::new(jobs);
1413        let pieces = std::thread::scope(|scope| {
1414            (0..workers)
1415                .map(|_| {
1416                    scope.spawn(|| {
1417                        let mut mine = Vec::new();
1418                        loop {
1419                            let taken = queue
1420                                .lock()
1421                                .map_err(|_| Error::internal("a native encode worker panicked"))?
1422                                .pop();
1423                            let Some((index, mut dictionary)) = taken else { break };
1424                            let encoded = Self::encode_column(index, held, &mut dictionary)?;
1425                            mine.push((index, dictionary, encoded));
1426                        }
1427                        Ok(mine)
1428                    })
1429                })
1430                .collect::<Vec<_>>()
1431                .into_iter()
1432                .map(|handle| {
1433                    handle.join().map_err(|_| Error::internal("a native encode worker panicked"))?
1434                })
1435                .collect::<Result<Vec<_>>>()
1436        })?;
1437        let mut dictionaries: Vec<Option<GlobalDictionary>> = (0..width).map(|_| None).collect();
1438        let mut encoded: Vec<Option<ColumnStripe>> = (0..width).map(|_| None).collect();
1439        for piece in pieces {
1440            for (index, dictionary, stripe) in piece {
1441                dictionaries[index] = dictionary;
1442                encoded[index] = Some(stripe);
1443            }
1444        }
1445        self.dictionaries = dictionaries;
1446        encoded
1447            .into_iter()
1448            .map(|stripe| stripe.ok_or_else(|| Error::internal("a column was never encoded")))
1449            .collect()
1450    }
1451
1452    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
1453    fn flush_pending(&mut self) -> Result<()> {
1454        if self.pending.is_empty() {
1455            return Ok(());
1456        }
1457        let width = self.table.fields.len();
1458        // Held here rather than read off the writer, because writing a page needs the writer and
1459        // the borrow checker is right that those are two different uses of it.
1460        let mut held = std::mem::take(&mut self.pending);
1461        let parts = held.len();
1462        let encoded = self.encode_columns(&held)?;
1463        let mut pages = Vec::with_capacity(width);
1464        let mut memberships = vec![None; width];
1465        let mut ranges = Vec::with_capacity(width);
1466        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
1467        for stripe in &encoded {
1468            let offset = self.at;
1469            let section = index.len();
1470            let mut length = 0_usize;
1471            for bytes in &stripe.pages {
1472                write_at(&self.file, self.at + length as u64, bytes)?;
1473                put_u32(
1474                    &mut index,
1475                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
1476                );
1477                put_u64(&mut index, checksum(bytes));
1478                length = length
1479                    .checked_add(bytes.len())
1480                    .ok_or_else(|| invalid("column page length overflow"))?;
1481            }
1482            let hash = checksum(&index[section..]);
1483            put_u64(&mut index, hash);
1484            if length > MAX_PAGE {
1485                return Err(invalid("column page exceeds the configured bound"));
1486            }
1487            self.at = self
1488                .at
1489                .checked_add(length as u64)
1490                .ok_or_else(|| invalid("native file length overflow"))?;
1491            pages.push(Span {
1492                offset,
1493                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
1494            });
1495            ranges.push(merged_range(stripe.ranges.iter().cloned()));
1496        }
1497        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
1498            if stripe.codes.iter().all(Option::is_none) {
1499                continue;
1500            }
1501            let lists = stripe
1502                .codes
1503                .iter()
1504                .map(|codes| codes.clone().unwrap_or_default())
1505                .collect::<Vec<_>>();
1506            let bytes = encode_membership(&merged_codes(lists));
1507            let offset = self.at;
1508            self.put(&bytes)?;
1509            *membership = Some(Page {
1510                offset,
1511                length: u32::try_from(bytes.len())
1512                    .map_err(|_| invalid("membership page length overflow"))?,
1513                hash: checksum(&bytes),
1514            });
1515        }
1516        let mut sieves = vec![None; width];
1517        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
1518            if stripe.sieves.iter().all(Option::is_none) {
1519                continue;
1520            }
1521            let bytes = encode_sieves(stripe.sieves.iter())?;
1522            let offset = self.at;
1523            self.put(&bytes)?;
1524            *page = Some(Page {
1525                offset,
1526                length: u32::try_from(bytes.len())
1527                    .map_err(|_| invalid("sieve page length overflow"))?,
1528                hash: checksum(&bytes),
1529            });
1530        }
1531        // A stripe of one part has the same rows in it as that part, so its own bounds are already
1532        // the part's and a page here would say what the directory says. Everywhere else the page is
1533        // written unless it comes to more than the column it indexes, which is the rule the sieves
1534        // go by and for the same reason: a reader reads this to decide whether to read the column,
1535        // so a page larger than the column has spent more than the read it is avoiding.
1536        let mut part_ranges = vec![None; width];
1537        if parts > 1 {
1538            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
1539                let bytes = encode_part_ranges(&stripe.ranges)?;
1540                if bytes.len() >= span.length as usize {
1541                    continue;
1542                }
1543                let offset = self.at;
1544                self.put(&bytes)?;
1545                *page = Some(Page {
1546                    offset,
1547                    length: u32::try_from(bytes.len())
1548                        .map_err(|_| invalid("part range page length overflow"))?,
1549                    hash: checksum(&bytes),
1550                });
1551            }
1552        }
1553        let offset = self.at;
1554        self.put(&index)?;
1555        let index = Span {
1556            offset,
1557            length: u32::try_from(index.len())
1558                .map_err(|_| invalid("index page length overflow"))?,
1559        };
1560        let mut rows = 0_usize;
1561        let mut lengths = Vec::with_capacity(parts);
1562        let mut span = None;
1563        for pending in held.drain(..) {
1564            let part = pending.chunk.len();
1565            rows = rows.checked_add(part).ok_or_else(|| invalid("row count overflow"))?;
1566            lengths.push(u32::try_from(part).map_err(|_| invalid("part row count overflow"))?);
1567            span = Some(
1568                span.map_or((pending.order, pending.order), |(first, _)| (first, pending.order)),
1569            );
1570        }
1571        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
1572        self.table.stripes.push(Stripe {
1573            rows,
1574            parts: lengths,
1575            index,
1576            pages,
1577            memberships,
1578            sieves,
1579            part_ranges,
1580            zone: Zone::from_ranges(ranges),
1581        });
1582        // Back where it came from, empty, so the next stripe buffers into the same allocation.
1583        self.pending = held;
1584        Ok(())
1585    }
1586
1587    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
1588    /// load is live. The pages are already in the target file, so one column at a time uses a
1589    /// bounded Misra-Gries candidate table and then recounts only those candidates.
1590    fn numeric_frequency(&self, column: usize) -> Result<Option<FrequencySummary>> {
1591        let ty = &self.table.fields[column].ty;
1592        if !matches!(
1593            ty,
1594            LogicalType::TinyInt
1595                | LogicalType::SmallInt
1596                | LogicalType::Integer
1597                | LogicalType::BigInt
1598                | LogicalType::UTinyInt
1599                | LogicalType::USmallInt
1600                | LogicalType::UInteger
1601                | LogicalType::UBigInt
1602                | LogicalType::Date
1603                | LogicalType::Timestamp
1604        ) {
1605            return Ok(None);
1606        }
1607        let mut candidates: HashMap<FrequencyValue, u32> = HashMap::new();
1608        let mut decrements = 0_u64;
1609        self.visit_numeric(column, |_, value| {
1610            if let Some(count) = candidates.get_mut(&value) {
1611                *count = count.saturating_add(1);
1612            } else if candidates.len() < FREQUENCY_CANDIDATES {
1613                candidates.insert(value, 1);
1614            } else {
1615                candidates.retain(|_, count| {
1616                    *count -= 1;
1617                    *count != 0
1618                });
1619                decrements = decrements.saturating_add(1);
1620            }
1621        })?;
1622        let (exact, ordinals) = if decrements == 0 {
1623            (
1624                candidates
1625                    .into_iter()
1626                    .map(|(value, count)| (value, u64::from(count)))
1627                    .collect::<HashMap<_, _>>(),
1628                Vec::new(),
1629            )
1630        } else {
1631            let mut lower = candidates.values().copied().collect::<Vec<_>>();
1632            lower.sort_unstable_by(|left, right| right.cmp(left));
1633            if lower.len() < FREQUENCY_BUILD_RANK
1634                || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
1635            {
1636                return Ok(None);
1637            }
1638            let mut exact =
1639                candidates.into_keys().map(|value| (value, 0_u64)).collect::<HashMap<_, _>>();
1640            let mut ordinals = Vec::new();
1641            let mut exceeded = false;
1642            self.visit_numeric(column, |ordinal, value| {
1643                if let Some(count) = exact.get_mut(&value) {
1644                    *count = count.saturating_add(1);
1645                    if !exceeded {
1646                        if ordinals.len() < FREQUENCY_ORDINALS {
1647                            ordinals.push(ordinal);
1648                        } else {
1649                            ordinals.clear();
1650                            exceeded = true;
1651                        }
1652                    }
1653                }
1654            })?;
1655            (exact, ordinals)
1656        };
1657        let mut entries = exact
1658            .into_iter()
1659            .map(|(value, count)| FrequencyEntry { value, count })
1660            .collect::<Vec<_>>();
1661        entries.sort_unstable_by(|left, right| {
1662            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
1663        });
1664        let omitted_max =
1665            entries.get(FREQUENCY_ENTRIES).map_or(decrements, |entry| decrements.max(entry.count));
1666        entries.truncate(FREQUENCY_ENTRIES);
1667        Ok(Some(FrequencySummary { entries, omitted_max, ordinals }))
1668    }
1669
1670    fn visit_numeric(
1671        &self,
1672        column: usize,
1673        mut visit: impl FnMut(u64, FrequencyValue),
1674    ) -> Result<()> {
1675        let ty = &self.table.fields[column].ty;
1676        let mut start = 0_u64;
1677        for stripe in &self.table.stripes {
1678            let spans = read_index(&self.file, stripe, column)?;
1679            let page = stripe.pages[column];
1680            let mut bytes = vec![0; page.length as usize];
1681            read_at(&self.file, page.offset, &mut bytes)?;
1682            for (span, &rows) in spans.iter().zip(&stripe.parts) {
1683                let part = part_bytes(&bytes, *span)?;
1684                if checksum(part) != span.hash {
1685                    return Err(invalid("column page checksum differs while building frequencies"));
1686                }
1687                let rows = rows as usize;
1688                let vector = decode(ty, rows, part, None)?;
1689                // row at a time: frequency construction visits decoded values to update bounded candidates.
1690                for row in 0..rows {
1691                    let value = if vector.is_null_at(row) {
1692                        FrequencyValue::Null
1693                    } else {
1694                        // An unsigned column has no signed reading, and the documented fallback is
1695                        // the value itself. Every unsigned width the format stores fits in the
1696                        // `i128` a candidate is keyed by, so nothing is lost on the way through.
1697                        let widened = match vector.signed_at(row) {
1698                            Some(value) => Some(value),
1699                            None => match vector.value_at(row) {
1700                                Value::UTinyInt(value) => Some(i128::from(value)),
1701                                Value::USmallInt(value) => Some(i128::from(value)),
1702                                Value::UInteger(value) => Some(i128::from(value)),
1703                                Value::UBigInt(value) => Some(i128::from(value)),
1704                                _ => None,
1705                            },
1706                        };
1707                        FrequencyValue::Integer(widened.ok_or_else(|| {
1708                            invalid("numeric frequency page did not contain an integer value")
1709                        })?)
1710                    };
1711                    visit(start.saturating_add(row as u64), value);
1712                }
1713                start = start.saturating_add(rows as u64);
1714            }
1715        }
1716        Ok(())
1717    }
1718
1719    /// Builds independent numeric synopses concurrently after all column pages are committed.
1720    ///
1721    /// The columns go through a queue rather than being cut into equal runs, because they are not
1722    /// equally expensive and they are not shuffled. A `BIGINT` column carries eight times the bytes
1723    /// of a `TINYINT` through the decode, and a run of them sits together in a schema the way it
1724    /// sits together in `hits`, so a worker that was handed the wrong six columns finishes long
1725    /// after one that was handed the right six and the whole phase waits for it.
1726    fn numeric_frequencies(&self) -> Result<Vec<Option<FrequencySummary>>> {
1727        let mut columns = self
1728            .table
1729            .fields
1730            .iter()
1731            .enumerate()
1732            .filter_map(|(column, field)| {
1733                matches!(
1734                    field.ty,
1735                    LogicalType::TinyInt
1736                        | LogicalType::SmallInt
1737                        | LogicalType::Integer
1738                        | LogicalType::BigInt
1739                        | LogicalType::UTinyInt
1740                        | LogicalType::USmallInt
1741                        | LogicalType::UInteger
1742                        | LogicalType::UBigInt
1743                        | LogicalType::Date
1744                        | LogicalType::Timestamp
1745                )
1746                .then_some(column)
1747            })
1748            .collect::<Vec<_>>();
1749        let workers = std::thread::available_parallelism()
1750            .map_or(1, usize::from)
1751            .min(MAX_FREQUENCY_WORKERS)
1752            .min(columns.len());
1753        if workers <= 1 {
1754            let mut frequencies = vec![None; self.table.fields.len()];
1755            for column in columns {
1756                frequencies[column] = self.numeric_frequency(column)?;
1757            }
1758            return Ok(frequencies);
1759        }
1760        // Popped from the back, so the expensive columns are the ones taken first and the cheap ones
1761        // are what is left to fill in behind them.
1762        columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
1763        let queue = Mutex::new(columns);
1764        let pieces = std::thread::scope(|scope| {
1765            (0..workers)
1766                .map(|_| {
1767                    scope.spawn(|| {
1768                        let mut mine = Vec::new();
1769                        loop {
1770                            let taken = queue
1771                                .lock()
1772                                .map_err(|_| Error::internal("a native frequency worker panicked"))?
1773                                .pop();
1774                            let Some(column) = taken else { break };
1775                            mine.push((column, self.numeric_frequency(column)?));
1776                        }
1777                        Ok(mine)
1778                    })
1779                })
1780                .collect::<Vec<_>>()
1781                .into_iter()
1782                .map(|handle| {
1783                    handle
1784                        .join()
1785                        .map_err(|_| Error::internal("a native frequency worker panicked"))?
1786                })
1787                .collect::<Result<Vec<_>>>()
1788        })?;
1789        let mut frequencies = vec![None; self.table.fields.len()];
1790        for piece in pieces {
1791            for (column, summary) in piece {
1792                frequencies[column] = summary;
1793            }
1794        }
1795        Ok(frequencies)
1796    }
1797
1798    /// Writes the directory of the table this writer is on and says where it went.
1799    ///
1800    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
1801    /// is what lets a second table follow a first: the bytes of a closed table are complete and
1802    /// addressable while nothing yet points at them, and the pointer is the last write of the
1803    /// commit.
1804    ///
1805    /// # Errors
1806    ///
1807    /// If directory encoding or writing fails.
1808    fn close(&mut self) -> Result<Entry> {
1809        self.flush_pending()?;
1810        let mut stripes = std::mem::take(&mut self.order)
1811            .into_iter()
1812            .zip(std::mem::take(&mut self.table.stripes))
1813            .collect::<Vec<_>>();
1814        stripes.sort_by_key(|(order, _)| order.0);
1815        let mut previous: Option<(u64, u64)> = None;
1816        for ((first, last), _) in &stripes {
1817            if previous.is_some_and(|previous| previous >= *first) {
1818                return Err(invalid("chunks did not arrive in source order"));
1819            }
1820            previous = Some(*last);
1821        }
1822        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
1823        self.table.frequencies = self.numeric_frequencies()?;
1824        let dictionaries = std::mem::take(&mut self.dictionaries);
1825        let orders = rankings(&dictionaries)?;
1826        for (index, (dictionary, order)) in dictionaries.into_iter().zip(orders).enumerate() {
1827            let Some(dictionary) = dictionary else { continue };
1828            // A code nothing counted is a code no non-null row of this column holds, which is the
1829            // empty string a null was written as and nothing else, because a code is only ever made
1830            // by a row asking for one.
1831            self.table.distincts[index] =
1832                Some(dictionary.counts.iter().filter(|count| **count != 0).count() as u64);
1833            self.table.frequencies[index] = Some(code_frequency(&dictionary));
1834            let encoded = encode_global_dictionary(dictionary, &order)?;
1835            let offset = self.at;
1836            self.put(&encoded.index)?;
1837            self.put(&encoded.ranks)?;
1838            for block in &encoded.payload {
1839                self.put(block)?;
1840            }
1841            let payload_len =
1842                encoded.payload.iter().try_fold(0_usize, |len, block| len.checked_add(block.len()));
1843            let length = payload_len
1844                .and_then(|len| len.checked_add(encoded.index.len()))
1845                .and_then(|len| len.checked_add(encoded.ranks.len()))
1846                .ok_or_else(|| invalid("dictionary page length overflow"))?;
1847            self.table.dictionaries[index] = Some(Page {
1848                offset,
1849                length: u32::try_from(length)
1850                    .map_err(|_| invalid("dictionary page length overflow"))?,
1851                hash: checksum(&encoded.index),
1852            });
1853        }
1854        let directory = encode_directory(&self.table)?;
1855        if directory.len() > MAX_DIRECTORY {
1856            return Err(invalid("directory exceeds the configured bound"));
1857        }
1858        let offset = self.at;
1859        self.put(&directory)?;
1860        Ok(Entry {
1861            name: self.table.name.clone(),
1862            fields: self.table.fields.clone(),
1863            rows: self.table.rows,
1864            directory: Page {
1865                offset,
1866                length: u32::try_from(directory.len())
1867                    .map_err(|_| invalid("directory length overflow"))?,
1868                hash: checksum(&directory),
1869            },
1870        })
1871    }
1872
1873    /// Commits every table this writer has written and syncs the file before publishing its header
1874    /// slot.
1875    ///
1876    /// The table handed back is the one the writer was on, which is the last of them. Callers that
1877    /// wrote several already know the others, since they named them.
1878    ///
1879    /// # Errors
1880    ///
1881    /// If directory encoding, writing, or syncing fails.
1882    pub fn finish(mut self) -> Result<Table> {
1883        let entry = self.close()?;
1884        let mut tables = std::mem::take(&mut self.closed);
1885        tables.push(entry);
1886        let catalog = encode_catalog(&tables, &self.views)?;
1887        if catalog.len() > MAX_DIRECTORY {
1888            return Err(invalid("catalog exceeds the configured bound"));
1889        }
1890        let offset = self.at;
1891        self.put(&catalog)?;
1892        // Every page and every table directory is on the disk before anything points at them. The
1893        // slot write below is what makes this generation the one a reader picks, so the order of
1894        // these two syncs is the whole of the commit.
1895        self.file.sync_all().map_err(io)?;
1896        let slot = Slot {
1897            offset,
1898            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1899            generation: self.generation,
1900            hash: checksum(&catalog),
1901        };
1902        // The one write that is not an append, and the last one. It goes back over the slot in the
1903        // header, so it names its offset rather than going through `put`, and `at` does not move.
1904        // Which of the two slots it is alternates with the generation, so the one naming the
1905        // generation before this is still intact and still valid until this write lands.
1906        write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
1907        self.file.sync_all().map_err(io)?;
1908        Ok(self.table)
1909    }
1910
1911    /// Commits a generation that changes the views and leaves every table exactly where it is.
1912    ///
1913    /// There was no way to do this before views existed, because everything that could change the
1914    /// catalog also wrote a table, so the only way to say something new about a file was to go
1915    /// through a table. A view is the first thing that can change on its own. Without this, adding
1916    /// a view to a database with eight tables in it would rewrite all eight, since the append path
1917    /// needs a table to append and the fallback is the whole file.
1918    ///
1919    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
1920    /// entries are carried forward by directory pointer the way an append carries them, the new
1921    /// catalog goes on the end, and the slot write at the end is what publishes it.
1922    ///
1923    /// # Errors
1924    ///
1925    /// If the file has no valid committed directory, is not this build's format, or cannot be
1926    /// written.
1927    pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1928        let path = path.as_ref();
1929        let (_, size, slot, bytes, _) = slot_bytes(path)?;
1930        let (closed, _) = decode_catalog(&bytes, size)?;
1931        let generation = slot
1932            .generation
1933            .checked_add(1)
1934            .ok_or_else(|| invalid("native file generation overflow"))?;
1935        let catalog = encode_catalog(&closed, views)?;
1936        if catalog.len() > MAX_DIRECTORY {
1937            return Err(invalid("catalog exceeds the configured bound"));
1938        }
1939        let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1940        write_at(&file, size, &catalog)?;
1941        file.sync_all().map_err(io)?;
1942        let slot = Slot {
1943            offset: size,
1944            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1945            generation,
1946            hash: checksum(&catalog),
1947        };
1948        write_at(&file, slot_offset(generation), &slot.bytes())?;
1949        file.sync_all().map_err(io)?;
1950        Ok(())
1951    }
1952}
1953
1954/// Appends one run of bytes at `at` and moves it past them, answering where they went.
1955///
1956/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
1957/// table. Every byte a section costs goes through here, so the offsets in an extent table come
1958/// from one place.
1959fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
1960    let offset = *at;
1961    write_at(file, offset, bytes)?;
1962    *at =
1963        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
1964    Ok(offset)
1965}
1966
1967/// Writes one attachment's payload as extents and returns the entry that names it.
1968///
1969/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
1970/// whose extents should break on a row boundary instead will want to hand its extents over already
1971/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
1972fn write_section(
1973    file: &File,
1974    at: &mut u64,
1975    one: &section::Attachment<'_>,
1976    generation: u64,
1977) -> Result<Section> {
1978    if one.header_bytes as usize > one.bytes.len() {
1979        return Err(invalid("a section's header is longer than its payload"));
1980    }
1981    let mut extents = Vec::new();
1982    let mut first = 0_u64;
1983    for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
1984        let offset = append(file, at, chunk)?;
1985        extents.push(section::Extent {
1986            offset,
1987            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
1988            hash: checksum(chunk),
1989            first,
1990        });
1991        first += chunk.len() as u64;
1992    }
1993    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
1994    section::encode_extents(&extents, &mut table)?;
1995    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
1996    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
1997    // relationship that did not fit the budget is recorded as not built rather than forgotten.
1998    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
1999    Ok(Section {
2000        kind: one.kind,
2001        id: one.id,
2002        generation,
2003        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
2004        extent_page,
2005        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
2006        hash: checksum(&table),
2007        flags: one.flags,
2008        header_bytes: one.header_bytes,
2009    })
2010}
2011
2012/// Attaches graph sections to a table already committed in a file, without rewriting a page.
2013///
2014/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
2015/// exist before the link that uses it can be built, and it is built by reading the key column back,
2016/// so the structures of a table cannot be written during the load that wrote the table. They are
2017/// written afterwards, by this, and the file in between the two is a correct file that answers
2018/// every query more slowly.
2019///
2020/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
2021/// the new catalog all go on the end of the file past the committed generation, and the last write
2022/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
2023/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
2024/// writes past.
2025///
2026/// An attachment replaces any section of the same kind and id, and every other section is carried
2027/// through untouched, including one whose kind this build does not know. The table's own generation
2028/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
2029///
2030/// # Errors
2031///
2032/// If the file has no valid committed directory, is an older format than this build writes, holds
2033/// no table of that name, names a section whose payload cannot be written, or would end up naming
2034/// more sections than the format allows.
2035pub fn attach(
2036    path: impl AsRef<Path>,
2037    table: &str,
2038    attachments: &[section::Attachment<'_>],
2039) -> Result<Table> {
2040    let path = path.as_ref();
2041    let (_, size, slot, bytes, _) = slot_bytes(path)?;
2042    let (mut entries, views) = decode_catalog(&bytes, size)?;
2043    let at = entries
2044        .iter()
2045        .position(|entry| entry.name == table)
2046        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
2047    let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
2048    let mut version = [0; 4];
2049    read_at(&file, 8, &mut version)?;
2050    let version = u32::from_le_bytes(version);
2051    // Readable is not the same as writable. A format 22 file has no section table, and giving its
2052    // directory one without moving the number in its header would leave a file that claims to be
2053    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
2054    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
2055    // just make.
2056    if version != FORMAT {
2057        return Err(invalid(&format!(
2058            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
2059             to be written again"
2060        )));
2061    }
2062    let mut directory = vec![0; entries[at].directory.length as usize];
2063    read_at(&file, entries[at].directory.offset, &mut directory)?;
2064    if checksum(&directory) != entries[at].directory.hash {
2065        return Err(invalid(&format!("the directory of table {table} does not checksum")));
2066    }
2067    let mut held = decode_directory(&directory, size)?;
2068    let mut cursor = size;
2069    for one in attachments {
2070        let written = write_section(&file, &mut cursor, one, held.generation)?;
2071        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
2072        held.sections.push(written);
2073    }
2074    if held.sections.len() > MAX_SECTIONS {
2075        return Err(invalid("the table would name more sections than the bound allows"));
2076    }
2077    let encoded = encode_directory(&held)?;
2078    if encoded.len() > MAX_DIRECTORY {
2079        return Err(invalid("directory exceeds the configured bound"));
2080    }
2081    let offset = append(&file, &mut cursor, &encoded)?;
2082    entries[at].directory = Page {
2083        offset,
2084        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
2085        hash: checksum(&encoded),
2086    };
2087    // The views the file already had, written back unchanged. Attaching a section to a table says
2088    // nothing about a view and must not drop one.
2089    let catalog = encode_catalog(&entries, &views)?;
2090    if catalog.len() > MAX_DIRECTORY {
2091        return Err(invalid("catalog exceeds the configured bound"));
2092    }
2093    let offset = append(&file, &mut cursor, &catalog)?;
2094    file.sync_all().map_err(io)?;
2095    let generation =
2096        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
2097    let committed = Slot {
2098        offset,
2099        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2100        generation,
2101        hash: checksum(&catalog),
2102    };
2103    write_at(&file, slot_offset(generation), &committed.bytes())?;
2104    file.sync_all().map_err(io)?;
2105    Ok(held)
2106}
2107
2108/// Reads committed native column pages without holding the table in memory.
2109#[derive(Debug, Clone)]
2110pub struct Reader {
2111    file: Arc<File>,
2112    table: Arc<Table>,
2113    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
2114    /// Held while a global dictionary is being opened, one per column.
2115    ///
2116    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
2117    /// already has it needs answered and is free. It does not say whether one is being opened, and
2118    /// the difference matters because every worker of a scan wants the same dictionary at the same
2119    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
2120    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
2121    /// entries, and was paying for it twice.
2122    loading: Arc<Vec<Mutex<()>>>,
2123    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
2124    /// dictionary once however many workers it has, and the test that says so is the only thing
2125    /// keeping it that way.
2126    opened: Arc<AtomicUsize>,
2127    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
2128    /// first time a probe asks about them. A query filters on one or two columns and never looks at
2129    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
2130    sieves: Arc<Vec<Vec<SieveSlot>>>,
2131    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
2132    /// first time something compares that column and kept after that.
2133    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
2134    /// Which stripe and which part of it every part of the table is, by table wide part number.
2135    places: Arc<Vec<Place>>,
2136    cache: Arc<Vec<Mutex<Cached>>>,
2137    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
2138    /// scan of a column should read each of its stripes once however many workers it has.
2139    pages: Arc<AtomicUsize>,
2140    /// How many index sections have been read. A scan of a column should read each of its stripes
2141    /// once here too, and the test that says so is the only thing keeping it that way.
2142    indexes: Arc<AtomicUsize>,
2143    /// How many stripes of one column the page cache keeps. See [`CACHED_STRIPES_PER_COLUMN`] for
2144    /// what sets it and [`Reader::keep_stripes`] for who raises it.
2145    kept: Arc<AtomicUsize>,
2146    /// The file's size when it was opened, for [`Reader::layout`].
2147    size: u64,
2148    /// The committed directory's size, for [`Reader::layout`].
2149    directory: u64,
2150    /// What opening the file cost, which is a number rather than a claim.
2151    opening: Opening,
2152}
2153
2154/// What [`Reader::open`] read before it returned.
2155///
2156/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
2157/// and nothing else, and once that document's statistics are in the file the tempting change is to
2158/// load a column summary or two on the way past, because they are small and the next query will
2159/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
2160/// embedded database is opened by processes that are about to run one trivial query.
2161///
2162/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
2163/// independent of how many rows the file holds, and the test that says so is what stops the
2164/// tempting change from landing quietly.
2165#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2166pub struct Opening {
2167    /// How many times the file was read. The header, then each directory slot that looked valid
2168    /// enough to check, so three at the most.
2169    pub reads: u32,
2170    /// How many bytes those reads asked for.
2171    pub bytes: u64,
2172}
2173
2174/// What a reader has read, while it was being opened and since.
2175#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2176pub struct Reads {
2177    /// What opening cost, before any query had been planned.
2178    pub opening: Opening,
2179    /// Whole stripe pages read since.
2180    pub pages: usize,
2181    /// Index sections read since.
2182    pub indexes: usize,
2183    /// Global dictionaries opened since. One per dictionary column that a query touched, however
2184    /// many workers touched it, which is a claim only a test can keep true.
2185    pub dictionaries: usize,
2186}
2187
2188/// Where one table wide part number lands.
2189#[derive(Debug, Clone, Copy)]
2190struct Place {
2191    stripe: u32,
2192    part: u32,
2193    rows: u32,
2194}
2195
2196/// One part's bytes inside one column page.
2197#[derive(Debug, Clone, Copy)]
2198struct PartSpan {
2199    start: usize,
2200    length: usize,
2201    hash: u64,
2202}
2203
2204/// What a reader holds for one stripe of one column.
2205///
2206/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
2207/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
2208/// four thousand would be reading sixty four times what it uses.
2209#[derive(Debug, Clone)]
2210struct CachedColumn {
2211    stripe: usize,
2212    index: Arc<Vec<PartSpan>>,
2213    page: Option<Arc<Vec<u8>>>,
2214}
2215
2216/// One column's stripes a reader holds, and which of them somebody is reading right now.
2217///
2218/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
2219/// finding a page is an index and not a walk. That matters because the walk happened under the
2220/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
2221/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
2222/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
2223/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
2224/// first, because that is the one thing the slots cannot say by themselves.
2225///
2226/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
2227/// a set because it holds at most one stripe per worker on the column and is walked far less often
2228/// than a hash of it would be built.
2229///
2230/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
2231/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
2232/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
2233/// stripe after its page had been evicted read the index again with it, which on the full
2234/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
2235#[derive(Debug, Default)]
2236struct Cached {
2237    pages: Vec<Option<Arc<Vec<u8>>>>,
2238    order: VecDeque<usize>,
2239    loading: Vec<usize>,
2240    index: Vec<Option<Arc<Vec<PartSpan>>>>,
2241}
2242
2243/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
2244///
2245/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
2246/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
2247/// needs, because then every worker is within a few parts of every other and at most a couple of
2248/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
2249/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
2250/// than paying for sixteen slots on every table that is read one part at a time.
2251///
2252/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
2253/// the number of columns a query touches.
2254const CACHED_STRIPES_PER_COLUMN: usize = 4;
2255
2256/// The sieves of one stripe of one column, once somebody has asked for them.
2257type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
2258
2259type RangeSlot = OnceLock<Arc<Vec<Range>>>;
2260
2261#[derive(Debug)]
2262struct NativeText {
2263    file: Arc<File>,
2264    /// How many values the dictionary holds.
2265    values: usize,
2266    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
2267    /// [`TEXT_OFFSET_RUN`].
2268    ///
2269    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
2270    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
2271    /// starts at zero by construction. Relative to the block rather than to the payload, because a
2272    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
2273    /// would have to subtract a base from anyway.
2274    offsets: Vec<u8>,
2275    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
2276    /// same for every block of it.
2277    offset_bits: usize,
2278    /// How many entries the sorted order has, which is the value count.
2279    ranks: usize,
2280    /// Where the sorted order starts in the file. It is read a block at a time and only when
2281    /// something searches it, so a query that never compares this column against a literal never
2282    /// touches it at all.
2283    rank_at: u64,
2284    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
2285    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
2286    /// arithmetic on the block number.
2287    rank_ends: Vec<u64>,
2288    rank_hashes: Vec<u64>,
2289    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2290    /// Bits one code is packed at, which is what the value count needs and is the same for every
2291    /// block of the column.
2292    code_bits: usize,
2293    /// The sorted order turned round, built the first time a reader asks for it.
2294    ///
2295    /// Four bytes per value against the four the offsets already hold, so a column that has this is
2296    /// carrying half again what it carried before rather than something of a new order. It is built
2297    /// only when something asks, which is a grouped min or max over this column and nothing else,
2298    /// and that reader was going to read the payload of this column once per row otherwise.
2299    code_ranks: OnceLock<Option<Vec<u32>>>,
2300    payload: u64,
2301    /// Where each block of the payload ends in the file, as a byte offset from `payload`. The
2302    /// blocks are stored back to back, so a block starts where the one before it ended.
2303    ends: Vec<u64>,
2304    hashes: Vec<u64>,
2305    /// The payload, read and decoded a block at a time and kept after that.
2306    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2307    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
2308    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
2309    keep_budget: usize,
2310    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
2311    /// is measured against.
2312    ///
2313    /// Roughly, because two threads that keep the same block at the same time both add its length
2314    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
2315    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
2316    /// than a lock on the path every scan of a string column goes through.
2317    payload_kept: AtomicUsize,
2318    /// The boundaries this dictionary has already been searched for, by the value searched for.
2319    ///
2320    /// A search is the expensive thing this type does. It settles a probe on the stored head where
2321    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
2322    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
2323    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
2324    /// worst candidate, and the worst candidate settles long before the chunks run out.
2325    ///
2326    /// Shared across the instances of a scan rather than kept per instance, because each of them has
2327    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
2328    /// is nothing next to a probe of a file.
2329    ///
2330    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
2331    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
2332    /// bound is there for the filter that searches for a different literal every chunk rather than
2333    /// for anything this is meant to help.
2334    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
2335}
2336
2337/// How many searched for values a column's dictionary remembers the boundary of.
2338///
2339/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
2340/// larger one would be wrong.
2341const TEXT_SEARCH_MEMO: usize = 64;
2342
2343/// How many values of a dictionary go in one block of the payload.
2344///
2345/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
2346/// reader has to decode to get at a single value, so it is the one number the payload format turns
2347/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
2348/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
2349///
2350/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
2351/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
2352/// better all the way up, because front coding and the LZ matcher have more to look back at and
2353/// because the per chunk setup is spread over more values. What stops it is the point read: a query
2354/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
2355/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
2356/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
2357/// Going down to 512 gives up five to nine percent.
2358const TEXT_PAYLOAD_VALUES: usize = 1024;
2359
2360/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
2361///
2362/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
2363/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
2364/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
2365/// asking the same thing decodes all of it again, and on the same column at a million rows that
2366/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
2367/// is now paid by every statement in it. Neither end is the answer. A bound is.
2368///
2369/// So a sweep keeps what it decodes until the column is holding this much and decodes without
2370/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
2371/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
2372/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
2373/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
2374///
2375/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
2376/// what should replace it: this wants to be a buffer pool over the whole database, sized against
2377/// the memory limit the session was given, with the blocks of every column competing for it and the
2378/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
2379/// without an eviction order, which is a ceiling.
2380const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
2381
2382/// How many offsets go in one packed run.
2383///
2384/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
2385/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
2386/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
2387/// a run starts where a multiply says it does and nothing is padded.
2388const TEXT_OFFSET_RUN: usize = 512;
2389
2390/// Bytes at the front of a global dictionary index: the value count, the values a payload block
2391/// holds, the block count and the bits an offset is packed at.
2392const DICTIONARY_HEADER: usize = 16;
2393
2394/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
2395/// unit.
2396///
2397/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
2398/// columns, which is well under a page. A binary search over half a million entries makes nineteen
2399/// probes, and the first ten land in ten different blocks while the last nine land in the one block
2400/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
2401/// smaller block would save a little on the early probes, cost a checksum and an end list four times
2402/// as long, and give the heads less to share a base with. A larger one would read more than it uses
2403/// on every probe.
2404const TEXT_RANK_BLOCK: usize = 512;
2405
2406/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
2407/// at.
2408///
2409/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
2410/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
2411/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
2412/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
2413/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
2414/// dictionary of eighteen million, which is twenty five bits and not thirty two.
2415///
2416/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
2417/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
2418/// and the codes.
2419const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
2420
2421impl NativeText {
2422    /// One block of the payload, read and decoded the first time anything asks for a value in it.
2423    ///
2424    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
2425    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
2426    /// file is the only thing the caller cannot work out for itself, because the stored form is
2427    /// shorter than the decoded one and by a different amount in every block.
2428    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
2429        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
2430        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
2431        Ok(Some(bytes.as_slice()))
2432    }
2433
2434    /// Reads and decodes one block of the payload, without deciding who keeps it.
2435    ///
2436    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
2437    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
2438    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
2439        let start = if block == 0 { 0 } else { self.ends[block - 1] };
2440        let end = self.ends[block];
2441        let len = end
2442            .checked_sub(start)
2443            .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
2444        let mut stored = vec![
2445            0;
2446            usize::try_from(len).map_err(|_| invalid(
2447                "global dictionary block does not fit in memory"
2448            ))?
2449        ];
2450        read_at(&self.file, self.payload + start, &mut stored)?;
2451        if checksum(&stored) != self.hashes[block] {
2452            return Err(invalid("global dictionary payload checksum differs"));
2453        }
2454        let first = block * TEXT_PAYLOAD_VALUES;
2455        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
2456        let want = self.end_within(last - 1)? as usize;
2457        let values = string::decode_flat(&stored)?;
2458        if values.len() != last - first {
2459            return Err(invalid("global dictionary block holds the wrong value count"));
2460        }
2461        let bytes = values.into_bytes();
2462        if bytes.len() != want {
2463            return Err(invalid("global dictionary block decodes to the wrong length"));
2464        }
2465        Ok(bytes)
2466    }
2467
2468    /// Where the value at `index` ends inside its payload block.
2469    fn end_within(&self, index: usize) -> Result<u32> {
2470        let run = index / TEXT_OFFSET_RUN;
2471        let bytes = self
2472            .offsets
2473            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2474            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2475        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
2476            .map_err(|_| invalid("global dictionary offsets are short"))?;
2477        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
2478    }
2479
2480    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
2481    ///
2482    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
2483    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
2484    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
2485    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
2486    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
2487    ///
2488    /// [`bitpack::unpack_tail`] walks the run instead, which makes the window a fixed width and so
2489    /// an unaligned load, and reads the bit position off a counter. A run is five hundred and twelve
2490    /// values and a block is two of them, so a block of a thousand and twenty four values costs two
2491    /// calls here and nothing per value.
2492    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
2493        let mut ends = Vec::with_capacity(last.saturating_sub(first));
2494        let mut at = first;
2495        while at < last {
2496            let run = at / TEXT_OFFSET_RUN;
2497            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
2498            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
2499            let bytes = self
2500                .offsets
2501                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2502                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2503            let run_ends = bitpack::unpack_tail(bytes, self.offset_bits, held)
2504                .map_err(|_| invalid("global dictionary offsets are short"))?;
2505            let within = run_ends
2506                .get(at % TEXT_OFFSET_RUN..stop - run * TEXT_OFFSET_RUN)
2507                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2508            ends.extend_from_slice(within);
2509            at = stop;
2510        }
2511        Ok(ends)
2512    }
2513
2514    /// Where the value at `index` starts inside its payload block, which is where the value before
2515    /// it ended unless it is the first of the block.
2516    fn start_within(&self, index: usize) -> Result<u32> {
2517        if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
2518    }
2519
2520    /// Where the value at `index` starts and ends inside its payload block.
2521    ///
2522    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
2523    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
2524    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
2525    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
2526    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
2527    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
2528        let within = index % TEXT_OFFSET_RUN;
2529        let (start, end) = if within == 0 {
2530            (self.start_within(index)?, self.end_within(index)?)
2531        } else {
2532            let run = index / TEXT_OFFSET_RUN;
2533            let bytes = self
2534                .offsets
2535                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2536                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2537            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
2538                .map_err(|_| invalid("global dictionary offsets are short"))?;
2539            let ends = u32::try_from(end)
2540                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2541            let starts = u32::try_from(start)
2542                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2543            (starts, ends)
2544        };
2545        if start > end {
2546            return Err(invalid("global dictionary value ends before it starts"));
2547        }
2548        Ok((start, end))
2549    }
2550
2551    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
2552    ///
2553    /// The block is read from the file and checked against the hash the index carries for it the
2554    /// first time anything asks, and kept after that, the same way a payload block is. A search
2555    /// makes about as many probes as the order has bits, so the whole search reads a handful of
2556    /// these and never the rest.
2557    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
2558        let slot = self
2559            .rank_blocks
2560            .get(rank / TEXT_RANK_BLOCK)
2561            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
2562        let block = slot
2563            .get_or_init(|| {
2564                let which = rank / TEXT_RANK_BLOCK;
2565                let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
2566                let end = self.rank_ends[which];
2567                let mut bytes = vec![0; (end - start) as usize];
2568                read_at(&self.file, self.rank_at + start, &mut bytes)?;
2569                if checksum(&bytes)
2570                    != *self
2571                        .rank_hashes
2572                        .get(rank / TEXT_RANK_BLOCK)
2573                        .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
2574                {
2575                    return Err(invalid("global dictionary rank checksum differs"));
2576                }
2577                Ok(bytes)
2578            })
2579            .as_ref()
2580            .map_err(Clone::clone)?;
2581        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
2582    }
2583
2584    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
2585    fn head_at(&self, rank: usize) -> Result<u64> {
2586        let (block, within) = self.rank_parts(rank)?;
2587        let (base, width, packed) = rank_heads(block)?;
2588        let above = bitpack::tail_at(packed, width, within)
2589            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
2590        Ok(base.wrapping_add(above))
2591    }
2592
2593    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
2594    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
2595        let (_, width, packed) = rank_heads(block)?;
2596        packed
2597            .get(bitpack::tail_len(count, width)..)
2598            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
2599    }
2600
2601    /// How many entries the block holding `rank` has, which is a full block except at the end.
2602    fn rank_block_len(&self, rank: usize) -> usize {
2603        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
2604        TEXT_RANK_BLOCK.min(self.ranks - first)
2605    }
2606}
2607
2608/// The base, the width and the packed bytes of one rank block's heads.
2609fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
2610    let header = block
2611        .get(..RANK_BLOCK_HEADER)
2612        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
2613    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
2614    let width = header[8] as usize;
2615    if width > 64 {
2616        return Err(invalid("global dictionary rank block packs heads past a word"));
2617    }
2618    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
2619}
2620
2621/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
2622///
2623/// One width for the whole column rather than one a block. A block is 1,024 values of the same
2624/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
2625/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
2626/// the arithmetic that finds where a block starts.
2627fn offset_width(offsets: &[u32]) -> usize {
2628    let values = offsets.len() - 1;
2629    let mut span = 0;
2630    for first in (0..values).step_by(TEXT_PAYLOAD_VALUES) {
2631        let last = (first + TEXT_PAYLOAD_VALUES).min(values);
2632        span = span.max(offsets[last] - offsets[first]);
2633    }
2634    (u32::BITS - span.leading_zeros()) as usize
2635}
2636
2637/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
2638/// has read any of them.
2639fn offset_bytes(values: usize, bits: usize) -> usize {
2640    let full = values / TEXT_OFFSET_RUN;
2641    let rest = values % TEXT_OFFSET_RUN;
2642    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
2643}
2644
2645/// The end of every value within its payload block, packed a run at a time.
2646fn encode_offsets(offsets: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
2647    let values = offsets.len() - 1;
2648    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
2649    for first in (0..values).step_by(TEXT_OFFSET_RUN) {
2650        let last = (first + TEXT_OFFSET_RUN).min(values);
2651        let base = offsets[first / TEXT_PAYLOAD_VALUES * TEXT_PAYLOAD_VALUES];
2652        run.clear();
2653        run.extend((first..last).map(|value| u64::from(offsets[value + 1] - base)));
2654        bitpack::pack_tail(&run, bits, out)
2655            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
2656    }
2657    Ok(())
2658}
2659
2660/// How many bits a code of a dictionary of `values` entries takes.
2661fn code_width(values: usize) -> usize {
2662    match u64::try_from(values).unwrap_or(u64::MAX) {
2663        0 | 1 => 0,
2664        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
2665    }
2666}
2667
2668impl TextSource for NativeText {
2669    fn len(&self) -> usize {
2670        self.values
2671    }
2672
2673    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
2674        if index >= self.values {
2675            return Ok(None);
2676        }
2677        let (start, end) = self.span_within(index)?;
2678        if start == end {
2679            return Ok(Some(&[]));
2680        }
2681        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
2682        // is in one block and the offsets already say where in it.
2683        let block = index / TEXT_PAYLOAD_VALUES;
2684        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
2685        Ok(bytes.get(start as usize..end as usize))
2686    }
2687
2688    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
2689        if index >= self.values {
2690            return Ok(None);
2691        }
2692        let (start, end) = self.span_within(index)?;
2693        Ok(Some((end - start) as usize))
2694    }
2695
2696    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
2697    ///
2698    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
2699    /// every block whatever it does. The question is whether it keeps them, and both answers are
2700    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
2701    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
2702    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
2703    /// the same question decode all of it again, which on the same column at a million rows is a
2704    /// `LIKE` going from 2.7 ms to 16.2 ms.
2705    ///
2706    /// So a sweep keeps what it decodes while the column is under [`TEXT_KEEP_BUDGET`] and drops it
2707    /// after that. A block already in hand is used where it is there and costs nothing either way.
2708    fn sweep(
2709        &self,
2710        first: usize,
2711        limit: usize,
2712        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
2713    ) -> Result<usize> {
2714        let limit = limit.min(self.values);
2715        if first >= limit {
2716            return Ok(first);
2717        }
2718        let block = first / TEXT_PAYLOAD_VALUES;
2719        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
2720        let decoded;
2721        let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
2722            Some(Ok(kept)) => kept,
2723            _ if self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
2724                let kept = self
2725                    .payload_block(block)?
2726                    .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
2727                self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
2728                kept
2729            }
2730            _ => {
2731                decoded = self.decode_block(block)?;
2732                &decoded
2733            }
2734        };
2735        let ends = self.ends_within(first, last)?;
2736        if ends.len() != last - first {
2737            return Err(invalid("global dictionary offsets are short"));
2738        }
2739        let mut start = u64::from(self.start_within(first)?);
2740        // row at a time: the caller is handed one value after another, and what it does with one is
2741        // its own business, so there is no shape here for anything but a walk.
2742        for (index, &end) in (first..last).zip(&ends) {
2743            let value = usize::try_from(start)
2744                .ok()
2745                .zip(usize::try_from(end).ok())
2746                .and_then(|(from, to)| bytes.get(from..to))
2747                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
2748            body(index, value)?;
2749            start = end;
2750        }
2751        Ok(last)
2752    }
2753
2754    fn ranks(&self) -> Option<usize> {
2755        (self.ranks > 0).then_some(self.ranks)
2756    }
2757
2758    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
2759    /// it is not.
2760    ///
2761    /// The lock is held over the search rather than dropped and taken again, so that two threads
2762    /// asking for the same value at the same time do the work once between them. That is the shape
2763    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
2764    /// improving their bound over the same early chunks.
2765    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
2766        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
2767        if let Some(&answer) = memo.get(wanted) {
2768            return Ok(answer);
2769        }
2770        let answer = search_below(self, ranks, wanted)?;
2771        if memo.len() >= TEXT_SEARCH_MEMO {
2772            memo.clear();
2773        }
2774        memo.insert(wanted.to_vec(), answer);
2775        Ok(answer)
2776    }
2777
2778    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
2779        // The head settles the probe unless the two values start with the same eight bytes, and
2780        // only then is a value read. On a column of URLs that is the difference between a search
2781        // that touches one block of the payload and a search that touches nineteen of them.
2782        let settled = self.head_at(rank)?.cmp(&head(wanted));
2783        if settled != Ordering::Equal {
2784            return Ok(settled);
2785        }
2786        let code = self.code_at_rank(rank)?;
2787        let bytes = self
2788            .bytes_at(code as usize)?
2789            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
2790        Ok(bytes.cmp(wanted))
2791    }
2792
2793    fn code_at_rank(&self, rank: usize) -> Result<u32> {
2794        let (block, within) = self.rank_parts(rank)?;
2795        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
2796        let code = bitpack::tail_at(codes, self.code_bits, within)
2797            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
2798        let code = u32::try_from(code)
2799            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
2800        if code as usize >= self.len() {
2801            return Err(invalid("global dictionary order names a code it does not have"));
2802        }
2803        Ok(code)
2804    }
2805
2806    fn code_ranks(&self) -> Option<&[u32]> {
2807        // The order is a permutation of the positions, so inverting it needs every position to be
2808        // named exactly once. Anything else and the slice would have holes, and a caller indexing
2809        // it by a code would read a rank that belongs to nothing.
2810        if self.ranks == 0 || self.ranks != self.len() {
2811            return None;
2812        }
2813        self.code_ranks
2814            .get_or_init(|| {
2815                let mut ranks = vec![u32::MAX; self.ranks];
2816                // A block at a time rather than a rank at a time, because reading it per rank pays
2817                // for the bounds check, the division and the lock on every one of them.
2818                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
2819                    let (block, _) = self.rank_parts(first).ok()?;
2820                    let count = self.rank_block_len(first);
2821                    let codes = self.rank_codes(block, count).ok()?;
2822                    for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
2823                        .ok()?
2824                        .into_iter()
2825                        .enumerate()
2826                    {
2827                        let code = usize::try_from(code).ok()?;
2828                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
2829                    }
2830                }
2831                if ranks.contains(&u32::MAX) {
2832                    return None;
2833                }
2834                Some(ranks)
2835            })
2836            .as_deref()
2837    }
2838
2839    fn footprint(&self) -> usize {
2840        self.offsets.capacity()
2841            + self
2842                .code_ranks
2843                .get()
2844                .and_then(Option::as_ref)
2845                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
2846            + self.rank_hashes.capacity() * size_of::<u64>()
2847            + self.rank_ends.capacity() * size_of::<u64>()
2848            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2849            + self
2850                .rank_blocks
2851                .iter()
2852                .filter_map(OnceLock::get)
2853                .filter_map(|result| result.as_ref().ok())
2854                .map(Vec::capacity)
2855                .sum::<usize>()
2856            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2857            + self.hashes.capacity() * size_of::<u64>()
2858            + self.ends.capacity() * size_of::<u64>()
2859            + self
2860                .blocks
2861                .iter()
2862                .filter_map(OnceLock::get)
2863                .filter_map(|result| result.as_ref().ok())
2864                .map(Vec::capacity)
2865                .sum::<usize>()
2866    }
2867}
2868
2869/// Every table wide part number in order, with the stripe it belongs to.
2870fn places(table: &Table) -> Result<Vec<Place>> {
2871    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
2872    for (at, stripe) in table.stripes.iter().enumerate() {
2873        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
2874        for (part, &rows) in stripe.parts.iter().enumerate() {
2875            places.push(Place {
2876                stripe: index,
2877                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
2878                rows,
2879            });
2880        }
2881    }
2882    Ok(places)
2883}
2884
2885/// Reads one column's section of a stripe's index page.
2886///
2887/// The section carries its own checksum, so a reader that wants one column out of a hundred and
2888/// five preads a few hundred bytes and still knows that what it got is what was written.
2889fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
2890    let parts = stripe.parts.len();
2891    let section = index_section(parts)?;
2892    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
2893    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
2894    if end > stripe.index.length as usize {
2895        return Err(invalid("index page is shorter than its columns"));
2896    }
2897    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
2898    let mut bytes = vec![0; section];
2899    let offset = stripe
2900        .index
2901        .offset
2902        .checked_add(at as u64)
2903        .ok_or_else(|| invalid("index page offset overflow"))?;
2904    read_at(file, offset, &mut bytes)?;
2905    let entries = section - size_of::<u64>();
2906    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
2907    if checksum(&bytes[..entries]) != stored {
2908        // With where it was read from, because the two ways this fires look identical from the
2909        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
2910        return Err(invalid(&format!(
2911            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
2912             wanted {stored:016x} and got {:016x}",
2913            checksum(&bytes[..entries]),
2914        )));
2915    }
2916    let mut spans = Vec::with_capacity(parts);
2917    let mut start = 0_usize;
2918    for part in 0..parts {
2919        let at = part * INDEX_ENTRY;
2920        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
2921        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
2922        spans.push(PartSpan { start, length, hash });
2923        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
2924    }
2925    if start != page.length as usize {
2926        return Err(invalid("column page length differs from its index"));
2927    }
2928    Ok(spans)
2929}
2930
2931/// One part's bytes out of a whole column page.
2932fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
2933    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
2934    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
2935}
2936
2937/// Puts one stripe of one column in the cache, dropping the stripe that has been there longest.
2938///
2939/// The index goes in its own slot and stays. Only the page is under the budget, and `kept` is how
2940/// many pages that budget is.
2941fn remember(cached: &mut Cached, held: &CachedColumn, kept: usize) {
2942    if let Some(slot) = cached.index.get_mut(held.stripe) {
2943        if slot.is_none() {
2944            *slot = Some(Arc::clone(&held.index));
2945        }
2946    }
2947    let Some(page) = held.page.clone() else { return };
2948    let Some(slot) = cached.pages.get_mut(held.stripe) else { return };
2949    if slot.is_none() {
2950        cached.order.push_back(held.stripe);
2951    }
2952    *slot = Some(page);
2953    while cached.order.len() > kept.max(1) {
2954        let Some(oldest) = cached.order.pop_front() else { break };
2955        if let Some(slot) = cached.pages.get_mut(oldest) {
2956            *slot = None;
2957        }
2958    }
2959}
2960
2961/// Every table a native file holds, without the directory of any of them.
2962///
2963/// This is what opening a database reads. It is the small level of the directory, so the cost is
2964/// proportional to how many tables there are rather than to how much data they hold, and a session
2965/// that touches two tables of eight decodes two table directories.
2966///
2967/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
2968/// file descriptor, not eight, which is the other thing one file buys over a file per table.
2969#[derive(Debug, Clone)]
2970pub struct Catalog {
2971    file: Arc<File>,
2972    size: u64,
2973    entries: Arc<Vec<Entry>>,
2974    /// The views the file holds, whole, since a view has no second level to read later.
2975    views: Arc<Vec<ViewEntry>>,
2976    opening: Opening,
2977}
2978
2979impl Catalog {
2980    /// Reads the highest valid catalog slot and nothing under it.
2981    ///
2982    /// # Errors
2983    ///
2984    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
2985    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
2986        let (file, size, _, bytes, opening) = slot_bytes(path)?;
2987        let (entries, views) = decode_catalog(&bytes, size)?;
2988        Ok(Self {
2989            file: Arc::new(file),
2990            size,
2991            entries: Arc::new(entries),
2992            views: Arc::new(views),
2993            opening,
2994        })
2995    }
2996
2997    /// The tables in the file, in the order they were written.
2998    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
2999        self.entries.iter().map(|entry| entry.name.as_str())
3000    }
3001
3002    /// The same tables with how many rows each of them holds.
3003    ///
3004    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
3005    /// A load asks a second question: whether a table already in the file is really in the way of
3006    /// the one it wants to write. A table with no rows is not, because it has no pages the next
3007    /// generation would have to carry, so the count has to come out of the catalog beside the name.
3008    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
3009        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
3010    }
3011
3012    /// The views in the file, in the order they were written.
3013    ///
3014    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
3015    /// by one. A view is a few strings and a column list and it was all read at open, so there is
3016    /// nothing left to go and fetch and no reason to make the caller ask twice.
3017    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
3018        self.views.iter()
3019    }
3020
3021    /// How many tables the file holds.
3022    #[must_use]
3023    pub fn len(&self) -> usize {
3024        self.entries.len()
3025    }
3026
3027    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
3028    /// database somebody dropped the last table out of comes back as.
3029    #[must_use]
3030    pub fn is_empty(&self) -> bool {
3031        self.entries.is_empty()
3032    }
3033
3034    /// Opens one table by name, decoding its directory now.
3035    ///
3036    /// # Errors
3037    ///
3038    /// If there is no table by that name, or its directory is torn or points outside the file.
3039    pub fn table(&self, name: &str) -> Result<Reader> {
3040        let entry = self
3041            .entries
3042            .iter()
3043            .find(|entry| entry.name == name)
3044            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
3045        let mut bytes = vec![0; entry.directory.length as usize];
3046        read_at(&self.file, entry.directory.offset, &mut bytes)?;
3047        if checksum(&bytes) != entry.directory.hash {
3048            return Err(invalid(&format!("the directory of table {name} does not checksum")));
3049        }
3050        let mut opening = self.opening;
3051        opening.reads += 1;
3052        opening.bytes += u64::from(entry.directory.length);
3053        Reader::build(
3054            Arc::clone(&self.file),
3055            self.size,
3056            decode_directory(&bytes, self.size)?,
3057            u64::from(entry.directory.length),
3058            opening,
3059        )
3060    }
3061}
3062
3063/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
3064///
3065/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
3066/// before there was a second generation to write.
3067fn slot_offset(generation: u64) -> u64 {
3068    16 + (generation - 1) % 2 * SLOT_BYTES as u64
3069}
3070
3071/// The header and the bytes the highest valid slot points at.
3072///
3073/// Both levels of the directory are reached this way, so the magic check, the version check and the
3074/// choice between the two slots live here rather than being written out twice.
3075fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
3076    let mut file = File::open(path).map_err(io)?;
3077    let size = file.metadata().map_err(io)?.len();
3078    if size < HEADER {
3079        return Err(invalid("file is shorter than its header"));
3080    }
3081    let mut header = [0; HEADER as usize];
3082    file.read_exact(&mut header).map_err(io)?;
3083    let mut opening = Opening { reads: 1, bytes: HEADER };
3084    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
3085    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
3086    // the answer is to look at the path. A wrong version is our own file from another build,
3087    // and the number this build wants is the only thing that tells the reader whether to
3088    // rebuild the file or to go back to the binary that wrote it.
3089    if &header[..8] != MAGIC {
3090        return Err(invalid("the header does not begin with a rudb native magic"));
3091    }
3092    if !READABLE.contains(&version) {
3093        return Err(invalid(&format!(
3094            "the file is format {version} and this build reads format {FORMAT}, so it has to \
3095                 be written again"
3096        )));
3097    }
3098    let mut selected = None;
3099    for start in [16, 16 + SLOT_BYTES] {
3100        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
3101        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
3102            continue;
3103        }
3104        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
3105        if slot.offset < HEADER || end > size {
3106            continue;
3107        }
3108        let mut bytes = vec![0; slot.length as usize];
3109        file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
3110        file.read_exact(&mut bytes).map_err(io)?;
3111        opening.reads += 1;
3112        opening.bytes += u64::from(slot.length);
3113        if checksum(&bytes) == slot.hash
3114            && selected
3115                .as_ref()
3116                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
3117        {
3118            selected = Some((slot, bytes));
3119        }
3120    }
3121    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
3122    Ok((file, size, slot, bytes, opening))
3123}
3124
3125impl Reader {
3126    /// Opens a file that holds exactly one table.
3127    ///
3128    /// # Errors
3129    ///
3130    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
3131    /// file holds more than one table, which is a file that has to be opened by name.
3132    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
3133        let catalog = Catalog::open(path)?;
3134        let mut names = catalog.names();
3135        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
3136        if names.next().is_some() {
3137            return Err(invalid(
3138                "the file holds more than one table, so it has to be opened by name",
3139            ));
3140        }
3141        catalog.table(&name)
3142    }
3143
3144    /// Builds a reader over one decoded table directory.
3145    fn build(
3146        file: Arc<File>,
3147        size: u64,
3148        table: Table,
3149        directory: u64,
3150        opening: Opening,
3151    ) -> Result<Self> {
3152        let places = places(&table)?;
3153        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
3154        let table_fields = table.fields.len();
3155        let stripes = table.stripes.len();
3156        let cache = (0..table.fields.len())
3157            .map(|_| {
3158                Mutex::new(Cached {
3159                    pages: (0..stripes).map(|_| None).collect(),
3160                    index: (0..stripes).map(|_| None).collect(),
3161                    ..Cached::default()
3162                })
3163            })
3164            .collect::<Vec<_>>();
3165        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
3166            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3167            .collect();
3168        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
3169            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3170            .collect();
3171        Ok(Self {
3172            file,
3173            table: Arc::new(table),
3174            dictionaries: Arc::new(dictionaries),
3175            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
3176            opened: Arc::new(AtomicUsize::new(0)),
3177            sieves: Arc::new(sieves),
3178            part_ranges: Arc::new(part_ranges),
3179            places: Arc::new(places),
3180            cache: Arc::new(cache),
3181            pages: Arc::new(AtomicUsize::new(0)),
3182            indexes: Arc::new(AtomicUsize::new(0)),
3183            kept: Arc::new(AtomicUsize::new(CACHED_STRIPES_PER_COLUMN)),
3184            size,
3185            directory,
3186            opening,
3187        })
3188    }
3189
3190    /// What this reader has read so far, and what opening it cost.
3191    ///
3192    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
3193    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
3194    /// file touched the data asks here, and gets an answer that does not depend on what the page
3195    /// cache happened to hold.
3196    #[must_use]
3197    pub fn reads(&self) -> Reads {
3198        Reads {
3199            opening: self.opening,
3200            pages: self.pages.load(Atomic::Relaxed),
3201            indexes: self.indexes.load(Atomic::Relaxed),
3202            dictionaries: self.opened.load(Atomic::Relaxed),
3203        }
3204    }
3205
3206    /// Where the file's bytes went, from the directory alone.
3207    ///
3208    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
3209    /// for what is charged where and for why the three things that are not columns stay separate.
3210    #[must_use]
3211    pub fn layout(&self) -> Layout {
3212        let table = &self.table;
3213        let stripes = table.stripes.as_slice();
3214        let columns = table
3215            .fields
3216            .iter()
3217            .enumerate()
3218            .map(|(at, field)| ColumnLayout {
3219                name: field.name.clone(),
3220                kind: field.ty.to_string(),
3221                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
3222                memberships: sum(stripes.iter().map(|stripe| page_bytes(&stripe.memberships, at))),
3223                sieves: sum(stripes.iter().map(|stripe| page_bytes(&stripe.sieves, at))),
3224                part_ranges: sum(stripes.iter().map(|stripe| page_bytes(&stripe.part_ranges, at))),
3225                dictionary: page_bytes(&table.dictionaries, at),
3226            })
3227            .collect();
3228        Layout {
3229            file: self.size,
3230            rows: table.rows,
3231            stripes: stripes.len(),
3232            parts: self.places.len(),
3233            columns,
3234            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
3235            directory: self.directory,
3236            header: HEADER,
3237        }
3238    }
3239
3240    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
3241    ///
3242    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
3243    /// nowhere else. The directory says how many bytes a column took and says nothing about what
3244    /// shape they are in, and the shape is the question worth asking: the same rows in a different
3245    /// order come back bit packed on one file and plain on another, and that is the difference a
3246    /// clustered load makes to a scan.
3247    ///
3248    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
3249    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
3250    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
3251    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
3252    ///
3253    /// # Errors
3254    ///
3255    /// If the column is outside the schema, or a page, index section or checksum is invalid.
3256    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
3257        let field = self
3258            .table
3259            .fields
3260            .get(column)
3261            .ok_or_else(|| invalid("stored column index out of range"))?;
3262        let mut stored = Vec::with_capacity(self.places.len());
3263        let mut row = 0;
3264        for (at, stripe) in self.table.stripes.iter().enumerate() {
3265            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3266            let index = read_index(&self.file, stripe, column)?;
3267            let mut bytes = vec![0; page.length as usize];
3268            read_at(&self.file, page.offset, &mut bytes)?;
3269            let ranges = self.stripe_part_ranges(at, column);
3270            for (part, &rows) in stripe.parts.iter().enumerate() {
3271                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
3272                let held = part_bytes(&bytes, span)?;
3273                let range = ranges.and_then(|held| held.get(part));
3274                stored.push(StoredPart {
3275                    stripe: at,
3276                    part,
3277                    row,
3278                    rows: rows as usize,
3279                    encoding: page_encoding(&field.ty, rows as usize, held),
3280                    bytes: span.length as u64,
3281                    page: page.offset,
3282                    offset: span.start as u64,
3283                    low: range
3284                        .and_then(|range| range.low.clone())
3285                        .and_then(|bound| bound.into_value(&field.ty)),
3286                    high: range
3287                        .and_then(|range| range.high.clone())
3288                        .and_then(|bound| bound.into_value(&field.ty)),
3289                    nulls: range.map(|range| range.nulls),
3290                });
3291                row += rows as usize;
3292            }
3293        }
3294        Ok(stored)
3295    }
3296
3297    /// How many parts the table has, which is how many chunks a scan of it reads.
3298    #[must_use]
3299    pub fn parts(&self) -> usize {
3300        self.places.len()
3301    }
3302
3303    /// The parts of each stripe, in table wide part numbers.
3304    ///
3305    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
3306    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
3307    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
3308    /// directory rather than worked out from a constant.
3309    #[must_use]
3310    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
3311        let mut runs = Vec::with_capacity(self.table.stripes.len());
3312        let mut start = 0;
3313        for stripe in &self.table.stripes {
3314            let end = start + stripe.parts.len();
3315            runs.push(start..end);
3316            start = end;
3317        }
3318        runs
3319    }
3320
3321    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
3322    ///
3323    /// Off the directory, which is already in memory, rather than by the caller asking for each
3324    /// part in turn through the catalog. Nothing past the end holds any rows.
3325    #[must_use]
3326    pub fn stripe_rows(&self, stripe: usize) -> usize {
3327        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
3328    }
3329
3330    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
3331    ///
3332    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
3333    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
3334    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
3335    /// reads a quarter of a megabyte for every part it takes out of it.
3336    pub fn keep_stripes(&self, stripes: usize) {
3337        self.kept.fetch_max(stripes, Atomic::Relaxed);
3338    }
3339
3340    /// Rows in one part, or zero when the part number is past the table.
3341    #[must_use]
3342    pub fn part_rows(&self, at: usize) -> usize {
3343        self.places.get(at).map_or(0, |place| place.rows as usize)
3344    }
3345
3346    /// The committed table directory.
3347    #[must_use]
3348    pub fn table(&self) -> &Table {
3349        &self.table
3350    }
3351
3352    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
3353    ///
3354    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
3355    /// additional ordering keys without losing a value tied with the requested boundary.
3356    ///
3357    /// # Errors
3358    ///
3359    /// If the column is outside the schema or a stored value does not fit its declared type.
3360    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
3361        let field = self
3362            .table
3363            .fields
3364            .get(column)
3365            .ok_or_else(|| invalid("frequency column index out of range"))?;
3366        let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3367            return Ok(None);
3368        };
3369        if top == 0 || summary.entries.len() < top {
3370            return Ok(None);
3371        }
3372        let boundary = summary.entries[top - 1].count;
3373        if boundary <= summary.omitted_max {
3374            return Ok(None);
3375        }
3376        self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
3377    }
3378
3379    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
3380    ///
3381    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
3382    /// out of room, so what it usually ends with is the leading values and a bound on everything it
3383    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
3384    /// the entries did not overflow the stored budget, so the list is every distinct value of the
3385    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
3386    ///
3387    /// That makes a whole class of question answerable without reading a row. How many rows hold a
3388    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
3389    /// all in here. It is only ever true of a column with few enough distinct values, which is the
3390    /// case worth having, because that is exactly the column a grouping or an equality filter would
3391    /// otherwise walk every row to answer.
3392    ///
3393    /// `None` when the column has no synopsis, or has one that dropped anything.
3394    ///
3395    /// # Errors
3396    ///
3397    /// If the column is outside the schema or a stored value does not fit its declared type.
3398    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
3399        let Some(prefix) = self.frequency_prefix(column)? else {
3400            return Ok(None);
3401        };
3402        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
3403    }
3404
3405    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
3406    ///
3407    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
3408    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
3409    /// made it into the list carries the number of rows that really hold it rather than whatever the
3410    /// pass had left over. What the pass loses is values, not counts.
3411    ///
3412    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
3413    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
3414    /// leading values of the column and everything else is somewhere between no rows and that bound.
3415    ///
3416    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
3417    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
3418    /// the rows by the distinct count is furthest from the truth.
3419    ///
3420    /// `None` when the column has no synopsis.
3421    ///
3422    /// # Errors
3423    ///
3424    /// If the column is outside the schema or a stored value does not fit its declared type.
3425    ///
3426    /// [`exact_frequencies`]: Self::exact_frequencies
3427    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
3428        let field = self
3429            .table
3430            .fields
3431            .get(column)
3432            .ok_or_else(|| invalid("frequency column index out of range"))?;
3433        let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3434            return Ok(None);
3435        };
3436        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
3437        Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
3438    }
3439
3440    /// Turns stored frequency entries into values of the column's own type.
3441    fn decode_frequencies(
3442        &self,
3443        column: usize,
3444        ty: &LogicalType,
3445        entries: &[FrequencyEntry],
3446    ) -> Result<Vec<(Value, u64)>> {
3447        let dictionary = if *ty == LogicalType::Varchar { self.dictionary(column)? } else { None };
3448        let mut out = Vec::with_capacity(entries.len());
3449        for entry in entries {
3450            let value = match entry.value {
3451                FrequencyValue::Null => Value::Null,
3452                FrequencyValue::Integer(value) => match *ty {
3453                    LogicalType::TinyInt => Value::TinyInt(
3454                        i8::try_from(value)
3455                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
3456                    ),
3457                    LogicalType::UTinyInt => Value::UTinyInt(
3458                        u8::try_from(value)
3459                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
3460                    ),
3461                    LogicalType::USmallInt => Value::USmallInt(
3462                        u16::try_from(value)
3463                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
3464                    ),
3465                    LogicalType::UInteger => Value::UInteger(
3466                        u32::try_from(value)
3467                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
3468                    ),
3469                    LogicalType::UBigInt => Value::UBigInt(
3470                        u64::try_from(value)
3471                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
3472                    ),
3473                    LogicalType::SmallInt => Value::SmallInt(
3474                        i16::try_from(value)
3475                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
3476                    ),
3477                    LogicalType::Integer => Value::Integer(
3478                        i32::try_from(value)
3479                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
3480                    ),
3481                    LogicalType::BigInt => Value::BigInt(
3482                        i64::try_from(value)
3483                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
3484                    ),
3485                    LogicalType::Date => Value::Date(
3486                        i32::try_from(value)
3487                            .map_err(|_| invalid("frequency DATE is out of range"))?,
3488                    ),
3489                    LogicalType::Timestamp => Value::Timestamp(
3490                        i64::try_from(value)
3491                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
3492                    ),
3493                    _ => return Err(invalid("integer frequency belongs to another type")),
3494                },
3495                FrequencyValue::Code(code) => dictionary
3496                    .as_ref()
3497                    .ok_or_else(|| invalid("frequency code has no dictionary"))?
3498                    .try_value_at(code as usize)?,
3499            };
3500            out.push((value, entry.count));
3501        }
3502        Ok(out)
3503    }
3504
3505    /// Sparse rows belonging to the bounded numeric frequency candidate set.
3506    ///
3507    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
3508    /// aggregate may accept a result over these rows only when its requested boundary is strictly
3509    /// greater than `omitted_max`.
3510    ///
3511    /// # Errors
3512    ///
3513    /// If the column is outside the schema.
3514    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
3515        self.table
3516            .fields
3517            .get(column)
3518            .ok_or_else(|| invalid("frequency column index out of range"))?;
3519        let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3520            return Ok(None);
3521        };
3522        if summary.ordinals.is_empty() {
3523            return Ok(None);
3524        }
3525        Ok(Some(FrequencyOccurrences {
3526            omitted_max: summary.omitted_max,
3527            ordinals: summary.ordinals.clone(),
3528        }))
3529    }
3530
3531    /// How many distinct values one column holds, counting a null as no value.
3532    ///
3533    /// A string column of this format is written against one dictionary that covers the whole table.
3534    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
3535    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
3536    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
3537    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
3538    /// every row.
3539    ///
3540    /// A null in the column used to make this `None` and no longer does. A null row is written as
3541    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
3542    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
3543    /// The writer does know, because it counts the non-null rows that use each code on its way to
3544    /// the frequency summary, so it records how many codes any row holds and the directory carries
3545    /// that number. This reads it rather than the size of the dictionary, which also means the
3546    /// dictionary page is not opened to answer.
3547    ///
3548    /// `None` for a column the file has no dictionary for, which is every column that is not a
3549    /// string. A sketch would answer that approximately and SQL asked for the exact number.
3550    ///
3551    /// # Errors
3552    ///
3553    /// If the column is outside the schema.
3554    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
3555        self.table
3556            .distincts
3557            .get(column)
3558            .copied()
3559            .ok_or_else(|| invalid("distinct column index out of range"))
3560    }
3561
3562    /// How many rows of one column are null, added up over the stripes.
3563    ///
3564    /// Every stripe records this exactly when it is written, because a null count is not a bound
3565    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
3566    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
3567    /// already in memory is what makes `COUNT(column)` over a whole table free.
3568    ///
3569    /// # Errors
3570    ///
3571    /// If the column is outside the schema.
3572    pub fn null_count(&self, column: usize) -> Result<u64> {
3573        if column >= self.table.fields.len() {
3574            return Err(invalid("null count column index out of range"));
3575        }
3576        let mut nulls = 0_u64;
3577        for stripe in &self.table.stripes {
3578            let range = stripe
3579                .zone
3580                .column(column)
3581                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3582            nulls = nulls
3583                .checked_add(range.nulls as u64)
3584                .ok_or_else(|| invalid("null count overflow"))?;
3585        }
3586        Ok(nulls)
3587    }
3588
3589    /// The smallest and the largest value of one string column, from the order beside its values.
3590    ///
3591    /// The dictionary holds exactly the values the column holds, so the first and the last of them
3592    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
3593    /// otherwise walks a million rows.
3594    ///
3595    /// `None` when the column is not a string, when the file was written before version 9 and so has
3596    /// no order, when the column has no values at all, or when it has a null in it, which is the
3597    /// placeholder again: the empty string a null is written as would sort ahead of every real
3598    /// value and be reported as the minimum.
3599    ///
3600    /// # Errors
3601    ///
3602    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
3603    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
3604        if self.null_count(column)? > 0 {
3605            return Ok(None);
3606        }
3607        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
3608        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
3609        if ranks == 0 {
3610            return Ok(None);
3611        }
3612        let low = text_at_rank(&dictionary, 0)?;
3613        let high = text_at_rank(&dictionary, ranks - 1)?;
3614        Ok(Some((low, high)))
3615    }
3616
3617    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
3618    ///
3619    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
3620    /// chunk that could not match is still correct when it rules out nothing. That is what makes
3621    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
3622    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
3623    /// all of them walked their rows.
3624    ///
3625    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
3626    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
3627    ///
3628    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
3629    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
3630    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
3631    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
3632    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
3633    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
3634    /// and the fix is a row count per part rather than anything here.
3635    ///
3636    /// # Errors
3637    ///
3638    /// If the column is outside the schema.
3639    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
3640        if column >= self.table.fields.len() {
3641            return Err(invalid("extremes column index out of range"));
3642        }
3643        let mut low: Option<Bound> = None;
3644        let mut high: Option<Bound> = None;
3645        for stripe in &self.table.stripes {
3646            let range = stripe
3647                .zone
3648                .column(column)
3649                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3650            if !range.exact {
3651                return Ok(None);
3652            }
3653            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
3654            // is why this skips it rather than giving up on the whole column. A stripe that has
3655            // rows and still has no end is a layout whose values this cannot see, and skipping that
3656            // one would answer with an end taken from the other stripes, so it gives up instead.
3657            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
3658                if stripe.rows > range.nulls {
3659                    return Ok(None);
3660                }
3661                continue;
3662            };
3663            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
3664            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
3665        }
3666        Ok(low.zip(high))
3667    }
3668
3669    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
3670    ///
3671    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
3672    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
3673    /// count would be doing the same walk twice.
3674    ///
3675    /// `None` for anything that is not an integer column, for a file written by something that did
3676    /// not record it, and when adding the stripes together would overflow.
3677    ///
3678    /// # Errors
3679    ///
3680    /// If the column is outside the schema.
3681    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
3682        if column >= self.table.fields.len() {
3683            return Err(invalid("sum column index out of range"));
3684        }
3685        let mut total = 0_i128;
3686        let mut rows = 0_u64;
3687        for stripe in &self.table.stripes {
3688            let range = stripe
3689                .zone
3690                .column(column)
3691                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3692            let Some(part) = range.sum else { return Ok(None) };
3693            let Some(sum) = total.checked_add(part) else { return Ok(None) };
3694            total = sum;
3695            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
3696        }
3697        Ok(Some((total, rows)))
3698    }
3699
3700    /// The global dictionary of a column, opened once however many workers ask for it at once.
3701    ///
3702    /// The unlocked look is first because it is the answer every time after the first and it costs a
3703    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
3704    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
3705    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
3706    /// dictionary that can hold half a million entries, and the alternative is every worker of the
3707    /// scan doing all of it and all but one dropping the result on the floor.
3708    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
3709        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
3710        if let Some(dictionary) = self.dictionaries[column].get() {
3711            return Ok(Some(Arc::clone(dictionary)));
3712        }
3713        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
3714        if let Some(dictionary) = self.dictionaries[column].get() {
3715            return Ok(Some(Arc::clone(dictionary)));
3716        }
3717        self.opened.fetch_add(1, Atomic::Relaxed);
3718        let dictionary = Arc::new(open_global_dictionary(
3719            Arc::clone(&self.file),
3720            page,
3721            &self.table.fields[column].ty,
3722            TEXT_KEEP_BUDGET,
3723        )?);
3724        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
3725        Ok(Some(dictionary))
3726    }
3727
3728    /// Reads one section's extent table and checks it against the entry that names it.
3729    ///
3730    /// # Errors
3731    ///
3732    /// If the entry points outside the file, the table does not checksum, or it does not decode as
3733    /// a run of extents in element order.
3734    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
3735        if of.extent_bytes == 0 {
3736            return Ok(Vec::new());
3737        }
3738        let mut bytes = vec![0; of.extent_bytes as usize];
3739        read_at(&self.file, of.extent_page, &mut bytes)?;
3740        if checksum(&bytes) != of.hash {
3741            return Err(invalid("a section's extent table does not checksum"));
3742        }
3743        let extents = section::decode_extents(&bytes)?;
3744        if extents.len() != of.extents as usize {
3745            return Err(invalid("a section's extent table is not the length the entry says"));
3746        }
3747        Ok(extents)
3748    }
3749
3750    /// Reads and verifies one extent of a section.
3751    ///
3752    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
3753    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
3754    /// difference between a structure that works at SF100 and issue #745.
3755    ///
3756    /// # Errors
3757    ///
3758    /// If the extent points outside the file, or its bytes do not checksum.
3759    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
3760        let end = of
3761            .offset
3762            .checked_add(u64::from(of.length))
3763            .ok_or_else(|| invalid("an extent overflows the file"))?;
3764        if of.offset < HEADER || end > self.size {
3765            return Err(invalid("an extent is outside the file"));
3766        }
3767        let mut bytes = vec![0; of.length as usize];
3768        read_at(&self.file, of.offset, &mut bytes)?;
3769        if checksum(&bytes) != of.hash {
3770            return Err(invalid("an extent does not checksum"));
3771        }
3772        Ok(bytes)
3773    }
3774
3775    /// Reads a whole section's payload, every extent of it, in order.
3776    ///
3777    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
3778    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
3779    ///
3780    /// # Errors
3781    ///
3782    /// If the extent table or any extent fails its check.
3783    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
3784        let extents = self.extents(of)?;
3785        let mut bytes =
3786            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
3787        for one in &extents {
3788            if one.first != bytes.len() as u64 {
3789                return Err(invalid("a section's extents do not join up"));
3790            }
3791            bytes.extend_from_slice(&self.extent(one)?);
3792        }
3793        if of.header_bytes as usize > bytes.len() {
3794            return Err(invalid("a section's header is longer than its payload"));
3795        }
3796        Ok(bytes)
3797    }
3798
3799    /// Reads only the named columns from one part.
3800    ///
3801    /// The whole stripe page each column lives in is read and kept, because a scan asks for the
3802    /// parts of a stripe one after another and this is what turns sixty four reads into one.
3803    ///
3804    /// # Errors
3805    ///
3806    /// If a part, column, page, or checksum is invalid.
3807    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3808        self.read_impl(part, columns, true)
3809    }
3810
3811    /// Reads named columns from one part without keeping the stripe page it came out of.
3812    ///
3813    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
3814    /// a stripe rather than all of them. A caller that will read most of a stripe should use
3815    /// [`Self::read`] instead, because this reads and discards the page index every time.
3816    ///
3817    /// # Errors
3818    ///
3819    /// If a part, column, page, or checksum is invalid.
3820    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3821        self.read_impl(part, columns, false)
3822    }
3823
3824    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
3825    /// contain any of the sorted candidate codes.
3826    ///
3827    /// # Errors
3828    ///
3829    /// If the part, column, index page, checksum, or delta stream is invalid.
3830    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
3831        if candidates.is_empty() {
3832            return Ok(true);
3833        }
3834        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
3835            return Err(Error::internal("native code candidates are not sorted and unique"));
3836        }
3837        let stripe = self.stripe_of(part)?;
3838        let Some(page) = stripe.memberships.get(column).copied().flatten() else {
3839            return Ok(false);
3840        };
3841        let mut bytes = vec![0; page.length as usize];
3842        read_at(&self.file, page.offset, &mut bytes)?;
3843        if checksum(&bytes) != page.hash {
3844            return Err(invalid("membership page checksum differs"));
3845        }
3846        let codes = decode_membership(&bytes)?;
3847        let mut left = 0;
3848        let mut right = 0;
3849        while left < codes.len() && right < candidates.len() {
3850            match codes[left].cmp(&candidates[right]) {
3851                Ordering::Less => left += 1,
3852                Ordering::Greater => right += 1,
3853                Ordering::Equal => return Ok(false),
3854            }
3855        }
3856        Ok(true)
3857    }
3858
3859    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
3860        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
3861        self.table
3862            .stripes
3863            .get(place.stripe as usize)
3864            .ok_or_else(|| invalid("stripe index out of range"))
3865    }
3866
3867    /// The page index of one column of one stripe, and its page when the caller wants all of it.
3868    ///
3869    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
3870    /// a few parts of the others and they all want the same page at the same moment. This used to
3871    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
3872    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
3873    /// look at 400 MB of column.
3874    ///
3875    /// A worker that finds the page it wants already being read neither waits for it nor reads it
3876    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
3877    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
3878    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
3879    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
3880    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
3881    ///
3882    /// The file is never read under the lock.
3883    fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
3884        let cache = self.cache.get(column).ok_or_else(|| invalid("column index out of range"))?;
3885        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3886        let known = cached.index.get(at).and_then(Clone::clone);
3887        let page = cached.pages.get(at).and_then(Clone::clone);
3888        if let Some(index) = known.clone() {
3889            if !whole || page.is_some() {
3890                return Ok(CachedColumn { stripe: at, index, page });
3891            }
3892        }
3893        if cached.loading.contains(&at) {
3894            drop(cached);
3895            // The index is almost always already here, because somebody read this stripe to get
3896            // into the loading list in the first place, so this branch usually costs no read at
3897            // all and the one part read in `read_impl` is all the losing worker pays for.
3898            if let Some(index) = known {
3899                return Ok(CachedColumn { stripe: at, index, page: None });
3900            }
3901            let held = self.page_of(stripe, column, at, false, None)?;
3902            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3903            remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3904            return Ok(held);
3905        }
3906        cached.loading.push(at);
3907        drop(cached);
3908
3909        let read = self.page_of(stripe, column, at, whole, known);
3910
3911        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
3912        // them separately would leave a moment where another worker sees neither and reads the
3913        // page a second time, which is the whole thing this is here to stop.
3914        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3915        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
3916            cached.loading.remove(position);
3917        }
3918        let held = read?;
3919        remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3920        Ok(held)
3921    }
3922
3923    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
3924    ///
3925    /// `known` is the index when the reader has already read it, which after the first worker
3926    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
3927    /// reader. Without that a scan reads the index again on every part that misses the page cache.
3928    fn page_of(
3929        &self,
3930        stripe: &Stripe,
3931        column: usize,
3932        at: usize,
3933        whole: bool,
3934        known: Option<Arc<Vec<PartSpan>>>,
3935    ) -> Result<CachedColumn> {
3936        let index = match known {
3937            Some(index) => index,
3938            None => {
3939                self.indexes.fetch_add(1, Atomic::Relaxed);
3940                Arc::new(read_index(&self.file, stripe, column)?)
3941            }
3942        };
3943        let page = if whole {
3944            self.pages.fetch_add(1, Atomic::Relaxed);
3945            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3946            let mut bytes = vec![0; span.length as usize];
3947            read_at(&self.file, span.offset, &mut bytes)?;
3948            Some(Arc::new(bytes))
3949        } else {
3950            None
3951        };
3952        Ok(CachedColumn { stripe: at, index, page })
3953    }
3954
3955    fn read_impl(&self, at: usize, columns: &[usize], whole: bool) -> Result<Chunk> {
3956        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
3957        let index = place.stripe as usize;
3958        let stripe =
3959            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
3960        let rows = place.rows as usize;
3961        let mut picked = Vec::with_capacity(columns.len());
3962        for &column in columns {
3963            let field = self
3964                .table
3965                .fields
3966                .get(column)
3967                .ok_or_else(|| invalid("column index out of range"))?;
3968            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3969            let held = self.held(index, stripe, column, whole)?;
3970            let span = *held
3971                .index
3972                .get(place.part as usize)
3973                .ok_or_else(|| invalid("part index out of range"))?;
3974            let owned;
3975            let bytes = match &held.page {
3976                Some(held) => part_bytes(held, span)?,
3977                None => {
3978                    let offset = page
3979                        .offset
3980                        .checked_add(span.start as u64)
3981                        .ok_or_else(|| invalid("part range overflow"))?;
3982                    let mut bytes = vec![0; span.length];
3983                    read_at(&self.file, offset, &mut bytes)?;
3984                    owned = bytes;
3985                    &owned
3986                }
3987            };
3988            if checksum(bytes) != span.hash {
3989                return Err(invalid(&format!(
3990                    "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
3991                     wanted {:016x} and got {:016x}",
3992                    place.part,
3993                    page.offset,
3994                    span.start,
3995                    span.length,
3996                    span.hash,
3997                    checksum(bytes),
3998                )));
3999            }
4000            let dictionary = self.dictionary(column)?;
4001            // Held as a page, because a column that came out of a file is handed out more than
4002            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
4003            // projection of a bare column name does the same, and a cut of a flat run copies unless
4004            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
4005            // run into the `Arc` without touching a value.
4006            picked.push(decode(&field.ty, rows, bytes, dictionary)?.into_pages());
4007        }
4008        Chunk::with_rows(picked, rows)
4009    }
4010
4011    /// Whether persisted statistics prove that a part cannot match the predicates.
4012    ///
4013    /// Three of them, asked cheapest first.
4014    ///
4015    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
4016    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
4017    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
4018    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
4019    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
4020    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
4021    /// really hold the value.
4022    ///
4023    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
4024    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
4025    /// and the part bounds leave thirty parts of nine hundred and seventy four.
4026    #[must_use]
4027    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
4028        let Some(place) = self.places.get(part).copied() else { return false };
4029        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
4030        if stripe.zone.skips(probes) {
4031            return true;
4032        }
4033        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
4034    }
4035
4036    /// Whether the bounds of one part rule out one probe.
4037    ///
4038    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
4039    /// time this is asked about a column. A column with no page here answers `false`, which is the
4040    /// answer a caller got before there were any.
4041    fn outside(&self, place: Place, probe: &Probe) -> bool {
4042        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
4043            Some(ranges) => ranges
4044                .get(place.part as usize)
4045                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
4046            None => false,
4047        }
4048    }
4049
4050    /// The per part ranges of one stripe of one column, read once and kept.
4051    ///
4052    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
4053    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
4054    /// cannot read one reads the rows and gets the right answer slowly.
4055    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
4056        let slot = self.part_ranges.get(column)?.get(stripe)?;
4057        if let Some(held) = slot.get() {
4058            return Some(held);
4059        }
4060        let page = self.table.stripes.get(stripe)?.part_ranges.get(column).copied().flatten()?;
4061        let mut bytes = vec![0; page.length as usize];
4062        read_at(&self.file, page.offset, &mut bytes).ok()?;
4063        if checksum(&bytes) != page.hash {
4064            return None;
4065        }
4066        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
4067        let _ = slot.set(ranges);
4068        slot.get().map(|held| held.as_slice())
4069    }
4070
4071    /// Whether persisted statistics prove that every row of a part matches the predicates.
4072    ///
4073    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
4074    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
4075    /// through.
4076    ///
4077    /// The stripe first and the part after it, the same two steps and in the same order as
4078    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
4079    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
4080    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
4081    /// stretch where everything passes contains no narrower stretch where something fails, and a
4082    /// stripe with no nulls has no nulls in any of its parts.
4083    ///
4084    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
4085    /// wider than its rows really are as well. That is the same safe direction for the same reason,
4086    /// and it is why this asks the two ends rather than anything `exact` says.
4087    #[must_use]
4088    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
4089        let Some(place) = self.places.get(part).copied() else { return false };
4090        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
4091        if stripe.zone.certain(probes) {
4092            return true;
4093        }
4094        probes
4095            .iter()
4096            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
4097    }
4098
4099    /// Whether one part's own two ends prove that every row of it passes `probe`.
4100    ///
4101    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
4102    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
4103    /// part's and the caller has already asked them.
4104    fn inside(&self, place: Place, probe: &Probe) -> bool {
4105        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
4106            Some(ranges) => ranges
4107                .get(place.part as usize)
4108                .is_some_and(|range| range.certain(probe.op, &probe.value)),
4109            None => false,
4110        }
4111    }
4112
4113    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
4114    ///
4115    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
4116    /// directory and are already in memory, so this answers without touching the file, and that is
4117    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
4118    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
4119    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
4120    ///
4121    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
4122    /// it to be wrong: the parts are still checked when they are read.
4123    #[must_use]
4124    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
4125        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
4126    }
4127
4128    /// Whether the sieve of one part rules out one probe.
4129    ///
4130    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
4131    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
4132    /// sieve gets anyway.
4133    fn sifted(&self, place: Place, probe: &Probe) -> bool {
4134        if probe.op != Op::Equal {
4135            return false;
4136        }
4137        match self.stripe_sieves(place.stripe as usize, probe.column) {
4138            Some(sieves) => sieves
4139                .get(place.part as usize)
4140                .and_then(Option::as_ref)
4141                .is_some_and(|sieve| sieve.excludes(&probe.value)),
4142            None => false,
4143        }
4144    }
4145
4146    /// The sieves of one stripe of one column, read once and kept.
4147    ///
4148    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
4149    /// bytes are not a page this version can read. A sieve is an index over data that is still there
4150    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
4151    /// a bad checksum is a slow query rather than an error.
4152    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
4153        let slot = self.sieves.get(column)?.get(stripe)?;
4154        if let Some(held) = slot.get() {
4155            return Some(held);
4156        }
4157        let page = self.table.stripes.get(stripe)?.sieves.get(column).copied().flatten()?;
4158        let mut bytes = vec![0; page.length as usize];
4159        read_at(&self.file, page.offset, &mut bytes).ok()?;
4160        if checksum(&bytes) != page.hash {
4161            return None;
4162        }
4163        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
4164        let _ = slot.set(sieves);
4165        slot.get().map(|held| held.as_slice())
4166    }
4167}
4168
4169/// The value sitting at one position of a dictionary's sorted order.
4170fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
4171    let code = dictionary.code_at_rank(rank)? as usize;
4172    let text = dictionary
4173        .try_text_at(code)?
4174        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4175    Ok(Value::Varchar(text.into()))
4176}
4177
4178/// Writes one span of a file at an offset, without depending on where the cursor is.
4179///
4180/// The writer owns an offset of its own and passes it in here, so that nothing it writes depends on
4181/// a cursor that a read is entitled to move. Both of these can come back short and both loop.
4182#[cfg(unix)]
4183fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4184    use std::os::unix::fs::FileExt;
4185    while !bytes.is_empty() {
4186        let written = file.write_at(bytes, offset).map_err(io)?;
4187        if written == 0 {
4188            return Err(invalid("a write to the native file wrote nothing"));
4189        }
4190        offset += written as u64;
4191        bytes = &bytes[written..];
4192    }
4193    Ok(())
4194}
4195
4196/// The same write, on the call Windows spells differently.
4197#[cfg(windows)]
4198fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4199    use std::os::windows::fs::FileExt;
4200    while !bytes.is_empty() {
4201        let written = file.seek_write(bytes, offset).map_err(io)?;
4202        if written == 0 {
4203            return Err(invalid("a write to the native file wrote nothing"));
4204        }
4205        offset += written as u64;
4206        bytes = &bytes[written..];
4207    }
4208    Ok(())
4209}
4210
4211/// Somewhere that is neither, where the cursor is all there is.
4212#[cfg(not(any(unix, windows)))]
4213fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
4214    use std::io::Write;
4215    let mut file = file.try_clone().map_err(io)?;
4216    file.seek(SeekFrom::Start(offset)).map_err(io)?;
4217    file.write_all(bytes).map_err(io)
4218}
4219
4220/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
4221///
4222/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
4223/// pages from several threads at once, so this has to be positional. Seeking and then reading is
4224/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
4225/// comes back with somebody else's bytes.
4226///
4227/// Both of these can come back short, so both loop. A read of zero bytes before the span is filled
4228/// means the file stops earlier than the directory said it does.
4229#[cfg(unix)]
4230fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4231    use std::os::unix::fs::FileExt;
4232    while !bytes.is_empty() {
4233        let read = file.read_at(bytes, offset).map_err(io)?;
4234        if read == 0 {
4235            return Err(invalid("column page ends before its declared length"));
4236        }
4237        offset += read as u64;
4238        bytes = &mut bytes[read..];
4239    }
4240    Ok(())
4241}
4242
4243/// The same read, on the call Windows spells differently.
4244///
4245/// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave the
4246/// way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is why
4247/// nothing in this file may read that cursor.
4248#[cfg(windows)]
4249fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4250    use std::os::windows::fs::FileExt;
4251    while !bytes.is_empty() {
4252        let read = file.seek_read(bytes, offset).map_err(io)?;
4253        if read == 0 {
4254            return Err(invalid("column page ends before its declared length"));
4255        }
4256        offset += read as u64;
4257        bytes = &mut bytes[read..];
4258    }
4259    Ok(())
4260}
4261
4262/// Somewhere that is neither, where the cursor is all there is.
4263///
4264/// This one does race, and there is no way to write it so it does not. Nothing we build for runs
4265/// here, so it exists to keep the crate compiling rather than to be correct under threads.
4266#[cfg(not(any(unix, windows)))]
4267fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
4268    let mut file = file.try_clone().map_err(io)?;
4269    file.seek(SeekFrom::Start(offset)).map_err(io)?;
4270    file.read_exact(bytes).map_err(io)
4271}
4272
4273/// What a column type is called in the directory.
4274///
4275/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
4276/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
4277/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
4278/// rather than in an order that means anything.
4279fn type_tag(ty: &LogicalType) -> Result<u8> {
4280    match ty {
4281        LogicalType::SmallInt => Ok(1),
4282        LogicalType::Integer => Ok(2),
4283        LogicalType::BigInt => Ok(3),
4284        LogicalType::Varchar => Ok(4),
4285        LogicalType::Date => Ok(5),
4286        LogicalType::Timestamp => Ok(6),
4287        LogicalType::Boolean => Ok(7),
4288        LogicalType::TinyInt => Ok(8),
4289        LogicalType::UTinyInt => Ok(9),
4290        LogicalType::USmallInt => Ok(10),
4291        LogicalType::UInteger => Ok(11),
4292        LogicalType::UBigInt => Ok(12),
4293        LogicalType::Decimal { .. } => Ok(13),
4294        LogicalType::Float => Ok(14),
4295        LogicalType::Double => Ok(15),
4296        LogicalType::HugeInt => Ok(16),
4297        LogicalType::UHugeInt => Ok(17),
4298        LogicalType::Time => Ok(18),
4299        LogicalType::TimeTz => Ok(19),
4300        LogicalType::TimestampTz => Ok(20),
4301        LogicalType::Interval => Ok(21),
4302        LogicalType::Uuid => Ok(22),
4303        LogicalType::Blob => Ok(23),
4304        LogicalType::Bit => Ok(24),
4305        LogicalType::TimestampS => Ok(25),
4306        LogicalType::TimestampMs => Ok(26),
4307        LogicalType::TimestampNs => Ok(27),
4308        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
4309    }
4310}
4311
4312/// The tag of a column type, and the parameters of the ones that have any.
4313///
4314/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
4315/// because they are what says how wide a value is on disk, and a reader that guessed would read the
4316/// wrong number of bytes per row rather than the wrong number of digits.
4317fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
4318    out.push(type_tag(ty)?);
4319    if let LogicalType::Decimal { width, scale } = ty {
4320        out.push(*width);
4321        out.push(*scale);
4322    }
4323    Ok(())
4324}
4325
4326/// The other half of [`put_type`], reading the parameters the tag says are there.
4327fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
4328    let tag = cur.u8()?;
4329    if tag == 13 {
4330        let width = cur.u8()?;
4331        let scale = cur.u8()?;
4332        return LogicalType::decimal(width, scale)
4333            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
4334    }
4335    tag_type(tag)
4336}
4337
4338fn tag_type(tag: u8) -> Result<LogicalType> {
4339    match tag {
4340        1 => Ok(LogicalType::SmallInt),
4341        2 => Ok(LogicalType::Integer),
4342        3 => Ok(LogicalType::BigInt),
4343        4 => Ok(LogicalType::Varchar),
4344        5 => Ok(LogicalType::Date),
4345        6 => Ok(LogicalType::Timestamp),
4346        7 => Ok(LogicalType::Boolean),
4347        8 => Ok(LogicalType::TinyInt),
4348        9 => Ok(LogicalType::UTinyInt),
4349        10 => Ok(LogicalType::USmallInt),
4350        11 => Ok(LogicalType::UInteger),
4351        12 => Ok(LogicalType::UBigInt),
4352        14 => Ok(LogicalType::Float),
4353        15 => Ok(LogicalType::Double),
4354        16 => Ok(LogicalType::HugeInt),
4355        17 => Ok(LogicalType::UHugeInt),
4356        18 => Ok(LogicalType::Time),
4357        19 => Ok(LogicalType::TimeTz),
4358        20 => Ok(LogicalType::TimestampTz),
4359        21 => Ok(LogicalType::Interval),
4360        22 => Ok(LogicalType::Uuid),
4361        23 => Ok(LogicalType::Blob),
4362        24 => Ok(LogicalType::Bit),
4363        25 => Ok(LogicalType::TimestampS),
4364        26 => Ok(LogicalType::TimestampMs),
4365        27 => Ok(LogicalType::TimestampNs),
4366        _ => Err(invalid("column type tag is unknown")),
4367    }
4368}
4369
4370fn put_u16(out: &mut Vec<u8>, value: u16) {
4371    out.extend_from_slice(&value.to_le_bytes());
4372}
4373fn put_u32(out: &mut Vec<u8>, value: u32) {
4374    out.extend_from_slice(&value.to_le_bytes());
4375}
4376fn put_u64(out: &mut Vec<u8>, value: u64) {
4377    out.extend_from_slice(&value.to_le_bytes());
4378}
4379fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
4380    while value >= 0x80 {
4381        out.push((value as u8 & 0x7f) | 0x80);
4382        value >>= 7;
4383    }
4384    out.push(value as u8);
4385}
4386
4387fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
4388    match (left, right) {
4389        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
4390        (FrequencyValue::Null, _) => Ordering::Less,
4391        (_, FrequencyValue::Null) => Ordering::Greater,
4392        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
4393        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
4394        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
4395        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
4396    }
4397}
4398
4399fn code_frequency(dictionary: &GlobalDictionary) -> FrequencySummary {
4400    let mut entries = dictionary
4401        .counts
4402        .iter()
4403        .enumerate()
4404        .filter(|(_, count)| **count != 0)
4405        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
4406        .collect::<Vec<_>>();
4407    if dictionary.nulls != 0 {
4408        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
4409    }
4410    entries.sort_unstable_by(|left, right| {
4411        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
4412    });
4413    let omitted_max = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
4414    entries.truncate(FREQUENCY_ENTRIES);
4415    FrequencySummary { entries, omitted_max, ordinals: Vec::new() }
4416}
4417
4418fn encode_directory(table: &Table) -> Result<Vec<u8>> {
4419    let mut out = DIRECTORY.to_vec();
4420    let name = table.name.as_bytes();
4421    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4422    out.extend_from_slice(name);
4423    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
4424    for field in &table.fields {
4425        let name = field.name.as_bytes();
4426        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
4427        out.extend_from_slice(name);
4428        put_type(&mut out, &field.ty)?;
4429        out.push(u8::from(field.not_null));
4430    }
4431    for dictionary in &table.dictionaries {
4432        match dictionary {
4433            None => out.push(0),
4434            Some(page) => {
4435                out.push(1);
4436                put_u64(&mut out, page.offset);
4437                put_u32(&mut out, page.length);
4438                put_u64(&mut out, page.hash);
4439            }
4440        }
4441    }
4442    for distinct in &table.distincts {
4443        match distinct {
4444            None => out.push(0),
4445            Some(count) => {
4446                out.push(1);
4447                put_u64(&mut out, *count);
4448            }
4449        }
4450    }
4451    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
4452    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
4453    for stripe in &table.stripes {
4454        put_u32(
4455            &mut out,
4456            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
4457        );
4458        for &rows in &stripe.parts {
4459            put_u32(&mut out, rows);
4460        }
4461        put_u64(&mut out, stripe.index.offset);
4462        put_u32(&mut out, stripe.index.length);
4463        for page in &stripe.pages {
4464            put_u64(&mut out, page.offset);
4465            put_u32(&mut out, page.length);
4466        }
4467        // A membership index says which of a dictionary's codes a part holds, so a column the writer
4468        // decided against giving a dictionary has nothing for it to be about and writes none. Every
4469        // file written before that decision existed has a dictionary on every varchar column, so
4470        // this reads those files byte for byte the way it always did.
4471        for ((field, dictionary), membership) in
4472            table.fields.iter().zip(&table.dictionaries).zip(&stripe.memberships)
4473        {
4474            if field.ty != LogicalType::Varchar || dictionary.is_none() {
4475                continue;
4476            }
4477            let page =
4478                membership.ok_or_else(|| invalid("string page has no code membership index"))?;
4479            put_u64(&mut out, page.offset);
4480            put_u32(&mut out, page.length);
4481            put_u64(&mut out, page.hash);
4482        }
4483        for sieve in &stripe.sieves {
4484            match sieve {
4485                None => out.push(0),
4486                Some(page) => {
4487                    out.push(1);
4488                    put_u64(&mut out, page.offset);
4489                    put_u32(&mut out, page.length);
4490                    put_u64(&mut out, page.hash);
4491                }
4492            }
4493        }
4494        for held in &stripe.part_ranges {
4495            match held {
4496                None => out.push(0),
4497                Some(page) => {
4498                    out.push(1);
4499                    put_u64(&mut out, page.offset);
4500                    put_u32(&mut out, page.length);
4501                    put_u64(&mut out, page.hash);
4502                }
4503            }
4504        }
4505        for range in stripe.zone.columns() {
4506            put_bound(&mut out, range.low.as_ref())?;
4507            put_bound(&mut out, range.high.as_ref())?;
4508            put_u32(
4509                &mut out,
4510                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
4511            );
4512            out.push(u8::from(range.exact));
4513            match range.sum {
4514                None => out.push(0),
4515                Some(total) => {
4516                    out.push(1);
4517                    out.extend_from_slice(&total.to_le_bytes());
4518                }
4519            }
4520        }
4521    }
4522    out.extend_from_slice(FREQUENCIES);
4523    put_u16(
4524        &mut out,
4525        u16::try_from(table.frequencies.len())
4526            .map_err(|_| invalid("too many frequency columns"))?,
4527    );
4528    for summary in &table.frequencies {
4529        let Some(summary) = summary else {
4530            out.push(0);
4531            continue;
4532        };
4533        out.push(1);
4534        put_u64(&mut out, summary.omitted_max);
4535        put_u32(
4536            &mut out,
4537            u32::try_from(summary.entries.len())
4538                .map_err(|_| invalid("too many frequency entries"))?,
4539        );
4540        for entry in &summary.entries {
4541            match entry.value {
4542                FrequencyValue::Null => out.push(0),
4543                FrequencyValue::Integer(value) => {
4544                    out.push(1);
4545                    out.extend_from_slice(&value.to_le_bytes());
4546                }
4547                FrequencyValue::Code(value) => {
4548                    out.push(2);
4549                    put_u32(&mut out, value);
4550                }
4551            }
4552            put_u64(&mut out, entry.count);
4553        }
4554        put_u32(
4555            &mut out,
4556            u32::try_from(summary.ordinals.len())
4557                .map_err(|_| invalid("too many frequency ordinals"))?,
4558        );
4559        let mut previous = 0_u64;
4560        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
4561            let delta = if at == 0 {
4562                ordinal
4563            } else {
4564                ordinal
4565                    .checked_sub(previous)
4566                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
4567            };
4568            if at != 0 && delta == 0 {
4569                return Err(invalid("frequency ordinals are not unique"));
4570            }
4571            put_var_u64(&mut out, delta);
4572            previous = ordinal;
4573        }
4574    }
4575    // Written only when there is a declaration, so that the common file is the same bytes it was
4576    // and the section is not a byte of zero on every table in the world that never asked for one.
4577    if let Some(clustering) = &table.clustering {
4578        out.extend_from_slice(CLUSTERING);
4579        out.push(clustering.width().tag());
4580        put_u16(
4581            &mut out,
4582            u16::try_from(clustering.columns().len())
4583                .map_err(|_| invalid("too many clustering columns"))?,
4584        );
4585        for &column in clustering.columns() {
4586            put_u16(
4587                &mut out,
4588                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
4589            );
4590        }
4591    }
4592    // The section table, last, behind its own magic, for the same reason the frequency block is
4593    // behind its own: a reader that stops before it gets a table with no sections, and a table with
4594    // no sections is a correct table. The one difference from the blocks before it is that this one
4595    // is written even when it is empty, so that a file written by this build always says which
4596    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
4597    out.extend_from_slice(SECTIONS);
4598    put_u64(&mut out, table.generation);
4599    put_u16(
4600        &mut out,
4601        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
4602    );
4603    for held in &table.sections {
4604        held.encode(&mut out)?;
4605    }
4606    Ok(out)
4607}
4608
4609/// The small level of the directory, naming every table in the file.
4610///
4611/// This is what a footer slot points at. Each entry carries its own checksum over its table
4612/// directory, so a table whose directory is torn is found when that table is first touched rather
4613/// than being trusted because the catalog around it checksummed.
4614///
4615/// The views go after the tables and are whole here, since a view is text and a column list and has
4616/// no pages for a second level to point at.
4617fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
4618    let mut out = CATALOG.to_vec();
4619    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
4620    for entry in entries {
4621        let name = entry.name.as_bytes();
4622        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4623        out.extend_from_slice(name);
4624        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
4625        put_u16(
4626            &mut out,
4627            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
4628        );
4629        for field in &entry.fields {
4630            let name = field.name.as_bytes();
4631            put_u16(
4632                &mut out,
4633                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
4634            );
4635            out.extend_from_slice(name);
4636            put_type(&mut out, &field.ty)?;
4637            out.push(u8::from(field.not_null));
4638        }
4639        put_u64(&mut out, entry.directory.offset);
4640        put_u32(&mut out, entry.directory.length);
4641        put_u64(&mut out, entry.directory.hash);
4642    }
4643    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
4644    for view in views {
4645        let name = view.name.as_bytes();
4646        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
4647        out.extend_from_slice(name);
4648        put_long_text(&mut out, &view.sql, "view body")?;
4649        put_long_text(&mut out, &view.statement, "view statement")?;
4650        put_u16(
4651            &mut out,
4652            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
4653        );
4654        for alias in &view.aliases {
4655            let alias = alias.as_bytes();
4656            put_u16(
4657                &mut out,
4658                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
4659            );
4660            out.extend_from_slice(alias);
4661        }
4662        put_u16(
4663            &mut out,
4664            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
4665        );
4666        for field in &view.columns {
4667            let name = field.name.as_bytes();
4668            put_u16(
4669                &mut out,
4670                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
4671            );
4672            out.extend_from_slice(name);
4673            put_type(&mut out, &field.ty)?;
4674            out.push(u8::from(field.not_null));
4675        }
4676    }
4677    Ok(out)
4678}
4679
4680/// A length and that many bytes, for text that is allowed to be longer than a name.
4681fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
4682    let bytes = text.as_bytes();
4683    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
4684    out.extend_from_slice(bytes);
4685    Ok(())
4686}
4687
4688/// Reads the catalog directory back, checking every span against the file before anything is
4689/// allocated for it.
4690fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
4691    let mut cur = Cursor { bytes, at: 0 };
4692    if cur.take(8)? != CATALOG {
4693        return Err(invalid("catalog magic differs"));
4694    }
4695    let count = cur.u32()? as usize;
4696    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
4697    for _ in 0..count {
4698        let name = cur.text()?;
4699        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4700        let width = cur.u16()? as usize;
4701        let mut fields = Vec::with_capacity(width);
4702        for _ in 0..width {
4703            let name = cur.text()?;
4704            let ty = read_type(&mut cur)?;
4705            let not_null = match cur.u8()? {
4706                0 => false,
4707                1 => true,
4708                _ => return Err(invalid("nullability flag differs")),
4709            };
4710            fields.push(Field { name, ty, not_null });
4711        }
4712        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4713        let end = directory
4714            .offset
4715            .checked_add(u64::from(directory.length))
4716            .ok_or_else(|| invalid("table directory offset overflow"))?;
4717        if directory.offset < HEADER
4718            || end > size
4719            || directory.length as usize > MAX_DIRECTORY
4720            || directory.length == 0
4721        {
4722            return Err(invalid("table directory range is outside the file"));
4723        }
4724        if entries.iter().any(|held| held.name == name) {
4725            return Err(invalid("two tables in the catalog have the same name"));
4726        }
4727        entries.push(Entry { name, fields, rows, directory });
4728    }
4729    // A catalog that ends where the tables end is a catalog with no views in it, which is every
4730    // file written before format 25. That is why the count is allowed to be missing rather than
4731    // read as a zero that has to be there: an older file has nothing after the last table entry at
4732    // all, and [`READABLE`] says those files still open.
4733    let count = if cur.done() { 0 } else { cur.u32()? as usize };
4734    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
4735    for _ in 0..count {
4736        let name = cur.text()?;
4737        let sql = cur.long_text()?;
4738        let statement = cur.long_text()?;
4739        let width = cur.u16()? as usize;
4740        let mut aliases = Vec::with_capacity(width);
4741        for _ in 0..width {
4742            aliases.push(cur.text()?);
4743        }
4744        let width = cur.u16()? as usize;
4745        let mut columns = Vec::with_capacity(width);
4746        for _ in 0..width {
4747            let name = cur.text()?;
4748            let ty = read_type(&mut cur)?;
4749            let not_null = match cur.u8()? {
4750                0 => false,
4751                1 => true,
4752                _ => return Err(invalid("nullability flag differs")),
4753            };
4754            columns.push(Field { name, ty, not_null });
4755        }
4756        // The same rule the tables above get, and for the same reason. Two entries under one name
4757        // is a catalog nothing can answer a lookup from, and finding that out here is better than
4758        // finding it out from whichever of the two a search happened to reach first.
4759        if views.iter().any(|held| held.name == name) {
4760            return Err(invalid("two views in the catalog have the same name"));
4761        }
4762        if entries.iter().any(|held| held.name == name) {
4763            return Err(invalid("a table and a view in the catalog have the same name"));
4764        }
4765        views.push(ViewEntry { name, sql, statement, aliases, columns });
4766    }
4767    Ok((entries, views))
4768}
4769
4770struct Cursor<'a> {
4771    bytes: &'a [u8],
4772    at: usize,
4773}
4774impl<'a> Cursor<'a> {
4775    fn take(&mut self, len: usize) -> Result<&'a [u8]> {
4776        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
4777        let bytes =
4778            self.bytes.get(self.at..end).ok_or_else(|| invalid("directory is truncated"))?;
4779        self.at = end;
4780        Ok(bytes)
4781    }
4782    fn u8(&mut self) -> Result<u8> {
4783        Ok(self.take(1)?[0])
4784    }
4785    fn u16(&mut self) -> Result<u16> {
4786        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
4787    }
4788    fn u32(&mut self) -> Result<u32> {
4789        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
4790    }
4791    fn u64(&mut self) -> Result<u64> {
4792        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
4793    }
4794    fn var_u64(&mut self) -> Result<u64> {
4795        let mut value = 0_u64;
4796        for shift in (0..=63).step_by(7) {
4797            let byte = self.u8()?;
4798            let part = u64::from(byte & 0x7f);
4799            if shift == 63 && part > 1 {
4800                return Err(invalid("frequency ordinal varint overflows"));
4801            }
4802            value |= part << shift;
4803            if byte & 0x80 == 0 {
4804                return Ok(value);
4805            }
4806        }
4807        Err(invalid("frequency ordinal varint is too long"))
4808    }
4809    /// A zone map's end, in the layout `rudb_common::bounds` defines.
4810    ///
4811    /// The bytes are the ones this directory has written since format 10 and the codec moved to
4812    /// rank zero rather than being copied, because a column summary now writes the same two ends
4813    /// and two encodings of one type is how the two quietly stop agreeing.
4814    fn bound(&mut self) -> Result<Option<Bound>> {
4815        bounds::get(self.bytes, &mut self.at)
4816    }
4817    fn text(&mut self) -> Result<String> {
4818        let len = self.u16()? as usize;
4819        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
4820    }
4821    /// Whether everything has been read, which is how a section that an older file does not have at
4822    /// all is told from one that is there and empty.
4823    fn done(&self) -> bool {
4824        self.at >= self.bytes.len()
4825    }
4826    /// The same, for text that is a query rather than a name.
4827    ///
4828    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
4829    /// kilobyte identifier by accident and people do write generated queries that long, and a view
4830    /// that could not be written down because its body was too big would be a limit invented here
4831    /// rather than one anything else in the engine has.
4832    fn long_text(&mut self) -> Result<String> {
4833        let len = self.u32()? as usize;
4834        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
4835    }
4836}
4837
4838fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
4839    let mut cur = Cursor { bytes, at: 0 };
4840    if cur.take(8)? != DIRECTORY {
4841        return Err(invalid("directory magic differs"));
4842    }
4843    let name = cur.text()?;
4844    let width = cur.u16()? as usize;
4845    let mut fields = Vec::with_capacity(width);
4846    for _ in 0..width {
4847        let name = cur.text()?;
4848        let ty = read_type(&mut cur)?;
4849        let not_null = match cur.u8()? {
4850            0 => false,
4851            1 => true,
4852            _ => return Err(invalid("nullability flag differs")),
4853        };
4854        fields.push(Field { name, ty, not_null });
4855    }
4856    let mut dictionaries = Vec::with_capacity(width);
4857    for _ in 0..width {
4858        dictionaries.push(match cur.u8()? {
4859            0 => None,
4860            1 => {
4861                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4862                let end = page
4863                    .offset
4864                    .checked_add(u64::from(page.length))
4865                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
4866                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
4867                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
4868                // pages are capped there. `Writer::finish` has already bounded this length by the
4869                // on-disk `u32`, and the range check below keeps it inside the file.
4870                if page.offset < HEADER || end > size {
4871                    return Err(invalid("dictionary page range is outside the file"));
4872                }
4873                Some(page)
4874            }
4875            _ => return Err(invalid("dictionary page tag differs")),
4876        });
4877    }
4878    let mut distincts = Vec::with_capacity(width);
4879    for _ in 0..width {
4880        distincts.push(match cur.u8()? {
4881            0 => None,
4882            1 => Some(cur.u64()?),
4883            _ => return Err(invalid("distinct count tag differs")),
4884        });
4885    }
4886    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4887    let count = cur.u32()? as usize;
4888    let mut stripes = Vec::with_capacity(count);
4889    let mut total = 0_usize;
4890    for _ in 0..count {
4891        let count = cur.u32()? as usize;
4892        if count == 0 || count > STRIPE_PARTS {
4893            return Err(invalid("stripe part count is outside its bound"));
4894        }
4895        let mut parts = Vec::with_capacity(count);
4896        let mut stripe_rows = 0_usize;
4897        for _ in 0..count {
4898            let rows = cur.u32()?;
4899            if rows == 0 {
4900                return Err(invalid("empty part"));
4901            }
4902            parts.push(rows);
4903            stripe_rows = stripe_rows
4904                .checked_add(rows as usize)
4905                .ok_or_else(|| invalid("stripe row count overflow"))?;
4906        }
4907        total =
4908            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
4909        let index = Span { offset: cur.u64()?, length: cur.u32()? };
4910        let section = index_section(count)?;
4911        let wanted = section
4912            .checked_mul(width)
4913            .and_then(|bytes| u32::try_from(bytes).ok())
4914            .ok_or_else(|| invalid("index page length overflow"))?;
4915        let end = index
4916            .offset
4917            .checked_add(u64::from(index.length))
4918            .ok_or_else(|| invalid("index page offset overflow"))?;
4919        if index.offset < HEADER || end > size || index.length != wanted {
4920            return Err(invalid("index page range is outside the file"));
4921        }
4922        let mut pages = Vec::with_capacity(width);
4923        for _ in 0..width {
4924            let offset = cur.u64()?;
4925            let length = cur.u32()?;
4926            let end = offset
4927                .checked_add(u64::from(length))
4928                .ok_or_else(|| invalid("page offset overflow"))?;
4929            if offset < HEADER || end > size || length as usize > MAX_PAGE {
4930                return Err(invalid("page range is outside the file"));
4931            }
4932            pages.push(Span { offset, length });
4933        }
4934        let mut memberships = vec![None; width];
4935        for (column, field) in fields.iter().enumerate() {
4936            if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
4937                continue;
4938            }
4939            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4940            let end = page
4941                .offset
4942                .checked_add(u64::from(page.length))
4943                .ok_or_else(|| invalid("membership page offset overflow"))?;
4944            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4945                return Err(invalid("membership page range is outside the file"));
4946            }
4947            memberships[column] = Some(page);
4948        }
4949        let mut sieves = vec![None; width];
4950        for sieve in sieves.iter_mut().take(width) {
4951            match cur.u8()? {
4952                0 => continue,
4953                1 => {}
4954                _ => return Err(invalid("a sieve page has an unknown tag")),
4955            }
4956            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4957            let end = page
4958                .offset
4959                .checked_add(u64::from(page.length))
4960                .ok_or_else(|| invalid("sieve page offset overflow"))?;
4961            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4962                return Err(invalid("sieve page range is outside the file"));
4963            }
4964            *sieve = Some(page);
4965        }
4966        let mut part_ranges = vec![None; width];
4967        for held in part_ranges.iter_mut().take(width) {
4968            match cur.u8()? {
4969                0 => continue,
4970                1 => {}
4971                _ => return Err(invalid("a part range page has an unknown tag")),
4972            }
4973            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4974            let end = page
4975                .offset
4976                .checked_add(u64::from(page.length))
4977                .ok_or_else(|| invalid("part range page offset overflow"))?;
4978            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4979                return Err(invalid("part range page range is outside the file"));
4980            }
4981            *held = Some(page);
4982        }
4983        let mut ranges = Vec::with_capacity(width);
4984        for column in 0..width {
4985            let low = cur.bound()?;
4986            let high = cur.bound()?;
4987            let nulls = cur.u32()? as usize;
4988            if nulls > stripe_rows {
4989                return Err(invalid("null count exceeds stripe rows"));
4990            }
4991            let exact = cur.u8()? != 0;
4992            let sum = match cur.u8()? {
4993                0 => None,
4994                1 => Some(i128::from_le_bytes(
4995                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
4996                )),
4997                _ => return Err(invalid("a stripe sum has an unknown tag")),
4998            };
4999            // Files written before the ends of a decimal or a timestamp column carried their power
5000            // of ten hold a bare integer here, and that integer is the one the column holds, which
5001            // is what the power is over. So the type puts it back on the way in and an old file
5002            // prunes as well as a new one. A file that already wrote the power keeps it, because
5003            // this leaves anything that is not an integer alone.
5004            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
5005            let low = low.map(|bound| scaled_as(bound, ty));
5006            let high = high.map(|bound| scaled_as(bound, ty));
5007            ranges.push(Range { low, high, nulls, exact, sum });
5008        }
5009        stripes.push(Stripe {
5010            rows: stripe_rows,
5011            parts,
5012            index,
5013            pages,
5014            memberships,
5015            sieves,
5016            part_ranges,
5017            zone: Zone::from_ranges(ranges),
5018        });
5019    }
5020    if total != rows {
5021        return Err(invalid("table row count differs from stripes"));
5022    }
5023    let frequencies = if cur.at == bytes.len() {
5024        vec![None; width]
5025    } else {
5026        if cur.take(8)? != FREQUENCIES {
5027            return Err(invalid("directory extension magic differs"));
5028        }
5029        if cur.u16()? as usize != width {
5030            return Err(invalid("frequency column count differs"));
5031        }
5032        let mut frequencies = Vec::with_capacity(width);
5033        for field in &fields {
5034            let summary = match cur.u8()? {
5035                0 => None,
5036                1 => {
5037                    let omitted_max = cur.u64()?;
5038                    let count = cur.u32()? as usize;
5039                    if count > FREQUENCY_ENTRIES {
5040                        return Err(invalid("frequency entry count exceeds its bound"));
5041                    }
5042                    let mut entries = Vec::with_capacity(count);
5043                    // row at a time: directory decoding validates each persisted bounded frequency entry.
5044                    for _ in 0..count {
5045                        let value = match cur.u8()? {
5046                            0 => FrequencyValue::Null,
5047                            1 => FrequencyValue::Integer(i128::from_le_bytes(
5048                                cur.take(16)?.try_into().expect("sixteen bytes"),
5049                            )),
5050                            2 => FrequencyValue::Code(cur.u32()?),
5051                            _ => return Err(invalid("frequency value tag differs")),
5052                        };
5053                        let valid = matches!(
5054                            (&field.ty, value),
5055                            (_, FrequencyValue::Null)
5056                                | (LogicalType::Varchar, FrequencyValue::Code(_))
5057                                | (
5058                                    LogicalType::TinyInt
5059                                        | LogicalType::SmallInt
5060                                        | LogicalType::Integer
5061                                        | LogicalType::BigInt
5062                                        | LogicalType::UTinyInt
5063                                        | LogicalType::USmallInt
5064                                        | LogicalType::UInteger
5065                                        | LogicalType::UBigInt
5066                                        | LogicalType::Date
5067                                        | LogicalType::Timestamp,
5068                                    FrequencyValue::Integer(_),
5069                                )
5070                        );
5071                        if !valid {
5072                            return Err(invalid("frequency value does not match its column"));
5073                        }
5074                        let count = cur.u64()?;
5075                        if count == 0 || count > rows as u64 {
5076                            return Err(invalid("frequency count is outside the table"));
5077                        }
5078                        entries.push(FrequencyEntry { value, count });
5079                    }
5080                    if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
5081                        return Err(invalid("frequency entries are not descending"));
5082                    }
5083                    let ordinals = {
5084                        let ordinal_count = cur.u32()? as usize;
5085                        if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
5086                            return Err(invalid("frequency ordinal count exceeds its bound"));
5087                        }
5088                        let mut ordinals = Vec::with_capacity(ordinal_count);
5089                        let mut previous = 0_u64;
5090                        for at in 0..ordinal_count {
5091                            let delta = cur.var_u64()?;
5092                            if at != 0 && delta == 0 {
5093                                return Err(invalid("frequency ordinals are not increasing"));
5094                            }
5095                            let ordinal = if at == 0 {
5096                                delta
5097                            } else {
5098                                previous
5099                                    .checked_add(delta)
5100                                    .ok_or_else(|| invalid("frequency ordinal overflows"))?
5101                            };
5102                            if ordinal >= rows as u64 {
5103                                return Err(invalid("frequency ordinal is outside the table"));
5104                            }
5105                            ordinals.push(ordinal);
5106                            previous = ordinal;
5107                        }
5108                        ordinals
5109                    };
5110                    Some(FrequencySummary { entries, omitted_max, ordinals })
5111                }
5112                _ => return Err(invalid("frequency summary tag differs")),
5113            };
5114            frequencies.push(summary);
5115        }
5116        frequencies
5117    };
5118    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
5119    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
5120    // independently: a format 22 directory ends here and has neither, a directory written before
5121    // the section table has only the clustering declaration, and each one still opens without a
5122    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
5123    // a file that predates them and answers every query, only without the graph path.
5124    //
5125    // A repeated block is refused rather than allowed to win, because two clustering declarations
5126    // in one directory is a torn directory and the only question is which of them is the lie.
5127    let mut clustering = None;
5128    let mut sections = Vec::new();
5129    let mut seen_sections = false;
5130    // Zero until a section table says otherwise, which is what a format 22 table gets and what
5131    // makes every section stamp fail to match on one, because real generations start at one.
5132    let mut generation = 0;
5133    while cur.at != bytes.len() {
5134        let mut tag = [0u8; 8];
5135        tag.copy_from_slice(cur.take(8)?);
5136        if &tag == CLUSTERING {
5137            if clustering.is_some() {
5138                return Err(invalid("directory names two clustering declarations"));
5139            }
5140            let bucket = Width::from_tag(cur.u8()?)
5141                .ok_or_else(|| invalid("clustering width tag differs"))?;
5142            let count = cur.u16()? as usize;
5143            let mut columns = Vec::with_capacity(count.min(fields.len()));
5144            for _ in 0..count {
5145                columns.push(u32::from(cur.u16()?));
5146            }
5147            // Through the constructor and not built by hand, so that a file claiming a column the
5148            // table does not have is caught at open rather than at the first scan that trusted it.
5149            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
5150                invalid("stored clustering declaration does not match the table it is on")
5151            })?);
5152        } else if &tag == SECTIONS {
5153            if seen_sections {
5154                return Err(invalid("directory names two section tables"));
5155            }
5156            seen_sections = true;
5157            generation = cur.u64()?;
5158            let count = cur.u16()? as usize;
5159            if count > MAX_SECTIONS {
5160                return Err(invalid("section count exceeds its bound"));
5161            }
5162            sections = Vec::with_capacity(count);
5163            // entry at a time: a malformed section entry is refused rather than turned into an
5164            // offset.
5165            for _ in 0..count {
5166                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
5167            }
5168            for held in &sections {
5169                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
5170                    return Err(invalid("a section's extent table overflows the file"));
5171                };
5172                // The bound check is here and not in `section`, because only the caller knows how
5173                // big the file is. A section pointing past the end is a torn directory, and reading
5174                // the payload it names would be reading whatever else is at that offset.
5175                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
5176                    return Err(invalid("a section's extent table is outside the file"));
5177                }
5178                if held.extents == 0 && held.extent_bytes != 0 {
5179                    return Err(invalid("a section with no extents names an extent table"));
5180                }
5181            }
5182        } else {
5183            return Err(invalid("directory extension magic differs"));
5184        }
5185    }
5186    if cur.at != bytes.len() {
5187        return Err(invalid("directory has trailing bytes"));
5188    }
5189    Ok(Table {
5190        name,
5191        fields,
5192        stripes,
5193        rows,
5194        dictionaries,
5195        distincts,
5196        frequencies,
5197        clustering,
5198        generation,
5199        sections,
5200    })
5201}
5202
5203/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
5204fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
5205    bounds::put(out, bound)
5206}
5207
5208/// Which cascades are worth trying on a run of dictionary codes.
5209///
5210/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
5211/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
5212/// three candidates were always going to win. It is the right default for a crate that does not
5213/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
5214/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
5215/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
5216///
5217/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
5218/// already the dictionary, and it is also the most expensive one to try. Below the top level the
5219/// streams are an RLE's run values and run lengths, which are integers in their own right with no
5220/// runs left in them, so only the two flat candidates go down there.
5221///
5222/// This is size given up for time on purpose, and the ablation is this chooser against
5223/// [`chooser::EXHAUSTIVE`] on the same file.
5224#[derive(Debug)]
5225struct Codes;
5226
5227impl chooser::Chooser for Codes {
5228    fn name(&self) -> &'static str {
5229        "codes"
5230    }
5231
5232    fn narrow_strings(
5233        &self,
5234        _values: &[&[u8]],
5235        offered: &[string::Kind],
5236        _depth: u8,
5237    ) -> Vec<string::Kind> {
5238        // Never reached, because nothing here encodes strings through the cascade. The trait asks
5239        // for it and the honest answer to a question we have no opinion on is the whole list.
5240        offered.to_vec()
5241    }
5242
5243    fn narrow_integers(
5244        &self,
5245        _values: &[i64],
5246        offered: &[integer::Kind],
5247        depth: u8,
5248    ) -> Vec<integer::Kind> {
5249        let keep: &[integer::Kind] = if depth == 0 {
5250            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
5251        } else {
5252            &[integer::Kind::Constant, integer::Kind::Packed]
5253        };
5254        let narrowed: Vec<integer::Kind> =
5255            offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5256        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
5257        // this has no opinion about rather than one that cannot be written.
5258        if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5259    }
5260}
5261
5262/// Which cascades are worth trying on a part of plain integers.
5263///
5264/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
5265/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
5266/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
5267/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
5268/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
5269/// every value. A column that is one value with a handful of exceptions is sparse. What is still
5270/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
5271/// expensive candidate to try and this file already puts the columns that want one through a
5272/// dictionary of their own before they ever reach here.
5273#[derive(Debug)]
5274struct Fixed;
5275
5276impl chooser::Chooser for Fixed {
5277    fn name(&self) -> &'static str {
5278        "fixed"
5279    }
5280
5281    fn narrow_strings(
5282        &self,
5283        _values: &[&[u8]],
5284        offered: &[string::Kind],
5285        _depth: u8,
5286    ) -> Vec<string::Kind> {
5287        offered.to_vec()
5288    }
5289
5290    fn narrow_integers(
5291        &self,
5292        _values: &[i64],
5293        offered: &[integer::Kind],
5294        depth: u8,
5295    ) -> Vec<integer::Kind> {
5296        let keep: &[integer::Kind] = if depth == 0 {
5297            &[
5298                integer::Kind::Constant,
5299                integer::Kind::Packed,
5300                integer::Kind::Delta,
5301                integer::Kind::Rle,
5302                integer::Kind::Sparse,
5303                integer::Kind::Strided,
5304            ]
5305        } else {
5306            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
5307        };
5308        let narrowed: Vec<integer::Kind> =
5309            offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5310        if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5311    }
5312}
5313
5314/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
5315/// losing one.
5316///
5317/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
5318/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
5319/// integers and have their own ways of being small.
5320fn widened(data: &Data) -> Option<Vec<i64>> {
5321    match data {
5322        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5323        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5324        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5325        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5326        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5327        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5328        Data::Int64(values) => Some(values.to_vec()),
5329        _ => None,
5330    }
5331}
5332
5333/// An integer type a cascaded page can be read back into, and how to tell whether a value fits.
5334///
5335/// This exists so that the check and the conversion can be two loops instead of one. `TryFrom` puts
5336/// them together, which is the right shape for one value and the wrong one for a page: a fallible
5337/// conversion a value at a time is a branch a value at a time, the branch decides whether the loop
5338/// keeps going, and a loop like that is one no compiler will widen.
5339trait Narrow: Copy {
5340    /// How wide this type is, and what to add to a value to put its range at the bottom of a `u64`.
5341    ///
5342    /// Half the width for a signed type, which is what moves its smallest value to zero, and nothing
5343    /// for an unsigned one, whose smallest value is already there.
5344    const BIASED: (u32, u64);
5345
5346    /// The value narrowed, which the caller has already shown fits.
5347    fn narrow(value: i64) -> Self;
5348}
5349
5350/// The bits of `value` a `T` cannot hold, and zero when the value fits.
5351///
5352/// The question is asked this way round because the answers or together. A page fits when every
5353/// residue in it is zero, so the loop is an or into an accumulator and the decision is one test
5354/// after it, where asking whether each value is between a floor and a ceiling gives an answer that
5355/// does not combine and turns into a running minimum and maximum.
5356///
5357/// Biasing and shifting is what the answer is made of, rather than anything that reads more like the
5358/// question, because those are the operations a machine has four of. A 64 bit integer minimum is
5359/// AVX-512. So is a 64 bit arithmetic shift right, which is how the sign extension this could be
5360/// written as would have to be done. An add and a logical shift right are AVX2 and are on every
5361/// machine this runs on, so this is the form that gets four values a cycle instead of one.
5362///
5363/// Adding the bias moves the type's range to `0..=2^bits`, wrapping, so everything in range shifts
5364/// away to nothing and everything outside it leaves something behind. A negative value under an
5365/// unsigned type is caught by the same shift, because a negative `i64` read as a `u64` is enormous.
5366#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
5367fn residue<T: Narrow>(value: i64) -> u64 {
5368    let (bits, bias) = T::BIASED;
5369    (value as u64).wrapping_add(bias) >> bits
5370}
5371
5372/// Says a primitive integer narrows with `as`, and where the bottom of its range is.
5373///
5374/// `as` is a truncation and is the right operation here only because [`fit`] has already found every
5375/// residue zero, and it is what makes the second loop a narrowing store with no branch in it.
5376macro_rules! narrows {
5377    ($($ty:ty => $bias:expr),* $(,)?) => {$(
5378        impl Narrow for $ty {
5379            const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
5380
5381            #[allow(
5382                clippy::cast_possible_truncation,
5383                clippy::cast_sign_loss,
5384                reason = "the caller has checked the bits this truncates away"
5385            )]
5386            fn narrow(value: i64) -> Self {
5387                value as Self
5388            }
5389        }
5390    )*};
5391}
5392
5393narrows! {
5394    i8 => 1 << 7,
5395    u8 => 0,
5396    i16 => 1 << 15,
5397    u16 => 0,
5398    i32 => 1 << 31,
5399    u32 => 0,
5400}
5401
5402/// Narrows a page's values, refusing the page if any of them does not fit.
5403///
5404/// The check first and the conversion second, rather than a fallible conversion a value at a time.
5405/// Both loops here are ones a compiler widens: [`residue`] is three instructions a lane and a
5406/// narrowing store is one. The version before this was a `TryFrom` and a `collect` into a `Result`,
5407/// which is a compare, a branch and a short circuit a value at a time, and on ClickBench 39 it was
5408/// seven percent of the query. The version after that kept a running minimum and maximum, which is
5409/// the obvious way to ask and needs a 64 bit integer minimum that AVX2 does not have, so it stayed
5410/// a value at a time and was still ten percent of the same query.
5411///
5412/// An empty page has nothing to refuse, which falls out of the accumulator starting at zero rather
5413/// than needing a case of its own.
5414fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
5415    let mut spilled = 0u64;
5416    for value in values {
5417        spilled |= residue::<T>(*value);
5418    }
5419    if spilled != 0 {
5420        return Err(invalid("page value is not of its type"));
5421    }
5422    Ok(values.iter().map(|value| T::narrow(*value)).collect())
5423}
5424
5425/// The same values back in the width the column is declared at.
5426///
5427/// A value that does not fit is a page that disagrees with the directory about what the column is,
5428/// which is a damaged file rather than a caller error, so it is refused rather than truncated.
5429fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
5430    Ok(match ty {
5431        LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
5432        LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
5433        LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
5434        LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
5435        LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
5436        LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
5437        LogicalType::BigInt
5438        | LogicalType::Timestamp
5439        | LogicalType::Time
5440        | LogicalType::TimeTz
5441        | LogicalType::TimestampTz
5442        | LogicalType::TimestampS
5443        | LogicalType::TimestampMs
5444        | LogicalType::TimestampNs => Data::Int64(values.into()),
5445        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
5446        // integer the declared width says the column is stored as.
5447        LogicalType::Decimal { .. } => match ty.physical() {
5448            PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
5449            PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
5450            PhysicalType::Int64 => Data::Int64(values.into()),
5451            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
5452        },
5453        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
5454    })
5455}
5456
5457/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
5458/// beat before it is worth the decode.
5459fn plain_width(ty: &LogicalType) -> Option<usize> {
5460    Some(match ty {
5461        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
5462        LogicalType::SmallInt | LogicalType::USmallInt => 2,
5463        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
5464        LogicalType::BigInt
5465        | LogicalType::Timestamp
5466        | LogicalType::Time
5467        | LogicalType::TimeTz
5468        | LogicalType::TimestampTz
5469        | LogicalType::TimestampS
5470        | LogicalType::TimestampMs
5471        | LogicalType::TimestampNs => 8,
5472        LogicalType::Decimal { .. } => match ty.physical() {
5473            PhysicalType::Int16 => 2,
5474            PhysicalType::Int32 => 4,
5475            PhysicalType::Int64 => 8,
5476            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
5477            // they take the plain path and there is nothing here to compare against.
5478            _ => return None,
5479        },
5480        _ => return None,
5481    })
5482}
5483
5484/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
5485///
5486/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
5487/// where there is one and the plain width where there is not. Both are cheaper to decode than a
5488/// cascade, so a tie goes to them.
5489fn cascaded(
5490    flat: &Vector,
5491    ty: &LogicalType,
5492    packed: Option<&Packed<'_>>,
5493) -> Result<Option<Vec<u8>>> {
5494    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
5495    let Some(values) = widened(data) else { return Ok(None) };
5496    let plain = values.len().saturating_mul(width);
5497    let best = match packed {
5498        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
5499        Some(packed) => plain.min(21 + size_of_val(packed.words())),
5500        None => plain,
5501    };
5502    let out = integer::encode_with(&values, &Fixed)?;
5503    Ok((out.len() < best).then_some(out))
5504}
5505
5506/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
5507///
5508/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
5509/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
5510/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
5511/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
5512/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
5513///
5514/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
5515/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
5516/// values, and there is no reason to pay for the decode when it does.
5517/// A varchar page as one FSST layer, or `None` when it did not pay.
5518///
5519/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
5520/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
5521/// a page of values with nothing in common and the wrong one for a page of English, and a column of
5522/// comments is the case this exists for.
5523///
5524/// One layer and not the full string cascade, which is what the payload blocks of a global
5525/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
5526/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
5527/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
5528/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
5529/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
5530/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
5531/// what the page has to be put back together from.
5532///
5533/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
5534/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
5535/// already lays them out, and what the reader hands a chunk is views over that buffer.
5536///
5537/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
5538/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
5539/// page that was being written raw.
5540///
5541/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
5542/// nothing at read time for having been offered.
5543fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
5544    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
5545    let mut payload = 0_usize;
5546    for row in 0..flat.len() {
5547        let text = flat.text_at(row).unwrap_or("").as_bytes();
5548        payload = payload.saturating_add(text.len());
5549        values.push(text);
5550    }
5551    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
5552    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
5553    let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
5554        return Ok(None);
5555    };
5556    Ok((out.len() < plain).then_some(out))
5557}
5558
5559fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
5560    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
5561    let coded = integer::encode_with(&wide, &Codes)?;
5562    let plain = codes.len().saturating_mul(size_of::<u32>());
5563    Ok((coded.len() < plain).then_some(coded))
5564}
5565
5566fn encode(
5567    vector: &Vector,
5568    global: Option<&mut GlobalDictionary>,
5569) -> Result<(Vec<u8>, Option<Vec<u32>>)> {
5570    let ty = vector.logical_type();
5571    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
5572    let flat = vector.flatten()?;
5573    let mut out = Vec::new();
5574    let mut global_codes = None;
5575    if let Some(global) = global {
5576        let mut codes = Vec::with_capacity(flat.len());
5577        for row in 0..flat.len() {
5578            let text = flat.text_at(row).unwrap_or("");
5579            let code = global.code(text)?;
5580            global.observe(code, flat.is_null_at(row))?;
5581            codes.push(code);
5582        }
5583        global_codes = Some(codes);
5584    }
5585    let membership = global_codes.as_deref().map(unique_codes);
5586    let dictionary = if global_codes.is_none() && ty == &LogicalType::Varchar {
5587        string_dictionary(&flat)?
5588    } else {
5589        None
5590    };
5591    let compressed_text =
5592        if global_codes.is_none() && dictionary.is_none() && ty == &LogicalType::Varchar {
5593            text_compressed(&flat)?
5594        } else {
5595            None
5596        };
5597    let packed_vector = if dictionary.is_none() && global_codes.is_none() {
5598        Some(flat.bit_packed()?)
5599    } else {
5600        None
5601    };
5602    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
5603    let coded = match global_codes.as_deref() {
5604        Some(codes) => encoded_codes(codes)?,
5605        None => None,
5606    };
5607    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
5608    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
5609    // when it halves it, so a column that shrinks by a third was coming out whole.
5610    let cascade = if dictionary.is_none() && global_codes.is_none() {
5611        cascaded(&flat, ty, packed.as_ref())?
5612    } else {
5613        None
5614    };
5615    out.push(if coded.is_some() {
5616        4
5617    } else if cascade.is_some() {
5618        5
5619    } else if global_codes.is_some() {
5620        3
5621    } else if dictionary.is_some() {
5622        1
5623    } else if compressed_text.is_some() {
5624        6
5625    } else if packed.is_some() {
5626        2
5627    } else {
5628        0
5629    });
5630    let nulls = flat.validity();
5631    let flag = match nulls {
5632        Validity::AllValid => 0,
5633        Validity::AllInvalid => 1,
5634        Validity::Mask(_) => 2,
5635    };
5636    out.push(flag);
5637    if flag == 2 {
5638        for group in (0..vector.len()).step_by(8) {
5639            let mut bits = 0_u8;
5640            for bit in 0..8 {
5641                if group + bit < vector.len() && !flat.is_null_at(group + bit) {
5642                    bits |= 1 << bit;
5643                }
5644            }
5645            out.push(bits);
5646        }
5647    }
5648    if let Some(coded) = coded {
5649        out.extend_from_slice(&coded);
5650        return Ok((out, membership));
5651    }
5652    if let Some(cascade) = cascade {
5653        out.extend_from_slice(&cascade);
5654        return Ok((out, membership));
5655    }
5656    if let Some(codes) = global_codes {
5657        for code in codes {
5658            put_u32(&mut out, code);
5659        }
5660        return Ok((out, membership));
5661    }
5662    if let Some(dictionary) = dictionary {
5663        out.extend_from_slice(&dictionary);
5664        return Ok((out, membership));
5665    }
5666    if let Some(compressed_text) = compressed_text {
5667        out.extend_from_slice(&compressed_text);
5668        return Ok((out, membership));
5669    }
5670    if let Some(packed) = packed {
5671        if packed.offset() != 0 {
5672            return Err(invalid("writer received a sliced packed vector"));
5673        }
5674        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
5675        out.extend_from_slice(&packed.base().to_le_bytes());
5676        put_u32(
5677            &mut out,
5678            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
5679        );
5680        for word in packed.words() {
5681            put_u64(&mut out, *word);
5682        }
5683        return Ok((out, membership));
5684    }
5685    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
5686    match (ty, data) {
5687        (LogicalType::TinyInt, Data::Int8(values)) => {
5688            for value in &**values {
5689                out.extend_from_slice(&value.to_le_bytes());
5690            }
5691        }
5692        (LogicalType::UTinyInt, Data::UInt8(values)) => {
5693            for value in &**values {
5694                out.extend_from_slice(&value.to_le_bytes());
5695            }
5696        }
5697        (LogicalType::SmallInt, Data::Int16(values)) => {
5698            for value in &**values {
5699                out.extend_from_slice(&value.to_le_bytes());
5700            }
5701        }
5702        (LogicalType::USmallInt, Data::UInt16(values)) => {
5703            for value in &**values {
5704                out.extend_from_slice(&value.to_le_bytes());
5705            }
5706        }
5707        (LogicalType::UInteger, Data::UInt32(values)) => {
5708            for value in &**values {
5709                out.extend_from_slice(&value.to_le_bytes());
5710            }
5711        }
5712        (LogicalType::UBigInt, Data::UInt64(values)) => {
5713            for value in &**values {
5714                out.extend_from_slice(&value.to_le_bytes());
5715            }
5716        }
5717        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
5718            for value in &**values {
5719                out.extend_from_slice(&value.to_le_bytes());
5720            }
5721        }
5722        (
5723            LogicalType::BigInt
5724            | LogicalType::Timestamp
5725            | LogicalType::Time
5726            | LogicalType::TimeTz
5727            | LogicalType::TimestampTz
5728            | LogicalType::TimestampS
5729            | LogicalType::TimestampMs
5730            | LogicalType::TimestampNs,
5731            Data::Int64(values),
5732        ) => {
5733            for value in &**values {
5734                out.extend_from_slice(&value.to_le_bytes());
5735            }
5736        }
5737        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
5738        // the engine already carries it in, so nothing about the value changes on the way down.
5739        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
5740            for value in &**values {
5741                out.extend_from_slice(&value.to_le_bytes());
5742            }
5743        }
5744        (LogicalType::UHugeInt, Data::UInt128(values)) => {
5745            for value in &**values {
5746                out.extend_from_slice(&value.to_le_bytes());
5747            }
5748        }
5749        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
5750        // float codecs is worth having before somebody has measured a corpus of them.
5751        (LogicalType::Float, Data::Float32(values)) => {
5752            for value in &**values {
5753                out.extend_from_slice(&value.to_le_bytes());
5754            }
5755        }
5756        (LogicalType::Double, Data::Float64(values)) => {
5757            for value in &**values {
5758                out.extend_from_slice(&value.to_le_bytes());
5759            }
5760        }
5761        // Three counts and not one number. Months, days and microseconds stay apart on disk because
5762        // they are apart in the value: a month is not a fixed number of days and a day is not a
5763        // fixed number of microseconds, which is the whole reason the type has three fields.
5764        (LogicalType::Interval, Data::Interval(values)) => {
5765            for (months, days, micros) in &**values {
5766                out.extend_from_slice(&months.to_le_bytes());
5767                out.extend_from_slice(&days.to_le_bytes());
5768                out.extend_from_slice(&micros.to_le_bytes());
5769            }
5770        }
5771        (LogicalType::Boolean, Data::Bool(values)) => {
5772            for value in &**values {
5773                out.push(u8::from(*value));
5774            }
5775        }
5776        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
5777        // directory already, so writing it a value at a time would be paying for it twice.
5778        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
5779            for value in &**values {
5780                out.extend_from_slice(&value.to_le_bytes());
5781            }
5782        }
5783        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
5784            for value in &**values {
5785                out.extend_from_slice(&value.to_le_bytes());
5786            }
5787        }
5788        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
5789            for value in &**values {
5790                out.extend_from_slice(&value.to_le_bytes());
5791            }
5792        }
5793        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
5794            for value in &**values {
5795                out.extend_from_slice(&value.to_le_bytes());
5796            }
5797        }
5798        // A blob and a bit string go down the way a varchar does, because the layout is the same
5799        // one: an offset a value and then the bytes. What is not the same is that nothing here may
5800        // read the payload as text, which is why this arm asks the column for bytes rather than for
5801        // a string, and why the codecs above that do read text are all asked of a varchar by name.
5802        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
5803            let mut bytes = Vec::new();
5804            put_u32(&mut out, 0);
5805            for row in 0..vector.len() {
5806                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
5807                bytes.extend_from_slice(value);
5808                put_u32(
5809                    &mut out,
5810                    u32::try_from(bytes.len())
5811                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
5812                );
5813            }
5814            out.extend_from_slice(&bytes);
5815        }
5816        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
5817    }
5818    Ok((out, membership))
5819}
5820
5821fn put_varint(out: &mut Vec<u8>, mut value: u32) {
5822    while value >= 0x80 {
5823        out.push((value as u8 & 0x7f) | 0x80);
5824        value >>= 7;
5825    }
5826    out.push(value as u8);
5827}
5828
5829/// The distinct codes of one part, which is what a stripe's membership index is merged from.
5830fn unique_codes(codes: &[u32]) -> Vec<u32> {
5831    let mut unique = codes.to_vec();
5832    unique.sort_unstable();
5833    unique.dedup();
5834    unique
5835}
5836
5837/// The union of the sorted distinct codes of every part in a stripe.
5838///
5839/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
5840/// work on paper and the tree is the one that does not sort what is already in order: sixty four
5841/// sorted lists become one in six passes over the values.
5842fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
5843    let mut lists = lists;
5844    while lists.len() > 1 {
5845        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
5846        for pair in lists.chunks(2) {
5847            match pair {
5848                [left, right] => next.push(merged_pair(left, right)),
5849                [only] => next.push(only.clone()),
5850                _ => {}
5851            }
5852        }
5853        lists = next;
5854    }
5855    lists.pop().unwrap_or_default()
5856}
5857
5858fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
5859    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
5860    let mut at = 0;
5861    let mut to = 0;
5862    while at < left.len() && to < right.len() {
5863        match left[at].cmp(&right[to]) {
5864            Ordering::Less => {
5865                out.push(left[at]);
5866                at += 1;
5867            }
5868            Ordering::Greater => {
5869                out.push(right[to]);
5870                to += 1;
5871            }
5872            Ordering::Equal => {
5873                out.push(left[at]);
5874                at += 1;
5875                to += 1;
5876            }
5877        }
5878    }
5879    out.extend_from_slice(&left[at..]);
5880    out.extend_from_slice(&right[to..]);
5881    out
5882}
5883
5884/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
5885///
5886/// A bound that is missing from any part is missing from the stripe, because a missing bound means
5887/// nothing is known and a stripe that holds an unknown cannot claim one.
5888fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
5889    let mut merged = Range::default();
5890    let mut first = true;
5891    for range in ranges {
5892        merged.nulls = merged.nulls.saturating_add(range.nulls);
5893        // Both of these have to survive every part, so one part that could not say anything makes
5894        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
5895        // which leaves the stripe with exact ends and no total, which is a true thing to say.
5896        merged.sum = match (merged.sum.take(), range.sum) {
5897            (Some(held), Some(next)) if !first => held.checked_add(next),
5898            (_, next) if first => next,
5899            _ => None,
5900        };
5901        merged.exact = if first { range.exact } else { merged.exact && range.exact };
5902        if first {
5903            merged.low = range.low;
5904            merged.high = range.high;
5905            first = false;
5906            continue;
5907        }
5908        merged.low = match (merged.low.take(), range.low) {
5909            (Some(held), Some(next)) => Some(held.smaller(next)),
5910            _ => None,
5911        };
5912        merged.high = match (merged.high.take(), range.high) {
5913            (Some(held), Some(next)) => Some(held.larger(next)),
5914            _ => None,
5915        };
5916    }
5917    merged
5918}
5919
5920/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
5921///
5922/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
5923/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
5924/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
5925/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
5926///
5927/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
5928/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
5929/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
5930/// bound rather than claiming one that is too small. Anything that is not a string is already a
5931/// fixed width and is left alone.
5932fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
5933    match bound {
5934        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
5935            value.truncate(PART_BOUND_BYTES);
5936            if !high {
5937                return Some(Bound::Bytes(value));
5938            }
5939            while let Some(last) = value.pop() {
5940                if last < u8::MAX {
5941                    value.push(last + 1);
5942                    return Some(Bound::Bytes(value));
5943                }
5944            }
5945            None
5946        }
5947        other => other,
5948    }
5949}
5950
5951/// The ranges of one column's parts of one stripe, as a page.
5952///
5953/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
5954/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
5955/// number costs sixty times less to keep. What a part range is for is skipping the part, and
5956/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
5957/// string end that was cut down anyway.
5958fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
5959    let mut out = Vec::new();
5960    put_u32(
5961        &mut out,
5962        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
5963    );
5964    for range in ranges {
5965        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
5966        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
5967        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
5968    }
5969    Ok(out)
5970}
5971
5972/// The ranges one encoded page holds, one entry per part of the stripe.
5973fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
5974    let mut cur = Cursor { bytes, at: 0 };
5975    let parts = cur.u32()? as usize;
5976    let mut out = Vec::new();
5977    for _ in 0..parts {
5978        let low = cur.bound()?;
5979        let high = cur.bound()?;
5980        let nulls = cur.u32()? as usize;
5981        out.push(Range { low, high, nulls, exact: false, sum: None });
5982    }
5983    Ok(out)
5984}
5985
5986fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
5987    let held: Vec<&Option<Sieve>> = sieves.collect();
5988    let mut out = Vec::new();
5989    put_u32(
5990        &mut out,
5991        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
5992    );
5993    for sieve in &held {
5994        let length = sieve.as_ref().map_or(0, Sieve::len);
5995        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
5996    }
5997    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
5998    for sieve in held.into_iter().flatten() {
5999        out.extend_from_slice(&sieve.to_bytes());
6000    }
6001    Ok(out)
6002}
6003
6004/// The sieves one encoded page holds, one entry per part of the stripe.
6005///
6006/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
6007/// that gets read. That is how a file written by a later version of the sieve stays readable rather
6008/// than being a corrupt page.
6009fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
6010    let parts = u32::from_le_bytes(
6011        bytes
6012            .get(..4)
6013            .ok_or_else(|| invalid("sieve page is truncated"))?
6014            .try_into()
6015            .map_err(|_| invalid("sieve page is truncated"))?,
6016    ) as usize;
6017    let mut lengths = Vec::with_capacity(parts);
6018    for part in 0..parts {
6019        let at = 4 + part * 4;
6020        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
6021        lengths.push(u32::from_le_bytes(
6022            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
6023        ) as usize);
6024    }
6025    let mut at = 4 + parts * 4;
6026    let mut out = Vec::with_capacity(parts);
6027    for length in lengths {
6028        if length == 0 {
6029            out.push(None);
6030            continue;
6031        }
6032        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
6033        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
6034        out.push(Sieve::from_bytes(field));
6035        at = end;
6036    }
6037    if at != bytes.len() {
6038        return Err(invalid("sieve page has trailing bytes"));
6039    }
6040    Ok(out)
6041}
6042
6043/// One stripe's membership index: the code count and then the codes as ascending deltas.
6044///
6045/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
6046/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
6047/// a step a caller can skip.
6048fn encode_membership(unique: &[u32]) -> Vec<u8> {
6049    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
6050    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
6051    let mut previous = 0;
6052    for (at, &code) in unique.iter().enumerate() {
6053        put_varint(&mut out, if at == 0 { code } else { code - previous });
6054        previous = code;
6055    }
6056    out
6057}
6058
6059fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
6060    let mut value = 0_u32;
6061    for shift in (0..35).step_by(7) {
6062        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
6063        *at += 1;
6064        let part = u32::from(byte & 0x7f);
6065        if shift == 28 && part > 0x0f {
6066            return Err(invalid("membership varint overflow"));
6067        }
6068        value = value
6069            .checked_add(
6070                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
6071            )
6072            .ok_or_else(|| invalid("membership varint overflow"))?;
6073        if byte & 0x80 == 0 {
6074            return Ok(value);
6075        }
6076    }
6077    Err(invalid("membership varint is too long"))
6078}
6079
6080fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
6081    let mut at = 0;
6082    let count = take_varint(bytes, &mut at)? as usize;
6083    let mut codes = Vec::with_capacity(count);
6084    let mut previous = 0_u32;
6085    for index in 0..count {
6086        let delta = take_varint(bytes, &mut at)?;
6087        let code = if index == 0 {
6088            delta
6089        } else {
6090            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
6091        };
6092        if index > 0 && code <= previous {
6093            return Err(invalid("membership codes are not increasing"));
6094        }
6095        codes.push(code);
6096        previous = code;
6097    }
6098    if at != bytes.len() {
6099        return Err(invalid("membership page has trailing bytes"));
6100    }
6101    Ok(codes)
6102}
6103
6104fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
6105    let mut by_text = HashMap::new();
6106    let mut values = Vec::new();
6107    let mut codes = Vec::with_capacity(vector.len());
6108    let mut plain_bytes = 0_usize;
6109    for row in 0..vector.len() {
6110        let text = vector.text_at(row).unwrap_or("");
6111        plain_bytes = plain_bytes.saturating_add(text.len());
6112        let code = match by_text.get(text) {
6113            Some(&code) => code,
6114            None => {
6115                let code = u32::try_from(values.len())
6116                    .map_err(|_| invalid("too many dictionary values"))?;
6117                by_text.insert(text, code);
6118                values.push(text);
6119                code
6120            }
6121        };
6122        codes.push(code);
6123    }
6124    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
6125    let encoded = 8_usize
6126        .saturating_add((values.len() + 1).saturating_mul(4))
6127        .saturating_add(dictionary_bytes)
6128        .saturating_add(codes.len().saturating_mul(4));
6129    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
6130    if encoded >= plain {
6131        return Ok(None);
6132    }
6133    let mut out = Vec::with_capacity(encoded);
6134    put_u32(
6135        &mut out,
6136        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
6137    );
6138    put_u32(
6139        &mut out,
6140        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
6141    );
6142    let mut offset = 0_u32;
6143    put_u32(&mut out, offset);
6144    for value in &values {
6145        offset = offset
6146            .checked_add(
6147                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
6148            )
6149            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
6150        put_u32(&mut out, offset);
6151    }
6152    for value in values {
6153        out.extend_from_slice(value.as_bytes());
6154    }
6155    for code in codes {
6156        put_u32(&mut out, code);
6157    }
6158    Ok(Some(out))
6159}
6160
6161struct EncodedDictionary {
6162    index: Vec<u8>,
6163    ranks: Vec<u8>,
6164    /// The payload as the blocks it is written as, kept apart rather than joined because joining
6165    /// them is a second copy of a thing that is already gigabytes on the columns that matter.
6166    payload: Vec<Vec<u8>>,
6167}
6168
6169/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
6170fn head(bytes: &[u8]) -> u64 {
6171    let mut word = [0; 8];
6172    let take = bytes.len().min(8);
6173    word[..take].copy_from_slice(&bytes[..take]);
6174    u64::from_be_bytes(word)
6175}
6176
6177/// The sorted order of every global dictionary, one entry per column and empty where there is no
6178/// dictionary.
6179///
6180/// One column's sort has nothing to do with another's, and a table like `hits` has fifteen string
6181/// columns, so this runs across threads the way the numeric synopses above do. It is the only part
6182/// of committing a file that is more than bookkeeping, and doing it serially would show up as a
6183/// pause at the end of a load that thirty two threads had been busy with until then.
6184fn rankings(dictionaries: &[Option<GlobalDictionary>]) -> Result<Vec<Vec<(u64, u32)>>> {
6185    let present =
6186        dictionaries.iter().enumerate().filter(|(_, held)| held.is_some()).map(|(at, _)| at);
6187    let present = present.collect::<Vec<_>>();
6188    let mut orders = vec![Vec::new(); dictionaries.len()];
6189    let workers = std::thread::available_parallelism()
6190        .map_or(1, usize::from)
6191        .min(MAX_FREQUENCY_WORKERS)
6192        .min(present.len());
6193    if workers <= 1 {
6194        for at in present {
6195            if let Some(dictionary) = &dictionaries[at] {
6196                orders[at] = dictionary.ranked();
6197            }
6198        }
6199        return Ok(orders);
6200    }
6201    let width = present.len().div_ceil(workers);
6202    let pieces = std::thread::scope(|scope| {
6203        present
6204            .chunks(width)
6205            .map(|columns| {
6206                scope.spawn(|| {
6207                    columns
6208                        .iter()
6209                        .filter_map(|&at| dictionaries[at].as_ref().map(|held| (at, held.ranked())))
6210                        .collect::<Vec<_>>()
6211                })
6212            })
6213            .collect::<Vec<_>>()
6214            .into_iter()
6215            .map(|handle| {
6216                handle.join().map_err(|_| Error::internal("a dictionary sort worker panicked"))
6217            })
6218            .collect::<Result<Vec<_>>>()
6219    })?;
6220    for piece in pieces {
6221        for (at, order) in piece {
6222            orders[at] = order;
6223        }
6224    }
6225    Ok(orders)
6226}
6227
6228fn encode_global_dictionary(
6229    dictionary: GlobalDictionary,
6230    order: &[(u64, u32)],
6231) -> Result<EncodedDictionary> {
6232    let values = dictionary.offsets.len() - 1;
6233    if order.len() != values {
6234        return Err(invalid("global dictionary order does not cover its values"));
6235    }
6236    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6237    let payload = encode_payload(&dictionary)?;
6238    if payload.len() != blocks {
6239        return Err(invalid("global dictionary payload is not the blocks it says it is"));
6240    }
6241    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
6242    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
6243    let offset_bits = offset_width(&dictionary.offsets);
6244    let mut index = Vec::with_capacity(
6245        DICTIONARY_HEADER + offset_bytes(values, offset_bits) + (blocks + rank_blocks) * 16,
6246    );
6247    put_u32(
6248        &mut index,
6249        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
6250    );
6251    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
6252    put_u32(
6253        &mut index,
6254        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
6255    );
6256    put_u32(&mut index, offset_bits as u32);
6257    encode_offsets(&dictionary.offsets, offset_bits, &mut index)?;
6258    // Where each block ends, so a reader can find one. The stored blocks are shorter than the
6259    // decoded ones and by a different amount each, so this is the one thing the offsets above no
6260    // longer say.
6261    let mut at = 0_u64;
6262    for block in &payload {
6263        at = at
6264            .checked_add(block.len() as u64)
6265            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
6266        put_u64(&mut index, at);
6267    }
6268    for block in &payload {
6269        put_u64(&mut index, checksum(block));
6270    }
6271    // The same two lists for the sorted order. A rank block is packed at whatever width its own
6272    // heads need, so where one ends is no longer arithmetic on the block number.
6273    if rank_ends.len() != rank_blocks {
6274        return Err(invalid("global dictionary order is not the blocks it says it is"));
6275    }
6276    for end in &rank_ends {
6277        put_u64(&mut index, *end);
6278    }
6279    let mut at = 0_usize;
6280    for end in &rank_ends {
6281        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
6282        put_u64(&mut index, checksum(&ranks[at..end]));
6283        at = end;
6284    }
6285    Ok(EncodedDictionary { index, ranks, payload })
6286}
6287
6288/// How many blocks of the payload the shape is settled on.
6289///
6290/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
6291/// the same reason. They are spread across the dictionary rather than taken off the front, because
6292/// a dictionary is in the order values were first seen and the front of it is the first morsel of
6293/// the load.
6294const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
6295
6296/// The shapes the payload encoder picks between.
6297///
6298/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
6299/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
6300/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
6301/// settles the outer level and the one below it, which is where almost all of that hour goes, and
6302/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
6303/// to cost nothing.
6304///
6305/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
6306/// block, against the exhaustive search over the same blocks:
6307///
6308/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
6309/// |---|---|---|---|---|
6310/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
6311/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
6312/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
6313/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
6314/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
6315///
6316/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
6317/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
6318/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
6319/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
6320/// rather than searched for an answer that does not exist.
6321fn payload_shapes() -> Vec<chooser::Settled> {
6322    let integers = vec![integer::Kind::Packed];
6323    [
6324        vec![string::Kind::Front, string::Kind::Lz],
6325        vec![string::Kind::Lz, string::Kind::Fsst],
6326        vec![string::Kind::Lz, string::Kind::Plain],
6327        vec![string::Kind::Fsst],
6328        vec![string::Kind::Plain],
6329    ]
6330    .into_iter()
6331    .map(|strings| chooser::Settled::new(strings, integers.clone()))
6332    .collect()
6333}
6334
6335/// The payload as encoded blocks of [`TEXT_PAYLOAD_VALUES`] values each.
6336///
6337/// Across threads because this is the only part of committing a file that is real work rather than
6338/// bookkeeping. The blocks are the same size and cost about the same, so an index each is enough of
6339/// a queue and there is nothing to weight the way the numeric synopses are weighted.
6340fn encode_payload(dictionary: &GlobalDictionary) -> Result<Vec<Vec<u8>>> {
6341    let values = dictionary.offsets.len() - 1;
6342    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6343    let run = |block: usize| {
6344        let first = block * TEXT_PAYLOAD_VALUES;
6345        let last = (first + TEXT_PAYLOAD_VALUES).min(values);
6346        (first..last)
6347            .map(|value| {
6348                let from = dictionary.offsets[value] as usize;
6349                let to = dictionary.offsets[value + 1] as usize;
6350                &dictionary.payload[from..to]
6351            })
6352            .collect::<Vec<_>>()
6353    };
6354    // A dictionary small enough to be the sample is small enough to search in full, and searching
6355    // it costs less than deciding not to.
6356    let shape = (blocks > PAYLOAD_SAMPLE_BLOCKS).then(|| settle_shape(&run, blocks)).transpose()?;
6357    let one = |block: usize| match &shape {
6358        Some(shape) => string::encode_with(&run(block), shape),
6359        None => string::encode(&run(block)),
6360    };
6361    let workers = std::thread::available_parallelism()
6362        .map_or(1, usize::from)
6363        .min(MAX_FREQUENCY_WORKERS)
6364        .min(blocks);
6365    if workers <= 1 {
6366        return (0..blocks).map(one).collect();
6367    }
6368    let next = AtomicUsize::new(0);
6369    let pieces = std::thread::scope(|scope| {
6370        (0..workers)
6371            .map(|_| {
6372                scope.spawn(|| {
6373                    let mut mine = Vec::new();
6374                    loop {
6375                        let block = next.fetch_add(1, Atomic::Relaxed);
6376                        if block >= blocks {
6377                            break;
6378                        }
6379                        mine.push((block, one(block)?));
6380                    }
6381                    Ok(mine)
6382                })
6383            })
6384            .collect::<Vec<_>>()
6385            .into_iter()
6386            .map(|handle| {
6387                handle.join().map_err(|_| Error::internal("a dictionary encode worker panicked"))?
6388            })
6389            .collect::<Result<Vec<_>>>()
6390    })?;
6391    let mut payload = vec![Vec::new(); blocks];
6392    for piece in pieces {
6393        for (block, bytes) in piece {
6394            payload[block] = bytes;
6395        }
6396    }
6397    Ok(payload)
6398}
6399
6400/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
6401///
6402/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
6403/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
6404/// sample is spread across the dictionary so that the first and last blocks are both in it, because
6405/// a dictionary written in first seen order has its common values at the front and its long tail at
6406/// the back, and those do not compress alike.
6407fn settle_shape<'a>(
6408    run: &dyn Fn(usize) -> Vec<&'a [u8]>,
6409    blocks: usize,
6410) -> Result<chooser::Settled> {
6411    let last = blocks - 1;
6412    let sample = (0..PAYLOAD_SAMPLE_BLOCKS)
6413        .map(|region| run(region * last / (PAYLOAD_SAMPLE_BLOCKS - 1)))
6414        .collect::<Vec<_>>();
6415    let mut best: Option<(chooser::Settled, usize)> = None;
6416    for shape in payload_shapes() {
6417        let mut size = 0;
6418        for block in &sample {
6419            size += string::encode_with(block, &shape)?.len();
6420        }
6421        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
6422            best = Some((shape, size));
6423        }
6424    }
6425    best.map(|(shape, _)| shape)
6426        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
6427}
6428
6429/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
6430///
6431/// Each block holds its heads first and then its codes, rather than pairing them, because a search
6432/// asks for a head at every probe and for a code about once a search. Keeping the heads together
6433/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
6434/// probes of a search, which are the ones that land in the same block, touch the same cache line.
6435fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
6436    let mut out = Vec::with_capacity(order.len() * 4);
6437    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
6438    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
6439    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
6440    for block in order.chunks(TEXT_RANK_BLOCK) {
6441        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
6442        // rise, the smallest is the first and the largest is the last.
6443        let base = block.first().map_or(0, |&(head, _)| head);
6444        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
6445        let width = (u64::BITS - span.leading_zeros()) as usize;
6446        heads.clear();
6447        codes.clear();
6448        for &(head, code) in block {
6449            heads.push(head.wrapping_sub(base));
6450            codes.push(u64::from(code));
6451        }
6452        put_u64(&mut out, base);
6453        out.push(width as u8);
6454        bitpack::pack_tail(&heads, width, &mut out)
6455            .map_err(|_| invalid("global dictionary heads do not pack"))?;
6456        bitpack::pack_tail(&codes, code_bits, &mut out)
6457            .map_err(|_| invalid("global dictionary codes do not pack"))?;
6458        ends.push(out.len() as u64);
6459    }
6460    Ok((out, ends))
6461}
6462
6463/// Opens a column's global dictionary, which reads its index and none of its payload.
6464///
6465/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
6466/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
6467/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
6468/// a quarter of a gigabyte of dictionary to reach it.
6469fn open_global_dictionary(
6470    file: Arc<File>,
6471    page: Page,
6472    ty: &LogicalType,
6473    keep_budget: usize,
6474) -> Result<Vector> {
6475    if ty != &LogicalType::Varchar {
6476        return Err(invalid("global dictionary belongs to a non-string column"));
6477    }
6478    let mut header = [0; DICTIONARY_HEADER];
6479    read_at(&file, page.offset, &mut header)?;
6480    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
6481    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
6482    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
6483    let offset_bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
6484    if per_block != TEXT_PAYLOAD_VALUES {
6485        return Err(invalid("global dictionary block width differs"));
6486    }
6487    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
6488        return Err(invalid("global dictionary block count differs from its value count"));
6489    }
6490    if offset_bits > u32::BITS as usize {
6491        return Err(invalid("global dictionary packs offsets past a payload"));
6492    }
6493    let offset_len = offset_bytes(count, offset_bits);
6494    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
6495    // full the moment the column is first touched, and the order is half again the size of the
6496    // offsets, so putting it there would make every query that reads a string column pay for a
6497    // search that most of them never make.
6498    let ranks = count;
6499    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
6500    // Two words a payload block, one for where it ends in the file and one for its checksum, and the
6501    // same two a rank block.
6502    let hash_len = blocks
6503        .checked_add(rank_blocks)
6504        .and_then(|words| words.checked_mul(16))
6505        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
6506    let index_len = DICTIONARY_HEADER
6507        .checked_add(offset_len)
6508        .and_then(|len| len.checked_add(hash_len))
6509        .ok_or_else(|| invalid("global dictionary header overflow"))?;
6510    if index_len > page.length as usize {
6511        return Err(invalid("global dictionary offset index exceeds its page"));
6512    }
6513    let mut index = vec![0; index_len];
6514    index[..DICTIONARY_HEADER].copy_from_slice(&header);
6515    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
6516    if checksum(&index) != page.hash {
6517        return Err(invalid("global dictionary index checksum differs"));
6518    }
6519    let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
6520    let mut words = index[DICTIONARY_HEADER + offset_len..]
6521        .chunks_exact(8)
6522        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
6523        .collect::<Vec<_>>();
6524    let mut hashes = words.split_off(blocks);
6525    let mut rank_ends = hashes.split_off(blocks);
6526    let rank_hashes = rank_ends.split_off(rank_blocks);
6527    let ends = words;
6528    // A rank block packs its heads at whatever width its own values need, so its length is no longer
6529    // arithmetic on the block number and the reader has to be told where each one ends.
6530    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
6531        return Err(invalid("global dictionary order blocks do not rise"));
6532    }
6533    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
6534        .map_err(|_| invalid("global dictionary rank overflow"))?;
6535    let body_len = index_len
6536        .checked_add(rank_len)
6537        .ok_or_else(|| invalid("global dictionary header overflow"))?;
6538    if body_len > page.length as usize {
6539        return Err(invalid("global dictionary order exceeds its page"));
6540    }
6541    // What the offsets bound is the decoded payload, and what the page holds is the stored one, so
6542    // the last block end is the only thing that ties the index to the length of the page.
6543    let stored_len = page.length as usize - body_len;
6544    if ends.last().copied().unwrap_or_default() as usize != stored_len
6545        || ends.windows(2).any(|pair| pair[0] > pair[1])
6546    {
6547        return Err(invalid("global dictionary blocks do not bound the payload"));
6548    }
6549    Vector::external_text(
6550        LogicalType::Varchar,
6551        Arc::new(NativeText {
6552            file,
6553            values: count,
6554            offsets,
6555            offset_bits,
6556            ranks,
6557            rank_at: page.offset + index_len as u64,
6558            rank_ends,
6559            rank_hashes,
6560            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
6561            code_bits: code_width(count),
6562            code_ranks: OnceLock::new(),
6563            payload: page.offset + body_len as u64,
6564            ends,
6565            hashes,
6566            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
6567            keep_budget,
6568            payload_kept: AtomicUsize::new(0),
6569            searched: Mutex::new(HashMap::new()),
6570        }),
6571    )
6572}
6573
6574/// What a stored page is, without decoding a value out of it.
6575///
6576/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
6577/// the format's own choice, and it is what says whether the column came back as codes into a table
6578/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
6579/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
6580/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
6581///
6582/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
6583/// cannot walk comes back as text rather than as an error, because a caller asking what a file
6584/// looks like is usually asking because something is wrong with it, and a report that stops at the
6585/// first bad page is a report that says nothing about the other nine hundred.
6586fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
6587    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
6588    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
6589        let mut cur = Cursor { bytes, at: 0 };
6590        let codec = cur.u8()?;
6591        if cur.u8()? == 2 {
6592            cur.take(rows.div_ceil(8))?;
6593        }
6594        Ok((codec, cur.at))
6595    }
6596    let Ok((codec, at)) = cascade_at(rows, bytes) else {
6597        return "UNREADABLE".to_string();
6598    };
6599    let tail = &bytes[at..];
6600    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
6601    match codec {
6602        0 => match ty {
6603            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
6604            _ => "FIXED".to_string(),
6605        },
6606        1 => "DICT(PLAIN)".to_string(),
6607        2 => "FOR+BITPACK".to_string(),
6608        3 => "TABLE DICT".to_string(),
6609        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
6610        5 => described(integer::describe(tail)),
6611        6 => described(string::describe(tail)),
6612        other => format!("CODEC {other}"),
6613    }
6614}
6615
6616fn decode(
6617    ty: &LogicalType,
6618    rows: usize,
6619    bytes: &[u8],
6620    global: Option<Arc<Vector>>,
6621) -> Result<Vector> {
6622    let mut cur = Cursor { bytes, at: 0 };
6623    let codec = cur.u8()?;
6624    let flag = cur.u8()?;
6625    let validity = match flag {
6626        0 => Validity::AllValid,
6627        1 => Validity::AllInvalid,
6628        2 => {
6629            let mask = cur.take(rows.div_ceil(8))?;
6630            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
6631        }
6632        _ => return Err(invalid("page validity tag differs")),
6633    };
6634    if codec == 1 {
6635        if ty != &LogicalType::Varchar {
6636            return Err(invalid("dictionary codec belongs to a non-string page"));
6637        }
6638        let count = cur.u32()? as usize;
6639        let payload_len = cur.u32()? as usize;
6640        let offset_bytes = cur.take(
6641            (count + 1)
6642                .checked_mul(4)
6643                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
6644        )?;
6645        let offsets = offset_bytes
6646            .chunks_exact(4)
6647            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6648            .collect::<Vec<_>>();
6649        let payload = cur.take(payload_len)?.to_vec();
6650        if offsets.first() != Some(&0)
6651            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6652            || offsets.windows(2).any(|pair| pair[0] > pair[1])
6653        {
6654            return Err(invalid("dictionary offsets do not bound the payload"));
6655        }
6656        // A page, because every chunk cut out of this dictionary points at the same payload and a
6657        // page is what lets a cut be the views and nothing else.
6658        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
6659        for pair in offsets.windows(2) {
6660            strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
6661        }
6662        let mut codes = Vec::with_capacity(rows);
6663        for _ in 0..rows {
6664            codes.push(cur.u32()?);
6665        }
6666        if codes.iter().any(|code| *code as usize >= count) {
6667            return Err(invalid("dictionary code is out of range"));
6668        }
6669        if cur.at != bytes.len() {
6670            return Err(invalid("dictionary page has trailing bytes"));
6671        }
6672        let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
6673        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
6674    }
6675    if codec == 3 || codec == 4 {
6676        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
6677        let codes = if codec == 4 {
6678            // The cascade holds the whole tail of the page and says how long it is itself, so the
6679            // check that nothing is left over is the one the decoder already makes.
6680            let wide = integer::decode(&bytes[cur.at..])?;
6681            if wide.len() != rows {
6682                return Err(invalid("encoded code page holds the wrong number of rows"));
6683            }
6684            // Converted in one pass and checked in the same one, rather than a fallible conversion
6685            // per code. A `Result` an element is a short circuit the loop cannot be vectorized past,
6686            // and it was costing about twelve instructions a row to narrow a number that already
6687            // fits. Every code a file holds is inside a `u32` or the file is corrupt, so the check
6688            // belongs once at the end: or the codes together and the answer has a bit set above the
6689            // low thirty two, or the sign bit, exactly when one of them did.
6690            let mut codes = Vec::with_capacity(wide.len());
6691            let mut seen = 0_i64;
6692            for &code in &wide {
6693                seen |= code;
6694                codes.push(code as u32);
6695            }
6696            if seen < 0 || seen > i64::from(u32::MAX) {
6697                return Err(invalid("code is not a code"));
6698            }
6699            codes
6700        } else {
6701            let mut codes = Vec::with_capacity(rows);
6702            for _ in 0..rows {
6703                codes.push(cur.u32()?);
6704            }
6705            if cur.at != bytes.len() {
6706                return Err(invalid("global code page has trailing bytes"));
6707            }
6708            codes
6709        };
6710        let highest = codes.iter().copied().max();
6711        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
6712            .with_validity(validity));
6713    }
6714    if codec == 6 {
6715        if ty != &LogicalType::Varchar {
6716            return Err(invalid("compressed text codec belongs to a non-string page"));
6717        }
6718        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
6719        // It comes back as one buffer with the values laid end to end and where each one ends, which
6720        // is the raw form's layout, so what is left to do here is what codec 0 does.
6721        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
6722        if ends.len() != rows {
6723            return Err(invalid("compressed text page holds the wrong number of rows"));
6724        }
6725        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
6726        // payload moves views rather than bytes.
6727        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6728        let mut start = 0;
6729        for end in ends {
6730            let len = end
6731                .checked_sub(start)
6732                .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
6733            values.push_in_place(start, len)?;
6734            start = end;
6735        }
6736        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
6737    }
6738    if codec == 5 {
6739        // The cascade holds the whole tail of the page and says how long it is itself.
6740        let values = integer::decode(&bytes[cur.at..])?;
6741        if values.len() != rows {
6742            return Err(invalid("cascade page holds the wrong number of rows"));
6743        }
6744        let data = narrowed(ty, values)?;
6745        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
6746    }
6747    if codec == 2 {
6748        let width = u32::from(cur.u8()?);
6749        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
6750        let count = cur.u32()? as usize;
6751        let mut words = Vec::with_capacity(count);
6752        for _ in 0..count {
6753            words.push(cur.u64()?);
6754        }
6755        if cur.at != bytes.len() {
6756            return Err(invalid("packed page has trailing bytes"));
6757        }
6758        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
6759    }
6760    if codec != 0 {
6761        return Err(invalid("page codec is unknown"));
6762    }
6763    let data = match ty {
6764        LogicalType::TinyInt => {
6765            let values = cur.take(rows)?;
6766            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
6767        }
6768        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
6769        LogicalType::SmallInt => {
6770            let values =
6771                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6772            Data::Int16(
6773                values
6774                    .chunks_exact(2)
6775                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6776                    .collect::<Vec<_>>()
6777                    .into(),
6778            )
6779        }
6780        LogicalType::USmallInt => {
6781            let values =
6782                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6783            Data::UInt16(
6784                values
6785                    .chunks_exact(2)
6786                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
6787                    .collect::<Vec<_>>()
6788                    .into(),
6789            )
6790        }
6791        LogicalType::UInteger => {
6792            let values =
6793                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6794            Data::UInt32(
6795                values
6796                    .chunks_exact(4)
6797                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
6798                    .collect::<Vec<_>>()
6799                    .into(),
6800            )
6801        }
6802        LogicalType::UBigInt => {
6803            let values =
6804                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6805            Data::UInt64(
6806                values
6807                    .chunks_exact(8)
6808                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
6809                    .collect::<Vec<_>>()
6810                    .into(),
6811            )
6812        }
6813        LogicalType::Integer | LogicalType::Date => {
6814            let values =
6815                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6816            Data::Int32(
6817                values
6818                    .chunks_exact(4)
6819                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6820                    .collect::<Vec<_>>()
6821                    .into(),
6822            )
6823        }
6824        LogicalType::BigInt
6825        | LogicalType::Timestamp
6826        | LogicalType::Time
6827        | LogicalType::TimeTz
6828        | LogicalType::TimestampTz
6829        | LogicalType::TimestampS
6830        | LogicalType::TimestampMs
6831        | LogicalType::TimestampNs => {
6832            let values =
6833                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6834            Data::Int64(
6835                values
6836                    .chunks_exact(8)
6837                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6838                    .collect::<Vec<_>>()
6839                    .into(),
6840            )
6841        }
6842        LogicalType::HugeInt | LogicalType::Uuid => {
6843            let values =
6844                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6845            Data::Int128(
6846                values
6847                    .chunks_exact(16)
6848                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6849                    .collect::<Vec<_>>()
6850                    .into(),
6851            )
6852        }
6853        LogicalType::UHugeInt => {
6854            let values =
6855                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6856            Data::UInt128(
6857                values
6858                    .chunks_exact(16)
6859                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6860                    .collect::<Vec<_>>()
6861                    .into(),
6862            )
6863        }
6864        LogicalType::Float => {
6865            let values =
6866                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6867            Data::Float32(
6868                values
6869                    .chunks_exact(4)
6870                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
6871                    .collect::<Vec<_>>()
6872                    .into(),
6873            )
6874        }
6875        LogicalType::Double => {
6876            let values =
6877                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6878            Data::Float64(
6879                values
6880                    .chunks_exact(8)
6881                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
6882                    .collect::<Vec<_>>()
6883                    .into(),
6884            )
6885        }
6886        LogicalType::Interval => {
6887            let values =
6888                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6889            Data::Interval(
6890                values
6891                    .chunks_exact(16)
6892                    .map(|item| {
6893                        (
6894                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
6895                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
6896                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
6897                        )
6898                    })
6899                    .collect::<Vec<_>>()
6900                    .into(),
6901            )
6902        }
6903        LogicalType::Boolean => {
6904            let values = cur.take(rows)?;
6905            if values.iter().any(|value| *value > 1) {
6906                return Err(invalid("boolean page has another value"));
6907            }
6908            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
6909        }
6910        // Whichever integer the declared width says, which is the mapping the rest of the engine
6911        // already uses for a decimal in memory.
6912        LogicalType::Decimal { .. } => match ty.physical() {
6913            PhysicalType::Int16 => {
6914                let values =
6915                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6916                Data::Int16(
6917                    values
6918                        .chunks_exact(2)
6919                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6920                        .collect::<Vec<_>>()
6921                        .into(),
6922                )
6923            }
6924            PhysicalType::Int32 => {
6925                let values =
6926                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6927                Data::Int32(
6928                    values
6929                        .chunks_exact(4)
6930                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6931                        .collect::<Vec<_>>()
6932                        .into(),
6933                )
6934            }
6935            PhysicalType::Int64 => {
6936                let values =
6937                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6938                Data::Int64(
6939                    values
6940                        .chunks_exact(8)
6941                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6942                        .collect::<Vec<_>>()
6943                        .into(),
6944                )
6945            }
6946            _ => {
6947                let values =
6948                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6949                Data::Int128(
6950                    values
6951                        .chunks_exact(16)
6952                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6953                        .collect::<Vec<_>>()
6954                        .into(),
6955                )
6956            }
6957        },
6958        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
6959            let offset_bytes = cur
6960                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
6961            let offsets = offset_bytes
6962                .chunks_exact(4)
6963                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6964                .collect::<Vec<_>>();
6965            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
6966            if offsets.first() != Some(&0)
6967                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6968                || offsets.windows(2).any(|pair| pair[0] > pair[1])
6969            {
6970                return Err(invalid("string offsets do not bound the payload"));
6971            }
6972            // A page for the reason the dictionary payload above is one: the page is read once and
6973            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
6974            // bytes.
6975            //
6976            // A varchar is checked for text on the way in and a blob and a bit string are not,
6977            // because the second pair never claimed to hold any. Reading them through the checking
6978            // seam would refuse a column for holding exactly what it was told to hold.
6979            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6980            let text = ty == &LogicalType::Varchar;
6981            for pair in offsets.windows(2) {
6982                let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
6983                if text {
6984                    values.push_in_place(at, len)?;
6985                } else {
6986                    values.push_bytes_in_place(at, len)?;
6987                }
6988            }
6989            Data::Varlen(values)
6990        }
6991        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
6992    };
6993    if cur.at != bytes.len() {
6994        return Err(invalid("page has trailing bytes"));
6995    }
6996    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
6997}
6998
6999#[cfg(test)]
7000mod tests {
7001    use std::fs;
7002    use std::io::{Seek, SeekFrom, Write};
7003    use std::path::PathBuf;
7004    use std::time::{SystemTime, UNIX_EPOCH};
7005
7006    use rudb_common::Stat;
7007    use rudb_common::Value;
7008    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
7009    use rudb_common::stat::Provenance;
7010
7011    use super::*;
7012
7013    #[test]
7014    fn checksum_matches_fixed_vectors() {
7015        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
7016        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
7017        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
7018    }
7019
7020    fn path(label: &str) -> PathBuf {
7021        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
7022        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
7023    }
7024
7025    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
7026    #[test]
7027    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
7028        const SPANS: usize = 64;
7029        const SPAN: usize = 512;
7030        let path = path("positional");
7031        let content: Vec<u8> =
7032            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
7033        fs::write(&path, &content).expect("the file is written");
7034        let file = Arc::new(File::open(&path).expect("the file opens"));
7035        std::thread::scope(|scope| {
7036            for _ in 0..8 {
7037                let file = Arc::clone(&file);
7038                scope.spawn(move || {
7039                    for _ in 0..64 {
7040                        for span in 0..SPANS {
7041                            let mut bytes = [0_u8; SPAN];
7042                            read_at(&file, (span * SPAN) as u64, &mut bytes)
7043                                .expect("the span reads");
7044                            assert!(
7045                                bytes.iter().all(|byte| *byte == span as u8),
7046                                "span {span} came back as {}",
7047                                bytes[0],
7048                            );
7049                        }
7050                    }
7051                });
7052            }
7053        });
7054        let mut past = [0_u8; SPAN];
7055        let end = (SPANS * SPAN) as u64;
7056        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
7057        assert!(error.message().contains("ends before its declared length"), "{error}");
7058        drop(file);
7059        let _ = fs::remove_file(&path);
7060    }
7061
7062    /// The writer records where it put a page and puts it there, whatever the cursor is doing.
7063    ///
7064    /// The cursor is moved between the steps that record an offset, which is what reading the pages
7065    /// back to build the frequencies does on a platform with no `pread`. Without the fix the
7066    /// directory lands on top of a page and the file fails to reopen.
7067    #[test]
7068    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
7069        let path = path("cursor");
7070        let mut writer = Writer::create(
7071            &path,
7072            "items",
7073            vec![
7074                Field::required("id", LogicalType::Integer),
7075                Field::new("text", LogicalType::Varchar),
7076            ],
7077        )
7078        .expect("new file");
7079        writer.append(&sample()).expect("first part");
7080        writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
7081        writer.append(&sample()).expect("second part");
7082        writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
7083        writer.finish().expect("commit");
7084        let reader = Reader::open(&path).expect("reopen from disk");
7085        assert_eq!(reader.table().rows(), 6);
7086        let ids = reader.read(0, &[0]).expect("the integer page reads back");
7087        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
7088        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
7089        let text = reader.read(1, &[1]).expect("the text page reads back");
7090        assert_eq!(text.value_at(1, 0), Value::Null);
7091        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7092        // Nothing the directory points at may run past the end of the file, which is the shape the
7093        // failure took: a page recorded at an offset the directory had already been written over.
7094        let end = reader.table().stripes().iter().flat_map(|stripe| {
7095            stripe
7096                .pages
7097                .iter()
7098                .map(|page| page.offset + u64::from(page.length))
7099                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
7100        });
7101        let last = end.fold(HEADER, u64::max);
7102        let directory = fs::metadata(&path).expect("the file is there").len();
7103        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
7104        fs::remove_file(path).expect("remove scratch file");
7105    }
7106
7107    /// How long a global dictionary index is, read out of the page's own header.
7108    ///
7109    /// The tests below damage a byte of the order or of the payload, so they need to know where each
7110    /// one starts, and working it out here rather than writing a number down means adding something
7111    /// to the index does not quietly turn one of them into a test that damages the index instead.
7112    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
7113        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
7114        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
7115        let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
7116        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
7117        DICTIONARY_HEADER as u64
7118            + offset_bytes(count as usize, bits) as u64
7119            + (blocks + rank_blocks) * 16
7120    }
7121
7122    /// How long the sorted order is, which is where its last block ends.
7123    fn last_rank_end(file: &File, offset: u64, header: &[u8; DICTIONARY_HEADER]) -> u64 {
7124        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
7125        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
7126        let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
7127        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
7128        let at = offset
7129            + DICTIONARY_HEADER as u64
7130            + offset_bytes(count as usize, bits) as u64
7131            + blocks * 16
7132            + (rank_blocks - 1) * 8;
7133        let mut end = [0; 8];
7134        read_at(file, at, &mut end).expect("the last rank block end");
7135        u64::from_le_bytes(end)
7136    }
7137
7138    fn sample() -> Chunk {
7139        Chunk::new(vec![
7140            Vector::from_values(
7141                LogicalType::Integer,
7142                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
7143            )
7144            .expect("integers"),
7145            Vector::from_values(
7146                LogicalType::Varchar,
7147                &[
7148                    Value::Varchar("alpha".into()),
7149                    Value::Null,
7150                    Value::Varchar("long text after a slash".into()),
7151                ],
7152            )
7153            .expect("strings"),
7154        ])
7155        .expect("matching rows")
7156    }
7157
7158    fn sample_ids() -> Chunk {
7159        Chunk::new(vec![
7160            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
7161                .expect("integers"),
7162        ])
7163        .expect("one column")
7164    }
7165
7166    #[test]
7167    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
7168        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
7169        // condition gets, and the number was in the stripe entry next to the bounds all along.
7170        let path = path("nulls_for_the_planner");
7171        let mut writer =
7172            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
7173                .expect("new file");
7174        let rows = Chunk::new(vec![
7175            Vector::from_values(
7176                LogicalType::Integer,
7177                &[
7178                    Value::Integer(4),
7179                    Value::Null,
7180                    Value::Integer(9),
7181                    Value::Null,
7182                    Value::Integer(1),
7183                    Value::Integer(2),
7184                ],
7185            )
7186            .expect("integers"),
7187        ])
7188        .expect("one column");
7189        writer.append(&rows).expect("the only part");
7190        writer.finish().expect("commit");
7191        let reader = Reader::open(&path).expect("reopen from disk");
7192        let stripes = Stripes::new(reader);
7193        let column = stripes.column("a").expect("the file has that column");
7194        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
7195        // A column the file does not have. Zero here would be a fact about a column that is not
7196        // there, which the planner would then divide by.
7197        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
7198        fs::remove_file(&path).expect("clean up");
7199    }
7200
7201    #[test]
7202    fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
7203        // The whole of the frequency half of #1106, end to end over a real file. Six rows, three
7204        // of one value and two of another, and a complete synopsis because six rows is well inside
7205        // what the writer can account for. The estimate for `id = 4` is three rows rather than a
7206        // sixth of the table, and for a value the file does not hold it is none.
7207        let path = path("frequencies_for_the_planner");
7208        let mut writer =
7209            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7210                .expect("new file");
7211        let rows = Chunk::new(vec![
7212            Vector::from_values(
7213                LogicalType::Integer,
7214                &[
7215                    Value::Integer(4),
7216                    Value::Integer(4),
7217                    Value::Integer(4),
7218                    Value::Integer(9),
7219                    Value::Integer(9),
7220                    Value::Integer(1),
7221                ],
7222            )
7223            .expect("integers"),
7224        ])
7225        .expect("one column");
7226        writer.append(&rows).expect("the only part");
7227        writer.finish().expect("commit");
7228        let reader = Reader::open(&path).expect("reopen from disk");
7229        let common = Common::new(reader);
7230        assert_eq!(common.rows(), 6);
7231        let column = common.column("id").expect("the file has that column");
7232        assert_eq!(common.column("nothing"), None);
7233        assert_eq!(
7234            common.rows_with(column, &Bound::Int(4)),
7235            Stat::exact(3, Provenance::FrequencySynopsis)
7236        );
7237        // Not in the file, and a synopsis that accounts for all six rows proves it.
7238        assert_eq!(
7239            common.rows_with(column, &Bound::Int(7)),
7240            Stat::exact(0, Provenance::FrequencySynopsis)
7241        );
7242        // A constant of another domain against an integer column. Nothing in the list compares
7243        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
7244        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
7245        // A complete list has no remainder. Answering one of no rows over no values would hand the
7246        // caller a division to special case, and the counts above already answer this column.
7247        assert_eq!(common.remainder(column), None);
7248        fs::remove_file(&path).expect("clean up");
7249    }
7250
7251    /// A table directory with nothing in it but a name and one column, for the section tests.
7252    ///
7253    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
7254    /// say so by starting from the emptiest table that encodes.
7255    fn bare_table(sections: Vec<Section>) -> Table {
7256        Table {
7257            name: "linked".to_owned(),
7258            fields: vec![Field::required("id", LogicalType::Integer)],
7259            stripes: Vec::new(),
7260            rows: 0,
7261            dictionaries: vec![None],
7262            distincts: vec![None],
7263            frequencies: vec![None],
7264            clustering: None,
7265            generation: 1,
7266            sections,
7267        }
7268    }
7269
7270    fn a_key_map_section() -> Section {
7271        Section {
7272            kind: *section::KEY_MAP,
7273            id: 1,
7274            generation: 3,
7275            extents: 1,
7276            extent_page: HEADER,
7277            extent_bytes: section::EXTENT_BYTES as u32,
7278            hash: 0x1234_5678_9abc_def0,
7279            flags: 0,
7280            header_bytes: 24,
7281        }
7282    }
7283
7284    #[test]
7285    fn a_section_table_round_trips_through_a_directory() {
7286        let mut later = a_key_map_section();
7287        later.kind = *b"RUDBZZ9\0";
7288        later.id = 2;
7289        let table = bare_table(vec![a_key_map_section(), later]);
7290        let directory = encode_directory(&table).expect("directory");
7291        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7292        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
7293        // The second is a kind this build has no name for, and it survived the round trip anyway.
7294        // That is what keeps an old build from silently discarding a newer build's work when it
7295        // rewrites a directory.
7296        assert!(decoded.sections()[0].known());
7297        assert!(!decoded.sections()[1].known());
7298    }
7299
7300    #[test]
7301    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
7302        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
7303        // build's directory with the trailing section block cut off, so cutting it off is the
7304        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
7305        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7306        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
7307        let older = &directory[..directory.len() - block];
7308        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
7309        assert!(decoded.sections().is_empty());
7310        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
7311        assert_eq!(decoded.name(), "linked");
7312        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
7313    }
7314
7315    #[test]
7316    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
7317        // The same criterion end to end, which is the one the milestone actually asks for: a build
7318        // that knows about sections opens a file written by a build that did not, with no rewrite
7319        // and no repair, and answers from it. The version field is patched rather than a file
7320        // committed by an old binary because the bytes either side of it are identical: format 22
7321        // and format 23 differ only in a trailing directory block, and a reader that stops before
7322        // that block gets a table with no sections.
7323        let path = path("format_twenty_two");
7324        let mut writer =
7325            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7326                .expect("new file");
7327        let rows = Chunk::new(vec![
7328            Vector::from_values(
7329                LogicalType::Integer,
7330                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
7331            )
7332            .expect("integers"),
7333        ])
7334        .expect("one column");
7335        writer.append(&rows).expect("the only part");
7336        writer.finish().expect("commit");
7337
7338        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7339        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7340        drop(file);
7341
7342        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
7343        assert_eq!(reader.table().rows(), 3);
7344        assert!(reader.table().sections().is_empty());
7345
7346        // And a format this build has never written is still refused, so the accept set is a list
7347        // and not an absence of a check.
7348        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7349        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
7350        drop(file);
7351        let error = Reader::open(&path).expect_err("format 21 is not readable");
7352        assert!(error.to_string().contains("format 21"), "{error}");
7353
7354        fs::remove_file(&path).expect("clean up");
7355    }
7356
7357    #[test]
7358    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
7359        // The bound the format has to check and `section` cannot, because only the reader knows how
7360        // big the file is. Reading the payload a section like this names would be reading whatever
7361        // else happens to be at that offset, which is the one way a graph section could turn into a
7362        // wrong answer rather than a slow one.
7363        let mut past = a_key_map_section();
7364        past.extent_page = 1 << 30;
7365        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
7366        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
7367        assert!(error.to_string().contains("outside the file"), "{error}");
7368
7369        let mut inside_the_header = a_key_map_section();
7370        inside_the_header.extent_page = 8;
7371        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
7372        assert!(
7373            decode_directory(&directory, 1 << 20).is_err(),
7374            "a section may not overlap a header"
7375        );
7376    }
7377
7378    #[test]
7379    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
7380        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
7381        // that `rudb_links()` can report what a larger budget would buy. That record is a section
7382        // entry with no extents, so it has to survive a round trip while naming nothing.
7383        let not_built = Section {
7384            kind: *section::FORWARD_LINK,
7385            id: 9,
7386            generation: 3,
7387            extents: 0,
7388            extent_page: 0,
7389            extent_bytes: 0,
7390            hash: 0,
7391            flags: 0,
7392            header_bytes: 0,
7393        };
7394        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
7395        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7396        assert_eq!(decoded.sections(), &[not_built]);
7397
7398        // But a section with no extents that still names an extent table is incoherent, and an
7399        // incoherent entry is a torn directory rather than a relationship that was skipped.
7400        let mut incoherent = not_built;
7401        incoherent.extent_bytes = 28;
7402        incoherent.extent_page = HEADER;
7403        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
7404        assert!(decode_directory(&directory, 1 << 20).is_err());
7405    }
7406
7407    #[test]
7408    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
7409        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7410        let mut torn = directory.clone();
7411        let count_at = torn.len() - size_of::<u16>();
7412        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
7413        // Not an allocation of sixty five thousand entries off a torn count: either the bound
7414        // refuses it or the bytes run out, and both are errors rather than a read past the end.
7415        assert!(decode_directory(&torn, 1 << 20).is_err());
7416    }
7417
7418    /// A committed one column file of `rows` integers, for the attach tests.
7419    fn linked_file(label: &str, rows: i32) -> PathBuf {
7420        let path = path(label);
7421        let mut writer =
7422            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7423                .expect("new file");
7424        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
7425        let chunk =
7426            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
7427                .expect("one column");
7428        writer.append(&chunk).expect("the only part");
7429        writer.finish().expect("commit");
7430        path
7431    }
7432
7433    fn a_key_map_payload() -> Vec<u8> {
7434        // Shaped like one without being one: this crate never reads a payload, so what matters here
7435        // is that every byte comes back and that the header the entry measures is at the front.
7436        (0..512_u32).flat_map(u32::to_le_bytes).collect()
7437    }
7438
7439    #[test]
7440    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
7441        let path = linked_file("attach", 64);
7442        let payload = a_key_map_payload();
7443        let table = attach(
7444            &path,
7445            "items",
7446            &[section::Attachment {
7447                kind: *section::KEY_MAP,
7448                id: 0,
7449                flags: 2,
7450                header_bytes: 40,
7451                bytes: &payload,
7452            }],
7453        )
7454        .expect("attach a key map");
7455        assert_eq!(table.sections().len(), 1);
7456
7457        let reader = Reader::open(&path).expect("reopen after the attach");
7458        let held = reader.table().sections();
7459        assert_eq!(held.len(), 1);
7460        assert_eq!(held[0].kind, *section::KEY_MAP);
7461        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
7462        assert_eq!(held[0].header_bytes, 40);
7463        // The generation is the one the pages were written at, not the one the attach committed at.
7464        // Attaching a section moved no row, so a section written by it is current, and a second
7465        // table added to this file later would not make it stale.
7466        assert_eq!(held[0].generation, 1);
7467        assert!(held[0].usable(reader.table().generation()));
7468        assert_eq!(reader.payload(&held[0]).expect("read the payload"), payload);
7469        assert_eq!(reader.extents(&held[0]).expect("extent table").len(), 1);
7470
7471        fs::remove_file(&path).expect("clean up");
7472    }
7473
7474    #[test]
7475    fn attaching_a_section_answers_every_row_exactly_as_before() {
7476        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
7477        // file with a section in it and the same file without one have to agree row for row, so the
7478        // comparison is made against the answers taken before the attach rather than against a
7479        // constant somebody typed.
7480        let path = linked_file("attach_changes_nothing", 300);
7481        let before = Reader::open(&path).expect("open before");
7482        let rows = before.table().rows();
7483        let first = before.read(0, &[0]).expect("read before");
7484        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
7485        let layout = before.layout().columns_total();
7486        drop(before);
7487
7488        let payload = a_key_map_payload();
7489        attach(
7490            &path,
7491            "items",
7492            &[section::Attachment {
7493                kind: *section::KEY_MAP,
7494                id: 0,
7495                flags: 0,
7496                header_bytes: 0,
7497                bytes: &payload,
7498            }],
7499        )
7500        .expect("attach");
7501
7502        let after = Reader::open(&path).expect("open after");
7503        assert_eq!(after.table().rows(), rows);
7504        let read = after.read(0, &[0]).expect("read after");
7505        for (at, value) in values.iter().enumerate() {
7506            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
7507        }
7508        assert_eq!(
7509            after.layout().columns_total(),
7510            layout,
7511            "an attach appends and does not rewrite a column page"
7512        );
7513
7514        fs::remove_file(&path).expect("clean up");
7515    }
7516
7517    #[test]
7518    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
7519        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
7520        // replaced, a table rebuilt a few times would name several maps for one column and a reader
7521        // would have to pick, which is a decision with no right answer in it.
7522        let path = linked_file("attach_twice", 32);
7523        let one = a_key_map_payload();
7524        let two = vec![7_u8; 1024];
7525        let entry = |bytes| section::Attachment {
7526            kind: *section::KEY_MAP,
7527            id: 4,
7528            flags: 1,
7529            header_bytes: 0,
7530            bytes,
7531        };
7532        attach(&path, "items", &[entry(&one)]).expect("first build");
7533        attach(&path, "items", &[entry(&two)]).expect("rebuild");
7534
7535        let reader = Reader::open(&path).expect("reopen");
7536        let held = reader.table().sections();
7537        assert_eq!(held.len(), 1, "one map per column and not one per build");
7538        assert_eq!(reader.payload(&held[0]).expect("payload"), two);
7539
7540        fs::remove_file(&path).expect("clean up");
7541    }
7542
7543    #[test]
7544    fn an_attach_carries_through_a_kind_it_does_not_know() {
7545        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
7546        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
7547        // an older binary and attaching one section quietly deletes the work of a newer one.
7548        let path = linked_file("attach_unknown", 16);
7549        let payload = vec![3_u8; 96];
7550        attach(
7551            &path,
7552            "items",
7553            &[section::Attachment {
7554                kind: *b"RUDBZZ9\0",
7555                id: 1,
7556                flags: 0,
7557                header_bytes: 0,
7558                bytes: &payload,
7559            }],
7560        )
7561        .expect("a kind this build does not know still writes");
7562        let key_map = a_key_map_payload();
7563        attach(
7564            &path,
7565            "items",
7566            &[section::Attachment {
7567                kind: *section::KEY_MAP,
7568                id: 0,
7569                flags: 0,
7570                header_bytes: 0,
7571                bytes: &key_map,
7572            }],
7573        )
7574        .expect("attach beside it");
7575
7576        let reader = Reader::open(&path).expect("reopen");
7577        let held = reader.table().sections();
7578        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
7579        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
7580        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
7581
7582        fs::remove_file(&path).expect("clean up");
7583    }
7584
7585    #[test]
7586    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
7587        let path = linked_file("attach_not_built", 8);
7588        attach(
7589            &path,
7590            "items",
7591            &[section::Attachment {
7592                kind: *section::FORWARD_LINK,
7593                id: 2,
7594                flags: 0,
7595                header_bytes: 0,
7596                bytes: &[],
7597            }],
7598        )
7599        .expect("record a link that did not fit the budget");
7600
7601        let reader = Reader::open(&path).expect("reopen");
7602        let held = reader.table().sections();
7603        assert_eq!(held.len(), 1);
7604        assert_eq!(held[0].extents, 0);
7605        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
7606        assert!(reader.extents(&held[0]).expect("no extent table").is_empty());
7607        assert!(reader.payload(&held[0]).expect("no payload").is_empty());
7608
7609        fs::remove_file(&path).expect("clean up");
7610    }
7611
7612    #[test]
7613    fn a_payload_past_one_extent_is_split_and_joined_back() {
7614        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
7615        // payload that has to be two extents, and it is the case a split written for the common
7616        // size gets wrong.
7617        let path = linked_file("attach_two_extents", 8);
7618        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
7619        attach(
7620            &path,
7621            "items",
7622            &[section::Attachment {
7623                kind: *section::KEY_MAP,
7624                id: 0,
7625                flags: 0,
7626                header_bytes: 0,
7627                bytes: &payload,
7628            }],
7629        )
7630        .expect("attach a payload past the bound");
7631
7632        let reader = Reader::open(&path).expect("reopen");
7633        let held = reader.table().sections();
7634        let extents = reader.extents(&held[0]).expect("extent table");
7635        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
7636        assert_eq!(extents[0].length, section::MAX_EXTENT);
7637        assert_eq!(extents[1].length, 1);
7638        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
7639        // And the extent the caller wants is readable on its own, which is the point of the split.
7640        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
7641        assert_eq!(reader.payload(&held[0]).expect("the whole payload").len(), payload.len());
7642
7643        fs::remove_file(&path).expect("clean up");
7644    }
7645
7646    #[test]
7647    fn a_torn_extent_is_refused_rather_than_decoded() {
7648        let path = linked_file("attach_torn", 8);
7649        let payload = a_key_map_payload();
7650        attach(
7651            &path,
7652            "items",
7653            &[section::Attachment {
7654                kind: *section::KEY_MAP,
7655                id: 0,
7656                flags: 0,
7657                header_bytes: 0,
7658                bytes: &payload,
7659            }],
7660        )
7661        .expect("attach");
7662
7663        let reader = Reader::open(&path).expect("reopen");
7664        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
7665        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
7666        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
7667        drop(file);
7668
7669        let reader = Reader::open(&path).expect("the table still opens");
7670        let error = reader
7671            .payload(&reader.table().sections()[0])
7672            .expect_err("a corrupt payload is not handed out");
7673        assert!(error.to_string().contains("checksum"), "{error}");
7674        // And the table is still readable, which is section 3.1: a section that cannot be trusted
7675        // costs the query its shortcut and nothing else.
7676        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
7677
7678        fs::remove_file(&path).expect("clean up");
7679    }
7680
7681    #[test]
7682    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
7683        // Readable is not writable. A format 22 directory has no section block, and adding one
7684        // without moving the number in the header would leave a file claiming a format it is not.
7685        let path = linked_file("attach_old_format", 8);
7686        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7687        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7688        drop(file);
7689
7690        let payload = a_key_map_payload();
7691        let error = attach(
7692            &path,
7693            "items",
7694            &[section::Attachment {
7695                kind: *section::KEY_MAP,
7696                id: 0,
7697                flags: 0,
7698                header_bytes: 0,
7699                bytes: &payload,
7700            }],
7701        )
7702        .expect_err("format 22 cannot gain a section");
7703        assert!(error.to_string().contains("format 22"), "{error}");
7704        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
7705
7706        fs::remove_file(&path).expect("clean up");
7707    }
7708
7709    #[test]
7710    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
7711        let path = linked_file("attach_bad_header", 8);
7712        let error = attach(
7713            &path,
7714            "items",
7715            &[section::Attachment {
7716                kind: *section::KEY_MAP,
7717                id: 0,
7718                flags: 0,
7719                header_bytes: 40,
7720                bytes: &[1, 2, 3],
7721            }],
7722        )
7723        .expect_err("a writer's bug stops at the write");
7724        assert!(error.to_string().contains("header is longer"), "{error}");
7725
7726        fs::remove_file(&path).expect("clean up");
7727    }
7728
7729    #[test]
7730    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
7731        let path = linked_file("attach_wrong_name", 8);
7732        let error = attach(&path, "orders", &[]).expect_err("no such table");
7733        assert!(error.to_string().contains("orders"), "{error}");
7734        fs::remove_file(&path).expect("clean up");
7735    }
7736
7737    #[test]
7738    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
7739        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
7740        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
7741        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
7742        // the tail is outside it. The counts inside it are still exact, because the pass recounts
7743        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
7744        // twenty six a distinct count of 601 would divide its way to.
7745        let path = path("frequency_prefix_for_the_planner");
7746        let mut writer =
7747            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7748                .expect("new file");
7749        let mut values = vec![Value::Integer(1); 10_000];
7750        for _ in 0..10 {
7751            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
7752        }
7753        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
7754        // synopsis walks the whole column rather than a part, so the counts are the same either way.
7755        for part in values.chunks(8_000) {
7756            let rows = Chunk::new(vec![
7757                Vector::from_values(LogicalType::Integer, part).expect("integers"),
7758            ])
7759            .expect("one column");
7760            writer.append(&rows).expect("a part");
7761        }
7762        writer.finish().expect("commit");
7763        let reader = Reader::open(&path).expect("reopen from disk");
7764        let prefix =
7765            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
7766        // A prefix and not the whole column, and the writer said how many rows anything left out of
7767        // it can hold.
7768        assert_eq!(prefix.entries.len(), 512);
7769        assert_eq!(prefix.omitted_max, 10);
7770        let common = Common::new(reader);
7771        assert_eq!(common.rows(), 16_000);
7772        let column = common.column("id").expect("the file has that column");
7773        assert_eq!(
7774            common.rows_with(column, &Bound::Int(1)),
7775            Stat::exact(10_000, Provenance::FrequencySynopsis)
7776        );
7777        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
7778        assert_eq!(
7779            common.rows_with(column, &Bound::Int(1_100)),
7780            Stat::exact(10, Provenance::FrequencySynopsis)
7781        );
7782        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
7783        // what a complete list would say, and the file holds ten rows of this one.
7784        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
7785        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
7786        // two apart, which is the whole of what it gives up.
7787        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
7788        // What the prefix left out, which is what turns the unknown above into a number. The 512
7789        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
7790        // and 890 over 89 is the ten rows each of them really holds.
7791        let remainder = common.remainder(column).expect("the list is a prefix");
7792        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
7793        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
7794        fs::remove_file(&path).expect("clean up");
7795    }
7796
7797    /// A file with no table in it is a file, and opening it says so rather than failing.
7798    #[test]
7799    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
7800        let path = path("empty");
7801        Writer::empty(&path, &[]).expect("a file with nothing in it");
7802        let catalog = Catalog::open(&path).expect("the empty file opens");
7803        assert_eq!(catalog.len(), 0);
7804        assert!(catalog.is_empty());
7805        assert_eq!(catalog.names().count(), 0);
7806        // The next generation goes over the top of it the way it goes over any other, which is what
7807        // says this is a committed file and not a special case somebody has to know about.
7808        let mut writer =
7809            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7810                .expect("a table goes into the empty file");
7811        writer.append(&sample_ids()).expect("rows");
7812        writer.finish().expect("commit");
7813        let catalog = Catalog::open(&path).expect("the file opens again");
7814        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7815        fs::remove_file(&path).expect("clean up");
7816    }
7817
7818    /// A committed table with no rows is a name the next generation takes over, and one with rows
7819    /// is a name it refuses.
7820    ///
7821    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
7822    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
7823    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
7824    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
7825    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
7826    /// instead of through memory.
7827    #[test]
7828    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
7829        let path = path("empty-name");
7830        let field = || vec![Field::required("id", LogicalType::Integer)];
7831        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
7832        let catalog = Catalog::open(&path).expect("the file opens");
7833        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
7834
7835        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
7836        writer.append(&sample_ids()).expect("rows");
7837        writer.finish().expect("commit");
7838        let catalog = Catalog::open(&path).expect("the file opens again");
7839        // One entry and not two. The generation replaced the empty table rather than joining it.
7840        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7841        let held = catalog.rows().collect::<Vec<_>>();
7842        assert_eq!(held.len(), 1);
7843        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
7844
7845        // The same call against the same name now that it holds rows, which is still refused.
7846        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
7847        assert!(error.to_string().contains("same name"), "{error}");
7848        fs::remove_file(&path).expect("clean up");
7849    }
7850
7851    /// A view, with everything about it that a reopened catalog has to be able to answer from.
7852    fn sample_view(name: &str) -> ViewEntry {
7853        ViewEntry {
7854            name: name.to_string(),
7855            sql: "SELECT id FROM items WHERE id > 0".to_string(),
7856            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
7857            aliases: vec!["n".to_string()],
7858            columns: vec![Field::new("n", LogicalType::Integer)],
7859        }
7860    }
7861
7862    #[test]
7863    fn a_view_written_into_the_catalog_comes_back_whole() {
7864        let path = path("views");
7865        let mut writer =
7866            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7867                .expect("new file");
7868        writer.append(&sample_ids()).expect("rows");
7869        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
7870        let catalog = Catalog::open(&path).expect("reopen");
7871        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
7872        // The tables are still there and are still read the same way, so the section on the end did
7873        // not move anything in front of it.
7874        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7875        fs::remove_file(&path).expect("clean up");
7876    }
7877
7878    /// A writer opened to append a table says nothing about views and must not lose them.
7879    #[test]
7880    fn appending_a_table_carries_the_views_forward() {
7881        let path = path("viewscarry");
7882        let mut writer =
7883            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7884                .expect("new file");
7885        writer.append(&sample_ids()).expect("rows");
7886        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
7887        let mut writer =
7888            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
7889                .expect("a second table");
7890        writer.append(&sample_ids()).expect("rows");
7891        writer.finish().expect("commit");
7892        let catalog = Catalog::open(&path).expect("reopen");
7893        assert_eq!(catalog.views().count(), 1);
7894        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
7895        fs::remove_file(&path).expect("clean up");
7896    }
7897
7898    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
7899    #[test]
7900    fn restating_the_views_leaves_every_table_where_it_was() {
7901        let path = path("restate");
7902        let mut writer =
7903            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7904                .expect("new file");
7905        writer.append(&sample_ids()).expect("rows");
7906        writer.finish().expect("commit");
7907        let before = fs::metadata(&path).expect("the file is there").len();
7908        Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
7909        let catalog = Catalog::open(&path).expect("reopen");
7910        assert_eq!(catalog.views().count(), 2);
7911        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7912        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
7913        // than the size of the table.
7914        let after = fs::metadata(&path).expect("the file is there").len();
7915        assert!(after > before, "a generation was written");
7916        assert!(after - before < before, "the table was not written again");
7917        // The rows are still readable through the new generation, which is the part that would go
7918        // wrong if the catalog carried the wrong directory pointers forward.
7919        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
7920        assert_eq!(reader.table().rows, 3);
7921        // And a restate over a restate keeps working, because each one reads the slot that
7922        // checksummed rather than the highest number in the header.
7923        Writer::restate(&path, &[]).expect("no views at all");
7924        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
7925        fs::remove_file(&path).expect("clean up");
7926    }
7927
7928    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
7929    #[test]
7930    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
7931        let bytes = encode_catalog(
7932            &[Entry {
7933                name: "items".to_string(),
7934                fields: vec![Field::required("id", LogicalType::Integer)],
7935                rows: 1,
7936                directory: Page { offset: HEADER, length: 8, hash: 0 },
7937            }],
7938            &[sample_view("items")],
7939        )
7940        .expect("it encodes, because encoding does not look");
7941        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
7942        assert!(error.to_string().contains("same name"), "{error}");
7943    }
7944
7945    #[test]
7946    fn committed_file_reopens_and_reads_only_requested_columns() {
7947        let path = path("reopen");
7948        let mut writer = Writer::create(
7949            &path,
7950            "items",
7951            vec![
7952                Field::required("id", LogicalType::Integer),
7953                Field::new("text", LogicalType::Varchar),
7954            ],
7955        )
7956        .expect("new file");
7957        writer.append(&sample()).expect("first part");
7958        writer.append(&sample()).expect("second part");
7959        writer.finish().expect("commit");
7960        let reader = Reader::open(&path).expect("reopen from disk");
7961        assert_eq!(reader.table().rows(), 6);
7962        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
7963        // of the split: the directory describes the stripe and the scan still reads a part.
7964        assert_eq!(reader.table().stripes().len(), 1);
7965        assert_eq!(reader.parts(), 2);
7966        assert_eq!(reader.part_rows(0), 3);
7967        assert_eq!(reader.part_rows(1), 3);
7968        let text = reader.read(1, &[1]).expect("only text page");
7969        assert_eq!(text.width(), 1);
7970        assert_eq!(text.value_at(1, 0), Value::Null);
7971        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7972        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
7973        assert_eq!(sparse.width(), 1);
7974        assert_eq!(sparse.value_at(1, 0), Value::Null);
7975        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7976        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
7977        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
7978        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
7979        let count = reader.read(0, &[]).expect("no page is needed for count");
7980        assert_eq!(count.len(), 3);
7981        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
7982        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
7983        let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
7984        assert_eq!(
7985            integers,
7986            vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
7987        );
7988        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
7989        assert_eq!(strings.len(), 3);
7990        assert!(strings.contains(&(Value::Null, 2)));
7991        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
7992        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
7993        fs::remove_file(path).expect("remove scratch file");
7994    }
7995
7996    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
7997    /// instance.
7998    ///
7999    /// The runs arrive in the order the instances finished reading them rather than in source
8000    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
8001    /// a stripe of its own and the table still reads back in source order, which is the whole of
8002    /// what the writer promises about ordering.
8003    #[test]
8004    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
8005        let path = path("interleaved-runs");
8006        let mut writer =
8007            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
8008                .expect("new file");
8009        for morsel in [2_u64, 0, 3, 1] {
8010            let parts = (0..4_u64)
8011                .map(|chunk| {
8012                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
8013                    let values =
8014                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
8015                    let column =
8016                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
8017                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
8018                })
8019                .collect::<Vec<_>>();
8020            writer.append_stripe(parts).expect("a stripe");
8021        }
8022        writer.finish().expect("commit");
8023
8024        let reader = Reader::open(&path).expect("valid directory");
8025        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
8026        assert_eq!(reader.table().rows(), 128);
8027        for part in 0..16_usize {
8028            let read = reader.read(part, &[0]).expect("a part back");
8029            for row in 0..8_usize {
8030                let want = i64::try_from(part * 8 + row).expect("small");
8031                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
8032            }
8033        }
8034        fs::remove_file(path).expect("remove scratch file");
8035    }
8036
8037    /// Runs from different callers may interleave and may not overlap, and the commit is what
8038    /// catches an overlap.
8039    #[test]
8040    fn runs_that_overlap_each_other_are_refused_at_commit() {
8041        let path = path("overlapping-runs");
8042        let mut writer =
8043            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
8044                .expect("new file");
8045        let one = |order: (u64, u64)| {
8046            let column =
8047                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
8048            (order, Chunk::new(vec![column]).expect("one column"))
8049        };
8050        // The second run sits inside the first rather than after it, which is a thing no instance
8051        // holding its own contiguous run can produce and a thing the file cannot represent.
8052        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
8053        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
8054        let error = writer.finish().expect_err("the runs overlap");
8055        assert!(error.message().contains("source order"), "{error}");
8056        fs::remove_file(path).expect("remove scratch file");
8057    }
8058
8059    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
8060    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
8061    #[test]
8062    fn a_run_longer_than_a_stripe_is_refused() {
8063        let path = path("overlong-run");
8064        let mut writer =
8065            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
8066                .expect("new file");
8067        let parts = (0..=STRIPE_PARTS)
8068            .map(|at| {
8069                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
8070                    .expect("a column");
8071                let chunk = Chunk::new(vec![column]).expect("one column");
8072                ((0, u64::try_from(at).expect("small")), chunk)
8073            })
8074            .collect::<Vec<_>>();
8075        let error = writer.append_stripe(parts).expect_err("one part too many");
8076        assert!(error.message().contains("more parts than it holds"), "{error}");
8077        fs::remove_file(path).expect("remove scratch file");
8078    }
8079
8080    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
8081    ///
8082    /// This is the shape the format exists for, so both ends of the split are checked here. The
8083    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
8084    /// part still answers with that part's rows rather than with its whole stripe's.
8085    #[test]
8086    fn parts_past_the_stripe_bound_start_a_new_stripe() {
8087        let path = path("stripe-bound");
8088        let mut writer = Writer::create(
8089            &path,
8090            "items",
8091            vec![
8092                Field::required("id", LogicalType::Integer),
8093                Field::new("text", LogicalType::Varchar),
8094            ],
8095        )
8096        .expect("new file");
8097        let parts = STRIPE_PARTS * 2 + 3;
8098        for part in 0..parts {
8099            let id = part as i32;
8100            let chunk = Chunk::new(vec![
8101                Vector::from_values(
8102                    LogicalType::Integer,
8103                    &[Value::Integer(id), Value::Integer(-id)],
8104                )
8105                .expect("integers"),
8106                Vector::from_values(
8107                    LogicalType::Varchar,
8108                    &[Value::Varchar(format!("value {part}")), Value::Null],
8109                )
8110                .expect("strings"),
8111            ])
8112            .expect("matching rows");
8113            writer.append(&chunk).expect("one part");
8114        }
8115        writer.finish().expect("commit");
8116
8117        let reader = Reader::open(&path).expect("reopen from disk");
8118        assert_eq!(reader.parts(), parts);
8119        assert_eq!(reader.table().rows(), parts * 2);
8120        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
8121        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
8122        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
8123        assert_eq!(reader.table().stripes()[2].parts(), 3);
8124        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
8125        // table the other way is what catches a cache that only ever holds what it just read.
8126        for part in (0..parts).rev() {
8127            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
8128            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
8129            for chunk in [&dense, &sparse] {
8130                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
8131                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8132                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
8133                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
8134                assert_eq!(chunk.value_at(1, 1), Value::Null);
8135            }
8136        }
8137        // The bounds are merged over the stripe, so they answer for the range the whole stripe
8138        // covers and not for the part that was asked about.
8139        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
8140        assert!(reader.skips(0, &above), "the first stripe stops at 63");
8141        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
8142        fs::remove_file(path).expect("remove scratch file");
8143    }
8144
8145    /// A scattered value in the column that decides `WHERE UserID = ?`.
8146    fn scattered(n: i64) -> i64 {
8147        n.wrapping_mul(-7_046_029_254_386_353_131)
8148    }
8149
8150    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
8151    ///
8152    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
8153    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
8154    /// holds the value is the only one a scan has to read.
8155    #[test]
8156    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
8157        let path = path("sieve-skip");
8158        let mut writer =
8159            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
8160                .expect("new file");
8161        let parts = STRIPE_PARTS + 3;
8162        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
8163        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
8164        // that small costs about as much to read as the rows do and is no longer written.
8165        let per_part = 128;
8166        for part in 0..parts {
8167            let held: Vec<Value> = (0..per_part)
8168                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
8169                .collect();
8170            let chunk =
8171                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8172                    .expect("one column");
8173            writer.append(&chunk).expect("one part");
8174        }
8175        writer.finish().expect("commit");
8176
8177        let reader = Reader::open(&path).expect("reopen from disk");
8178        let probe = |value: i64| Probe {
8179            column: 0,
8180            op: Op::Equal,
8181            value: Bound::Int(i128::from(scattered(value))),
8182        };
8183        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
8184            let tests = [probe(wanted)];
8185            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
8186            let home = wanted as usize / per_part;
8187            assert!(kept.contains(&home), "the part holding {wanted} is read");
8188            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
8189            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
8190            // stray part across the whole file and that is what this leaves room for.
8191            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
8192        }
8193        let absent = [probe((parts * per_part) as i64 + 1)];
8194        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
8195        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
8196        // The same probes against the bounds alone, which is what this replaces. A column of
8197        // scattered numbers has a range per stripe that covers nearly the whole type.
8198        let tests = [probe(0)];
8199        assert!(
8200            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
8201            "the bounds rule out no stripe at all"
8202        );
8203        fs::remove_file(path).expect("remove scratch file");
8204    }
8205
8206    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
8207    ///
8208    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
8209    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
8210    /// rules out none of it and rules out all but a few parts.
8211    #[test]
8212    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
8213        let path = path("part-range-skip");
8214        let mut writer =
8215            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8216                .expect("new file");
8217        let parts = STRIPE_PARTS + 3;
8218        let per_part = 128;
8219        for part in 0..parts {
8220            // Scattered inside the part's own band rather than a run, because a run of
8221            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
8222            // costs more than reading the column it indexes, which is the case the writer declines.
8223            let held: Vec<Value> = (0..per_part)
8224                .map(|row| {
8225                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
8226                })
8227                .collect();
8228            let chunk =
8229                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8230                    .expect("one column");
8231            writer.append(&chunk).expect("one part");
8232        }
8233        writer.finish().expect("commit");
8234
8235        let reader = Reader::open(&path).expect("reopen from disk");
8236        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
8237        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
8238        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
8239        // The same question asked of the stripe alone, which is what this replaces.
8240        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
8241        fs::remove_file(path).expect("remove scratch file");
8242    }
8243
8244    /// The other half of the same page. A part whose own bounds put every row of it inside the
8245    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
8246    /// across every part and can prove nothing.
8247    #[test]
8248    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
8249        let path = path("part-range-certain");
8250        let mut writer =
8251            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8252                .expect("new file");
8253        let parts = STRIPE_PARTS + 3;
8254        let per_part = 128;
8255        for part in 0..parts {
8256            let held: Vec<Value> = (0..per_part)
8257                .map(|row| {
8258                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
8259                })
8260                .collect();
8261            let chunk =
8262                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8263                    .expect("one column");
8264            writer.append(&chunk).expect("one part");
8265        }
8266        writer.finish().expect("commit");
8267
8268        let reader = Reader::open(&path).expect("reopen from disk");
8269        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
8270        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
8271        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
8272        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
8273        // and settles nothing either way. The three yeses above are the parts' own ends talking.
8274        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
8275        fs::remove_file(path).expect("remove scratch file");
8276    }
8277
8278    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
8279    /// that has a single part, where the stripe bounds already are the part's.
8280    #[test]
8281    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
8282        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
8283            let path = path("part-range-page");
8284            let mut writer =
8285                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8286                    .expect("new file");
8287            for part in 0..parts {
8288                let held: Vec<Value> = (0..128)
8289                    .map(|row| {
8290                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
8291                    })
8292                    .collect();
8293                let chunk = Chunk::new(vec![
8294                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
8295                ])
8296                .expect("one column");
8297                writer.append(&chunk).expect("one part");
8298            }
8299            writer.finish().expect("commit");
8300            let reader = Reader::open(&path).expect("reopen from disk");
8301            let bytes = reader.layout().columns[0].part_ranges;
8302            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
8303            fs::remove_file(path).expect("remove scratch file");
8304        }
8305    }
8306
8307    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
8308    /// a shortened bound from turning a skip into a wrong answer.
8309    #[test]
8310    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
8311        let long = vec![b'a'; PART_BOUND_BYTES * 2];
8312        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
8313        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
8314        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
8315        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
8316        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
8317        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
8318        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
8319    }
8320
8321    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
8322    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
8323    #[test]
8324    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
8325        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
8326        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
8327        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
8328        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
8329    }
8330
8331    /// What a column is stored as, asked of two files holding the same rows in a different order.
8332    ///
8333    /// This is the question the report exists to answer and it is the one the directory cannot. The
8334    /// two files have the same rows, the same schema and the same number of parts, and the column
8335    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
8336    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
8337    /// says so, and reading it is what this does.
8338    ///
8339    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
8340    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
8341    /// pays for the wider ones.
8342    #[test]
8343    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
8344        let parts = 4;
8345        let per_part = 1024;
8346        let rows = parts * per_part;
8347        let written = |name: &str, keys: &[i64]| {
8348            let path = path(name);
8349            let fields = vec![Field::required("key", LogicalType::BigInt)];
8350            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
8351            for part in 0..parts {
8352                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
8353                    .iter()
8354                    .map(|key| Value::BigInt(*key))
8355                    .collect();
8356                let chunk = Chunk::new(vec![
8357                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
8358                ])
8359                .expect("one column");
8360                writer.append(&chunk).expect("one part");
8361            }
8362            writer.finish().expect("commit");
8363            path
8364        };
8365        // Ascending with a small irregular step, which is what a key column in arrival order looks
8366        // like: an order has one to seven line items, so the key repeats and then moves on by one.
8367        let climbing = |step: &dyn Fn(usize) -> i64| {
8368            let mut key = 0;
8369            (0..rows)
8370                .map(|row| {
8371                    key += step(row);
8372                    key
8373                })
8374                .collect::<Vec<i64>>()
8375        };
8376        let ascending = climbing(&|row| (row % 3) as i64);
8377        // The same rows in the same direction over a range a thousand times wider, which is what a
8378        // partition of a clustered table holds: still ascending, and far enough apart that the
8379        // deltas no longer fit in a handful of bits.
8380        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
8381        let near_path = written("stored-near", &ascending);
8382        let far_path = written("stored-far", &sparse);
8383
8384        let one = Reader::open(&near_path).expect("reopen from disk");
8385        let other = Reader::open(&far_path).expect("reopen from disk");
8386        let near = one.stored(0).expect("the column is stored");
8387        let far = other.stored(0).expect("the column is stored");
8388        assert_eq!(near.len(), parts, "one row per part");
8389        assert_eq!(far.len(), parts);
8390        // The bytes are the same bytes the directory totals, which is the check that this is
8391        // reading the pages the file really holds rather than some other pages.
8392        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
8393        assert_eq!(total(&near), one.layout().columns[0].pages);
8394        assert_eq!(total(&far), other.layout().columns[0].pages);
8395        assert!(
8396            total(&near) * 2 < total(&far),
8397            "the sparse keys cost more, {} against {}",
8398            total(&far),
8399            total(&near)
8400        );
8401        // Every part accounted for, in order, with the row it starts at following the one before.
8402        for (at, part) in near.iter().enumerate() {
8403            assert_eq!(part.part, at);
8404            assert_eq!(part.row, at * per_part);
8405            assert_eq!(part.rows, per_part);
8406            let held = &ascending[at * per_part..(at + 1) * per_part];
8407            assert_eq!(part.low, Some(Value::BigInt(held[0])));
8408            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
8409            assert_eq!(part.nulls, Some(0));
8410        }
8411        // And the encoding is a line of text that names what the encoder chose, which is the whole
8412        // point. Both are a cascade over deltas and the widths inside them are what differ.
8413        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
8414        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
8415        assert_ne!(near[0].encoding, far[0].encoding);
8416        fs::remove_file(near_path).expect("remove scratch file");
8417        fs::remove_file(far_path).expect("remove scratch file");
8418    }
8419
8420    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
8421    ///
8422    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
8423    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
8424    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
8425    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
8426    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
8427    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
8428    /// the part, every time, and that is the case this drops.
8429    #[test]
8430    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
8431        let path = path("sieve-pays");
8432        let fields = vec![
8433            Field::required("spread", LogicalType::BigInt),
8434            Field::required("repeated", LogicalType::BigInt),
8435        ];
8436        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
8437        let parts = 3;
8438        let per_part = 1024;
8439        for part in 0..parts {
8440            let base = (part * per_part) as i64;
8441            let spread: Vec<Value> =
8442                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
8443            let repeated: Vec<Value> =
8444                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
8445            let chunk = Chunk::new(vec![
8446                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
8447                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
8448            ])
8449            .expect("two columns");
8450            writer.append(&chunk).expect("one part");
8451        }
8452        writer.finish().expect("commit");
8453
8454        let reader = Reader::open(&path).expect("reopen from disk");
8455        let layout = reader.layout();
8456        let spread = &layout.columns[0];
8457        let repeated = &layout.columns[1];
8458        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
8459        assert_eq!(
8460            repeated.sieves, 0,
8461            "a column whose filter costs more than its parts keeps none"
8462        );
8463        // Per part this is the rule itself, so it holds over the column as well: a part without a
8464        // sieve adds to one side of this and to nothing on the other.
8465        for column in &layout.columns {
8466            assert!(
8467                column.sieves < column.pages,
8468                "{} spends {} on sieves over {} of data",
8469                column.name,
8470                column.sieves,
8471                column.pages
8472            );
8473        }
8474        // The filter that was kept still does what it is for.
8475        let absent = [Probe {
8476            column: 0,
8477            op: Op::Equal,
8478            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
8479        }];
8480        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
8481        fs::remove_file(path).expect("remove scratch file");
8482    }
8483
8484    /// A damaged sieve page is a part that gets read, not a query that fails.
8485    ///
8486    /// A sieve is an index over rows that are still there and still correct, so losing one costs
8487    /// time and costs no answers. That is the opposite of the membership index beside it, which is
8488    /// the only thing standing between a string page and a wrong answer.
8489    #[test]
8490    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
8491        let path = path("sieve-damaged");
8492        let mut writer =
8493            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
8494                .expect("new file");
8495        let rows = 128;
8496        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
8497        let chunk =
8498            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8499                .expect("one column");
8500        writer.append(&chunk).expect("one part");
8501        writer.finish().expect("commit");
8502
8503        let page =
8504            Reader::open(&path).expect("reopen").table.stripes[0].sieves[0].expect("a sieve page");
8505        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
8506        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
8507        file.write_all(&[0xff]).expect("damage one byte");
8508        drop(file);
8509
8510        let reader = Reader::open(&path).expect("reopen the damaged file");
8511        let absent =
8512            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
8513        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
8514        assert_eq!(
8515            reader.read(0, &[0]).expect("the rows are untouched").len(),
8516            usize::try_from(rows).expect("a small count")
8517        );
8518        fs::remove_file(path).expect("remove scratch file");
8519    }
8520
8521    /// Eight workers over one stripe read it once between them.
8522    ///
8523    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
8524    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
8525    /// started sharing the read every one of them read the whole page. On the full ClickBench file
8526    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
8527    /// column, which is most of what a first touch costs.
8528    ///
8529    /// The workers that lose the race still answer, out of the part reads they do instead, which is
8530    /// what the values below are checking.
8531    #[test]
8532    fn workers_that_want_the_same_stripe_read_it_once() {
8533        let path = path("single-flight");
8534        let mut writer =
8535            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8536                .expect("new file");
8537        for part in 0..STRIPE_PARTS {
8538            let id = part as i32;
8539            let chunk = Chunk::new(vec![
8540                Vector::from_values(
8541                    LogicalType::Integer,
8542                    &[Value::Integer(id), Value::Integer(-id)],
8543                )
8544                .expect("integers"),
8545            ])
8546            .expect("matching rows");
8547            writer.append(&chunk).expect("one part");
8548        }
8549        writer.finish().expect("commit");
8550
8551        let reader = Reader::open(&path).expect("reopen from disk");
8552        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
8553        let barrier = std::sync::Barrier::new(8);
8554        std::thread::scope(|scope| {
8555            for worker in 0..8 {
8556                let reader = &reader;
8557                let barrier = &barrier;
8558                scope.spawn(move || {
8559                    barrier.wait();
8560                    for part in (worker..STRIPE_PARTS).step_by(8) {
8561                        let chunk = reader.read(part, &[0]).expect("a whole page read");
8562                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8563                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
8564                    }
8565                });
8566            }
8567        });
8568        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
8569        fs::remove_file(path).expect("remove scratch file");
8570    }
8571
8572    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
8573    ///
8574    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
8575    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
8576    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
8577    /// the next query will want them, so read them on the way past. A process that opened the
8578    /// database to run one trivial query pays for all of it and gets nothing.
8579    ///
8580    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
8581    /// two openings cost the same. The stripe count is held equal so that the directory is the same
8582    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
8583    /// data would show up here.
8584    #[test]
8585    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
8586        let opened = |label: &str, rows_per_part: i32| {
8587            let path = path(label);
8588            let mut writer =
8589                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8590                    .expect("new file");
8591            for part in 0..STRIPE_PARTS * 3 {
8592                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
8593                // of consecutive integers encodes to almost nothing and would leave the two files
8594                // the same size, which would make this test pass for the wrong reason.
8595                let values = (0..rows_per_part)
8596                    .map(|row| {
8597                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
8598                    })
8599                    .collect::<Vec<_>>();
8600                let chunk = Chunk::new(vec![
8601                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
8602                ])
8603                .expect("matching rows");
8604                writer.append(&chunk).expect("one part");
8605            }
8606            writer.finish().expect("commit");
8607            let reader = Reader::open(&path).expect("reopen from disk");
8608            let size = fs::metadata(&path).expect("the file is there").len();
8609            let out = (reader.reads(), reader.table().stripes().len(), size);
8610            fs::remove_file(path).expect("remove scratch file");
8611            out
8612        };
8613
8614        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
8615        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
8616        assert_eq!(
8617            thin_stripes, fat_stripes,
8618            "the same stripe count is what makes this a fair ask"
8619        );
8620        assert!(
8621            fat_size > thin_size * 50,
8622            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
8623        );
8624
8625        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
8626        assert_eq!(thin.pages, 0, "opening read a page");
8627        assert_eq!(fat.pages, 0, "opening read a page");
8628        assert_eq!(thin.indexes, 0, "opening read an index");
8629        assert_eq!(fat.indexes, 0, "opening read an index");
8630        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
8631        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
8632        assert!(
8633            fat.opening.bytes < thin.opening.bytes * 2,
8634            "opening the thin file read {} bytes and the fat one read {}",
8635            thin.opening.bytes,
8636            fat.opening.bytes
8637        );
8638    }
8639
8640    /// The reads a file costs to open are fixed by its shape and not by what ran before.
8641    ///
8642    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
8643    /// the plan is a function of the data, the generation and the settings, and never of what
8644    /// happened to be in cache. Opening the same file twice in the same process has to cost the
8645    /// same, because a second open that read less would be an open that was about to plan
8646    /// differently.
8647    #[test]
8648    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
8649        let path = path("open-twice");
8650        let mut writer =
8651            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8652                .expect("new file");
8653        for part in 0..STRIPE_PARTS * 3 {
8654            let chunk = Chunk::new(vec![
8655                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8656                    .expect("integers"),
8657            ])
8658            .expect("matching rows");
8659            writer.append(&chunk).expect("one part");
8660        }
8661        writer.finish().expect("commit");
8662
8663        let first = Reader::open(&path).expect("open");
8664        // A whole scan in between, so the operating system's page cache is as warm as it gets and
8665        // anything that consulted it would show up in the second open.
8666        for part in 0..first.parts() {
8667            first.read(part, &[0]).expect("a part");
8668        }
8669        assert!(first.reads().pages > 0, "the scan has to have read something");
8670        let second = Reader::open(&path).expect("open again");
8671
8672        assert_eq!(first.reads().opening, second.reads().opening);
8673        assert_eq!(
8674            second.reads().pages,
8675            0,
8676            "the second open read a page off the back of the first"
8677        );
8678        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
8679        fs::remove_file(path).expect("remove scratch file");
8680    }
8681
8682    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
8683    ///
8684    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
8685    /// stripes than that read the index again every time a stripe came back around. The index is a
8686    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
8687    /// different budgets. This is the test that keeps them there, since the saving is small enough
8688    /// that nothing in a benchmark would notice it going away again.
8689    #[test]
8690    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
8691        let path = path("index-cache");
8692        let mut writer =
8693            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8694                .expect("new file");
8695        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
8696        for part in 0..parts {
8697            let id = part as i32;
8698            let chunk = Chunk::new(vec![
8699                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
8700            ])
8701            .expect("matching rows");
8702            writer.append(&chunk).expect("one part");
8703        }
8704        writer.finish().expect("commit");
8705
8706        let reader = Reader::open(&path).expect("reopen from disk");
8707        let stripes = reader.table().stripes().len();
8708        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
8709        // Twice over, so that the second pass finds every page evicted and every index kept.
8710        for _ in 0..2 {
8711            for part in 0..parts {
8712                let chunk = reader.read(part, &[0]).expect("a part");
8713                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8714            }
8715        }
8716        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
8717        assert!(
8718            reader.pages.load(Atomic::Relaxed) > stripes,
8719            "the pages are the ones that get read again, which is what makes the index count mean \
8720             something"
8721        );
8722        fs::remove_file(path).expect("remove scratch file");
8723    }
8724
8725    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
8726    ///
8727    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
8728    /// Nobody races for a page any more, but every worker holds a different one for the length of a
8729    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
8730    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
8731    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
8732    /// without it a worker can run a whole stripe before the next one starts and never collide.
8733    #[test]
8734    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
8735        let workers = CACHED_STRIPES_PER_COLUMN + 4;
8736        let path = path("stripe-per-worker");
8737        let mut writer =
8738            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8739                .expect("new file");
8740        for part in 0..STRIPE_PARTS * workers {
8741            let chunk = Chunk::new(vec![
8742                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8743                    .expect("integers"),
8744            ])
8745            .expect("matching rows");
8746            writer.append(&chunk).expect("one part");
8747        }
8748        writer.finish().expect("commit");
8749
8750        let read = |told: bool| {
8751            let reader = Reader::open(&path).expect("reopen from disk");
8752            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
8753            if told {
8754                reader.keep_stripes(workers);
8755            }
8756            let barrier = std::sync::Barrier::new(workers);
8757            std::thread::scope(|scope| {
8758                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
8759                    let reader = &reader;
8760                    let barrier = &barrier;
8761                    scope.spawn(move || {
8762                        for part in run {
8763                            barrier.wait();
8764                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
8765                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8766                        }
8767                        assert!(worker < workers);
8768                    });
8769                }
8770            });
8771            reader.pages.load(Atomic::Relaxed)
8772        };
8773
8774        assert_eq!(read(true), workers, "one page read per stripe and no more");
8775        assert!(read(false) > workers, "a cache that small is read again on every part");
8776        fs::remove_file(path).expect("remove scratch file");
8777    }
8778
8779    /// A damaged index page is caught before anything decodes a part out of it.
8780    ///
8781    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
8782    /// per column section rather than one for the page, and this is what says that check runs.
8783    #[test]
8784    fn a_damaged_index_page_is_an_error() {
8785        let path = path("damaged-index");
8786        let mut writer =
8787            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8788                .expect("new file");
8789        writer.append(&sample_ids()).expect("first part");
8790        writer.append(&sample_ids()).expect("second part");
8791        writer.finish().expect("commit");
8792
8793        let reader = Reader::open(&path).expect("valid directory");
8794        let index = reader.table.stripes[0].index;
8795        let mut byte = [0; 1];
8796        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
8797        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
8798        file.seek(SeekFrom::Start(index.offset)).expect("index start");
8799        file.write_all(&[!byte[0]]).expect("damage the first part length");
8800        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
8801        assert!(error.message().contains("index page section checksum differs"), "{error}");
8802        fs::remove_file(path).expect("remove scratch file");
8803    }
8804
8805    /// Every integer width the format knows about, written and read back.
8806    ///
8807    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
8808    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
8809    /// are in here on purpose, because a width that round trips through the wrong signedness only
8810    /// goes wrong at the end of its range.
8811    #[test]
8812    fn every_integer_width_round_trips_through_a_page() {
8813        let path = path("integer-widths");
8814        let columns = [
8815            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
8816            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
8817            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
8818            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
8819            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
8820            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
8821            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
8822            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
8823        ];
8824        let fields = columns
8825            .iter()
8826            .enumerate()
8827            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8828            .collect::<Vec<_>>();
8829        let vectors = columns
8830            .iter()
8831            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8832            .collect::<Vec<_>>();
8833        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
8834        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8835        writer.finish().expect("commit");
8836
8837        let reader = Reader::open(&path).expect("reopen from disk");
8838        let wanted = (0..columns.len()).collect::<Vec<_>>();
8839        let read = reader.read(0, &wanted).expect("every column");
8840        assert_eq!(read.len(), 2);
8841        // row at a time: each column has its own type and its own pair of extremes.
8842        for (at, (ty, values)) in columns.iter().enumerate() {
8843            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8844            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8845        }
8846        fs::remove_file(path).expect("remove scratch file");
8847    }
8848
8849    /// The rest of the fixed width types, and the byte strings, written and read back.
8850    ///
8851    /// The extremes again, and for a float that means more than the ends of the range. Negative
8852    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
8853    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
8854    /// `==`, which a NaN fails against itself.
8855    ///
8856    /// A blob is here beside them because it is the same round trip asked of bytes that are not
8857    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
8858    /// past turns this test red rather than turning a user's column into nulls.
8859    #[test]
8860    fn every_other_type_the_format_knows_round_trips_through_a_page() {
8861        let path = path("other-types");
8862        let columns = [
8863            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
8864            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
8865            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
8866            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
8867            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
8868            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
8869            (
8870                LogicalType::TimestampTz,
8871                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
8872            ),
8873            (
8874                LogicalType::Interval,
8875                vec![
8876                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
8877                    Value::Interval { months: 13, days: -1, micros: 1 },
8878                ],
8879            ),
8880            (
8881                LogicalType::Blob,
8882                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
8883            ),
8884        ];
8885        let fields = columns
8886            .iter()
8887            .enumerate()
8888            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8889            .collect::<Vec<_>>();
8890        let vectors = columns
8891            .iter()
8892            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8893            .collect::<Vec<_>>();
8894        let mut writer = Writer::create(&path, "others", fields).expect("new file");
8895        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8896        writer.finish().expect("commit");
8897
8898        let reader = Reader::open(&path).expect("reopen from disk");
8899        let wanted = (0..columns.len()).collect::<Vec<_>>();
8900        let read = reader.read(0, &wanted).expect("every column");
8901        assert_eq!(read.len(), 2);
8902        for (at, (ty, values)) in columns.iter().enumerate() {
8903            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8904            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8905        }
8906        // A float keeps its sign through a zero, which `==` says nothing about because negative
8907        // zero and zero compare equal.
8908        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
8909        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
8910
8911        fs::remove_file(path).expect("remove scratch file");
8912    }
8913
8914    /// A NaN is still a NaN after a trip through a page.
8915    ///
8916    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
8917    /// to itself, so a comparison against the value that was written passes for every NaN and for
8918    /// nothing else, which is the one assertion that would not catch a page that lost it.
8919    #[test]
8920    fn a_nan_survives_being_written_down() {
8921        let path = path("nan");
8922        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
8923            .expect("a NaN vector");
8924        let mut writer =
8925            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
8926                .expect("new file");
8927        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
8928        writer.finish().expect("commit");
8929        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
8930        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
8931        assert!(back.is_nan(), "a NaN came back as {back}");
8932        fs::remove_file(path).expect("remove scratch file");
8933    }
8934
8935    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
8936    ///
8937    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
8938    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
8939    /// whatever the file held. The data underneath is what the storage promise is about, so that is
8940    /// what this reads.
8941    #[test]
8942    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
8943        let path = path("uuid-and-bit");
8944        let uuids = vec![0_i128, i128::MIN, -1];
8945        let mut bits = StringColumn::new();
8946        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
8947            bits.push_bytes(value);
8948        }
8949        let expected = bits.clone();
8950        let fields =
8951            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
8952        let vectors = vec![
8953            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
8954            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
8955        ];
8956        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
8957        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8958        writer.finish().expect("commit");
8959
8960        let reader = Reader::open(&path).expect("reopen from disk");
8961        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
8962        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
8963            panic!("a uuid column is the 128 bit lane")
8964        };
8965        assert_eq!(back.as_slice(), uuids.as_slice());
8966        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
8967            panic!("a bit column is bytes")
8968        };
8969        for row in 0..expected.len() {
8970            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
8971        }
8972        fs::remove_file(path).expect("remove scratch file");
8973    }
8974
8975    #[test]
8976    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
8977        let path = path("frequency-ordinals");
8978        let mut writer =
8979            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
8980                .expect("new file");
8981        let mut values = Vec::new();
8982        for leader in 0..10_i64 {
8983            values.extend(std::iter::repeat_n(leader, 100));
8984        }
8985        values.extend(1_000_i64..41_000);
8986        for part in values.chunks(1_024) {
8987            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
8988                .expect("big integers");
8989            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
8990        }
8991        writer.finish().expect("commit");
8992
8993        let reader = Reader::open(&path).expect("reopen from disk");
8994        let occurrences =
8995            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
8996        assert!(occurrences.omitted_max < 100);
8997        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
8998        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
8999        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
9000        fs::remove_file(path).expect("remove scratch file");
9001    }
9002
9003    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
9004    /// format went from 11 to 12, every binary built after that said "magic or major version is
9005    /// unsupported" about the file, and there was no way to tell from the message whether the path
9006    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
9007    /// wants is the whole answer and it was the one thing the message did not carry.
9008    #[test]
9009    fn a_file_from_another_format_says_which_format_it_is() {
9010        let older = path("older-format");
9011        let mut writer =
9012            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
9013                .expect("new file");
9014        let chunk = Chunk::new(vec![
9015            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
9016                .expect("integers"),
9017        ])
9018        .expect("chunk");
9019        writer.append(&chunk).expect("page written");
9020        writer.finish().expect("commit");
9021
9022        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
9023        // more than one member now: format 22 is deliberately still readable, so the version that
9024        // has to be refused is the one under the oldest one accepted.
9025        let unreadable =
9026            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
9027        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
9028        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
9029        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
9030        drop(file);
9031        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
9032        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
9033        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
9034
9035        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
9036        file.seek(SeekFrom::Start(0)).expect("the magic is first");
9037        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
9038        drop(file);
9039        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
9040        assert!(complaint.contains("magic"), "{complaint}");
9041        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
9042        fs::remove_file(older).expect("remove scratch file");
9043    }
9044
9045    #[test]
9046    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
9047        let unfinished = path("unfinished");
9048        let mut writer =
9049            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
9050                .expect("new file");
9051        let chunk = Chunk::new(vec![
9052            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
9053                .expect("integers"),
9054        ])
9055        .expect("chunk");
9056        writer.append(&chunk).expect("page written");
9057        drop(writer);
9058        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
9059        fs::remove_file(unfinished).expect("remove scratch file");
9060
9061        let damaged = path("damaged");
9062        let mut writer =
9063            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
9064                .expect("new file");
9065        writer.append(&chunk).expect("page written");
9066        writer.finish().expect("commit");
9067        let reader = Reader::open(&damaged).expect("valid directory");
9068        let mut file =
9069            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
9070        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
9071        file.write_all(&[255]).expect("damage one byte");
9072        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
9073        fs::remove_file(damaged).expect("remove scratch file");
9074    }
9075
9076    #[test]
9077    fn damaged_lazy_dictionary_payload_is_an_error() {
9078        let path = path("damaged-dictionary");
9079        let mut writer = Writer::create(
9080            &path,
9081            "items",
9082            vec![
9083                Field::required("id", LogicalType::Integer),
9084                Field::new("text", LogicalType::Varchar),
9085            ],
9086        )
9087        .expect("new file");
9088        writer.append(&sample()).expect("stripe written");
9089        writer.finish().expect("commit");
9090
9091        let reader = Reader::open(&path).expect("valid directory");
9092        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
9093        // Read the count out of the page rather than writing it here, so that adding something
9094        // else to the index does not silently turn this into a test that damages the index.
9095        let mut header = [0; DICTIONARY_HEADER];
9096        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
9097        let index_len = dictionary_index_len(&header);
9098        let rank_len = last_rank_end(&reader.file, dictionary.offset, &header);
9099        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9100        file.seek(SeekFrom::Start(dictionary.offset + index_len + rank_len))
9101            .expect("inside dictionary payload");
9102        file.write_all(&[255]).expect("damage dictionary payload");
9103
9104        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
9105        let error =
9106            chunk.validate_external().expect_err("payload corruption must reach the caller");
9107        assert!(error.message().contains("payload checksum differs"), "{error}");
9108        fs::remove_file(path).expect("remove scratch file");
9109    }
9110
9111    /// A column whose values are all different is written without a dictionary, and one whose
9112    /// values repeat keeps it.
9113    ///
9114    /// The two columns go in the same table and hold the same number of rows, so the only thing
9115    /// separating them is how much of the first stripe was a value it had not seen before. Both have
9116    /// to read back the values that were written, because the decision is about cost and nothing
9117    /// else. The file size is the other half of it: a column written without a dictionary goes
9118    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
9119    /// column raw.
9120    #[test]
9121    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
9122        let path = path("dictionary-decide");
9123        let rows = 20_000;
9124        // Long enough that storing it raw would show, and different in every row.
9125        let unique =
9126            |row: usize| format!("{row:09} a value that appears exactly once in the table");
9127        // The same values in the same shape, each one used forty times over.
9128        let repeated = |row: usize| unique(row / 40);
9129        let mut writer = Writer::create(
9130            &path,
9131            "items",
9132            vec![
9133                Field::required("unique", LogicalType::Varchar),
9134                Field::required("repeated", LogicalType::Varchar),
9135            ],
9136        )
9137        .expect("new file");
9138        for part in (0..rows).step_by(1_000) {
9139            let span = part..(part + 1_000).min(rows);
9140            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
9141            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
9142            writer
9143                .append(
9144                    &Chunk::new(vec![
9145                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
9146                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
9147                    ])
9148                    .expect("two columns"),
9149                )
9150                .expect("a part");
9151        }
9152        writer.finish().expect("commit");
9153
9154        let reader = Reader::open(&path).expect("reopen from disk");
9155        assert!(
9156            reader.table.dictionaries[0].is_none(),
9157            "a column with no repeats has nothing to say twice"
9158        );
9159        assert!(
9160            reader.table.dictionaries[1].is_some(),
9161            "a column whose values come round again keeps its dictionary"
9162        );
9163        let mut first = 0;
9164        for part in 0..reader.parts() {
9165            let chunk = reader.read(part, &[0, 1]).expect("a part");
9166            for row in 0..chunk.len() {
9167                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
9168                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
9169            }
9170            first += chunk.len();
9171        }
9172        assert_eq!(first, rows, "every row was read back");
9173        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
9174        let size = fs::metadata(&path).expect("the file is there").len() as usize;
9175        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
9176        fs::remove_file(path).expect("remove scratch file");
9177    }
9178
9179    /// A payload of many blocks reads and checks every block of it.
9180    ///
9181    /// The test above has a dictionary of three values, which is one block, so it says nothing
9182    /// about a reader finding the right block among many. This one has thirty two thousand values,
9183    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
9184    /// the last and then damages the last and asks for it again.
9185    ///
9186    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
9187    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
9188    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
9189    /// The repeats are put at the front so that the values still arrive in order after them, which
9190    /// is what keeps the last part of the table on the last block of the payload.
9191    #[test]
9192    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
9193        let path = path("dictionary-blocks");
9194        let value = |row: usize| {
9195            let row = row.saturating_sub(8_000);
9196            format!("{row:07} a value long enough to be worth a payload block")
9197        };
9198        let parts = 40;
9199        let per_part = 1000;
9200        let mut writer =
9201            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9202                .expect("new file");
9203        for part in 0..parts {
9204            let values = (0..per_part)
9205                .map(|row| Value::Varchar(value(part * per_part + row)))
9206                .collect::<Vec<_>>();
9207            let chunk = Chunk::new(vec![
9208                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
9209            ])
9210            .expect("matching rows");
9211            writer.append(&chunk).expect("a part");
9212        }
9213        writer.finish().expect("commit");
9214
9215        let reader = Reader::open(&path).expect("reopen from disk");
9216        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
9217        assert!(
9218            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
9219            "the dictionary has to be several blocks for this to be testing anything"
9220        );
9221        for part in [0, parts - 1] {
9222            let chunk = reader.read(part, &[0]).expect("a part");
9223            chunk.validate_external().expect("every payload block checks out");
9224            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
9225        }
9226
9227        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9228        file.seek(SeekFrom::Start(dictionary.offset + u64::from(dictionary.length) - 4))
9229            .expect("the last bytes of the page are payload");
9230        file.write_all(&[255]).expect("damage the last payload block");
9231        let reader = Reader::open(&path).expect("the directory and the index are untouched");
9232        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
9233        let error = chunk.validate_external().expect_err("the damage must reach the caller");
9234        assert!(error.message().contains("payload checksum differs"), "{error}");
9235        fs::remove_file(path).expect("remove scratch file");
9236    }
9237
9238    /// Values of different lengths read back where the offsets say they do.
9239    ///
9240    /// The offsets are packed at one width for the column, they are relative to the payload block a
9241    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
9242    /// arithmetic could be off by one and neither shows up on values that are all the same length.
9243    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
9244    /// so the first value of a block, the last value of a run and the last value of a block are all
9245    /// covered several times over. An empty value is in the cycle because a zero length span is the
9246    /// case the reader short circuits.
9247    ///
9248    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
9249    /// distinct is written without a dictionary and then there are no packed offsets to be off by
9250    /// one in.
9251    #[test]
9252    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
9253        let path = path("dictionary-offsets");
9254        let value = |row: usize| {
9255            let row = row % 5_000;
9256            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
9257        };
9258        let rows = 6_000;
9259        let mut writer =
9260            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9261                .expect("new file");
9262        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
9263        for part in values.chunks(1_000) {
9264            let chunk =
9265                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
9266                    .expect("matching rows");
9267            writer.append(&chunk).expect("a part");
9268        }
9269        writer.finish().expect("commit");
9270
9271        let reader = Reader::open(&path).expect("reopen from disk");
9272        assert!(
9273            rows > TEXT_PAYLOAD_VALUES * 4,
9274            "the dictionary has to be several blocks for this to be testing anything"
9275        );
9276        for part in 0..rows / 1_000 {
9277            let chunk = reader.read(part, &[0]).expect("a part");
9278            for row in 0..1_000 {
9279                let row = part * 1_000 + row;
9280                assert_eq!(
9281                    chunk.value_at(row % 1_000, 0),
9282                    Value::Varchar(value(row)),
9283                    "value {row}"
9284                );
9285            }
9286        }
9287        fs::remove_file(path).expect("remove scratch file");
9288    }
9289
9290    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
9291    ///
9292    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
9293    /// the dictionary is asking and not the one a worker without it is asking, which is whether
9294    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
9295    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
9296    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
9297    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
9298    ///
9299    /// The barrier is what makes the test about that rather than about luck. Without it the first
9300    /// thread is usually finished before the last one starts and the count is one either way.
9301    #[test]
9302    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
9303        let path = path("dictionary-once");
9304        let parts = 8;
9305        let per_part = 500;
9306        let value =
9307            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
9308        let mut writer =
9309            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9310                .expect("new file");
9311        for part in 0..parts {
9312            let values = (0..per_part)
9313                .map(|row| Value::Varchar(value(part * per_part + row)))
9314                .collect::<Vec<_>>();
9315            let chunk = Chunk::new(vec![
9316                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
9317            ])
9318            .expect("matching rows");
9319            writer.append(&chunk).expect("a part");
9320        }
9321        writer.finish().expect("commit");
9322
9323        let reader = Reader::open(&path).expect("reopen from disk");
9324        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
9325        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
9326
9327        let workers = 16;
9328        let gate = std::sync::Barrier::new(workers);
9329        std::thread::scope(|scope| {
9330            for worker in 0..workers {
9331                let reader = reader.clone();
9332                let gate = &gate;
9333                scope.spawn(move || {
9334                    gate.wait();
9335                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
9336                    assert_eq!(
9337                        chunk.value_at(0, 0),
9338                        Value::Varchar(value((worker % parts) * per_part))
9339                    );
9340                });
9341            }
9342        });
9343
9344        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
9345        fs::remove_file(path).expect("remove scratch file");
9346    }
9347
9348    /// The sorted order sits outside the index the page checksum covers, because a query that
9349    /// never searches a dictionary should not read it, so it carries its own checksums and this is
9350    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
9351    /// rather than a slow one.
9352    #[test]
9353    fn a_damaged_sorted_order_is_an_error() {
9354        let path = path("damaged-order");
9355        let mut writer = Writer::create(
9356            &path,
9357            "items",
9358            vec![
9359                Field::required("id", LogicalType::Integer),
9360                Field::new("text", LogicalType::Varchar),
9361            ],
9362        )
9363        .expect("new file");
9364        writer.append(&sample()).expect("stripe written");
9365        writer.finish().expect("commit");
9366
9367        let reader = Reader::open(&path).expect("valid directory");
9368        let page = reader.table.dictionaries[1].expect("string dictionary page");
9369        let mut header = [0; DICTIONARY_HEADER];
9370        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
9371        let index_len = dictionary_index_len(&header);
9372        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9373        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
9374        file.write_all(&[255]).expect("damage the order");
9375
9376        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
9377        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
9378        assert!(error.message().contains("rank checksum differs"), "{error}");
9379        fs::remove_file(path).expect("remove scratch file");
9380    }
9381
9382    /// Codes stay in first appearance order and the sorted order is written beside them, so a
9383    /// reader can put the values back in order without the writer having had to know them all
9384    /// before it handed out the first code.
9385    #[test]
9386    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
9387        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
9388        // a nine byte prefix, one is a prefix of another, and one is empty.
9389        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
9390        let path = path("dictionary-order");
9391        let mut writer =
9392            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9393                .expect("new file");
9394        writer
9395            .append(
9396                &Chunk::new(vec![
9397                    Vector::from_values(
9398                        LogicalType::Varchar,
9399                        &spellings.map(|text| Value::Varchar(text.into())),
9400                    )
9401                    .expect("strings"),
9402                ])
9403                .expect("one column"),
9404            )
9405            .expect("stripe written");
9406        writer.finish().expect("commit");
9407
9408        let reader = Reader::open(&path).expect("valid directory");
9409        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9410        let count = dictionary.ranks().expect("a v10 file stores one");
9411        assert_eq!(count, spellings.len(), "every distinct value has a rank");
9412        let order = (0..count)
9413            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
9414            .collect::<Vec<_>>();
9415        let mut seen = order.clone();
9416        seen.sort_unstable();
9417        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
9418
9419        let ranked = order
9420            .iter()
9421            .map(|&code| {
9422                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
9423            })
9424            .collect::<Vec<_>>();
9425        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
9426        expected.sort();
9427        assert_eq!(ranked, expected, "rank order is value order");
9428
9429        // What a search asks, on the values themselves rather than through a kernel, so that a
9430        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
9431        for (rank, value) in expected.iter().enumerate() {
9432            assert_eq!(
9433                dictionary.compare_rank(rank, value).expect("compare"),
9434                Ordering::Equal,
9435                "rank {rank} is its own value"
9436            );
9437            if rank > 0 {
9438                assert_eq!(
9439                    dictionary.compare_rank(rank - 1, value).expect("compare"),
9440                    Ordering::Less,
9441                    "rank {rank} follows the one before it"
9442                );
9443            }
9444        }
9445        fs::remove_file(path).expect("remove scratch file");
9446    }
9447
9448    /// A sweep of the dictionary reads every value and keeps what it read, up to the budget.
9449    ///
9450    /// The point of the sweep is the resident size rather than the answer, so both are checked
9451    /// here. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so it keeps everything and
9452    /// a second sweep decodes nothing, which is what makes the second statement of a session asking
9453    /// the same question cost what it should. The ceiling is the other half of it and it has its own
9454    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
9455    #[test]
9456    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
9457        let path = path("dictionary-sweep");
9458        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
9459        // third, so the sweep has to be called more than once and the last call has to stop short.
9460        let spellings = (0..2_500)
9461            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9462            .collect::<Vec<_>>();
9463        let mut writer =
9464            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9465                .expect("new file");
9466        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
9467        // The dictionary is table wide and does not care where a value was written.
9468        for part in spellings.chunks(1_024) {
9469            writer
9470                .append(
9471                    &Chunk::new(vec![
9472                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9473                    ])
9474                    .expect("one column"),
9475                )
9476                .expect("stripe written");
9477        }
9478        writer.finish().expect("commit");
9479
9480        let reader = Reader::open(&path).expect("valid directory");
9481        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9482        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9483
9484        let resting = dictionary.footprint();
9485        let mut swept: Vec<Vec<u8>> = Vec::new();
9486        let mut at = 0;
9487        let mut calls = 0;
9488        while at < dictionary.len() {
9489            let stopped = dictionary
9490                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9491                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9492                    swept.push(text.to_vec());
9493                    Ok(())
9494                })
9495                .expect("a sweep reads");
9496            assert!(stopped > at, "a sweep moves");
9497            at = stopped;
9498            calls += 1;
9499        }
9500        assert_eq!(calls, 3, "a sweep hands over one block at a time");
9501        let after = dictionary.footprint();
9502        assert!(after > resting, "a sweep under the budget keeps what it decoded");
9503
9504        let read = (0..dictionary.len())
9505            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9506            .collect::<Vec<_>>();
9507        assert_eq!(swept, read, "a sweep answers what a point read answers");
9508        assert_eq!(dictionary.footprint(), after, "a point read of a kept block decodes nothing");
9509        fs::remove_file(path).expect("remove scratch file");
9510    }
9511
9512    /// A sweep over a block whose second run of offsets is short reads the same values as a point
9513    /// read does.
9514    ///
9515    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
9516    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
9517    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
9518    /// never puts a short run second in its block: the last block there begins on a run boundary and
9519    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
9520    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
9521    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
9522    #[test]
9523    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
9524        let path = path("dictionary-sweep-short-run");
9525        let spellings = (0..2_800)
9526            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9527            .collect::<Vec<_>>();
9528        let mut writer =
9529            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9530                .expect("new file");
9531        for part in spellings.chunks(1_024) {
9532            writer
9533                .append(
9534                    &Chunk::new(vec![
9535                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9536                    ])
9537                    .expect("one column"),
9538                )
9539                .expect("stripe written");
9540        }
9541        writer.finish().expect("commit");
9542
9543        let reader = Reader::open(&path).expect("valid directory");
9544        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9545        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9546        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
9547        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
9548        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
9549
9550        let mut swept: Vec<Vec<u8>> = Vec::new();
9551        let mut at = 0;
9552        while at < dictionary.len() {
9553            let stopped = dictionary
9554                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9555                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9556                    swept.push(text.to_vec());
9557                    Ok(())
9558                })
9559                .expect("a sweep reads");
9560            assert!(stopped > at, "a sweep moves");
9561            at = stopped;
9562        }
9563        let read = (0..dictionary.len())
9564            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9565            .collect::<Vec<_>>();
9566        assert_eq!(swept, read, "a sweep answers what a point read answers");
9567        fs::remove_file(path).expect("remove scratch file");
9568    }
9569
9570    /// Narrowing a page takes what fits and refuses the page for anything that does not.
9571    ///
9572    /// The edges of the range on both sides and one step past each of them, for every type, because
9573    /// checking a page separately from converting it is only right if the check refuses exactly what
9574    /// `TryFrom` would have refused, and off by one there is a file that reads back a different
9575    /// number than it was given. The check is a bit pattern rather than a comparison, so it is not
9576    /// the shape a reader would guess from the bounds, which is why all six are here. The empty page
9577    /// is here because a check written the obvious way starts with the extremes the wrong way round
9578    /// and refuses it.
9579    #[test]
9580    fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
9581        assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
9582        assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
9583        fit::<i8>(&[128]).expect_err("one past the top does not fit");
9584        fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
9585        assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
9586        fit::<u8>(&[256]).expect_err("one past the top does not fit");
9587        fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
9588        assert_eq!(
9589            fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
9590            vec![-32_768_i16, 0, 32_767]
9591        );
9592        fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
9593        fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
9594        assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
9595        fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
9596        fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
9597        assert_eq!(
9598            fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
9599            vec![i32::MIN, 0, i32::MAX]
9600        );
9601        fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
9602        fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
9603        assert_eq!(
9604            fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
9605            vec![0_u32, 4_294_967_295]
9606        );
9607        fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
9608        fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
9609
9610        // One value in a page that fits is still a page that does not, which is the thing an or
9611        // into an accumulator could get wrong in a way a page of one value would never show.
9612        fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
9613    }
9614
9615    /// The residue says yes to exactly what `TryFrom` says yes to.
9616    ///
9617    /// The edges above are the cases anyone would think to write down. This is the argument that
9618    /// there are no others, made by asking both questions about every value either narrow type could
9619    /// have an opinion about, and then about the values around the wide edges and the ends of an
9620    /// `i64`, which a range that size cannot reach.
9621    #[test]
9622    fn the_residue_agrees_with_a_checked_conversion_everywhere() {
9623        for value in -70_000_i64..70_000 {
9624            assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
9625            assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
9626            assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
9627            assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
9628        }
9629        let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
9630        for edge in wide {
9631            for step in -2_i64..=2 {
9632                let value = edge.saturating_add(step);
9633                assert_eq!(
9634                    fit::<i32>(&[value]).is_ok(),
9635                    i32::try_from(value).is_ok(),
9636                    "{value} as i32"
9637                );
9638                assert_eq!(
9639                    fit::<u32>(&[value]).is_ok(),
9640                    u32::try_from(value).is_ok(),
9641                    "{value} as u32"
9642                );
9643            }
9644        }
9645    }
9646
9647    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
9648    ///
9649    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
9650    /// column and no size at all for a test, so this opens the same dictionary a second time with a
9651    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
9652    /// somewhere in the middle of itself and everything past that point is read and dropped, which
9653    /// costs the decode again and holds none of it.
9654    #[test]
9655    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
9656        let path = path("dictionary-budget");
9657        let spellings = (0..2_500)
9658            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
9659            .collect::<Vec<_>>();
9660        let mut writer =
9661            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9662                .expect("new file");
9663        for part in spellings.chunks(1_024) {
9664            writer
9665                .append(
9666                    &Chunk::new(vec![
9667                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9668                    ])
9669                    .expect("one column"),
9670                )
9671                .expect("stripe written");
9672        }
9673        writer.finish().expect("commit");
9674
9675        let reader = Reader::open(&path).expect("valid directory");
9676        let page = reader.table.dictionaries[0].expect("a string column has one");
9677        let file = Arc::clone(&reader.file);
9678        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
9679            .expect("a dictionary opens whatever it may keep");
9680
9681        let resting = starved.footprint();
9682        let mut swept: Vec<Vec<u8>> = Vec::new();
9683        let mut at = 0;
9684        while at < starved.len() {
9685            at = starved
9686                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
9687                    swept.push(text.to_vec());
9688                    Ok(())
9689                })
9690                .expect("a sweep reads");
9691        }
9692        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
9693        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
9694
9695        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
9696        let read = (0..generous.len())
9697            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
9698            .collect::<Vec<_>>();
9699        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
9700        fs::remove_file(path).expect("remove scratch file");
9701    }
9702
9703    #[test]
9704    fn damaged_membership_cannot_skip_a_string_page() {
9705        let path = path("damaged-membership");
9706        let mut writer = Writer::create(
9707            &path,
9708            "items",
9709            vec![
9710                Field::required("id", LogicalType::Integer),
9711                Field::new("text", LogicalType::Varchar),
9712            ],
9713        )
9714        .expect("new file");
9715        writer.append(&sample()).expect("stripe written");
9716        writer.finish().expect("commit");
9717
9718        let reader = Reader::open(&path).expect("valid directory");
9719        let membership = reader.table.stripes[0].memberships[1].expect("string membership");
9720        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
9721        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
9722        file.write_all(&[255]).expect("damage membership");
9723        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
9724        assert!(error.message().contains("membership page checksum differs"), "{error}");
9725        fs::remove_file(path).expect("remove scratch file");
9726    }
9727
9728    #[test]
9729    fn membership_delta_stream_is_sorted_exact_and_bounded() {
9730        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
9731        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
9732        let encoded = encode_membership(&unique);
9733        assert_eq!(
9734            decode_membership(&encoded).expect("valid membership"),
9735            [4, 9, 72, 900, u32::MAX]
9736        );
9737        // A stripe's index is the union of its parts', so a code in two of them is in it once and
9738        // the result is still one ascending run of deltas.
9739        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
9740        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
9741        assert_eq!(
9742            decode_membership(&encode_membership(&merged)).expect("valid membership"),
9743            unique
9744        );
9745        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
9746        assert!(
9747            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
9748            "a value past u32 is invalid"
9749        );
9750    }
9751
9752    #[test]
9753    fn a_global_dictionary_may_be_larger_than_one_column_page() {
9754        let dictionary = Page {
9755            offset: HEADER,
9756            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
9757            hash: 0,
9758        };
9759        let table = Table {
9760            name: "items".to_owned(),
9761            fields: vec![Field::new("text", LogicalType::Varchar)],
9762            stripes: Vec::new(),
9763            rows: 0,
9764            dictionaries: vec![Some(dictionary)],
9765            distincts: vec![None],
9766            frequencies: vec![None],
9767            clustering: None,
9768            generation: 1,
9769            sections: Vec::new(),
9770        };
9771        let directory = encode_directory(&table).expect("directory");
9772        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
9773
9774        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
9775        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
9776    }
9777
9778    #[test]
9779    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
9780        let path = path("constant-codes");
9781        let mut writer =
9782            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9783                .expect("new file");
9784        let empty = vec![Value::Varchar(String::new()); 1024];
9785        for _ in 0..4 {
9786            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
9787            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
9788        }
9789        writer.finish().expect("commit");
9790
9791        let reader = Reader::open(&path).expect("valid directory");
9792        let pages = reader.layout().columns.first().expect("one column").pages;
9793        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
9794        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
9795        // a tag, a count and the value, and the row count stops being what drives the number.
9796        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
9797        let read = reader.read(3, &[0]).expect("the last part back");
9798        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
9799        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
9800        fs::remove_file(path).expect("remove scratch file");
9801    }
9802
9803    #[test]
9804    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
9805        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
9806        // truncated, but the values do not belong to the column the directory says they do.
9807        let over = vec![i64::from(i32::MAX) + 1];
9808        let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
9809        assert!(format!("{error}").contains("not of its type"), "{error}");
9810        assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
9811        assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
9812    }
9813
9814    #[test]
9815    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
9816        // A shift register rather than a run, because an arithmetic run is the one wide shape the
9817        // cascade does shrink. This is what a column with tens of millions of distinct values hands
9818        // over: full width codes with no order to them.
9819        let mut state: u32 = 0x9e37_79b9;
9820        let spread: Vec<u32> = (0..1024)
9821            .map(|_| {
9822                state ^= state << 13;
9823                state ^= state >> 17;
9824                state ^= state << 5;
9825                state
9826            })
9827            .collect();
9828        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
9829        let near: Vec<u32> = (0..1024).collect();
9830        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
9831        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
9832    }
9833
9834    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
9835    /// must not depend on which thread that was is the file. Two writes of the same rows are
9836    /// compared byte for byte rather than value for value, because a dictionary that two columns
9837    /// somehow shared would still read back correctly and would hand out its codes in the order the
9838    /// threads happened to run in, which is exactly what this is here to catch.
9839    #[test]
9840    fn two_writes_of_the_same_rows_give_the_same_bytes() {
9841        fn written(path: &PathBuf) {
9842            let fields = (0..40)
9843                .map(|column| {
9844                    let ty =
9845                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
9846                    Field::new(format!("c{column}"), ty)
9847                })
9848                .collect::<Vec<_>>();
9849            let mut writer = Writer::create(path, "wide", fields).expect("new file");
9850            for part in 0..70_u64 {
9851                let columns = (0..40)
9852                    .map(|column| {
9853                        let values = (0..64_u64)
9854                            .map(|row| {
9855                                let seed = part.wrapping_mul(31).wrapping_add(row);
9856                                if column % 4 == 0 {
9857                                    Value::Varchar(format!("v{}", seed % 17))
9858                                } else {
9859                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
9860                                }
9861                            })
9862                            .collect::<Vec<_>>();
9863                        let ty = if column % 4 == 0 {
9864                            LogicalType::Varchar
9865                        } else {
9866                            LogicalType::BigInt
9867                        };
9868                        Vector::from_values(ty, &values).expect("a column")
9869                    })
9870                    .collect::<Vec<_>>();
9871                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
9872            }
9873            writer.finish().expect("commit");
9874        }
9875
9876        let first = path("repeatable-one");
9877        let second = path("repeatable-two");
9878        written(&first);
9879        written(&second);
9880        let left = fs::read(&first).expect("the first file");
9881        let right = fs::read(&second).expect("the second file");
9882        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
9883        assert!(left == right, "two writes of the same rows differ in their bytes");
9884
9885        // And the rows are still there, since a pair of identically wrong files would pass the
9886        // comparison above on its own.
9887        let reader = Reader::open(&first).expect("valid directory");
9888        assert_eq!(reader.table().rows(), 70 * 64);
9889        let read = reader.read(0, &[0, 1]).expect("the first part back");
9890        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
9891        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
9892        fs::remove_file(first).expect("remove scratch file");
9893        fs::remove_file(second).expect("remove scratch file");
9894    }
9895
9896    /// Three tables of different shapes in one file, read back by name.
9897    fn three_tables(path: &PathBuf) {
9898        let writer = Writer::create(
9899            path,
9900            "region",
9901            vec![
9902                Field::new("r_key", LogicalType::Integer),
9903                Field::new("r_name", LogicalType::Varchar),
9904            ],
9905        )
9906        .expect("new file");
9907        let mut writer = writer;
9908        writer
9909            .append(
9910                &Chunk::new(vec![
9911                    Vector::from_values(
9912                        LogicalType::Integer,
9913                        &[Value::Integer(0), Value::Integer(1)],
9914                    )
9915                    .expect("keys"),
9916                    Vector::from_values(
9917                        LogicalType::Varchar,
9918                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
9919                    )
9920                    .expect("names"),
9921                ])
9922                .expect("two columns"),
9923            )
9924            .expect("a part");
9925        let mut writer = writer
9926            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
9927            .expect("a second table");
9928        writer
9929            .append(
9930                &Chunk::new(vec![
9931                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
9932                ])
9933                .expect("one column"),
9934            )
9935            .expect("a part");
9936        let mut writer =
9937            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
9938        for part in 0..70_i64 {
9939            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
9940            writer
9941                .append(
9942                    &Chunk::new(vec![
9943                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
9944                    ])
9945                    .expect("one column"),
9946                )
9947                .expect("a part");
9948        }
9949        writer.finish().expect("commit");
9950    }
9951
9952    #[test]
9953    fn three_tables_in_one_file_read_back_by_name() {
9954        let file = path("three-tables");
9955        three_tables(&file);
9956        let catalog = Catalog::open(&file).expect("a committed catalog");
9957        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
9958
9959        let region = catalog.table("region").expect("the first table");
9960        assert_eq!(region.table().rows(), 2);
9961        assert_eq!(
9962            region.read(0, &[1]).expect("names").value_at(1, 0),
9963            Value::Varchar("ASIA".to_owned())
9964        );
9965
9966        let wide = catalog.table("wide").expect("the third table");
9967        assert_eq!(wide.table().rows(), 70 * 64);
9968        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
9969
9970        // The middle table is reached without the one after it having been touched, which is what
9971        // a directory per table buys over one directory of everything.
9972        let empty = catalog.table("empty").expect("the second table");
9973        assert_eq!(empty.table().rows(), 1);
9974        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
9975
9976        fs::remove_file(file).expect("remove scratch file");
9977    }
9978
9979    #[test]
9980    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
9981        let file = path("three-tables-missing");
9982        three_tables(&file);
9983        let catalog = Catalog::open(&file).expect("a committed catalog");
9984        let error = catalog.table("nation").expect_err("no such table");
9985        assert!(error.message().contains("nation"), "{}", error.message());
9986        fs::remove_file(file).expect("remove scratch file");
9987    }
9988
9989    #[test]
9990    fn a_file_of_three_tables_will_not_open_as_one() {
9991        let file = path("three-tables-unnamed");
9992        three_tables(&file);
9993        let error = Reader::open(&file).expect_err("more than one table");
9994        assert!(error.message().contains("more than one table"), "{}", error.message());
9995        fs::remove_file(file).expect("remove scratch file");
9996    }
9997
9998    /// One column per storage width, because the width is what decides how many bytes a row costs.
9999    #[test]
10000    fn decimals_of_every_storage_width_round_trip() {
10001        let file = path("decimals");
10002        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
10003        let fields = widths
10004            .iter()
10005            .enumerate()
10006            .map(|(index, (width, scale))| {
10007                Field::new(
10008                    format!("d{index}"),
10009                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
10010                )
10011            })
10012            .collect::<Vec<_>>();
10013        let mut writer = Writer::create(&file, "money", fields).expect("new file");
10014        let rows: [i128; 3] = [-1234, 0, 999];
10015        let columns = widths
10016            .iter()
10017            .map(|(width, scale)| {
10018                let values = rows
10019                    .iter()
10020                    .map(|unscaled| Value::Decimal {
10021                        unscaled: *unscaled,
10022                        width: *width,
10023                        scale: *scale,
10024                    })
10025                    .collect::<Vec<_>>();
10026                Vector::from_values(
10027                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
10028                    &values,
10029                )
10030                .expect("a decimal column")
10031            })
10032            .collect::<Vec<_>>();
10033        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
10034        writer.finish().expect("commit");
10035
10036        let reader = Reader::open(&file).expect("a committed file");
10037        for (index, (width, scale)) in widths.iter().enumerate() {
10038            assert_eq!(
10039                reader.table().fields()[index].ty,
10040                LogicalType::decimal(*width, *scale).expect("a decimal type"),
10041                "column {index} came back as another type"
10042            );
10043            let column = reader.read(0, &[index]).expect("the column");
10044            for (row, unscaled) in rows.iter().enumerate() {
10045                assert_eq!(
10046                    column.value_at(row, 0),
10047                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
10048                    "column {index} row {row}"
10049                );
10050            }
10051        }
10052        fs::remove_file(file).expect("remove scratch file");
10053    }
10054
10055    #[test]
10056    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
10057        let file = path("two-of-a-name");
10058        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
10059            .expect("new file");
10060        let error = writer
10061            .next("t", vec![Field::new("a", LogicalType::BigInt)])
10062            .expect_err("the same name twice");
10063        assert!(error.message().contains("same name"), "{}", error.message());
10064        fs::remove_file(file).expect("remove scratch file");
10065    }
10066
10067    #[test]
10068    fn opening_the_catalog_reads_no_table_directory() {
10069        let file = path("catalog-only");
10070        three_tables(&file);
10071        let catalog = Catalog::open(&file).expect("a committed catalog");
10072        // The header and one slot, and nothing under it. The third table's directory covers seventy
10073        // stripes and reading it here would be the whole point of the two levels thrown away.
10074        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
10075        assert_eq!(catalog.names().len(), 3);
10076        fs::remove_file(file).expect("remove scratch file");
10077    }
10078
10079    /// The checksum answers what it has always answered, at every length its branches split on.
10080    ///
10081    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
10082    /// any particular function, but a file already on disk carries the answers the version that
10083    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
10084    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
10085    /// a block and a word, a word and a half word, and a half word and a byte.
10086    ///
10087    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
10088    /// also a check that this is the function it says it is.
10089    #[test]
10090    fn the_checksum_answers_what_it_has_always_answered() {
10091        let bytes: Vec<u8> =
10092            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
10093        for (length, expected) in [
10094            (0, 0xef46_db37_51d8_e999),
10095            (1, 0xa96c_7f0c_e858_bbb7),
10096            (3, 0x56e6_9576_32a4_87f9),
10097            (4, 0xc60d_15b1_e3ff_8f04),
10098            (5, 0x8088_1585_8624_dd4e),
10099            (7, 0xafbe_fc3d_6c6f_9a8e),
10100            (8, 0x3da5_c7aa_2696_83e0),
10101            (9, 0x465e_c429_b13c_3892),
10102            (15, 0xdee8_9d8a_065a_6233),
10103            (16, 0x1330_489a_7767_9c80),
10104            (31, 0x3391_303d_485e_846e),
10105            (32, 0x40b7_aff7_5d45_bbc8),
10106            (33, 0x4997_cae4_951c_17a5),
10107            (39, 0x5807_28fd_5c14_5739),
10108            (40, 0xf95c_f6f5_c08a_3d3b),
10109            (63, 0x2944_b4da_fc69_b206),
10110            (64, 0xbb76_f6ef_19bd_5a1b),
10111            (65, 0x814e_0c65_4a9f_d640),
10112            (127, 0x00de_aab1_31cf_f89b),
10113            (1000, 0x9e33_00c1_cde3_c58d),
10114        ] {
10115            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
10116        }
10117        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
10118    }
10119    /// A declared order survives the file, and a table that declared none stays as it was.
10120    ///
10121    /// The second half is the one worth a test. The clustering section is written only when there
10122    /// is a declaration, so a file of two tables where one is clustered exercises both the present
10123    /// and the absent branch of the decoder in one directory, which is where a length bug would
10124    /// show up as one table reading the other's bytes.
10125    #[test]
10126    fn a_declared_order_comes_back_out_of_the_file() {
10127        let path = path("clustered");
10128        let shipped = vec![
10129            Field::new("key", LogicalType::BigInt),
10130            Field::new("line", LogicalType::Integer),
10131            Field::new("shipdate", LogicalType::Date),
10132        ];
10133        let plain = vec![Field::new("a", LogicalType::Integer)];
10134        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
10135
10136        let mut writer = Writer::create(&path, "lineitem", shipped)
10137            .expect("new file")
10138            .declare(stage_zero.clone())
10139            .expect("the columns are the table's");
10140        let column = |ty: LogicalType, values: &[Value]| {
10141            Vector::from_values(ty, values).expect("the values match the type")
10142        };
10143        writer
10144            .append(
10145                &Chunk::new(vec![
10146                    column(
10147                        LogicalType::BigInt,
10148                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
10149                    ),
10150                    column(
10151                        LogicalType::Integer,
10152                        &[
10153                            Value::Integer(1),
10154                            Value::Integer(1),
10155                            Value::Integer(1),
10156                            Value::Integer(1),
10157                        ],
10158                    ),
10159                    column(
10160                        LogicalType::Date,
10161                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
10162                    ),
10163                ])
10164                .expect("three columns"),
10165            )
10166            .expect("four rows");
10167        let mut writer = writer.next("nation", plain).expect("a second table");
10168        writer
10169            .append(
10170                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
10171                    .expect("one column"),
10172            )
10173            .expect("one row");
10174        writer.finish().expect("commit");
10175
10176        let catalog = Catalog::open(&path).expect("reopen");
10177        let lineitem = catalog.table("lineitem").expect("the clustered table");
10178        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
10179        let nation = catalog.table("nation").expect("the plain table");
10180        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
10181
10182        // And the rows are still the rows, because the section goes on the end of the directory
10183        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
10184        assert_eq!(lineitem.table().rows(), 4);
10185        assert_eq!(nation.table().rows(), 1);
10186        fs::remove_file(&path).ok();
10187    }
10188
10189    /// A declaration naming a column the table does not have is refused where it is made.
10190    #[test]
10191    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
10192        let path = path("clustered-bad");
10193        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
10194            .expect("new file");
10195        let four =
10196            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
10197        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
10198        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
10199        fs::remove_file(&path).ok();
10200    }
10201}