Skip to main content

rudb_native/
lib.rs

1//! Rudb's single-file columnar snapshot format.
2//!
3//! A committed directory names independently readable column pages. It has two levels: a catalog
4//! directory naming every table in the file, which is what a footer slot points at and what opening
5//! a database reads, and one directory per table under it holding that table's stripes, pages and
6//! statistics. One slot write publishes all of them, so a commit is atomic across tables.
7//!
8//! This version handles scalar columns; the file header has two generation slots so an unfinished
9//! replacement directory cannot hide the last complete one. See
10//! `spec/storage-v3/12-many-tables-in-one-file.md`.
11//!
12//! # Parts and stripes
13//!
14//! A part is one appended chunk, which is a thousand rows, and it is the unit a scan decodes and
15//! hands to the pipeline. A stripe is sixty four parts, and it is the unit the directory describes
16//! and the unit the file is laid out in: one page per column per stripe, holding that column's
17//! sixty four part payloads end to end.
18//!
19//! The two are separate because they are sized by different pressures. A part wants to be small
20//! because it is a vector and vectors live in cache. A stripe wants to be large because everything
21//! the directory holds is per stripe and the directory is one buffer that has to be read and
22//! decoded before a single row can be answered. A hundred million rows of the hundred and five
23//! column ClickBench table is ninety seven thousand parts, and a directory with a page entry and a
24//! pair of bounds per part per column is several hundred megabytes, which is what made that load
25//! fail before this split existed. Sixty four parts to a stripe divides that by sixty four.
26//!
27//! Where the parts of a page start is not in the directory either, for the same reason. Each
28//! stripe writes one index page holding a length and a checksum per part per column, and a reader
29//! preads the sixty four entries belonging to the column it wants. A scan reads the whole column
30//! page once and slices it; a sparse row fetch reads the index entries and then only the part it
31//! needs.
32
33#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::sequence::Sequence;
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_io::{Filesystem, OpenMode, RealFilesystem};
50use rudb_metrics::{LoadProfile, Stage};
51use rudb_storage::sieve::Sieve;
52use rudb_storage::{Probe, Range, Zone};
53use rudb_vector::string::StringColumn;
54use rudb_vector::validity::Validity;
55use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
56
57mod distinct;
58pub mod graph;
59pub mod host;
60mod prepare;
61mod projection;
62mod run_projection;
63use prepare::Lent;
64pub mod section;
65pub mod stats;
66mod zones;
67
68pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
69pub use projection::build_sorted_projection;
70pub use run_projection::build_run_projection;
71pub use section::Section;
72pub use zones::{Common, Stripes, ascending, distincts, widths};
73
74const MAGIC: &[u8; 8] = b"RUDBNV10";
75const DIRECTORY: &[u8; 8] = b"RUDBDI10";
76const CATALOG: &[u8; 8] = b"RUDBCA10";
77const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
78const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
79const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
80const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
81const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
82const MAX_CATALOG_FREQUENCIES: usize = 64;
83const FORMAT: u32 = 29;
84
85/// Formats this build can open.
86///
87/// More than one, for the first time, and the reason is spec/graph/10-milestones.md's G1 exit
88/// criterion: a build with the section table in it has to open a file written before the section
89/// table existed, unchanged and without a rewrite. Formats 22 and 23 are those files, and both read
90/// as a table with an empty section table, which is exactly what section 3.1 says a table with no
91/// graph sections is.
92///
93/// All three of the older ones are readable for the same reason. What took the format from 22 to 23
94/// was tags for fourteen more column types, and a file written before that has none of them in it,
95/// so nothing in an older file is a tag this build cannot read. What took it from 23 to 24 is the
96/// section table, which a file written before it simply does not have. What takes it from 24 to 25
97/// is the view section on the end of the catalog, which an older file does not have either, and a
98/// catalog that ends where the tables end reads as a catalog with no views in it. What takes it
99/// from 25 to 26 is that a global dictionary's payload blocks now say where they are, and a file
100/// written before that has them behind one another, which [`open_global_dictionary`] reads by
101/// turning the ends it finds into the same places the newer files name outright. What takes it
102/// from 26 to 27 is that those blocks are written into the file as the load goes, between the
103/// stripes, rather than behind the dictionary's index at the end, so the dictionary's page is the
104/// index and the sorted order and nothing else. A format 26 file has its blocks inside the page,
105/// and the reader tells the two apart by whether the page has room left over for them.
106///
107/// Format 28 adds per-payload-block substring signatures to global string dictionaries. Older
108/// files have no signatures and use the ordinary exact string filter. Format 29 makes each
109/// signature four times as wide, which a dictionary says with [`DICTIONARY_WIDE_GRAMS`], and a
110/// format 28 file is read with the narrow ones it has.
111///
112/// This is not a general compatibility promise. Seven formats are readable because there was a
113/// specific reason for each, and the list shrinks again the moment the older ones stop being worth
114/// carrying.
115const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, FORMAT];
116
117const HEADER: u64 = 80;
118const SLOT_BYTES: usize = 28;
119const MAX_PAGE: usize = 256 * 1024 * 1024;
120const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
121const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
122const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
123/// Inline spellings for string entries in the bounded frequency synopsis.
124///
125/// A planner usually asks about one literal such as the empty string. Without this block it opens
126/// a multi-million-value global dictionary and visits the payload blocks of every retained entry
127/// merely to compare that literal with at most 512 heavy hitters. The spellings are already in
128/// memory while the writer sorts the dictionary, so storing this bounded copy makes planning a
129/// directory read and leaves the dictionary unopened.
130const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
131/// Certified host aggregate state for the version-one anchored replacement expression.
132const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
133/// Exact leading counts for a bounded pair of dictionary-backed grouping keys.
134///
135/// This is a separate optional directory block rather than another frequency format. Readers that
136/// predate it still understand every earlier directory, and a table without a pair worth keeping
137/// writes no block at all.
138const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
139/// The clustering declaration, written after the frequencies and only when there is one.
140///
141/// No format bump for this, which is the convention the frequency section set in #728: a new
142/// optional trailing section with its own magic leaves every file that does not use it byte for
143/// byte what it was, and the version is bumped for a change to a layout that already exists, as
144/// #1029 did. A file with no declaration is the same bytes this build wrote yesterday.
145///
146/// The width byte in this block gained a fifth value for #1285, for a declaration that leaves the
147/// bucket to the row count, and that did not bump the format either. It is the one case where the
148/// reasoning needs saying out loud, because it is a new value in a layout that already exists
149/// rather than a new section. A build without it reading one of these says `clustering width
150/// tag differs` and refuses the table, which is what that message was written for. Bumping the
151/// format instead would have made every file this build writes unreadable to an older one, whether
152/// it has a declaration in it or not, to warn about a case that only arises when it does.
153const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
154/// The string columns whose global dictionary stopped taking values partway through the load.
155///
156/// Section 5.5 of the encoding spec: a column whose stripes are nearly all new values, or the
157/// fastest growing one once the dictionaries together pass their cap, stops adding to its
158/// dictionary, and every stripe after that is written plainly. The stripes before keep their codes,
159/// so the dictionary is still written and still decodes them, but it no longer holds every value of
160/// the column, and nothing that reads it as if it did can be trusted: not the distinct count, not
161/// the frequencies, not the sorted order's first and last value, and not the codes as a group key
162/// or a membership index. A reader that finds a column named here decodes its coded pages to plain
163/// strings and answers everything else the way it answers a column with no dictionary.
164///
165/// Same convention as [`CLUSTERING`], written only when a column was demoted, so a file with none
166/// is the bytes it always was. A build that predates it refuses a file that has one with
167/// `directory extension magic differs`, which is the right answer, because that build would trust
168/// the dictionary.
169///
170/// A stripe written after the demotion has no membership index for the column. Its slot in the
171/// stripe is written as a page of no bytes, which no real membership index is, since the smallest
172/// one holds its code count.
173const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
174/// The graph section table, written after the clustering declaration and written even when empty.
175///
176/// Same convention and the same reason as the block above it, with one difference: this one is
177/// always there, so a file written by this build says which sections it has rather than leaving a
178/// reader to infer it from where the bytes ran out. Section 3.1 of the graph spec is what makes
179/// that safe to add without a format bump, because a table with no sections answers every query
180/// the way it did before, only without the graph path.
181const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
182/// How many bytes of each column's global dictionary live outside its page, written only when any do.
183///
184/// From format 27 a dictionary's payload blocks are written into the file while the load runs, so
185/// they sit between the stripes and the dictionary's page covers only its index and sorted order.
186/// Nothing needs the total to read the file, because the index names every block. It is here for
187/// what a file costs a column, which [`Reader::layout`] and the statistics budget both report, and
188/// which would otherwise lose most of the bytes of every large string column.
189const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
190
191/// The most sections one table's directory may name.
192///
193/// A relationship contributes at most three sections, so this bounds a table at a few thousand
194/// relationships, which is far past anything a schema has. The bound is here so that a torn
195/// directory naming four billion of them is refused at decode rather than turned into an
196/// allocation, the same reason the extent count has one.
197const MAX_SECTIONS: usize = 4096;
198const FREQUENCY_CANDIDATES: usize = 32_768;
199const FREQUENCY_ENTRIES: usize = 512;
200const FREQUENCY_BUILD_RANK: usize = 10;
201const FREQUENCY_ORDINALS: usize = 131_072;
202const MAX_PAIR_FREQUENCIES: usize = 1024;
203/// The most exact heavy-hitter text one column may copy into the directory.
204///
205/// A column with unusually large leading values keeps the old code-only synopsis instead. The
206/// optimization must never turn a valid load into a directory-size failure.
207const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
208/// The most threads the two per column passes at the end of a commit are spread over.
209///
210/// A table like `hits` has ninety numeric columns, so on a machine with more cores than this the
211/// cap is what decides how long the frequencies take rather than the columns are. It is here at all
212/// because each worker holds a candidate table and a decoded part, and a hundred of those at once
213/// on a narrow machine would be worse than waiting.
214const MAX_FREQUENCY_WORKERS: usize = 32;
215
216/// How many threads the passes at the end of a commit are spread over on this machine.
217fn close_workers() -> usize {
218    std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
219}
220
221/// How many bytes the columns closing at the same time may hold between them.
222///
223/// Closing a global dictionary decodes every value it holds, sorts them and drops them, and #1356
224/// took the columns one at a time so that five of them decoded at once were not the peak of a load.
225/// A numeric column's frequencies hold a candidate table and, past it, an exact set of its distinct
226/// values that reaches 512 MiB. The two used to run side by side with only the dictionaries under a
227/// bound, and on the ClickBench `hits` 10M load the close took a load that had held 3.1 GB to 4.8
228/// GB. A column is taken while the ones already closing leave room for it under this, and always
229/// when nothing else is closing, so every dictionary of `hits` at 10M rows closes at once and `URL`
230/// at 100M, which is past this alone, still closes on its own.
231const CLOSE_BYTES: usize = 1 << 30;
232
233/// What a numeric column's frequencies hold before its exact distinct set, which is the candidate
234/// table, its recount and the page being read, with room to spare.
235const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
236
237/// The most threads one stripe's encode is spread over.
238///
239/// Higher than the frequency cap because this is the load itself rather than a pass at the end of
240/// it, and the work is one column of sixty four parts, which is large enough that a thread that
241/// takes one is not a thread that was started for nothing. A machine with more cores than this has
242/// the rest of them on the Parquet read, which is still one thread and is the other half of #808.
243const MAX_ENCODE_WORKERS: usize = 32;
244
245/// How much a writer appends before it asks the kernel to start writing it to the device.
246///
247/// Without it every byte of a load waits in the page cache for the sync at the commit, and that
248/// sync was 1.3 to 1.7 s of a ClickBench `hits` 10M load of 8 to 9 s on the 32 core box. With it
249/// the device writes while the load is still encoding. Thirty two megabytes is a few stripes of
250/// `hits`, big enough that the call costs nothing next to the write, and small enough that what
251/// is left for the commit is one stretch.
252const WRITEBACK_STRETCH: u64 = 32 << 20;
253
254/// The most bytes one column of one part may spend on a membership sieve.
255///
256/// A part is a thousand rows, so a filter sized for every one of them being distinct is about
257/// thirteen hundred bytes and this never binds in practice. It is here so that a part that somehow
258/// arrives much wider than a vector cannot put an unbounded index in the file. What does bind is the
259/// rule in `Writer::encode_pages` that a sieve may not be as large as the part it indexes, which is a cap
260/// per column rather than one number for the whole file.
261const SIEVE_BUDGET: usize = 8 * 1024;
262
263/// The most bytes one end of a per part range may spend on a string.
264///
265/// A bound is allowed to be wider than the truth and never narrower, so a long string is cut down to
266/// this many bytes for the low end and cut down and then stepped up for the high end. The reason for
267/// a cap at all is that there are nine hundred and seventy four parts of a hundred and five columns
268/// in a million rows of ClickBench and `URL` runs to hundreds of bytes, so keeping every end whole
269/// would put more in the directory than the skipping is worth. Twenty four bytes is past the point
270/// where two URLs of the same site still look alike.
271const PART_BOUND_BYTES: usize = 24;
272
273fn io(error: std::io::Error) -> Error {
274    Error::io(error.to_string())
275}
276
277fn invalid(message: &str) -> Error {
278    Error::invalid_input(format!("invalid rudb native file: {message}"))
279}
280
281/// Adds a sequence of byte counts without an overflow the caller has to think about.
282fn sum(counts: impl Iterator<Item = u64>) -> u64 {
283    counts.fold(0, u64::saturating_add)
284}
285
286/// One column's span out of a per column list, or zero when the list is shorter than the column.
287fn span_bytes(spans: &[Span], at: usize) -> u64 {
288    spans.get(at).map_or(0, |span| u64::from(span.length))
289}
290
291/// One column's page out of a per column list, or zero when that column has no page at all.
292fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
293    pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
294}
295
296/// Everything one column's global dictionary costs the file, its page and the blocks outside it.
297fn dictionary_bytes(table: &Table, at: usize) -> u64 {
298    page_bytes(&table.dictionaries, at)
299        .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
300}
301
302/// The xxHash64 of `bytes`, which is what every span this format stores is checked against.
303///
304/// It walks the input as chunks rather than as offsets into it, and that is the only thing about it
305/// worth a comment. The offset form reads `bytes[at..at + 8]`, and neither the slicing nor the
306/// `try_into` behind it can be proved in range by a compiler that does not know where `at` stopped,
307/// so each of the four lanes paid for a bounds check and a length check on every thirty two bytes.
308/// A chunk carries its own length, so both fold away and the loop is the multiplies and rotates it
309/// was meant to be. That loop runs over every byte of every span a query reads, which on ClickBench
310/// 8 is about five percent of the query.
311fn checksum(bytes: &[u8]) -> u64 {
312    seeded_checksum(bytes, 0)
313}
314
315/// A hundred and twenty eight bit name for `bytes`, as two xxHash64 walks under different seeds,
316/// with the format this build writes folded in so that a name made by one format is never taken
317/// for the name of a file in another.
318///
319/// For a caller outside this crate that has to name a file by what went into it, which is what a
320/// Parquet mirror's key is. See the global dictionary's use of the same pair for the arithmetic.
321#[must_use]
322pub fn content_name(bytes: &[u8]) -> u128 {
323    let seed = u64::from(FORMAT);
324    u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
325}
326
327/// [`content_name`] of bytes that arrive in pieces, which gives the same name as the pieces joined.
328///
329/// A Parquet mirror is named for the file's footer, and that is 930 KB on the ten million row
330/// ClickBench file. Read whole to be hashed it is a freed megabyte in every process that opens the
331/// mirror, which the allocator keeps. Read a window at a time it is a window.
332#[derive(Debug, Clone)]
333pub struct ContentNamer {
334    seeds: [u64; 2],
335    lanes: [[u64; 4]; 2],
336    held: [u8; 32],
337    filled: usize,
338    length: u64,
339}
340
341impl Default for ContentNamer {
342    fn default() -> Self {
343        let seed = u64::from(FORMAT);
344        let seeds = [seed, !seed];
345        let lanes = seeds.map(|seed| {
346            [
347                seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
348                seed.wrapping_add(XXH_P2),
349                seed,
350                seed.wrapping_sub(XXH_P1),
351            ]
352        });
353        Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
354    }
355}
356
357impl ContentNamer {
358    /// Takes the next piece.
359    pub fn update(&mut self, mut bytes: &[u8]) {
360        self.length += bytes.len() as u64;
361        if self.filled > 0 {
362            let take = (32 - self.filled).min(bytes.len());
363            self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
364            self.filled += take;
365            bytes = &bytes[take..];
366            if self.filled < 32 {
367                return;
368            }
369            let block = self.held;
370            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
371            self.filled = 0;
372        }
373        let mut blocks = bytes.chunks_exact(32);
374        for block in blocks.by_ref() {
375            self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
376        }
377        let rest = blocks.remainder();
378        self.held[..rest.len()].copy_from_slice(rest);
379        self.filled = rest.len();
380    }
381
382    /// The name of everything taken so far.
383    #[must_use]
384    pub fn finish(&self) -> u128 {
385        let rest = &self.held[..self.filled];
386        let [first, second] = [0, 1].map(|at| {
387            if self.length < 32 {
388                checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
389            } else {
390                finish_checksum(self.lanes[at], rest, self.length)
391            }
392        });
393        u128::from(first) << 64 | u128::from(second)
394    }
395}
396
397/// The xxHash64 of `bytes` started from `seed`, which is the same walk with a different beginning.
398///
399/// A seed is here for one caller: a global dictionary decides whether two values are the same by
400/// their hashes rather than by their bytes, and one sixty four bit hash is not enough to do that
401/// with. Twenty million distinct values collide on sixty four bits about once in a hundred thousand
402/// loads, which for a wrong answer is far too often. Two hashes of the same value under different
403/// seeds are independent, so the pair is a hundred and twenty eight bits and the same arithmetic
404/// puts that at around one in 1e24.
405fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
406    // Asked for before the loop rather than after it, because a `ChunksExact` settles what it
407    // cannot divide when it is built and hands back the same tail whether it has been walked or not.
408    let mut blocks = bytes.chunks_exact(32);
409    let rest = blocks.remainder();
410    if bytes.len() < 32 {
411        return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
412    }
413    let mut lanes = [
414        seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
415        seed.wrapping_add(XXH_P2),
416        seed,
417        seed.wrapping_sub(XXH_P1),
418    ];
419    for block in blocks.by_ref() {
420        checksum_block(&mut lanes, block);
421    }
422    finish_checksum(lanes, rest, bytes.len() as u64)
423}
424
425const XXH_P1: u64 = 11_400_714_785_074_694_791;
426const XXH_P2: u64 = 14_029_467_366_897_019_727;
427const XXH_P3: u64 = 1_609_587_929_392_839_161;
428const XXH_P4: u64 = 9_650_029_242_287_828_579;
429const XXH_P5: u64 = 2_870_177_450_012_600_261;
430
431fn checksum_round(state: u64, word: u64) -> u64 {
432    state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
433}
434
435fn checksum_word(chunk: &[u8]) -> u64 {
436    u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
437}
438
439/// One thirty two byte block into the four lanes.
440fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
441    for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
442        *lane = checksum_round(*lane, checksum_word(chunk));
443    }
444}
445
446/// The lanes after every whole block, folded together with what was left over and the length.
447fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
448    let merge = |state: u64, lane: u64| {
449        (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
450    };
451    let [one, two, three, four] = lanes;
452    let combined = one
453        .rotate_left(1)
454        .wrapping_add(two.rotate_left(7))
455        .wrapping_add(three.rotate_left(12))
456        .wrapping_add(four.rotate_left(18));
457    let hash = merge(merge(merge(merge(combined, one), two), three), four);
458    checksum_tail(hash.wrapping_add(length), rest)
459}
460
461/// The fewer than thirty two bytes after the last whole block, and the final mix.
462fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
463    let mut words = rest.chunks_exact(8);
464    for chunk in words.by_ref() {
465        hash ^= checksum_round(0, checksum_word(chunk));
466        hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
467    }
468    rest = words.remainder();
469    if rest.len() >= 4 {
470        let (head, tail) = rest.split_at(4);
471        let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
472        hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
473        hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
474        rest = tail;
475    }
476    for &byte in rest {
477        hash ^= u64::from(byte).wrapping_mul(XXH_P5);
478        hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
479    }
480    hash ^= hash >> 33;
481    hash = hash.wrapping_mul(XXH_P2);
482    hash ^= hash >> 29;
483    hash = hash.wrapping_mul(XXH_P3);
484    hash ^ (hash >> 32)
485}
486
487/// The checksum of `length` bytes of `file` from `offset`, read [`DIRECTORY_WINDOW`] at a time.
488///
489/// The same xxHash64 as [`checksum`], carried across reads rather than over one buffer, so that a
490/// directory can be checked without all of it being in memory at once. The four lanes take whole
491/// thirty two byte blocks, and a read that ends partway through one keeps the tail for the next.
492fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
493    walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
494}
495
496/// Reads `length` bytes at `offset` a window at a time, hands each window to `each`, and answers
497/// the checksum of all of them.
498///
499/// `window` is a multiple of thirty two, so every window but the last is whole blocks of the hash
500/// and nothing has to be carried from one read to the next.
501fn walk_checksummed(
502    file: &File,
503    offset: u64,
504    length: usize,
505    window: usize,
506    mut each: impl FnMut(&[u8]) -> Result<()>,
507) -> Result<u64> {
508    debug_assert!(window % 32 == 0 && window > 0, "a window is whole blocks of the hash");
509    if length < 32 {
510        let mut bytes = vec![0; length];
511        read_at(file, offset, &mut bytes)?;
512        each(&bytes)?;
513        return Ok(checksum(&bytes));
514    }
515    let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
516    let mut buffer = vec![0; window.min(length)];
517    let mut read = 0;
518    let (mut whole, mut filled) = (0, 0);
519    while read < length {
520        filled = buffer.len().min(length - read);
521        read_at(file, offset + read as u64, &mut buffer[..filled])?;
522        read += filled;
523        each(&buffer[..filled])?;
524        whole = filled / 32 * 32;
525        for block in buffer[..whole].chunks_exact(32) {
526            checksum_block(&mut lanes, block);
527        }
528    }
529    Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
530}
531
532#[derive(Debug, Clone, Copy)]
533struct Slot {
534    offset: u64,
535    length: u32,
536    generation: u64,
537    hash: u64,
538}
539
540impl Slot {
541    fn bytes(self) -> [u8; SLOT_BYTES] {
542        let mut result = [0; SLOT_BYTES];
543        result[..8].copy_from_slice(&self.offset.to_le_bytes());
544        result[8..12].copy_from_slice(&self.length.to_le_bytes());
545        result[12..20].copy_from_slice(&self.generation.to_le_bytes());
546        result[20..28].copy_from_slice(&self.hash.to_le_bytes());
547        result
548    }
549
550    fn read(bytes: &[u8]) -> Self {
551        Self {
552            offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
553            length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
554            generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
555            hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
556        }
557    }
558}
559
560#[derive(Debug, Clone, Copy)]
561struct Page {
562    offset: u64,
563    length: u32,
564    hash: u64,
565}
566
567impl Page {
568    /// How much of the file this page takes, for [`Reader::layout`].
569    fn bytes(&self) -> u64 {
570        u64::from(self.length)
571    }
572}
573
574#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
575enum FrequencyValue {
576    Null,
577    Integer(i128),
578    Code(u32),
579}
580
581/// A table keyed by the sixty four bits of the values the numeric frequency pass counts.
582///
583/// Every integer of every numeric column goes through one of these at least once when a table
584/// closes, and with the standard hasher that was a fifth of the close on its own, all of it SipHash
585/// guarding against an attacker who would have to choose the rows of the file being written.
586type FrequencyMap<V> = HashMap<u64, V, Spread>;
587
588/// The first pass of [`Writer::numeric_frequency`]: a Misra-Gries candidate table keyed by a value's
589/// sixty four bits, with the null counted beside it.
590///
591/// The table is an open addressed one of its own rather than a `HashMap`. On a column that is near
592/// unique, which `hits` has a dozen of, nearly every row is a value the table has not seen, and a
593/// `HashMap` spent a lookup and then a second hash and probe to insert it, and a `retain` over every
594/// bucket each time the table filled. Those were 6 percent of the CPU of loading the 10m ClickBench
595/// file, and the slowest of those columns decided how long the whole frequency step took. Here a
596/// value is found or given the empty slot it stopped at in one probe, and a decrement rebuilds the
597/// table from the few candidates that outlive it.
598///
599/// What the table holds after a stream of rows is the same set of counts either way, since that is
600/// fixed by the algorithm and not by where the counts live.
601#[derive(Debug)]
602struct Candidates {
603    /// A power of two number of slots, at most half of them in use. A count of zero is an empty
604    /// slot, which no candidate ever is, because one whose count reaches zero is dropped.
605    slots: Vec<Candidate>,
606    held: usize,
607    nulls: u32,
608    decrements: u64,
609    /// The candidates that outlive a decrement, kept so that each decrement is not an allocation.
610    survivors: Vec<Candidate>,
611}
612
613/// One slot of [`Candidates`], the value's bits beside its count so a probe reads one line.
614#[derive(Debug, Default, Clone, Copy)]
615struct Candidate {
616    bits: u64,
617    count: u32,
618}
619
620/// The slots a candidate table starts with, grown by doubling as it fills.
621const FIRST_CANDIDATE_SLOTS: usize = 64;
622
623impl Default for Candidates {
624    fn default() -> Self {
625        Self {
626            slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
627            held: 0,
628            nulls: 0,
629            decrements: 0,
630            survivors: Vec::new(),
631        }
632    }
633}
634
635impl Candidates {
636    /// Counts `times` rows of `bits` and ends in the state `times` rows counted one at a time would.
637    ///
638    /// A value already held, or one there is room to hold, takes the whole run at once, because
639    /// every row after the first would find it held. A value the full table turns away goes a row
640    /// at a time, because each of its rows decrements every candidate and one of those decrements
641    /// can free the place the next row takes.
642    fn add(&mut self, bits: Option<u64>, mut times: u32) {
643        while times > 0 {
644            let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
645            match bits {
646                Some(bits) => {
647                    let (at, found) = self.find(bits);
648                    if found {
649                        self.slots[at].count = self.slots[at].count.saturating_add(times);
650                        return;
651                    }
652                    if room {
653                        self.place(at, bits, times);
654                        return;
655                    }
656                }
657                None if self.nulls != 0 => {
658                    self.nulls = self.nulls.saturating_add(times);
659                    return;
660                }
661                None if room => {
662                    self.nulls = times;
663                    return;
664                }
665                None => {}
666            }
667            self.decrement();
668            times -= 1;
669        }
670    }
671
672    /// The slot holding `bits` and `true`, or the empty slot a search for it stopped at and `false`.
673    fn find(&self, bits: u64) -> (usize, bool) {
674        let mask = self.slots.len() - 1;
675        let mut at = home(bits, self.slots.len());
676        loop {
677            let slot = self.slots[at];
678            if slot.count == 0 {
679                return (at, false);
680            }
681            if slot.bits == bits {
682                return (at, true);
683            }
684            at = (at + 1) & mask;
685        }
686    }
687
688    /// Where `bits` is held, for the recount, which reads the table without changing it.
689    fn position(&self, bits: u64) -> Option<usize> {
690        match self.find(bits) {
691            (at, true) => Some(at),
692            (_, false) => None,
693        }
694    }
695
696    /// Puts a new candidate in the empty slot `at`, which a search for it just stopped at, doubling
697    /// the table first when that would fill more than half of it.
698    fn place(&mut self, at: usize, bits: u64, count: u32) {
699        let at = if (self.held + 1) * 2 > self.slots.len() {
700            let wider = self.slots.len() * 2;
701            let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
702            for slot in old.into_iter().filter(|slot| slot.count != 0) {
703                let (to, _) = self.find(slot.bits);
704                self.slots[to] = slot;
705            }
706            self.find(bits).0
707        } else {
708            at
709        };
710        self.slots[at] = Candidate { bits, count };
711        self.held += 1;
712    }
713
714    /// Takes one from every candidate and the null, dropping the ones that reach zero.
715    fn decrement(&mut self) {
716        let mut survivors = std::mem::take(&mut self.survivors);
717        survivors.clear();
718        survivors.extend(
719            self.slots
720                .iter()
721                .filter(|slot| slot.count > 1)
722                .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
723        );
724        self.slots.fill(Candidate::default());
725        self.held = survivors.len();
726        for &slot in &survivors {
727            let (at, _) = self.find(slot.bits);
728            self.slots[at] = slot;
729        }
730        self.survivors = survivors;
731        self.nulls = self.nulls.saturating_sub(1);
732        self.decrements = self.decrements.saturating_add(1);
733    }
734
735    /// Every candidate's bits and count, in no particular order.
736    fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
737        self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
738    }
739}
740
741/// The slot a search for `bits` starts at in a table of `slots`, a power of two.
742///
743/// The top bits of a multiply by the golden ratio, which every bit of the value reaches, so a
744/// timestamp column whose values are all multiples of a million still spreads over the table.
745fn home(bits: u64, slots: usize) -> usize {
746    (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
747}
748
749/// Equal rows in a row, gathered so they are counted once.
750#[derive(Debug, Default)]
751struct Run {
752    bits: Option<u64>,
753    times: u32,
754}
755
756impl Run {
757    /// Adds one row, and hands back the run it ended if it was not the same value.
758    fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
759        if self.times != 0 && self.bits == bits && self.times < u32::MAX {
760            self.times += 1;
761            return None;
762        }
763        let ended = self.take();
764        self.bits = bits;
765        self.times = 1;
766        ended
767    }
768
769    /// The run being gathered, if there is one, leaving none.
770    fn take(&mut self) -> Option<(Option<u64>, u32)> {
771        let times = std::mem::take(&mut self.times);
772        (times != 0).then_some((self.bits, times))
773    }
774}
775
776/// Builds the hasher for [`FrequencyMap`].
777#[derive(Debug, Default, Clone, Copy)]
778struct Spread;
779
780impl std::hash::BuildHasher for Spread {
781    type Hasher = SpreadHasher;
782
783    fn build_hasher(&self) -> SpreadHasher {
784        SpreadHasher(0)
785    }
786}
787
788/// Folds each word in with a full width multiply whose two halves are xored together.
789///
790/// A plain multiply leaves the low bits of the hash as poor as the low bits of the key, and the
791/// table picks its bucket from the low bits, so a timestamp column, whose values are all multiples
792/// of a million microseconds, would pile into a sixty fourth of the buckets. Folding the high half
793/// of the product back in is what gives the low bits the whole word.
794#[derive(Debug)]
795struct SpreadHasher(u64);
796
797impl SpreadHasher {
798    fn mix(&mut self, word: u64) {
799        let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
800        self.0 = (product as u64) ^ ((product >> 64) as u64);
801    }
802}
803
804impl std::hash::Hasher for SpreadHasher {
805    fn write(&mut self, bytes: &[u8]) {
806        for part in bytes.chunks(8) {
807            let mut word = [0; 8];
808            word[..part.len()].copy_from_slice(part);
809            self.mix(u64::from_le_bytes(word));
810        }
811    }
812
813    fn write_u32(&mut self, value: u32) {
814        self.mix(u64::from(value));
815    }
816
817    fn write_u64(&mut self, value: u64) {
818        self.mix(value);
819    }
820
821    fn write_i128(&mut self, value: i128) {
822        self.mix(value as u64);
823        self.mix((value >> 64) as u64);
824    }
825
826    fn write_isize(&mut self, value: isize) {
827        self.mix(value as u64);
828    }
829
830    fn finish(&self) -> u64 {
831        self.0
832    }
833}
834
835#[derive(Debug, Clone)]
836struct FrequencyEntry {
837    value: FrequencyValue,
838    count: u64,
839}
840
841/// Exact leading frequencies for one column.
842///
843/// Values outside `entries` occur at most `omitted_max` times. This lets a count-descending TopN
844/// use the synopsis only when its last winner is strictly above every omitted value.
845#[derive(Debug, Clone)]
846struct FrequencySummary {
847    entries: Vec<FrequencyEntry>,
848    omitted_max: u64,
849    ordinals: Vec<u64>,
850    ordinal_entries: Vec<u16>,
851}
852
853#[derive(Debug, Clone)]
854struct PairFrequencyEntry {
855    first_entry: u16,
856    second: Option<u32>,
857    count: u64,
858}
859
860/// Exact leading counts for one numeric frequency anchor and one stable string code space.
861///
862/// `omitted_max` covers both first-key values outside the numeric synopsis and pairs below the
863/// retained prefix. A TopN may therefore use the entries only when its boundary strictly exceeds
864/// this number.
865#[derive(Debug, Clone)]
866struct PairFrequencySummary {
867    first: u16,
868    second: u16,
869    entries: Vec<PairFrequencyEntry>,
870    omitted_max: u64,
871}
872
873/// One column's frequency synopsis, in memory or left where it is in the file.
874///
875/// A writer holds what it counted. A reader leaves every synopsis in the file and reads one back
876/// when a query asks about its column, because they are the largest thing in a directory once they
877/// are decoded, forty eight bytes an entry and nearly twenty thousand entries over `hits`, and
878/// most queries ask about none of them. Where one sits is found at open, by reading it through and
879/// checking it, so a torn synopsis is still refused when the table is opened.
880#[derive(Debug, Clone)]
881enum Frequencies {
882    Held(FrequencySummary),
883    /// Where the synopsis sits, and whether it was written with the value of each ordinal, which
884    /// is what the directory's frequency magic says and the synopsis itself does not.
885    Stored {
886        span: Span,
887        values: bool,
888    },
889}
890
891/// The values one column's frequency synopsis lists, with a bound on everything it left out.
892///
893/// What [`Reader::frequency_prefix`] answers. The counts are exact, and `omitted_max` is how many
894/// rows any value not in the list can hold, which is zero when nothing was left out at all.
895#[derive(Debug, Clone)]
896pub struct FrequencyPrefix {
897    /// Every value the synopsis lists, with the number of rows holding it, count descending.
898    pub entries: Vec<(Value, u64)>,
899    /// How many rows the most common value outside the list holds, and zero for a complete list.
900    pub omitted_max: u64,
901}
902
903/// Sparse row ordinals covered by a numeric frequency candidate set.
904#[derive(Debug, Clone, PartialEq)]
905pub struct FrequencyOccurrences {
906    /// Upper bound for the frequency of every value absent from the fetched rows.
907    pub omitted_max: u64,
908    /// Table-wide row ordinals in ascending order.
909    pub ordinals: Vec<u64>,
910    /// The retained heavy-hitter values named by `anchor_indices`.
911    pub anchors: Vec<Value>,
912    /// The index in `anchors` at each ordinal, or empty for a legacy FQ2 directory.
913    pub anchor_indices: Vec<u16>,
914}
915
916/// Exact grouped counts for a pair of values, in descending count order.
917pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
918
919/// Where one column's page for one stripe sits in the file.
920///
921/// A column page has no checksum of its own because every part inside it carries one, and the
922/// stripe's index page holds those. Checking a part on the way out of the page covers exactly the
923/// bytes a reader is about to decode, and covers them once whether the reader took the whole page
924/// or pulled one part out of the middle of it.
925#[derive(Debug, Clone, Copy, Default)]
926struct Span {
927    offset: u64,
928    length: u32,
929}
930
931/// One optional page for each column of a stripe, holding only the pages that are there.
932///
933/// A stripe has three of these, the membership, sieve and part range pages. As a
934/// `Vec<Option<Page>>` each was thirty two bytes a column whether the page was there or not, and
935/// over the ten million rows of `hits` that is half a megabyte at open for 7171 pages out of 16380
936/// slots. Kept sparse and packed, a page that is there is twenty four bytes and one that is not is
937/// nothing.
938#[derive(Debug, Clone, Default)]
939struct Pages {
940    columns: usize,
941    held: Box<[StripePage]>,
942}
943
944/// A page and the column it is for, packed so that the column sits where the padding was.
945#[derive(Debug, Clone, Copy)]
946struct StripePage {
947    offset: u64,
948    hash: u64,
949    length: u32,
950    column: u32,
951}
952
953impl Pages {
954    /// The pages of `columns` columns, one slot each in column order.
955    fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
956        let mut held = Vec::with_capacity(slots.iter().flatten().count());
957        for (column, page) in slots.iter().enumerate() {
958            if let Some(page) = page {
959                let column =
960                    u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
961                held.push(StripePage {
962                    offset: page.offset,
963                    hash: page.hash,
964                    length: page.length,
965                    column,
966                });
967            }
968        }
969        Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
970    }
971
972    /// The page of one column, if it has one.
973    fn get(&self, column: usize) -> Option<Page> {
974        let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
975        let placed = self.held[at];
976        Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
977    }
978
979    /// One slot per column, in column order, the way the directory writes them.
980    fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
981        (0..self.columns).map(|column| self.get(column))
982    }
983
984    /// How much of the file one column's page takes, or zero when it has none.
985    fn bytes(&self, column: usize) -> u64 {
986        self.get(column).map_or(0, |page| page.bytes())
987    }
988}
989
990/// One independently readable stripe of a table.
991#[derive(Debug, Clone)]
992pub struct Stripe {
993    rows: usize,
994    /// Rows in each part, in source order. Kept in the directory so that mapping a row ordinal to a
995    /// part, which every sparse fetch does, never reads the file.
996    parts: Vec<u32>,
997    /// The index page: one section per column, holding a length and a checksum for every part and
998    /// then a checksum of the section itself, so that a reader can pread one column's section and
999    /// still know it is intact.
1000    index: Span,
1001    pages: Vec<Span>,
1002    memberships: Pages,
1003    /// One page per column holding the membership sieve of every part of the stripe, for the
1004    /// columns that have one. A column whose parts all declined a sieve has no page at all.
1005    sieves: Pages,
1006    /// One page per column holding the two ends and the null count of every part of the stripe.
1007    ///
1008    /// The stripe's own `zone` below covers sixty four times as many rows, and on a column that is
1009    /// not the one the rows are ordered by that is the difference between skipping half the file and
1010    /// skipping all but three percent of it. On ClickBench 24 the cutoff the answer settles at
1011    /// leaves eight stripes of sixteen alive and thirty parts of nine hundred and seventy four.
1012    ///
1013    /// A page per column rather than one page for the stripe, so that a query that compares one
1014    /// column reads the ends of that column and not of the hundred and four beside it. Read lazily
1015    /// for the same reason, like the sieves.
1016    part_ranges: Pages,
1017    zone: Zone,
1018}
1019
1020impl Stripe {
1021    /// Number of rows in this stripe.
1022    #[must_use]
1023    pub fn rows(&self) -> usize {
1024        self.rows
1025    }
1026
1027    /// Number of parts in this stripe.
1028    #[must_use]
1029    pub fn parts(&self) -> usize {
1030        self.parts.len()
1031    }
1032
1033    /// The two ends and the null count of every column over the whole stripe.
1034    ///
1035    /// In the directory and so in memory, which is what makes it the one a planner can ask. The
1036    /// finer ones are a page per column per stripe in the file, read by [`Reader::skips`] when a
1037    /// scan wants to know which parts to open.
1038    #[must_use]
1039    pub fn zone(&self) -> &Zone {
1040        &self.zone
1041    }
1042}
1043
1044/// The committed table directory.
1045#[derive(Debug, Clone)]
1046pub struct Table {
1047    name: String,
1048    fields: Vec<Field>,
1049    stripes: Vec<Stripe>,
1050    rows: usize,
1051    dictionaries: Vec<Option<Page>>,
1052    /// Bytes of each column's dictionary payload that are outside its page, which is all of them
1053    /// from format 27 and none of them before. See [`DICTIONARY_PAYLOADS`].
1054    ///
1055    /// Empty rather than a row of zeros on a table that has none, and read with `get` for that
1056    /// reason, so that a table built by hand in a test does not have to know about it.
1057    dictionary_payloads: Vec<u64>,
1058    /// The columns whose dictionary stopped taking values partway through the load, see
1059    /// [`DEMOTED`].
1060    ///
1061    /// Empty rather than a row of `false` on a table that has none, and read with `get`, for the
1062    /// same reason `dictionary_payloads` is.
1063    demoted: Vec<bool>,
1064    frequencies: Vec<Option<Frequencies>>,
1065    pair_frequencies: Vec<PairFrequencySummary>,
1066    /// String spellings aligned with each column's frequency entries.
1067    ///
1068    /// Empty for files written before `RUDBFT1`. A `None` entry is the null frequency entry; every
1069    /// code entry in a column named by the block has its exact bytes here.
1070    frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1071    /// Exact candidate host aggregates and an upper bound for every omitted host.
1072    host_groups: Option<host::HostSummary>,
1073    /// How many distinct values each column holds, for the columns that know.
1074    ///
1075    /// A dictionary entry is made the first time a value is seen and nothing ever removes one, so
1076    /// the size of the dictionary is the number of distinct values in the column. That is the whole
1077    /// story for a column with no null in it, and the wrong number by one for a column with a null
1078    /// in it, because a null row is written as the code for the empty string and makes an entry the
1079    /// dictionary would not otherwise have. The writer knows which case it is, since it counts the
1080    /// non-null rows that use each code while it builds the frequency summary, and the reader cannot
1081    /// work it out from the dictionary alone. So the writer settles it here.
1082    distincts: Vec<Option<u64>>,
1083    /// The order the rows of this table are meant to be stored in, if anybody declared one.
1084    ///
1085    /// A declaration and not a measurement. Nothing here checks that the stripes actually arrived
1086    /// in this order, and the reason it is worth storing anyway is that the order is the only thing
1087    /// about a table that a rewrite destroys without anybody noticing. The fragment ranges prune on
1088    /// whatever order the rows came in, so a table loaded sorted prunes and the same table after a
1089    /// checkpoint that did not know to keep the order quietly stops pruning and nothing says why.
1090    clustering: Option<Clustering>,
1091    /// The file generation of the commit that last wrote this table's column pages.
1092    ///
1093    /// This is what spec/graph/03-the-file-format.md section 3.2 calls the table generation, and
1094    /// the definition is deliberately about the pages rather than about the directory. A graph
1095    /// section is a restatement of a column in terms of row ids, so what invalidates one is the
1096    /// rows being renumbered, and nothing else. Adding a second table to the file, or attaching a
1097    /// section to this one, commits a new file generation without touching a single row of this
1098    /// table, and a definition that moved with those would declare every section in the file stale
1099    /// for no reason.
1100    ///
1101    /// Zero on a table written before format 23, where nothing recorded it. Real generations start
1102    /// at one, so zero can never match a section's stamp, and a table from format 22 has no
1103    /// sections for it to match anyway.
1104    generation: u64,
1105    /// The graph sections this table carries, per spec/graph/03-the-file-format.md section 3.2.
1106    ///
1107    /// Empty for every table written before the section table existed, and empty is not a
1108    /// degraded state: section 3.1 says deleting every graph section from a file changes no answer,
1109    /// only the time, so a table with none here answers every query the same way and slower. That
1110    /// is what lets this field arrive without a migration.
1111    sections: Vec<Section>,
1112}
1113
1114impl Table {
1115    /// The SQL table name held by this snapshot.
1116    #[must_use]
1117    pub fn name(&self) -> &str {
1118        &self.name
1119    }
1120
1121    /// Columns in their SQL order.
1122    #[must_use]
1123    pub fn fields(&self) -> &[Field] {
1124        &self.fields
1125    }
1126
1127    /// Committed row count.
1128    #[must_use]
1129    pub fn rows(&self) -> usize {
1130        self.rows
1131    }
1132
1133    /// Independently readable stripes.
1134    #[must_use]
1135    pub fn stripes(&self) -> &[Stripe] {
1136        &self.stripes
1137    }
1138
1139    /// The order the rows are meant to be stored in, if this table was declared with one.
1140    #[must_use]
1141    pub fn clustering(&self) -> Option<&Clustering> {
1142        self.clustering.as_ref()
1143    }
1144
1145    /// The generation every section of this table is judged against.
1146    ///
1147    /// See the field. A caller deciding whether to read a section asks [`Section::usable`] with
1148    /// this.
1149    #[must_use]
1150    pub fn generation(&self) -> u64 {
1151        self.generation
1152    }
1153
1154    /// Every graph section this table names, including the kinds this build does not know.
1155    ///
1156    /// Including them is the point. A caller that wants only the ones it can use asks
1157    /// [`Section::usable`], and a caller rewriting the directory carries the rest through, so a
1158    /// file opened by an older build and written again does not silently lose a section that build
1159    /// had no name for.
1160    #[must_use]
1161    pub fn sections(&self) -> &[Section] {
1162        &self.sections
1163    }
1164}
1165
1166/// One table's line in the catalog directory.
1167///
1168/// The small level of the two. It holds what opening a database needs and nothing else: the name to
1169/// bind, the shape to plan against, the row count, and where the table's own directory sits. A file
1170/// of eight tables is eight of these, and reading them costs the same whether the tables hold a
1171/// thousand rows or a billion.
1172///
1173/// The name, the fields and the row count are repeated here rather than pointed at inside the table
1174/// directory, which is the entire point of having two levels. A catalog that pointed at them would
1175/// have to read every table directory at open to answer what tables there are, which is the cost
1176/// this level exists to avoid.
1177#[derive(Debug, Clone)]
1178struct Entry {
1179    name: String,
1180    fields: Vec<Field>,
1181    rows: usize,
1182    /// Where this table's own directory sits, with the checksum it was committed under.
1183    directory: Page,
1184    /// Legacy nonzero counts. New files leave these empty and derive filtered counts from
1185    /// reusable column frequencies when a query needs them.
1186    nonzero: Vec<Option<u64>>,
1187    /// Exact sum and non-null count for signed integer columns.
1188    aggregates: Vec<Option<(i128, u64)>>,
1189    /// Exact non-null distinct values when the writer finished counting the column.
1190    distincts: Vec<Option<u64>>,
1191    /// Exact integer or date bounds; the inner `None` means every row is null.
1192    extremes: Vec<StoredIntegerExtremes>,
1193    /// Complete bounded numeric frequencies, including NULL when present.
1194    frequencies: Vec<StoredNumericFrequencies>,
1195}
1196
1197type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1198type StoredNumericFrequencies = Option<NumericFrequencies>;
1199
1200/// One view's line in the catalog directory.
1201///
1202/// A view has no pages, so unlike a table it is entirely here and there is no second level under it.
1203/// What it is made of is text: the body the binder binds again at every reference, and the whole
1204/// statement written back out, which is what `duckdb_views()` reports and nothing else reads.
1205///
1206/// The columns are a cache and they are written down anyway, which is worth saying out loud because
1207/// a cache in a file looks like a mistake. It is what the pin does. Create a view on a file, open
1208/// the file again in another process, and `duckdb_views()` answers `column_count` and `is_bound`
1209/// true without anything having bound the body, so the list survived the write. Not writing it
1210/// would answer null and false there, and the only way back would be to bind every view at open,
1211/// which is the thing the cache exists to avoid.
1212#[derive(Debug, Clone, PartialEq, Eq)]
1213pub struct ViewEntry {
1214    /// The view's own name, without the schema, the way a table entry holds its name.
1215    pub name: String,
1216    /// The query the view stands for, as the text that was written.
1217    pub sql: String,
1218    /// The whole `CREATE VIEW` written back out.
1219    pub statement: String,
1220    /// The column names the statement gave, which rename a prefix of what the body produces.
1221    pub aliases: Vec<String>,
1222    /// The columns the last bind of the body produced.
1223    pub columns: Vec<Field>,
1224}
1225
1226/// Where one column's bytes went, taken from the directory rather than by reading pages.
1227#[derive(Debug, Clone)]
1228pub struct ColumnLayout {
1229    /// The column's name, so a report does not have to carry the field list beside this.
1230    pub name: String,
1231    /// The type, spelled the way the catalog spells it.
1232    pub kind: String,
1233    /// Every stripe's page of this column added up, which is the encoded data itself.
1234    pub pages: u64,
1235    /// Every stripe's exact code membership page for this column.
1236    pub memberships: u64,
1237    /// Every stripe's membership sieve page for this column.
1238    pub sieves: u64,
1239    /// Every stripe's per part range page for this column.
1240    pub part_ranges: u64,
1241    /// The table wide dictionary of this column, if it has one.
1242    pub dictionary: u64,
1243}
1244
1245impl ColumnLayout {
1246    /// Everything this column costs, which is what the file would lose if the column went.
1247    #[must_use]
1248    pub fn total(&self) -> u64 {
1249        self.pages
1250            .saturating_add(self.memberships)
1251            .saturating_add(self.sieves)
1252            .saturating_add(self.part_ranges)
1253            .saturating_add(self.dictionary)
1254    }
1255}
1256
1257/// Where a whole file's bytes went.
1258///
1259/// Every number here comes out of the committed directory, so taking it costs one directory read
1260/// however large the file is. That is the point: a 45 GB table has to be able to say where it went
1261/// without being read, or nobody will ask.
1262///
1263/// The parts that are not a column are kept apart rather than shared out over the columns. The
1264/// stripe index page holds a section per column and could be split, and the directory and the
1265/// header cannot be, so splitting one of the three and not the others would read as if the columns
1266/// accounted for everything. They do not, and the gap is the thing worth looking at.
1267#[derive(Debug, Clone)]
1268pub struct Layout {
1269    /// The size of the file on disk.
1270    pub file: u64,
1271    /// Committed rows.
1272    pub rows: usize,
1273    /// Committed stripes.
1274    pub stripes: usize,
1275    /// Committed parts, which is how many chunks a scan reads.
1276    pub parts: usize,
1277    /// One entry per column, in the table's column order.
1278    pub columns: Vec<ColumnLayout>,
1279    /// Every stripe's index page, which carries a length and a checksum for every part of every
1280    /// column and is charged per stripe rather than per column.
1281    pub indexes: u64,
1282    /// The committed directory itself, the one that was read to build this.
1283    pub directory: u64,
1284    /// The fixed header, which holds the magic, the format and the two directory slots.
1285    pub header: u64,
1286}
1287
1288impl Layout {
1289    /// Everything the columns cost together.
1290    #[must_use]
1291    pub fn columns_total(&self) -> u64 {
1292        self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1293    }
1294
1295    /// What the file holds that this does not account for.
1296    ///
1297    /// A committed file is written once and never rewritten in place, so an earlier directory and
1298    /// the pages of an earlier snapshot are still in it. That is the honest place for them: they
1299    /// are bytes on disk that no column owns.
1300    #[must_use]
1301    pub fn unaccounted(&self) -> u64 {
1302        self.file
1303            .saturating_sub(self.columns_total())
1304            .saturating_sub(self.indexes)
1305            .saturating_sub(self.directory)
1306            .saturating_sub(self.header)
1307    }
1308}
1309
1310/// How one part of one column is stored, which is one row of `pragma_storage_info`.
1311///
1312/// Everything here is read off the file rather than worked out from the schema, because the whole
1313/// question this answers is what the encoder chose, and the encoder chooses per part. Two files
1314/// holding the same rows in a different order give different answers and that difference is the
1315/// reason to ask.
1316///
1317/// The encoding costs a read of the column's page, so this is not free the way [`Layout`] is. It is
1318/// one read per column per stripe rather than one per part, because a part is a few kilobytes out
1319/// of a page that is a quarter of a megabyte.
1320#[derive(Debug, Clone)]
1321pub struct StoredPart {
1322    /// Which stripe the part belongs to.
1323    pub stripe: usize,
1324    /// Which part of that stripe it is, counting from zero inside the stripe.
1325    pub part: usize,
1326    /// The table wide row number the part starts at.
1327    pub row: usize,
1328    /// How many rows it holds.
1329    pub rows: usize,
1330    /// What the encoder made of it, as a line of text like `DICT(PACKED, PACKED)`.
1331    pub encoding: String,
1332    /// The stored bytes of the part, which is what it costs in the file.
1333    pub bytes: u64,
1334    /// Where in the file the column page holding this part starts.
1335    pub page: u64,
1336    /// Where in that page the part starts.
1337    pub offset: u64,
1338    /// The smallest value the part holds, when the stored ranges say.
1339    pub low: Option<Value>,
1340    /// The largest, same.
1341    pub high: Option<Value>,
1342    /// How many of its rows are null, when the stored ranges say.
1343    pub nulls: Option<usize>,
1344}
1345
1346/// Seeds the second hash a global dictionary tells its values apart by.
1347///
1348/// Any value that is not zero does, since zero is the seed [`checksum`] already uses and the point
1349/// is only that the two hashes of one value are not the same number. This one is the fractional part
1350/// of the golden ratio in sixty four bits, which is the constant everything else here is built out
1351/// of and is as good a nothing-up-my-sleeve number as any.
1352const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1353
1354/// One column's table wide dictionary while the load is running.
1355///
1356/// The thing to understand about this is what it does not hold. A dictionary of `URL` at a hundred
1357/// million ClickBench rows has about eighteen million distinct values and 1.3 GB of bytes in them,
1358/// and five columns like it are twelve of the seventeen gigabytes a load of `hits` peaks at. So the
1359/// bytes are not kept. A value's bytes go into [`GlobalDictionary::filling`], and when that reaches
1360/// [`TEXT_PAYLOAD_VALUES`] values the block is sealed, handed out at the end of the merge that
1361/// sealed it to be encoded with the stripe's pages, and never seen in that form again. What is left
1362/// is the encoded block, which is two to three times smaller, and that is the same bytes the file
1363/// is going to hold anyway.
1364///
1365/// Two things needed the raw bytes and neither needs them now. Deciding whether a value has been
1366/// seen before was a hash lookup and then a comparison of the bytes, and is now a hash lookup and a
1367/// comparison of a second hash under a different seed, which is [`DICTIONARY_CHECK_SEED`] and the
1368/// argument for why that is sound. Sorting the values at the end needed all of them at once, and
1369/// now reads the blocks back through [`GlobalDictionary::decoded`] one column at a time, which is
1370/// one column's bytes rather than every column's.
1371///
1372/// The offsets going block relative comes free with it, and takes the four gigabyte wall with it.
1373/// They were `u32` into a per column payload, so a column could not hold more than four gigabytes of
1374/// values however much memory the machine had, and `URL` and `Referer` are within a small factor of
1375/// that at a hundred million rows. A `u32` into a block of 1,024 values is not a bound anything real
1376/// reaches. The stored form is unchanged, because [`encode_offsets`] was already subtracting a per
1377/// block base before writing.
1378#[derive(Debug)]
1379struct GlobalDictionary {
1380    /// Keyed by the value's hash, which is already well spread, so the maps hash it once more
1381    /// with a multiply rather than with SipHash. SipHash here was one percent of a ClickBench load,
1382    /// and every stripe's merge of a column waits on the one before it.
1383    primary: HashMap<u64, u32, Spread>,
1384    collisions: HashMap<u64, Vec<u32>, Spread>,
1385    /// Every value's hash under [`DICTIONARY_CHECK_SEED`], in code order.
1386    checks: Vec<u64>,
1387    /// Where every value ends inside the payload block it is in, in code order.
1388    ends: Vec<u32>,
1389    counts: Vec<u64>,
1390    nulls: u64,
1391    /// The values of the block being filled, back to back.
1392    filling: Vec<u8>,
1393    /// One conservative four-byte substring signature per encoded payload block, in block order.
1394    ///
1395    /// Made where the block is encoded rather than where it is sealed, because sealing is under the
1396    /// writer's lock and every byte of every value going through [`gram_bits`] was 2.9 of the 14
1397    /// seconds the 10m ClickBench load spent on the 32 core box.
1398    grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1399    /// Blocks that have filled and not been handed out to be encoded yet, each with its block number.
1400    ///
1401    /// Empty except inside the merge that filled them, and while the column is still too small to
1402    /// settle a shape on.
1403    waiting: Vec<(usize, Vec<u8>)>,
1404    /// Blocks kept raw to settle a shape on, spread across the column, each with its number.
1405    ///
1406    /// At most [`PAYLOAD_SAMPLE_BLOCKS`] of them and so at most a few megabytes. Spread rather than
1407    /// taken off the front for the reason [`settle_shape`] gives, and kept rather than read back
1408    /// because reading back is a decode and this is a sample of a column that is still growing.
1409    sample: Vec<(usize, Vec<u8>)>,
1410    /// How far apart the blocks in `sample` are, which doubles every time there are too many.
1411    stride: usize,
1412    /// What the blocks encoded so far were encoded with, once the column is big enough to settle it.
1413    shape: Option<chooser::Settled>,
1414    /// How many blocks had filled when that shape was settled.
1415    settled: usize,
1416    /// The blocks that are encoded and not yet in the file, in block order, following `placed`.
1417    ///
1418    /// Empty between stripes, because [`Writer::place_blocks`] writes them the moment they come
1419    /// back. Only a dictionary that never meets a writer, which is a test's, keeps them here.
1420    blocks: Vec<Vec<u8>>,
1421    /// Blocks that came back encoded ahead of a block before them, by block number.
1422    ///
1423    /// Two stripes merged one after the other can have their pages built in the other order, and a
1424    /// block cannot go into `blocks` until every block before it is there. They wait here until the
1425    /// gap closes, which is at most until the stripe merged just before this one is written.
1426    early: BTreeMap<usize, EncodedBlock>,
1427    /// Where every block already written to the file is, in block order.
1428    placed: Vec<Placed>,
1429    /// What the dictionary held the last time it was asked, see [`Self::recharge`], which is also
1430    /// what the load profile was told when there is one.
1431    charged: u64,
1432    /// Whether the dictionary stopped taking values, see [`Self::demote`].
1433    demoted: bool,
1434}
1435
1436/// Where one payload block of a global dictionary is in the file, and its checksum.
1437#[derive(Debug, Clone, Copy)]
1438struct Placed {
1439    start: u64,
1440    length: u64,
1441    hash: u64,
1442}
1443
1444/// Sorted `(head, code)` entries and the decoded bytes and block bases they were sorted over.
1445type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1446
1447impl GlobalDictionary {
1448    fn new() -> Self {
1449        Self {
1450            primary: HashMap::default(),
1451            collisions: HashMap::default(),
1452            checks: Vec::new(),
1453            ends: Vec::new(),
1454            counts: Vec::new(),
1455            nulls: 0,
1456            filling: Vec::new(),
1457            grams: Vec::new(),
1458            waiting: Vec::new(),
1459            sample: Vec::new(),
1460            stride: 1,
1461            shape: None,
1462            settled: 0,
1463            blocks: Vec::new(),
1464            early: BTreeMap::new(),
1465            placed: Vec::new(),
1466            charged: 0,
1467            demoted: false,
1468        }
1469    }
1470
1471    /// How many distinct values this dictionary holds, which is one past its largest code.
1472    fn values(&self) -> usize {
1473        self.ends.len()
1474    }
1475
1476    /// About how many bytes closing this dictionary holds at once: every value decoded, and a
1477    /// sort entry and a code for each.
1478    fn closing_bytes(&self) -> usize {
1479        let values = self.values();
1480        let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1481            .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1482            .sum::<usize>();
1483        decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1484    }
1485
1486    /// About what the dictionary holds in memory, by capacity rather than by length.
1487    ///
1488    /// A hash table is charged its buckets, which is a power of two over eight sevenths of what it
1489    /// says it can hold, and a byte of control per bucket. The blocks waiting to be encoded and the
1490    /// ones kept to settle a shape on are counted one by one, and there are only ever a few.
1491    fn held_bytes(&self) -> u64 {
1492        fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1493            (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1494        }
1495        fn spilled<T>(values: &Vec<T>) -> usize {
1496            values.capacity() * size_of::<T>()
1497        }
1498        let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1499            spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1500        };
1501        let bytes = table(&self.primary)
1502            + table(&self.collisions)
1503            + self.collisions.values().map(spilled).sum::<usize>()
1504            + spilled(&self.checks)
1505            + spilled(&self.ends)
1506            + spilled(&self.counts)
1507            + self.filling.capacity()
1508            + spilled(&self.grams)
1509            + raw(&self.waiting)
1510            + raw(&self.sample)
1511            + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1512            + spilled(&self.placed);
1513        bytes as u64
1514    }
1515
1516    /// Tells `profile` what the dictionary has grown or shrunk by since the last time, and hands
1517    /// back what it held then and what it holds now.
1518    fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1519        let before = self.charged;
1520        let now = self.held_bytes();
1521        if let Some(profile) = profile {
1522            if now >= before {
1523                profile.hold(now - before);
1524            } else {
1525                profile.release(before - now);
1526            }
1527        }
1528        self.charged = now;
1529        (before, now)
1530    }
1531
1532    /// Stops the dictionary taking values, for good.
1533    ///
1534    /// The block being filled is sealed so that it goes out with the others, and what the
1535    /// dictionary keeps for looking values up is let go of, which on a column of mostly new values
1536    /// is most of what it holds. What stays is what the close needs to write the dictionary's page:
1537    /// where every value ends, how often each was seen and where its blocks went. The stripes that
1538    /// were coded against it still need that page to be read. See [`DEMOTED`].
1539    fn demote(&mut self) {
1540        if self.demoted {
1541            return;
1542        }
1543        self.seal_rest();
1544        self.release_lookup();
1545        self.demoted = true;
1546    }
1547
1548    /// Frees what the dictionary keeps for coding new values, once none are coming.
1549    ///
1550    /// The hash tables, the check hash of every value and the blocks kept to settle a shape on are
1551    /// what a merge looks values up in. The close reads the counts, the ends and the written blocks
1552    /// and none of these, which are most of what the dictionary holds per value, so they go before
1553    /// the close takes memory of its own rather than after.
1554    fn release_lookup(&mut self) {
1555        self.primary = HashMap::default();
1556        self.collisions = HashMap::default();
1557        self.checks = Vec::new();
1558        self.sample = Vec::new();
1559        self.filling = Vec::new();
1560    }
1561
1562    /// How many blocks are encoded, written or not, which is the number the next one has to have.
1563    fn encoded(&self) -> usize {
1564        self.placed.len() + self.blocks.len()
1565    }
1566
1567    #[cfg(test)]
1568    fn code(&mut self, text: &str) -> Result<u32> {
1569        let bytes = text.as_bytes();
1570        self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1571    }
1572
1573    /// The code for a value whose two hashes the caller already has.
1574    ///
1575    /// A stripe prepared outside the writer's lock hashed every value it holds while it was coding
1576    /// them, and merging it into this dictionary is one of these a distinct value rather than two
1577    /// hashes of every row. See [`prepare`].
1578    fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1579        if let Some(&code) = self.primary.get(&hash) {
1580            if self.checks.get(code as usize) == Some(&check) {
1581                return Ok(code);
1582            }
1583            if let Some(codes) = self.collisions.get(&hash) {
1584                if let Some(code) =
1585                    codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1586                {
1587                    return Ok(code);
1588                }
1589            }
1590            let code = self.insert(text, check)?;
1591            self.collisions.entry(hash).or_default().push(code);
1592            return Ok(code);
1593        }
1594        let code = self.insert(text, check)?;
1595        self.primary.insert(hash, code);
1596        Ok(code)
1597    }
1598
1599    fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1600        if self.demoted {
1601            return Err(Error::internal("a value was coded against a demoted dictionary"));
1602        }
1603        let code = u32::try_from(self.ends.len())
1604            .map_err(|_| invalid("global dictionary has too many values"))?;
1605        self.filling.extend_from_slice(text);
1606        self.ends.push(
1607            u32::try_from(self.filling.len())
1608                .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1609        );
1610        self.checks.push(check);
1611        self.counts.push(0);
1612        if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1613            self.seal();
1614        }
1615        Ok(code)
1616    }
1617
1618    /// Closes the block being filled and puts it in the queue to be encoded.
1619    ///
1620    /// Also keeps a copy of it if it lands on the sample's stride, and halves the sample when that
1621    /// has left too many, which is what keeps the kept blocks spread evenly over however much of the
1622    /// column exists rather than bunched at whichever end was cheap to remember.
1623    fn seal(&mut self) {
1624        let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1625        let bytes = std::mem::take(&mut self.filling);
1626        if at % self.stride == 0 {
1627            self.sample.push((at, bytes.clone()));
1628            if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1629                self.stride *= 2;
1630                let stride = self.stride;
1631                self.sample.retain(|(at, _)| at % stride == 0);
1632            }
1633        }
1634        self.waiting.push((at, bytes));
1635    }
1636
1637    /// The values of one block, as slices into the bytes the block was filled with.
1638    fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1639        block_values(self.block_ends(at), bytes)
1640    }
1641
1642    /// Where every value of one block ends, relative to the block.
1643    fn block_ends(&self, at: usize) -> &[u32] {
1644        let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1645        let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1646        &self.ends[first..last]
1647    }
1648
1649    /// Takes every waiting block out to be encoded somewhere else, if the column has a shape to
1650    /// encode them with.
1651    ///
1652    /// This is what keeps the encoding out of the writer's lock. A block needs its bytes, where its
1653    /// values end and the shape, and nothing else of the dictionary, so it goes out with a copy of
1654    /// the four kilobytes of ends it has and comes back through [`GlobalDictionary::take_back`].
1655    fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1656        let Some(shape) = &self.shape else { return Vec::new() };
1657        let waiting = std::mem::take(&mut self.waiting);
1658        waiting
1659            .into_iter()
1660            .map(|(at, bytes)| Unencoded {
1661                column,
1662                at,
1663                ends: self.block_ends(at).to_vec(),
1664                bytes,
1665                shape: shape.clone(),
1666            })
1667            .collect()
1668    }
1669
1670    /// Takes back one block that was handed out, and moves every block that is now next in line
1671    /// into `blocks`.
1672    fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1673        if at < self.encoded() || self.early.insert(at, block).is_some() {
1674            return Err(Error::internal("a dictionary block came back twice"));
1675        }
1676        while let Some(block) = self.early.remove(&self.encoded()) {
1677            self.push_block(block);
1678        }
1679        Ok(())
1680    }
1681
1682    /// Appends the next encoded block and its signature.
1683    fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1684        self.blocks.push(bytes);
1685        self.grams.push(*grams);
1686    }
1687
1688    /// Settles the shape the waiting blocks are about to be encoded with, if there is enough column
1689    /// to settle one on.
1690    ///
1691    /// Settled again once the column has grown fourfold, because the sample it was settled on then
1692    /// covered a quarter of what exists now and a dictionary in first seen order does not look the
1693    /// same at both ends. Blocks already encoded keep the shape they were encoded with. They can,
1694    /// because a block says what it is: nothing reading one asks the column what shape to expect.
1695    fn settle(&mut self) -> Result<()> {
1696        if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1697            return Ok(());
1698        }
1699        self.settle_on_sample()
1700    }
1701
1702    /// Settles a shape on whatever sample there is, for a column the load ended before it had
1703    /// enough of to settle one the usual way.
1704    ///
1705    /// Such a column has fewer than [`PAYLOAD_SAMPLE_BLOCKS`] blocks, so the sample is every block
1706    /// it has. Trying every candidate on each of them instead runs at two to six megabytes a second,
1707    /// and once `hits` stored its string columns with a dictionary, the forty or so small ones were
1708    /// more than half the CPU of a million row load, all of it in the close.
1709    fn settle_rest(&mut self) -> Result<()> {
1710        if self.shape.is_some() || self.sample.is_empty() {
1711            return Ok(());
1712        }
1713        self.settle_on_sample()
1714    }
1715
1716    fn settle_on_sample(&mut self) -> Result<()> {
1717        let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1718        if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1719            return Ok(());
1720        }
1721        let sample =
1722            self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1723        self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1724        self.settled = complete;
1725        Ok(())
1726    }
1727
1728    /// Seals the part block at the end of the load, if there is one.
1729    fn seal_rest(&mut self) {
1730        // Asked of the values rather than of the bytes, because a block of empty strings has values
1731        // in it and no bytes, and a column of nulls is exactly that. A demoted dictionary sealed its
1732        // part block when it was demoted and has taken nothing since.
1733        if !self.demoted && self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1734            self.seal();
1735        }
1736    }
1737
1738    /// Encodes the waiting block at `at`, with the settled shape when there is one and by trying
1739    /// everything when the column was too small to settle one.
1740    fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1741        let (block, bytes) = &self.waiting[at];
1742        let values = self.slices(*block, bytes);
1743        let encoded = match &self.shape {
1744            Some(shape) => string::encode_with(&values, shape)?,
1745            None => string::encode(&values)?,
1746        };
1747        Ok((encoded, block_grams(&values)))
1748    }
1749
1750    /// [`finish_dictionaries`] for one dictionary on this thread, for the tests that hold one.
1751    #[cfg(test)]
1752    fn finish_blocks(&mut self) -> Result<()> {
1753        self.seal_rest();
1754        let made = (0..self.waiting.len())
1755            .map(|at| self.encode_waiting(at))
1756            .collect::<Result<Vec<_>>>()?;
1757        for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1758            if self.encoded() != at {
1759                return Err(Error::internal("a dictionary block was encoded out of order"));
1760            }
1761            self.push_block(block);
1762        }
1763        Ok(())
1764    }
1765
1766    /// Every value of this dictionary read back out of its encoded blocks, as the bytes back to back
1767    /// and where each block starts in them.
1768    ///
1769    /// This is the one place the whole column is in memory at once and the reason [`Writer::close`]
1770    /// takes the columns one at a time rather than across threads. One column's values is 1.3 GB on
1771    /// the worst ClickBench column, and five columns of that at once is the peak this was all meant
1772    /// to remove.
1773    ///
1774    /// The blocks are spread over threads instead. Each block's decoded length is already known from
1775    /// the ends of its values, so the answer is laid out before anything is decoded and every thread
1776    /// decodes its own run of blocks straight into its own part of it. On the 10m ClickBench sample
1777    /// this was a second of the close for `URL` alone, on one core of thirty two, and the close is
1778    /// what a load waits on once its stripes are written.
1779    ///
1780    /// The blocks already written are read back out of `file`, so what a thread holds beyond the
1781    /// answer is one encoded block. They were written moments or minutes ago and are almost always
1782    /// still in the page cache, so this is a copy rather than a read of the disk.
1783    fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1784        let count = self.placed.len() + self.blocks.len();
1785        if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1786            return Err(invalid("global dictionary blocks do not cover its values"));
1787        }
1788        let mut bases = Vec::with_capacity(count);
1789        let mut total = 0_usize;
1790        for block in 0..count {
1791            bases.push(total as u64);
1792            let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1793            total = total
1794                .checked_add(self.ends[last] as usize)
1795                .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1796        }
1797        let mut flat = vec![0_u8; total];
1798        let mut outs = Vec::with_capacity(count);
1799        let mut rest = flat.as_mut_slice();
1800        for block in 0..count {
1801            let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1802            let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1803            outs.push((block, out));
1804            rest = after;
1805        }
1806        let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1807            let mut stored = Vec::new();
1808            for (block, out) in run {
1809                let encoded = match self.placed.get(*block) {
1810                    Some(place) => {
1811                        let file = file.ok_or_else(|| {
1812                            Error::internal("a written dictionary block has no file")
1813                        })?;
1814                        let length = usize::try_from(place.length).map_err(|_| {
1815                            invalid("global dictionary block does not fit in memory")
1816                        })?;
1817                        stored.resize(length, 0);
1818                        read_at(file, place.start, &mut stored)?;
1819                        if checksum(&stored) != place.hash {
1820                            return Err(invalid(
1821                                "a global dictionary block did not read back as written",
1822                            ));
1823                        }
1824                        stored.as_slice()
1825                    }
1826                    None => &self.blocks[*block - self.placed.len()],
1827                };
1828                let decoded = string::decode_flat(encoded)?;
1829                if decoded.bytes().len() != out.len() {
1830                    return Err(invalid(
1831                        "a global dictionary block is not the length its ends say",
1832                    ));
1833                }
1834                out.copy_from_slice(decoded.bytes());
1835            }
1836            Ok(())
1837        };
1838        // Sixteen blocks a thread at the least, because a thread costs about what decoding a few
1839        // blocks does and most columns have one or two.
1840        let workers = close_workers().min(count / 16).max(1);
1841        if workers <= 1 {
1842            one(&mut outs)?;
1843        } else {
1844            let per = count.div_ceil(workers);
1845            std::thread::scope(|scope| {
1846                outs.chunks_mut(per)
1847                    .map(|run| scope.spawn(|| one(run)))
1848                    .collect::<Vec<_>>()
1849                    .into_iter()
1850                    .try_for_each(|handle| {
1851                        handle.join().map_err(|_| {
1852                            Error::internal("a global dictionary decode worker panicked")
1853                        })?
1854                    })
1855            })?;
1856        }
1857        drop(outs);
1858        Ok((flat, bases))
1859    }
1860
1861    /// Where the value at `code` sits in the bytes [`GlobalDictionary::decoded`] handed back.
1862    ///
1863    /// A block's first value starts at the block, and every other value starts where the one before
1864    /// it ended, which is what makes 1,024 values 1,024 numbers rather than 1,025.
1865    fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1866        let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1867        let Some(&end) = ends.get(code) else { return (0, 0) };
1868        let base = base as usize;
1869        let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1870        (base + from, base + end as usize)
1871    }
1872
1873    /// This dictionary's values in sorted order, each as the first eight bytes of the value and the
1874    /// code that holds it, so entry `rank` describes the value that sits at `rank` when the values
1875    /// are sorted by their bytes.
1876    ///
1877    /// Codes themselves stay in first appearance order, which is what lets the writer hand one out
1878    /// the moment it sees a value rather than waiting for the last stripe, and which also keeps a
1879    /// stripe's codes close together because the data is clustered. This is what puts the values
1880    /// back in order for anything that needs it, and it is separate from the codes so that getting
1881    /// it costs a sort of the distinct values at the end rather than a rewrite of every code page.
1882    ///
1883    /// The order is the byte order of the values and nothing else. The heads are attached after the
1884    /// sort rather than sorted on, because padding with zero on the right is order preserving for
1885    /// byte strings and so sorting by head and then by bytes lands in the same place as sorting by
1886    /// bytes: a shorter value differs from a longer one that starts the same way at a position
1887    /// where the shorter one has run out, and zero is below every byte that could be there.
1888    ///
1889    /// The heads are kept because a reader searching this order wants a comparison it can make out
1890    /// of the index alone. What they buy there depends entirely on the column and is much less than
1891    /// it looks on the columns that cost the most, which [`sort_by_value`] measures.
1892    fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1893        let (flat, bases) = self.decoded(file)?;
1894        let value = |code: u32| {
1895            let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1896            flat.get(from..to).unwrap_or_default()
1897        };
1898        let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1899        sort_by_value_across(&mut codes, value, close_workers());
1900        let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1901        Ok((order, flat, bases))
1902    }
1903
1904    #[cfg(test)]
1905    fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1906        self.ranked_with_values(file).map(|(order, _, _)| order)
1907    }
1908}
1909
1910/// Appends pages and commits a new directory.
1911///
1912/// One writer covers a whole file rather than one table. [`Writer::next`] closes the table it is on
1913/// and opens another over the same file, and [`Writer::finish`] commits every table it has closed in
1914/// one generation. That is what makes a checkpoint atomic across tables: there is one slot write at
1915/// the end of it and a reader sees every table at the generation before it or every table at the
1916/// generation after it.
1917#[derive(Debug)]
1918pub struct Writer {
1919    /// The file, through `rudb-io` rather than `std::fs`, so that a test can hand the writer a
1920    /// simulated filesystem and crash a load at every call it makes.
1921    file: Box<dyn rudb_io::File>,
1922    /// Where the next write goes, counted here rather than asked of the file.
1923    ///
1924    /// The file's own cursor is not ours. Building the numeric frequencies reads pages back through
1925    /// [`read_at`], and a positional read is only positional about where it reads from: `pread`
1926    /// leaves the cursor alone, and the call Windows has for it moves the cursor to the end of what
1927    /// it read. A writer that asked the file where it was would then write the directory over a
1928    /// page it had already written, which is what it did.
1929    at: u64,
1930    /// How far into the file the kernel has been asked to start writing, see [`WRITEBACK_STRETCH`].
1931    written_back: u64,
1932    table: Table,
1933    generation: u64,
1934    /// The first and the last source position in every stripe, in the order the stripes were
1935    /// written.
1936    order: Vec<((u64, u64), (u64, u64))>,
1937    next_order: u64,
1938    dictionaries: Vec<Option<GlobalDictionary>>,
1939    /// Which columns still have a global dictionary, shared with every [`Preparer`] this writer
1940    /// hands out so that a stripe prepared after a column lost its dictionary is not coded for it.
1941    coded: Arc<prepare::Coding>,
1942    /// One per column, folding the rows into a summary and a sketch as they go past.
1943    ///
1944    /// `None` for a column with no hash rule, which is the interval and the nested types. See
1945    /// [`stats::Gather`] for why the statistics are built here rather than by reading the file back
1946    /// once it is committed.
1947    gathers: Vec<Option<stats::Gather>>,
1948    /// The dictionaries and the statistics while a [`Merger`] has them, which is from
1949    /// [`Writer::merger`] until the table is closed. `dictionaries` and `gathers` are empty then.
1950    lent: Option<Arc<Lent>>,
1951    pending: Vec<PendingChunk>,
1952    /// The tables already closed in this generation, in the order they were written.
1953    closed: Vec<Entry>,
1954    /// The views the next commit writes down, which [`Writer::with_views`] sets.
1955    ///
1956    /// Carried forward from the committed generation by [`Writer::open`], so a writer that was only
1957    /// opened to append a table does not have to know about views to avoid dropping them.
1958    views: Vec<ViewEntry>,
1959    /// Where the stages this writer runs are charged, which [`Writer::with_profile`] sets.
1960    ///
1961    /// The writer runs the page builder, the dictionary blocks, the writes and the publish, and it
1962    /// charges them once per stripe and once per worker, never per chunk. See
1963    /// `rudb_metrics::LoadProfile` for why that is the grain.
1964    profile: Option<Arc<LoadProfile>>,
1965}
1966
1967/// A chunk that has arrived and is waiting for the rest of its stripe.
1968///
1969/// The rows are kept rather than the pages they encode to, which is the whole of #808's first half.
1970/// Encoding on arrival put every column of every part on the thread that called `append_at`, and
1971/// that thread is the only one the load has. Encoding at the flush instead means a stripe's worth
1972/// of work is on the table at once, and a stripe splits by column into a hundred and five pieces
1973/// that share nothing.
1974#[derive(Debug)]
1975struct PendingChunk {
1976    order: (u64, u64),
1977    chunk: Chunk,
1978}
1979
1980/// What the writer still needs of a part once its columns are encoded: where in the source it came
1981/// from, how many rows it has and how large those rows were.
1982///
1983/// A stripe waiting for the writer's lock carries these rather than its chunks, so its rows are
1984/// freed as soon as they are encoded and not after the stripe is written. See [`prepare`].
1985#[derive(Debug, Clone, Copy)]
1986struct Part {
1987    order: (u64, u64),
1988    rows: usize,
1989    footprint: usize,
1990}
1991
1992impl Part {
1993    fn of(pending: &PendingChunk) -> Self {
1994        Self {
1995            order: pending.order,
1996            rows: pending.chunk.len(),
1997            footprint: pending.chunk.footprint(),
1998        }
1999    }
2000}
2001
2002/// One column's share of a stripe, which is what one encode worker produces.
2003///
2004/// Indexed by part, so a stripe is a column of these and the write loop reads down one of them.
2005/// That is also the order the loop wanted: `flush_pending` walks a column at a time and lays its
2006/// parts next to each other, and it used to reach across a row of parts to do it.
2007#[derive(Debug, Default)]
2008struct ColumnStripe {
2009    pages: Vec<Vec<u8>>,
2010    codes: Vec<Option<Vec<u32>>>,
2011    sieves: Vec<Option<Sieve>>,
2012    ranges: Vec<Range>,
2013}
2014
2015/// Whether a column of this type is coded against a global dictionary.
2016///
2017/// A dictionary, its codes and the membership index beside them are about bytes and not about
2018/// text, so a blob gets one the same as a varchar does. ClickBench's `hits.parquet` stores every
2019/// string column as a plain byte array, which reads back as a blob, and those columns were being
2020/// written as a length and the bytes for every row: 533 MB for the first million rows where DuckDB
2021/// writes 142.
2022fn coded_type(ty: &LogicalType) -> bool {
2023    matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2024}
2025
2026/// The tag a directory gives a column's global dictionary.
2027///
2028/// A varchar's is 1, as it always was. A blob's is 2, so that a reader from before blobs had
2029/// dictionaries meets a tag it does not know and refuses the file, rather than laying the rest of
2030/// the directory out as if the blob columns had no dictionary and reading everything after the
2031/// first one from the wrong place.
2032fn dictionary_tag(ty: &LogicalType) -> u8 {
2033    if ty == &LogicalType::Blob { 2 } else { 1 }
2034}
2035
2036/// Roughly what encoding a column of this type costs, for ordering the encode queue.
2037///
2038/// Only the order matters and only roughly. A string column hashes and copies every value into a
2039/// dictionary and is in a different class from everything else, and among the fixed widths the wide
2040/// ones carry more bytes through the cascade than the narrow ones. Anything finer than that would
2041/// be a cost model, and the queue already absorbs a wrong guess: it only has to avoid finishing on
2042/// a column nobody else can help with.
2043fn weight(ty: &LogicalType) -> usize {
2044    match ty {
2045        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2046        LogicalType::HugeInt
2047        | LogicalType::UHugeInt
2048        | LogicalType::Uuid
2049        | LogicalType::Interval => 16,
2050        LogicalType::BigInt
2051        | LogicalType::UBigInt
2052        | LogicalType::Timestamp
2053        | LogicalType::Time
2054        | LogicalType::TimeTz
2055        | LogicalType::TimestampTz
2056        | LogicalType::TimestampS
2057        | LogicalType::TimestampMs
2058        | LogicalType::TimestampNs
2059        | LogicalType::Double
2060        | LogicalType::Decimal { .. } => 8,
2061        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2062        LogicalType::SmallInt | LogicalType::USmallInt => 2,
2063        _ => 1,
2064    }
2065}
2066
2067/// Parts in one stripe.
2068///
2069/// Sixty four thousand rows is the smallest stripe that keeps the ClickBench directory in single
2070/// digit megabytes at a hundred million rows, and it puts a four byte column's page at a quarter of
2071/// a megabyte, which is the size a sequential read wants. Larger stripes buy a smaller directory
2072/// and cost a sparse fetch, which has to read a page index before it can reach one part.
2073pub const STRIPE_PARTS: usize = 64;
2074
2075/// How many rows the writer wants to see before it decides whether a varchar column gets to keep
2076/// its global dictionary.
2077///
2078/// See [`prepare::drops_dictionary`]. A stripe is up to [`STRIPE_PARTS`] parts, so most tables give it
2079/// far more than this and it binds only on a table that is smaller than one stripe. A handful of
2080/// rows says nothing about whether a column repeats itself, and the answer that costs nothing when
2081/// the sample is that small is the one the writer has always given, which is to keep the dictionary.
2082const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2083
2084/// Out of ten. A varchar column loses its dictionary when more than this many rows in ten of the
2085/// first stripe held a value that stripe had not seen before.
2086///
2087/// See [`prepare::drops_dictionary`]. Nine and not five, because the properties a dictionary buys are
2088/// worth keeping everywhere they are real. On ClickBench the widest string column is `Referer` at
2089/// 0.131 of its first stripe and every other one is below that, so nothing there is near this and
2090/// every one of them keeps its dictionary, which is what a group by on codes wants. On TPC-H
2091/// `o_comment` and `c_comment` are at 0.97 and are what this catches.
2092///
2093/// `l_comment` sits at 0.883 and so keeps its dictionary. Eight was built and measured rather than
2094/// argued about, and it is not a clear win: it takes `select l_comment from lineitem` from 4.335 G
2095/// instructions to 3.473 G and the file from 280.2 MB to 260.4 MB, and it takes a `like` over the
2096/// same column from 3.29 G to 4.27 G, because a dictionary runs the predicate once a distinct value
2097/// and there are 3.6 M of those to 6.0 M rows. The 22 query suite came out 6.91 s against 7.07 s in
2098/// favour of nine. So nine stays until there is a reason to prefer one of those shapes. See #1137.
2099const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2100
2101/// Bytes one part takes in a stripe's index page: four for the length, eight for the checksum.
2102const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2103
2104/// Bytes one column's section of a stripe's index page takes, including its own trailing checksum.
2105fn index_section(parts: usize) -> Result<usize> {
2106    parts
2107        .checked_mul(INDEX_ENTRY)
2108        .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2109        .ok_or_else(|| invalid("index page length overflow"))
2110}
2111
2112impl Writer {
2113    /// Opens a committed file and starts a table in the generation after the one it holds.
2114    ///
2115    /// The tables already in the file are carried forward by name and by directory pointer, and
2116    /// their pages are not read. Nothing in the file is overwritten: the new table's pages and the
2117    /// new catalog go on the end, past the catalog the committed generation points at, and the one
2118    /// write that is not an append is the slot in the header that [`Writer::finish`] does last.
2119    ///
2120    /// That slot is the other one. A file committed at generation 1 is named by the slot at 16 and
2121    /// generation 2 writes the one at 44, so until the last four bytes of the commit land the file
2122    /// still reads as the generation before it, and a slot torn across a write fails its checksum
2123    /// and the reader falls back to the one beside it. This is what the second slot has always been
2124    /// for.
2125    ///
2126    /// # Errors
2127    ///
2128    /// If the file has no valid committed directory, is not this build's format, repeats the name
2129    /// of a table already in it that holds rows, has a field with no scalar encoding, or cannot be
2130    /// written.
2131    pub fn open(
2132        path: impl AsRef<Path>,
2133        name: impl Into<String>,
2134        fields: Vec<Field>,
2135    ) -> Result<Self> {
2136        Self::open_in(&RealFilesystem::new(), path, name, fields)
2137    }
2138
2139    /// [`Writer::open`] on a file in `fs`, which is how a crash test runs an append against the
2140    /// simulated filesystem.
2141    ///
2142    /// # Errors
2143    ///
2144    /// The same as [`Writer::open`].
2145    pub fn open_in(
2146        fs: &dyn Filesystem,
2147        path: impl AsRef<Path>,
2148        name: impl Into<String>,
2149        fields: Vec<Field>,
2150    ) -> Result<Self> {
2151        for field in &fields {
2152            type_tag(&field.ty)?;
2153        }
2154        let name = name.into();
2155        let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2156        let size = file.len()?;
2157        let (slot, bytes, _) = committed_slot(&*file, size)?;
2158        let (mut closed, views) = decode_catalog(&bytes, size)?;
2159        // A table already in the file under this name is only in the way if it holds rows. One that
2160        // holds none has no pages for this generation to carry and no reader that could lose
2161        // anything, so the table being started here takes its place in the catalog rather than
2162        // colliding with it, and `finish` writes the new entry where the old one was.
2163        //
2164        // That is not a corner. It is the shape every loading script writes: the schema goes in one
2165        // statement and the rows go in the next, and a checkpoint between them commits the empty
2166        // table. Before this, the second statement had to build the whole table in memory because
2167        // the first had already put the name in the file, which is how a load of a table larger
2168        // than memory became a load that needed memory the size of the table.
2169        if let Some(at) = closed.iter().position(|held| held.name == name) {
2170            if closed[at].rows > 0 {
2171                return Err(invalid("two tables in one native file have the same name"));
2172            }
2173            closed.remove(at);
2174        }
2175        // The generation of the slot whose bytes checksummed, and not the highest number in the
2176        // header. A slot torn across a write can hold any number at all, and taking that one would
2177        // be choosing which slot to overwrite from a value nothing has vouched for, which is how a
2178        // half written commit gets to destroy the one good copy beside it.
2179        let generation = slot
2180            .generation
2181            .checked_add(1)
2182            .ok_or_else(|| invalid("native file generation overflow"))?;
2183        Ok(Self {
2184            file,
2185            // The end of the file, so that the committed generation's catalog stays where its slot
2186            // says it is and keeps naming a file a reader can still open.
2187            at: size,
2188            written_back: size,
2189            dictionaries: fields
2190                .iter()
2191                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2192                .collect(),
2193            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2194            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2195            lent: None,
2196            table: Table {
2197                name,
2198                dictionaries: vec![None; fields.len()],
2199                dictionary_payloads: Vec::new(),
2200                demoted: Vec::new(),
2201                distincts: vec![None; fields.len()],
2202                fields,
2203                stripes: Vec::new(),
2204                rows: 0,
2205                frequencies: Vec::new(),
2206                pair_frequencies: Vec::new(),
2207                frequency_texts: Vec::new(),
2208                host_groups: None,
2209                clustering: None,
2210                generation,
2211                sections: Vec::new(),
2212            },
2213            generation,
2214            order: Vec::new(),
2215            next_order: 0,
2216            pending: Vec::with_capacity(STRIPE_PARTS),
2217            closed,
2218            views,
2219            profile: None,
2220        })
2221    }
2222
2223    /// Creates a new v10 file and its first table.
2224    ///
2225    /// # Errors
2226    ///
2227    /// If the file exists, a field has no scalar encoding, or the path cannot be written.
2228    pub fn create(
2229        path: impl AsRef<Path>,
2230        name: impl Into<String>,
2231        fields: Vec<Field>,
2232    ) -> Result<Self> {
2233        Self::create_in(&RealFilesystem::new(), path, name, fields)
2234    }
2235
2236    /// [`Writer::create`] with the file made in `fs` rather than on the real filesystem.
2237    ///
2238    /// Every call the writer makes on the file from here to [`Writer::finish`] goes to that
2239    /// filesystem, which is what lets a test built on `rudb_io::SimFilesystem` stop a load at any
2240    /// one of them and look at what a crash there would leave on the disk.
2241    ///
2242    /// # Errors
2243    ///
2244    /// The same as [`Writer::create`].
2245    pub fn create_in(
2246        fs: &dyn Filesystem,
2247        path: impl AsRef<Path>,
2248        name: impl Into<String>,
2249        fields: Vec<Field>,
2250    ) -> Result<Self> {
2251        for field in &fields {
2252            type_tag(&field.ty)?;
2253        }
2254        let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2255        let mut header = [0; HEADER as usize];
2256        header[..8].copy_from_slice(MAGIC);
2257        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2258        file.write_at(0, &header)?;
2259        Ok(Self {
2260            file,
2261            at: HEADER,
2262            written_back: HEADER,
2263            dictionaries: fields
2264                .iter()
2265                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2266                .collect(),
2267            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2268            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2269            lent: None,
2270            table: Table {
2271                name: name.into(),
2272                dictionaries: vec![None; fields.len()],
2273                dictionary_payloads: Vec::new(),
2274                demoted: Vec::new(),
2275                distincts: vec![None; fields.len()],
2276                fields,
2277                stripes: Vec::new(),
2278                rows: 0,
2279                frequencies: Vec::new(),
2280                pair_frequencies: Vec::new(),
2281                frequency_texts: Vec::new(),
2282                host_groups: None,
2283                clustering: None,
2284                generation: 1,
2285                sections: Vec::new(),
2286            },
2287            generation: 1,
2288            order: Vec::new(),
2289            next_order: 0,
2290            pending: Vec::with_capacity(STRIPE_PARTS),
2291            closed: Vec::new(),
2292            views: Vec::new(),
2293            profile: None,
2294        })
2295    }
2296
2297    /// Creates a new file that holds no table at all, committed and ready to open.
2298    ///
2299    /// A database somebody dropped the last table out of is still a database, and until this there
2300    /// was no way to write one down. Every other way into this file goes through a table, because
2301    /// [`Writer::create`] takes the first one and [`Writer::finish`] commits the one it is on, so a
2302    /// catalog with nothing in it could be read and not written. The format already allowed it: the
2303    /// catalog is a count and that many entries, and a count of nought encodes and decodes the same
2304    /// way every other count does, which is why nothing here is a version change.
2305    ///
2306    /// It hands back nothing rather than a writer, because a writer with no table is a writer with
2307    /// nothing to append to. A file that is going to hold a table is [`Writer::create`], and one
2308    /// that is going to have a table added to it later is [`Writer::open`], which reads what this
2309    /// wrote the same way it reads any other generation.
2310    ///
2311    /// It takes the views anyway, because a database with no table can still have views in it. A
2312    /// view over `range` or over another view names no table, so dropping the last table out of a
2313    /// database does not have to leave the catalog with nothing worth writing down.
2314    ///
2315    /// # Errors
2316    ///
2317    /// If the file exists or the path cannot be written.
2318    pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2319        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2320        let mut header = [0; HEADER as usize];
2321        header[..8].copy_from_slice(MAGIC);
2322        header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2323        file.write_at(0, &header)?;
2324        let catalog = encode_catalog(&[], views)?;
2325        file.write_at(HEADER, &catalog)?;
2326        // The same two syncs in the same order as [`Writer::finish`], and for the same reason. The
2327        // catalog is on the disk before the slot names it, so a file this is interrupted in the
2328        // middle of is a header with no valid slot rather than a slot pointing at nothing.
2329        file.sync()?;
2330        let slot = Slot {
2331            offset: HEADER,
2332            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2333            generation: 1,
2334            hash: checksum(&catalog),
2335        };
2336        file.write_at(slot_offset(1), &slot.bytes())?;
2337        file.sync()?;
2338        Ok(())
2339    }
2340
2341    /// Closes the table this writer is on and starts another one in the same file.
2342    ///
2343    /// Nothing is published here. The closed table's directory is written so that the bytes are on
2344    /// disk and its span is known, and the catalog that names it is only written by
2345    /// [`Writer::finish`], so a crash between two tables leaves the previous generation intact.
2346    ///
2347    /// # Errors
2348    ///
2349    /// If the name repeats a table already closed, a field has no scalar encoding, or the table
2350    /// being closed cannot be written.
2351    pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2352        for field in &fields {
2353            type_tag(&field.ty)?;
2354        }
2355        let name = name.into();
2356        let entry = self.close()?;
2357        if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2358            return Err(invalid("two tables in one native file have the same name"));
2359        }
2360        let Self { file, at, generation, mut closed, views, .. } = self;
2361        closed.push(entry);
2362        Ok(Self {
2363            file,
2364            written_back: at,
2365            at,
2366            generation,
2367            closed,
2368            views,
2369            profile: None,
2370            dictionaries: fields
2371                .iter()
2372                .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2373                .collect(),
2374            coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2375            gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2376            lent: None,
2377            table: Table {
2378                name,
2379                dictionaries: vec![None; fields.len()],
2380                dictionary_payloads: Vec::new(),
2381                demoted: Vec::new(),
2382                distincts: vec![None; fields.len()],
2383                fields,
2384                stripes: Vec::new(),
2385                rows: 0,
2386                frequencies: Vec::new(),
2387                pair_frequencies: Vec::new(),
2388                frequency_texts: Vec::new(),
2389                host_groups: None,
2390                clustering: None,
2391                generation,
2392                sections: Vec::new(),
2393            },
2394            order: Vec::new(),
2395            next_order: 0,
2396            pending: Vec::with_capacity(STRIPE_PARTS),
2397        })
2398    }
2399
2400    /// Sets the views the next commit writes down, replacing whatever was carried forward.
2401    ///
2402    /// It replaces rather than adds because the caller has the whole catalog in front of it and the
2403    /// writer does not. A view that was dropped is a view that is not in the list any more, and
2404    /// there is no other way for the writer to hear about that, since nothing else it is told about
2405    /// mentions views at all.
2406    ///
2407    /// A writer that is never told anything writes back the views it read at [`Writer::open`], so a
2408    /// checkpoint that only had a table to append does not quietly drop them.
2409    #[must_use]
2410    pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2411        self.views = views;
2412        self
2413    }
2414
2415    /// Charges the stages this writer runs to `profile`.
2416    ///
2417    /// For the table being written now. [`Writer::next`] starts the next table without one,
2418    /// because a second table's stripes charged to the first table's load would be a profile of
2419    /// neither.
2420    #[must_use]
2421    pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2422        self.profile = Some(profile);
2423        self
2424    }
2425
2426    /// Sets what the table's global dictionaries may hold between them before the one growing
2427    /// fastest stops taking values, which is [`DICTIONARY_CAP_BYTES`] unless this says
2428    /// otherwise. It applies to every [`Preparer`] and [`Merger`] this writer has handed out too.
2429    #[must_use]
2430    pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2431        self.coded.cap(bytes);
2432        self
2433    }
2434
2435    /// Records the order this table's rows are meant to be stored in.
2436    ///
2437    /// The declaration goes in the table directory and comes back out of
2438    /// [`Table::clustering`]. Nothing here sorts anything, and nothing here checks that the rows
2439    /// handed to [`Writer::append`] arrive in the order this claims. That is deliberate for now:
2440    /// the thing that was missing was a place to write the order down, and a loader that honours
2441    /// the declaration is the next piece rather than this one.
2442    ///
2443    /// The declaration applies to the table the writer is currently on, so it is set after
2444    /// [`Writer::next`] rather than once for the file.
2445    ///
2446    /// # Errors
2447    ///
2448    /// If the declaration names a column this table does not have.
2449    pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2450        // Rebuilt against this table's own column count rather than trusted, because the caller
2451        // built it against a catalog entry and the two could have drifted.
2452        self.table.clustering = Some(Clustering::new(
2453            clustering.columns().to_vec(),
2454            clustering.width(),
2455            &self.table.fields,
2456        )?);
2457        Ok(self)
2458    }
2459
2460    /// Appends bytes at the end of the file and moves the writer's own offset past them.
2461    ///
2462    /// Every write in here goes through this, so that [`Writer::at`] is the only answer to where
2463    /// anything is and the file's cursor is never consulted for it.
2464    fn put(&mut self, bytes: &[u8]) -> Result<()> {
2465        self.file.write_at(self.at, bytes)?;
2466        self.at = self
2467            .at
2468            .checked_add(bytes.len() as u64)
2469            .ok_or_else(|| invalid("native file length overflow"))?;
2470        if self.at - self.written_back >= WRITEBACK_STRETCH {
2471            self.file.start_writeback(self.written_back, self.at - self.written_back);
2472            self.written_back = self.at;
2473        }
2474        Ok(())
2475    }
2476
2477    /// Writes one chunk as independently readable column pages.
2478    ///
2479    /// # Errors
2480    ///
2481    /// If its width or types differ from the declared table, or a page exceeds its bound.
2482    pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2483        let order = (self.next_order, 0);
2484        self.next_order = self.next_order.saturating_add(1);
2485        self.append_at(order, chunk)
2486    }
2487
2488    /// Writes one chunk and records its source position for directory ordering.
2489    ///
2490    /// Pages may be encoded by parallel pipeline instances and reach the file in completion order.
2491    /// The stripe they land in is sorted by this key at commit, and [`Self::finish`] rejects a
2492    /// sequence whose parts do not come out in source order once the stripes are sorted, because a
2493    /// stripe groups whatever arrived together and cannot put a late part back where it belongs.
2494    ///
2495    /// # Errors
2496    ///
2497    /// The same as [`Self::append`].
2498    pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2499        if chunk.is_empty() {
2500            return Ok(());
2501        }
2502        self.admit(chunk)?;
2503        if self.pending.last().is_some_and(|last| last.order > order) {
2504            self.flush_pending()?;
2505        }
2506        // Cloned rather than encoded, and a clone of a chunk that owns its buffers is a copy of
2507        // them. Sixty four parts of a hundred and five columns is tens of megabytes held for the
2508        // length of a stripe and a few seconds of memory traffic over a whole ClickBench load,
2509        // against the hundreds of seconds of encode this is what lets off one thread.
2510        self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2511        if self.pending.len() == STRIPE_PARTS {
2512            self.flush_pending()?;
2513        }
2514        Ok(())
2515    }
2516
2517    /// Writes a run of chunks as one stripe of its own.
2518    ///
2519    /// [`Self::append_at`] decides where a stripe ends by watching the orders go past, which works
2520    /// when one caller hands over every chunk in source order and does not when several do. A
2521    /// writer being fed by more than one pipeline instance sees the orders interleave, and a stripe
2522    /// that ends every time two of them cross is a stripe of one or two parts.
2523    ///
2524    /// So the grouping moves to the caller. Whoever is buffering hands over a run it already knows
2525    /// is contiguous and in order, and gets a stripe holding exactly that run. The orders still
2526    /// have to come out in source order once the stripes are sorted, which [`Self::finish`] checks,
2527    /// so the runs from different callers may interleave with each other but may not overlap.
2528    ///
2529    /// # Errors
2530    ///
2531    /// The same as [`Self::append`], and if the run is longer than [`STRIPE_PARTS`].
2532    pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2533        if parts.len() > STRIPE_PARTS {
2534            return Err(invalid("a stripe was handed more parts than it holds"));
2535        }
2536        // Whatever an earlier caller left behind is its own stripe rather than the front of this
2537        // one, because the two runs are from different places in the source and a stripe is a run.
2538        self.flush_pending()?;
2539        for (order, chunk) in parts {
2540            if chunk.is_empty() {
2541                continue;
2542            }
2543            self.admit(&chunk)?;
2544            self.pending.push(PendingChunk { order, chunk });
2545        }
2546        self.flush_pending()
2547    }
2548
2549    /// Checks a chunk against the declared table and counts its rows in.
2550    fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2551        if chunk.width() != self.table.fields.len() {
2552            return Err(invalid("chunk width differs from table schema"));
2553        }
2554        for (index, field) in self.table.fields.iter().enumerate() {
2555            if chunk.column(index)?.logical_type() != &field.ty {
2556                return Err(invalid("chunk type differs from table schema"));
2557            }
2558        }
2559        self.table.rows = self
2560            .table
2561            .rows
2562            .checked_add(chunk.len())
2563            .ok_or_else(|| invalid("row count overflow"))?;
2564        Ok(())
2565    }
2566
2567    /// One column's parts of a stripe as pages, for a column with no global dictionary.
2568    fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2569        let mut stripe = ColumnStripe {
2570            pages: Vec::with_capacity(columns.len()),
2571            codes: Vec::with_capacity(columns.len()),
2572            sieves: Vec::with_capacity(columns.len()),
2573            ranges: Vec::with_capacity(columns.len()),
2574        };
2575        let mut settling = Settling::default();
2576        for &column in columns {
2577            Self::encode_page(&mut stripe, &mut settling, column)?;
2578        }
2579        Ok(stripe)
2580    }
2581
2582    /// One more part of a column with no global dictionary as a page, after the ones already in
2583    /// `stripe`. The parts have to come in order, since `settling` carries from one to the next.
2584    fn encode_page(
2585        stripe: &mut ColumnStripe,
2586        settling: &mut Settling,
2587        column: &Vector,
2588    ) -> Result<()> {
2589        let bytes = encode(column, settling)?;
2590        if bytes.len() > MAX_PAGE {
2591            return Err(invalid("column page exceeds the configured bound"));
2592        }
2593        // The range is built first because the sieve reads it rather than walking the column a
2594        // second time to find out how wide it is.
2595        let range = Range::of(column);
2596        // A sieve at least as large as the part it indexes is not written. A reader reads the
2597        // sieve to decide whether to read the part, so when the sieve is the larger of the two
2598        // it has already spent more than the read it is trying to avoid, and that holds even if
2599        // it rejects every time. It is a necessary condition rather than the whole rule, which
2600        // is that a sieve pays when its bytes are under the rejection rate times the part's,
2601        // but the rejection rate depends on what a query probes for and the writer does not
2602        // know that. The necessary half needs two numbers that are both in hand here.
2603        //
2604        // A column with a global dictionary gets none, because it already has an exact
2605        // membership index per stripe. Those do not come through here. See [`prepare`].
2606        let sieve =
2607            Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2608        stripe.pages.push(bytes);
2609        stripe.codes.push(None);
2610        stripe.sieves.push(sieve);
2611        stripe.ranges.push(range);
2612        Ok(())
2613    }
2614
2615    /// Writes every encoded dictionary block that is not in the file yet and forgets its bytes.
2616    ///
2617    /// This is what keeps a load from holding its dictionaries' payload. The blocks land between
2618    /// stripes wherever the writer is, which is fine because the index says where each one is.
2619    fn place_blocks(&mut self) -> Result<()> {
2620        if let Some(lent) = self.lent.clone() {
2621            return self.place_lent_blocks(&lent);
2622        }
2623        let mut dictionaries = std::mem::take(&mut self.dictionaries);
2624        let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2625            for block in std::mem::take(&mut dictionary.blocks) {
2626                let start = self.at;
2627                self.put(&block)?;
2628                dictionary.placed.push(Placed {
2629                    start,
2630                    length: block.len() as u64,
2631                    hash: checksum(&block),
2632                });
2633            }
2634            Ok(())
2635        });
2636        self.dictionaries = dictionaries;
2637        placed
2638    }
2639
2640    /// [`Writer::place_blocks`] while a [`Merger`] has the dictionaries.
2641    ///
2642    /// A column whose merge is running is passed over rather than waited for, because the writer's
2643    /// lock is held here and a merge of `URL` can take tens of milliseconds. Its blocks go out with
2644    /// a later stripe, or at the close.
2645    fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2646        for column in lent.columns() {
2647            let Ok(mut held) = column.try_lock() else { continue };
2648            let Some(dictionary) = held.dictionary.as_mut() else { continue };
2649            for block in std::mem::take(&mut dictionary.blocks) {
2650                let start = self.at;
2651                self.put(&block)?;
2652                dictionary.placed.push(Placed {
2653                    start,
2654                    length: block.len() as u64,
2655                    hash: checksum(&block),
2656                });
2657            }
2658        }
2659        Ok(())
2660    }
2661
2662    /// Takes the dictionaries and the statistics back from the [`Merger`] that has them.
2663    ///
2664    /// A merge that starts after this is refused, since whatever it merged would be lost.
2665    fn reclaim(&mut self) -> Result<()> {
2666        let Some(lent) = self.lent.take() else { return Ok(()) };
2667        let (dictionaries, gathers) = lent.reclaim()?;
2668        self.dictionaries = dictionaries;
2669        self.gathers = gathers;
2670        Ok(())
2671    }
2672
2673    /// Writes the buffered parts as one stripe, each column's parts contiguous on disk.
2674    ///
2675    /// The same four steps a caller holding this writer behind a lock takes, with nobody else
2676    /// waiting between them. See [`prepare`].
2677    fn flush_pending(&mut self) -> Result<()> {
2678        if self.pending.is_empty() {
2679            return Ok(());
2680        }
2681        let held = std::mem::take(&mut self.pending);
2682        let prepared = self.preparer().prepare_held(held)?;
2683        let merged = self.merge_held(prepared)?;
2684        let paged = merged.pages()?;
2685        self.write_paged(paged)
2686    }
2687
2688    /// Writes one stripe whose pages are built, each column's parts contiguous on disk.
2689    fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2690        let width = self.table.fields.len();
2691        let parts = held.len();
2692        if encoded.len() != width {
2693            return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2694        }
2695        let profile = self.profile.clone();
2696        if let Some(profile) = &profile {
2697            let rows = held.iter().map(|part| part.rows as u64).sum();
2698            let raw = held.iter().map(|part| part.footprint as u64).sum();
2699            let pages =
2700                encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2701            profile.moved(Stage::Pages, raw, pages, rows);
2702        }
2703        // Before a byte of the stripe is written, so that the blocks the stripe's pages were built
2704        // with, and any that were waiting on them, are let go of now rather than a stripe later.
2705        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2706        let before = self.at;
2707        self.place_blocks()?;
2708        drop(timing);
2709        if let Some(profile) = &profile {
2710            profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2711        }
2712        let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2713        let before = self.at;
2714        let mut pages = Vec::with_capacity(width);
2715        let mut memberships = vec![None; width];
2716        let mut ranges = Vec::with_capacity(width);
2717        let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2718        for stripe in &encoded {
2719            let offset = self.at;
2720            let section = index.len();
2721            let mut length = 0_usize;
2722            for bytes in &stripe.pages {
2723                self.file.write_at(self.at + length as u64, bytes)?;
2724                put_u32(
2725                    &mut index,
2726                    u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2727                );
2728                put_u64(&mut index, checksum(bytes));
2729                length = length
2730                    .checked_add(bytes.len())
2731                    .ok_or_else(|| invalid("column page length overflow"))?;
2732            }
2733            let hash = checksum(&index[section..]);
2734            put_u64(&mut index, hash);
2735            if length > MAX_PAGE {
2736                return Err(invalid("column page exceeds the configured bound"));
2737            }
2738            self.at = self
2739                .at
2740                .checked_add(length as u64)
2741                .ok_or_else(|| invalid("native file length overflow"))?;
2742            pages.push(Span {
2743                offset,
2744                length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2745            });
2746            ranges.push(merged_range(stripe.ranges.iter().cloned()));
2747        }
2748        for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2749            if stripe.codes.iter().all(Option::is_none) {
2750                continue;
2751            }
2752            let lists = stripe
2753                .codes
2754                .iter()
2755                .map(|codes| codes.clone().unwrap_or_default())
2756                .collect::<Vec<_>>();
2757            let bytes = encode_membership(&merged_codes(lists));
2758            let offset = self.at;
2759            self.put(&bytes)?;
2760            *membership = Some(Page {
2761                offset,
2762                length: u32::try_from(bytes.len())
2763                    .map_err(|_| invalid("membership page length overflow"))?,
2764                hash: checksum(&bytes),
2765            });
2766        }
2767        let mut sieves = vec![None; width];
2768        for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2769            if stripe.sieves.iter().all(Option::is_none) {
2770                continue;
2771            }
2772            let bytes = encode_sieves(stripe.sieves.iter())?;
2773            let offset = self.at;
2774            self.put(&bytes)?;
2775            *page = Some(Page {
2776                offset,
2777                length: u32::try_from(bytes.len())
2778                    .map_err(|_| invalid("sieve page length overflow"))?,
2779                hash: checksum(&bytes),
2780            });
2781        }
2782        // A stripe of one part has the same rows in it as that part, so its own bounds are already
2783        // the part's and a page here would say what the directory says. Everywhere else the page is
2784        // written unless it comes to more than the column it indexes, which is the rule the sieves
2785        // go by and for the same reason: a reader reads this to decide whether to read the column,
2786        // so a page larger than the column has spent more than the read it is avoiding.
2787        let mut part_ranges = vec![None; width];
2788        if parts > 1 {
2789            for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2790                let bytes = encode_part_ranges(&stripe.ranges)?;
2791                if bytes.len() >= span.length as usize {
2792                    continue;
2793                }
2794                let offset = self.at;
2795                self.put(&bytes)?;
2796                *page = Some(Page {
2797                    offset,
2798                    length: u32::try_from(bytes.len())
2799                        .map_err(|_| invalid("part range page length overflow"))?,
2800                    hash: checksum(&bytes),
2801                });
2802            }
2803        }
2804        let offset = self.at;
2805        self.put(&index)?;
2806        let index = Span {
2807            offset,
2808            length: u32::try_from(index.len())
2809                .map_err(|_| invalid("index page length overflow"))?,
2810        };
2811        let mut rows = 0_usize;
2812        let mut lengths = Vec::with_capacity(parts);
2813        let mut span = None;
2814        for part in held {
2815            rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2816            lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2817            span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2818        }
2819        self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2820        self.table.stripes.push(Stripe {
2821            rows,
2822            parts: lengths,
2823            index,
2824            pages,
2825            memberships: Pages::from_slots(memberships)?,
2826            sieves: Pages::from_slots(sieves)?,
2827            part_ranges: Pages::from_slots(part_ranges)?,
2828            zone: Zone::from_ranges(ranges),
2829        });
2830        drop(timing);
2831        if let Some(profile) = &profile {
2832            profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2833        }
2834        Ok(())
2835    }
2836
2837    /// Finds exact heavy hitters without keeping a hash table for every numeric column while the
2838    /// load is live. The pages are already in the target file, so one column at a time uses a
2839    /// bounded Misra-Gries candidate table and then recounts only those candidates.
2840    ///
2841    /// The first of those passes also counts the column's distinct values exactly, up to the cap in
2842    /// [`distinct`], which is the number a string column gets from its dictionary. It comes back
2843    /// beside the summary because a column whose heavy hitters cannot be proved can still have been
2844    /// counted.
2845    ///
2846    /// The tables are keyed by a value's sixty four bits rather than by [`FrequencyValue`], and a
2847    /// null is counted beside them. Every integer type the format stores fits in those bits, so
2848    /// within one column two values share bits only if they are the same value, and a sixteen byte
2849    /// entry keeps the whole candidate table in the second level cache where the forty eight byte
2850    /// one did not. The null takes part in the candidate table exactly as a key would: it holds a
2851    /// place while its count is above zero, and it is decremented with the rest.
2852    ///
2853    /// `counted` is false for a column whose sketch says its distinct values are far past what the
2854    /// exact set holds. It still gets its frequencies, and a count only if it turns out to have
2855    /// fewer values than the candidate table, which is the count that costs nothing.
2856    fn numeric_frequency(
2857        &self,
2858        column: usize,
2859        counted: bool,
2860        dense: Option<(u64, usize)>,
2861    ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2862        let signed = match self.table.fields[column].ty {
2863            LogicalType::TinyInt
2864            | LogicalType::SmallInt
2865            | LogicalType::Integer
2866            | LogicalType::BigInt
2867            | LogicalType::Date
2868            | LogicalType::Timestamp => true,
2869            LogicalType::UTinyInt
2870            | LogicalType::USmallInt
2871            | LogicalType::UInteger
2872            | LogicalType::UBigInt => false,
2873            _ => return Ok((None, None)),
2874        };
2875        let value_of = |bits: Option<u64>| match bits {
2876            None => FrequencyValue::Null,
2877            Some(bits) => integer_value(bits, signed),
2878        };
2879        // A column the writer's tally held whole has its exact counts already, gathered as the rows
2880        // went past, so the pages are not read back to count them again. On `hits` that is most of
2881        // the flag and enum columns. The tally only speaks for the whole column when it saw every
2882        // row, which is the same check the statistics make before they are written.
2883        let tallied = self
2884            .gathers
2885            .get(column)
2886            .and_then(Option::as_ref)
2887            .filter(|gather| gather.rows() == self.table.rows as u64)
2888            .and_then(stats::Gather::frequencies)
2889            .and_then(|(values, nulls)| {
2890                let entries = values
2891                    .iter()
2892                    .map(|(value, count)| {
2893                        let value = value_of(Some(frequency_bits(value)?));
2894                        Some(FrequencyEntry { value, count: *count })
2895                    })
2896                    .chain((nulls != 0).then_some(Some(FrequencyEntry {
2897                        value: FrequencyValue::Null,
2898                        count: nulls,
2899                    })))
2900                    .collect::<Option<Vec<_>>>()?;
2901                Some((entries, values.len() as u64))
2902            });
2903        // A column the sketch expects to fit the exact set is counted there, every value with the
2904        // rows holding it, which is its distinct count and its frequencies from one read of its
2905        // pages. Only a column past the set's cap goes through the candidate table.
2906        let exact = match (&tallied, counted) {
2907            (None, true) => self.exact_frequency(column, signed, dense)?,
2908            _ => None,
2909        };
2910        let (mut entries, decrements, distinct_count) = match (tallied, exact) {
2911            (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
2912            (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
2913            (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
2914            (None, None) => {
2915                // Rows arrive a run of equal values at a time, because a sorted column is runs and
2916                // a flag column is mostly one value, so a run is counted and inserted once rather
2917                // than per row.
2918                let mut first = Candidates::default();
2919                let mut run = Run::default();
2920                self.visit_numeric(column, signed, |_, bits| {
2921                    if let Some((ended, times)) = run.push(bits) {
2922                        first.add(ended, times);
2923                    }
2924                })?;
2925                if let Some((bits, times)) = run.take() {
2926                    first.add(bits, times);
2927                }
2928                // Until a candidate is turned away the table holds every value the column has, so
2929                // its size is the count.
2930                let (nulls, decrements) = (first.nulls, first.decrements);
2931                let distinct_count = (decrements == 0).then_some(first.held as u64);
2932                let (exact, null_count) = if decrements == 0 {
2933                    let exact = first
2934                        .pairs()
2935                        .map(|(bits, count)| (bits, u64::from(count)))
2936                        .collect::<FrequencyMap<_>>();
2937                    (exact, (nulls != 0).then_some(u64::from(nulls)))
2938                } else {
2939                    let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
2940                    if nulls != 0 {
2941                        lower.push(nulls);
2942                    }
2943                    lower.sort_unstable_by(|left, right| right.cmp(left));
2944                    if lower.len() < FREQUENCY_BUILD_RANK
2945                        || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2946                    {
2947                        return Ok((None, distinct_count));
2948                    }
2949                    // Counted beside the slot each candidate sits in, since the table is not
2950                    // changed again and a lookup in it is the one probe the first pass made.
2951                    let mut recounts = vec![0_u64; first.slots.len()];
2952                    let mut null_count = (nulls != 0).then_some(0_u64);
2953                    let mut recount = |bits: Option<u64>, times: u32| {
2954                        let held = match bits {
2955                            Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
2956                            None => null_count.as_mut(),
2957                        };
2958                        if let Some(count) = held {
2959                            *count = count.saturating_add(u64::from(times));
2960                        }
2961                    };
2962                    let mut run = Run::default();
2963                    self.visit_numeric(column, signed, |_, bits| {
2964                        if let Some((bits, times)) = run.push(bits) {
2965                            recount(bits, times);
2966                        }
2967                    })?;
2968                    if let Some((bits, times)) = run.take() {
2969                        recount(bits, times);
2970                    }
2971                    let exact = first
2972                        .slots
2973                        .iter()
2974                        .zip(&recounts)
2975                        .filter(|(slot, _)| slot.count != 0)
2976                        .map(|(slot, &count)| (slot.bits, count))
2977                        .collect::<FrequencyMap<_>>();
2978                    (exact, null_count)
2979                };
2980                let entries = exact
2981                    .into_iter()
2982                    .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2983                    .chain(
2984                        null_count
2985                            .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
2986                    )
2987                    .collect::<Vec<_>>();
2988                (entries, decrements, distinct_count)
2989            }
2990        };
2991        let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
2992        // A complete value-to-count table is also the result of grouping this column.
2993        // Keep up to two leading frequencies for selectivity and equality predicates,
2994        // but leave multi-value grouped counts to the encoded rows at query time.
2995        if omitted_max == 0 && entries.len() > 1 {
2996            let retained = entries.len().saturating_sub(1).min(2);
2997            omitted_max = entries[retained].count;
2998            entries.truncate(retained);
2999        }
3000        let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
3001            total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
3002        });
3003        let mut ordinals = Vec::new();
3004        let mut ordinal_entries = Vec::new();
3005        if let Some(kept_rows) = kept_rows {
3006            let mut kept = FrequencyMap::default();
3007            let mut null_kept = None;
3008            for (at, entry) in entries.iter().enumerate() {
3009                let at = u16::try_from(at)
3010                    .map_err(|_| invalid("too many retained frequency entries"))?;
3011                match entry.value {
3012                    FrequencyValue::Integer(value) => {
3013                        kept.insert(value as u64, at);
3014                    }
3015                    FrequencyValue::Null => null_kept = Some(at),
3016                    FrequencyValue::Code(_) => {}
3017                }
3018            }
3019            ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3020            ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3021            self.visit_numeric(column, signed, |ordinal, bits| {
3022                let held = match bits {
3023                    Some(bits) => kept.get(&bits).copied(),
3024                    None => null_kept,
3025                };
3026                if let Some(entry) = held {
3027                    ordinals.push(ordinal);
3028                    ordinal_entries.push(entry);
3029                }
3030            })?;
3031        }
3032        Ok((
3033            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
3034            distinct_count,
3035        ))
3036    }
3037
3038    /// The bits of a column's lowest value and how many values its range holds, when counting it in
3039    /// a [`distinct::DenseCounts`] would take no more memory than the set it would otherwise be
3040    /// charged, or a mebibyte, whichever is more.
3041    ///
3042    /// Only for a column the statistics saw every row of, since otherwise its ends may not be its
3043    /// ends, and a table of fewer than `u32::MAX` rows, so that a count fits in its slot.
3044    fn dense_range(&self, gather: &stats::Gather, set: usize) -> Option<(u64, usize)> {
3045        let rows = self.table.rows;
3046        if gather.rows() != rows as u64 || u32::try_from(rows).is_err() {
3047            return None;
3048        }
3049        let (low, high) = gather.span()?;
3050        let len = usize::try_from(high.checked_sub(low)?.checked_add(1)?).ok()?;
3051        #[allow(clippy::cast_sign_loss, clippy::cast_possible_truncation)]
3052        let bits = low as u64;
3053        (len.checked_mul(size_of::<u32>())? <= set.max(1 << 20)).then_some((bits, len))
3054    }
3055
3056    /// Counts every value of an integer column and the rows holding it, and hands back the
3057    /// frequency entries worth keeping beside the distinct count, or nothing for a column with more
3058    /// values than [`distinct::ExactCounts`] keeps.
3059    ///
3060    /// The entries are `None` for a column with no value common enough to be worth a synopsis. The
3061    /// rule is the one the candidate table applied. A column with more values than that table holds
3062    /// keeps its frequencies only if its tenth commonest value is held by more rows than a
3063    /// Misra-Gries table of [`FREQUENCY_CANDIDATES`] could have decremented it by, which is its rows
3064    /// over one more than the candidates. The counts kept are exact either way, so the largest one
3065    /// left out is exact too and not the table's bound on it.
3066    ///
3067    /// Only the commonest entries and the ones tied with the first left out are built, since on a
3068    /// column of a million values the rest are thrown away the moment they are ranked.
3069    fn exact_frequency(
3070        &self,
3071        column: usize,
3072        signed: bool,
3073        dense: Option<(u64, usize)>,
3074    ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3075        // A column whose ends are close together is counted in a flat array. A value outside the
3076        // ends it was given, which would be a bug in the statistics, sends it to the set instead.
3077        if let Some((low, len)) = dense {
3078            let mut counts = distinct::DenseCounts::new(low, len);
3079            let nulls =
3080                self.count_numeric(column, signed, |bits, times| counts.insert(bits, times))?;
3081            if let Some(distinct) = counts.count() {
3082                let Some(distinct) = distinct else { return Ok(None) };
3083                return Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3084                    counts.visit(visit);
3085                })));
3086            }
3087        }
3088        let mut set = distinct::ExactCounts::new();
3089        let nulls = self.count_numeric(column, signed, |bits, times| set.insert(bits, times))?;
3090        let Some(distinct) = set.count() else {
3091            return Ok(None);
3092        };
3093        Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3094            set.visit(visit);
3095        })))
3096    }
3097
3098    /// Hands every run of equal non-null values in an integer column to `add` as its bits and its
3099    /// length, and answers how many rows were null.
3100    fn count_numeric(
3101        &self,
3102        column: usize,
3103        signed: bool,
3104        mut add: impl FnMut(u64, u32),
3105    ) -> Result<u64> {
3106        let mut nulls = 0_u64;
3107        let mut run = Run::default();
3108        let mut take = |bits: Option<u64>, times: u32| match bits {
3109            Some(bits) => add(bits, times),
3110            None => nulls += u64::from(times),
3111        };
3112        self.visit_numeric(column, signed, |_, bits| {
3113            if let Some((bits, times)) = run.push(bits) {
3114                take(bits, times);
3115            }
3116        })?;
3117        if let Some((bits, times)) = run.take() {
3118            take(bits, times);
3119        }
3120        Ok(nulls)
3121    }
3122
3123    /// The frequency entries worth keeping out of a column's exact counts, which `visit` hands over
3124    /// as bits and rows once for each call it gets. See [`Self::exact_frequency`] for the rule.
3125    fn frequent_entries(
3126        &self,
3127        signed: bool,
3128        distinct: u64,
3129        nulls: u64,
3130        mut visit: impl FnMut(&mut dyn FnMut(u64, u64)),
3131    ) -> (Option<Vec<FrequencyEntry>>, u64) {
3132        // The commonest counts, one more than the entries kept so that the first left out is here.
3133        let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3134        let mut rank = |count: u64| {
3135            if top.len() <= FREQUENCY_ENTRIES {
3136                top.push(Reverse(count));
3137            } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3138                top.pop();
3139                top.push(Reverse(count));
3140            }
3141        };
3142        visit(&mut |_, count| rank(count));
3143        if nulls != 0 {
3144            rank(nulls);
3145        }
3146        let top = top.into_sorted_vec();
3147        let values = distinct + u64::from(nulls != 0);
3148        if values > FREQUENCY_CANDIDATES as u64 {
3149            let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3150            if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3151                return (None, distinct);
3152            }
3153        }
3154        let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3155        let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3156        visit(&mut |bits, count| {
3157            if count >= least {
3158                entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3159            }
3160        });
3161        if nulls != 0 && nulls >= least {
3162            entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3163        }
3164        (Some(entries), distinct)
3165    }
3166
3167    /// Hands every row of an integer column to `visit` as its ordinal and its sixty four bits, or
3168    /// `None` for a null.
3169    ///
3170    /// `signed` says which of the two readings the column has. A packed unsigned column would come
3171    /// back from `signed_block` as a base plus a code in `i64`, which wraps for a value past the top
3172    /// of `BIGINT`, so only a signed column takes the block path.
3173    fn visit_numeric(
3174        &self,
3175        column: usize,
3176        signed: bool,
3177        mut visit: impl FnMut(u64, Option<u64>),
3178    ) -> Result<()> {
3179        let ty = &self.table.fields[column].ty;
3180        let mut start = 0_u64;
3181        let mut block = Vec::new();
3182        for stripe in &self.table.stripes {
3183            let spans = read_index(&self.file, stripe, column)?;
3184            let page = stripe.pages[column];
3185            let mut bytes = vec![0; page.length as usize];
3186            read_at(&self.file, page.offset, &mut bytes)?;
3187            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3188                let part = part_bytes(&bytes, *span)?;
3189                if checksum(part) != span.hash {
3190                    return Err(invalid("column page checksum differs while building frequencies"));
3191                }
3192                let rows = rows as usize;
3193                let vector = decode(ty, rows, part, None)?;
3194                // Every signed layout a numeric column decodes to, which is every column of `hits`,
3195                // comes out as one run of `i64` and is walked as a slice. The row path below is for
3196                // the unsigned types and anything else that cannot be handed over that way.
3197                if signed && vector.signed_block(&mut block) && block.len() == rows {
3198                    if vector.none_null() {
3199                        for (row, &value) in block.iter().enumerate() {
3200                            visit(start.saturating_add(row as u64), Some(value as u64));
3201                        }
3202                    } else {
3203                        for (row, &value) in block.iter().enumerate() {
3204                            let bits = (!vector.is_null_at(row)).then_some(value as u64);
3205                            visit(start.saturating_add(row as u64), bits);
3206                        }
3207                    }
3208                    start = start.saturating_add(rows as u64);
3209                    continue;
3210                }
3211                // row at a time: frequency construction visits decoded values to update bounded candidates.
3212                for row in 0..rows {
3213                    let bits = if vector.is_null_at(row) {
3214                        None
3215                    } else {
3216                        // An unsigned column has no signed reading, and the documented fallback is
3217                        // the value itself. Every width the format stores fits in sixty four bits,
3218                        // so nothing is lost on the way through.
3219                        let widened = match vector.signed_at(row) {
3220                            Some(value) => Some(value as u64),
3221                            None => match vector.value_at(row) {
3222                                Value::UTinyInt(value) => Some(u64::from(value)),
3223                                Value::USmallInt(value) => Some(u64::from(value)),
3224                                Value::UInteger(value) => Some(u64::from(value)),
3225                                Value::UBigInt(value) => Some(value),
3226                                _ => None,
3227                            },
3228                        };
3229                        Some(widened.ok_or_else(|| {
3230                            invalid("numeric frequency page did not contain an integer value")
3231                        })?)
3232                    };
3233                    visit(start.saturating_add(row as u64), bits);
3234                }
3235                start = start.saturating_add(rows as u64);
3236            }
3237        }
3238        Ok(())
3239    }
3240
3241    /// The columns that get numeric frequencies, which are the integer, date and timestamp ones.
3242    fn numeric_columns(&self) -> Vec<usize> {
3243        self.table
3244            .fields
3245            .iter()
3246            .enumerate()
3247            .filter_map(|(column, field)| {
3248                matches!(
3249                    field.ty,
3250                    LogicalType::TinyInt
3251                        | LogicalType::SmallInt
3252                        | LogicalType::Integer
3253                        | LogicalType::BigInt
3254                        | LogicalType::UTinyInt
3255                        | LogicalType::USmallInt
3256                        | LogicalType::UInteger
3257                        | LogicalType::UBigInt
3258                        | LogicalType::Date
3259                        | LogicalType::Timestamp
3260                )
3261                .then_some(column)
3262            })
3263            .collect()
3264    }
3265
3266    /// Reads one stable dictionary code column only at sorted table-wide row ordinals.
3267    #[allow(dead_code)]
3268    fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3269        if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3270            return Ok(None);
3271        }
3272        if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3273            return Err(invalid("frequency ordinals are not sorted and unique"));
3274        }
3275        let mut out = Vec::with_capacity(ordinals.len());
3276        let mut wanted = 0;
3277        let mut stripe_start = 0_u64;
3278        for stripe in &self.table.stripes {
3279            let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3280            if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3281                stripe_start = stripe_end;
3282                continue;
3283            }
3284            let spans = read_index(&self.file, stripe, column)?;
3285            let page = stripe.pages[column];
3286            let mut bytes = vec![0; page.length as usize];
3287            read_at(&self.file, page.offset, &mut bytes)?;
3288            let mut part_start = stripe_start;
3289            for (span, &rows) in spans.iter().zip(&stripe.parts) {
3290                let part_end = part_start.saturating_add(u64::from(rows));
3291                if wanted < ordinals.len() && ordinals[wanted] < part_end {
3292                    let part = part_bytes(&bytes, *span)?;
3293                    if checksum(part) != span.hash {
3294                        return Err(invalid(
3295                            "column page checksum differs while building pair frequencies",
3296                        ));
3297                    }
3298                    let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3299                    let positions = ordinals[wanted..upto]
3300                        .iter()
3301                        .map(|&ordinal| {
3302                            usize::try_from(ordinal.saturating_sub(part_start))
3303                                .map_err(|_| invalid("frequency row offset does not fit in memory"))
3304                        })
3305                        .collect::<Result<Vec<_>>>()?;
3306                    if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3307                        return Ok(None);
3308                    }
3309                    wanted = upto;
3310                }
3311                part_start = part_end;
3312            }
3313            stripe_start = stripe_end;
3314        }
3315        if wanted != ordinals.len() {
3316            return Err(invalid("frequency ordinal is outside the table"));
3317        }
3318        Ok(Some(out))
3319    }
3320
3321    /// Derives bounded two-key leaders from numeric anchor ordinals and stable string codes.
3322    #[allow(dead_code)]
3323    fn pair_frequencies(
3324        &self,
3325        frequencies: &[Option<Frequencies>],
3326    ) -> Result<Vec<PairFrequencySummary>> {
3327        let anchors = frequencies
3328            .iter()
3329            .enumerate()
3330            .filter_map(|(column, summary)| {
3331                // A writer holds every synopsis it counted, so there is nothing stored to skip.
3332                match summary {
3333                    Some(Frequencies::Held(summary)) => Some(summary),
3334                    _ => None,
3335                }
3336                .filter(|summary| {
3337                    !summary.ordinals.is_empty()
3338                        && summary.ordinal_entries.len() == summary.ordinals.len()
3339                })
3340                .cloned()
3341                .map(|summary| (column, summary))
3342            })
3343            .collect::<Vec<_>>();
3344        let strings = self
3345            .dictionaries
3346            .iter()
3347            .enumerate()
3348            .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3349            .collect::<Vec<_>>();
3350        let mut summaries = Vec::new();
3351        for (first, anchors) in anchors {
3352            for &second in &strings {
3353                if summaries.len() == MAX_PAIR_FREQUENCIES {
3354                    return Ok(summaries);
3355                }
3356                let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3357                    continue;
3358                };
3359                if codes.len() != anchors.ordinal_entries.len() {
3360                    return Err(invalid("pair frequency columns have different lengths"));
3361                }
3362                let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3363                for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3364                    *counts.entry((anchor, code)).or_default() += 1;
3365                }
3366                let mut entries = counts
3367                    .into_iter()
3368                    .map(|((first_entry, second), count)| PairFrequencyEntry {
3369                        first_entry,
3370                        second,
3371                        count,
3372                    })
3373                    .collect::<Vec<_>>();
3374                entries.sort_unstable_by(|left, right| {
3375                    right
3376                        .count
3377                        .cmp(&left.count)
3378                        .then_with(|| left.first_entry.cmp(&right.first_entry))
3379                        .then_with(|| left.second.cmp(&right.second))
3380                });
3381                let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3382                entries.truncate(FREQUENCY_ENTRIES);
3383                summaries.push(PairFrequencySummary {
3384                    first: u16::try_from(first)
3385                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3386                    second: u16::try_from(second)
3387                        .map_err(|_| invalid("pair frequency column index overflows"))?,
3388                    entries,
3389                    omitted_max: anchors.omitted_max.max(pair_omitted),
3390                });
3391            }
3392        }
3393        Ok(summaries)
3394    }
3395
3396    /// Writes the directory of the table this writer is on and says where it went.
3397    ///
3398    /// Everything [`Writer::finish`] used to do except the two writes that publish. Pulling it out
3399    /// is what lets a second table follow a first: the bytes of a closed table are complete and
3400    /// addressable while nothing yet points at them, and the pointer is the last write of the
3401    /// commit.
3402    ///
3403    /// # Errors
3404    ///
3405    /// If directory encoding or writing fails.
3406    fn close(&mut self) -> Result<Entry> {
3407        self.reclaim()?;
3408        self.flush_pending()?;
3409        // The rest of a table is its statistics, its dictionaries and its directory. The dictionary
3410        // work is charged as its own stage, because ranking a global dictionary can be most of what
3411        // this costs, and the rest as publish.
3412        let profile = self.profile.clone();
3413        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3414        let before = self.at;
3415        let mut stripes = std::mem::take(&mut self.order)
3416            .into_iter()
3417            .zip(std::mem::take(&mut self.table.stripes))
3418            .collect::<Vec<_>>();
3419        stripes.sort_by_key(|(order, _)| order.0);
3420        let mut previous: Option<(u64, u64)> = None;
3421        for ((first, last), _) in &stripes {
3422            if previous.is_some_and(|previous| previous >= *first) {
3423                return Err(invalid("chunks did not arrive in source order"));
3424            }
3425            previous = Some(*last);
3426        }
3427        self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3428        drop(timing);
3429        let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3430        let placing = self.at;
3431        finish_dictionaries(&mut self.dictionaries)?;
3432        self.place_blocks()?;
3433        for dictionary in self.dictionaries.iter_mut().flatten() {
3434            dictionary.release_lookup();
3435            dictionary.recharge(profile.as_deref());
3436        }
3437        let (numeric, closed) = self.close_columns()?;
3438        let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3439            numeric.into_iter().unzip();
3440        let frequencies =
3441            frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3442        // Pair leaders are query results, not reusable column statistics.
3443        let pairs = Vec::new();
3444        self.table.frequencies = frequencies;
3445        self.table.distincts = distincts;
3446        self.table.pair_frequencies = pairs;
3447        if let Some(profile) = &profile {
3448            profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3449        }
3450        self.table.demoted = self
3451            .dictionaries
3452            .iter()
3453            .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3454            .collect();
3455        if !self.table.demoted.contains(&true) {
3456            self.table.demoted = Vec::new();
3457        }
3458        self.dictionaries = Vec::new();
3459        self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3460        self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3461        self.table.host_groups = None;
3462        for (index, closed) in closed.into_iter().enumerate() {
3463            let Some(closed) = closed else { continue };
3464            let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3465            self.table.distincts[index] = distinct;
3466            self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3467            self.table.frequency_texts[index] = texts;
3468            if hosts.is_some() {
3469                self.table.host_groups = hosts;
3470            }
3471            let offset = self.at;
3472            self.put(&encoded.index)?;
3473            self.put(&encoded.ranks)?;
3474            self.put(&encoded.grams)?;
3475            self.table.dictionary_payloads[index] = payload;
3476            let length = encoded
3477                .index
3478                .len()
3479                .checked_add(encoded.ranks.len())
3480                .and_then(|len| len.checked_add(encoded.grams.len()))
3481                .ok_or_else(|| invalid("dictionary page length overflow"))?;
3482            self.table.dictionaries[index] = Some(Page {
3483                offset,
3484                length: u32::try_from(length)
3485                    .map_err(|_| invalid("dictionary page length overflow"))?,
3486                hash: checksum(&encoded.index),
3487            });
3488        }
3489        drop(timing);
3490        let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3491        let placed = self.at - placing;
3492        self.write_stats()?;
3493        let directory = encode_directory(&self.table)?;
3494        if directory.len() > MAX_DIRECTORY {
3495            return Err(invalid("directory exceeds the configured bound"));
3496        }
3497        let offset = self.at;
3498        self.put(&directory)?;
3499        drop(timing);
3500        if let Some(profile) = &profile {
3501            profile.moved(Stage::Dictionary, 0, placed, 0);
3502            profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3503        }
3504        Ok(Entry {
3505            name: self.table.name.clone(),
3506            fields: self.table.fields.clone(),
3507            rows: self.table.rows,
3508            nonzero: vec![None; self.table.fields.len()],
3509            aggregates: table_aggregate_sums(&self.table),
3510            distincts: self.table.distincts.clone(),
3511            extremes: table_integer_extremes(&self.table),
3512            frequencies: table_complete_numeric_frequencies(&self.table),
3513            directory: Page {
3514                offset,
3515                length: u32::try_from(directory.len())
3516                    .map_err(|_| invalid("directory length overflow"))?,
3517                hash: checksum(&directory),
3518            },
3519        })
3520    }
3521
3522    /// Every numeric column's frequencies and every global dictionary's page and statistics, by
3523    /// column, as many columns at a time as [`CLOSE_BYTES`] allows.
3524    ///
3525    /// The two kinds read what is already written and write nothing, so they share one set of
3526    /// threads. Each was most of a second on `hits` with the other waiting for it, and neither keeps
3527    /// every core busy on its own. The most expensive column that fits is the one taken next, so
3528    /// the long ones start first and the short ones fill in behind them. A column that does not fit
3529    /// waits for one that is closing to finish, unless nothing is closing, in which case it goes
3530    /// alone.
3531    ///
3532    /// A numeric column is charged the exact distinct set its sketch says it will need, and one the
3533    /// sketch puts far past what that set can hold does not build it, because the set would fill,
3534    /// give up and have held 512 MiB for nothing. A column with no sketch is charged the whole set.
3535    /// Each job charges itself as its own span, publish for the numeric ones and dictionary for the
3536    /// rest, because it runs on a thread of its own and a span on this one would see the wall time
3537    /// and none of the CPU.
3538    #[allow(clippy::type_complexity)]
3539    fn close_columns(
3540        &self,
3541    ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3542        let numeric = self.numeric_columns().into_iter().map(|column| {
3543            let gather = self.gathers.get(column).and_then(Option::as_ref);
3544            let estimate = gather.and_then(stats::Gather::distinct);
3545            let counted = !estimate.is_some_and(distinct::beyond);
3546            let set =
3547                if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3548            let dense = gather.filter(|_| counted).and_then(|gather| self.dense_range(gather, set));
3549            let set = dense.map_or(set, |(_, len)| len * size_of::<u32>());
3550            let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3551            (Closing::Numeric { column, counted, dense }, NUMERIC_CLOSE_BYTES + set, cost)
3552        });
3553        let dictionaries =
3554            self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3555                let dictionary = dictionary.as_ref()?;
3556                let bytes = dictionary.closing_bytes();
3557                Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3558            });
3559        let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3560        jobs.sort_by_key(|&(_, _, cost)| cost);
3561        let columns = self.table.fields.len();
3562        let mut frequencies = vec![(None, None); columns];
3563        let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3564        let profile = self.profile.as_deref();
3565        let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3566            let _holding = profile.map(|profile| profile.holding(bytes as u64));
3567            match job {
3568                Closing::Numeric { column, counted, dense } => {
3569                    let _timing = profile.map(|profile| profile.span(Stage::Publish));
3570                    Ok(Closed::Numeric(column, self.numeric_frequency(column, counted, dense)?))
3571                }
3572                Closing::Dictionary { index, dictionary } => {
3573                    let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3574                    Ok(Closed::Dictionary(index, self.close_dictionary(index, dictionary)?))
3575                }
3576            }
3577        };
3578        let workers = close_workers().min(jobs.len());
3579        let pieces = if workers <= 1 {
3580            jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3581        } else {
3582            // The columns not taken yet, cheapest first, and the bytes the ones closing now hold.
3583            let state = Mutex::new((jobs, 0_usize));
3584            let finished = Condvar::new();
3585            std::thread::scope(|scope| {
3586                (0..workers)
3587                    .map(|_| {
3588                        scope.spawn(|| {
3589                            let mut mine = Vec::new();
3590                            loop {
3591                                let mut held = state.lock().map_err(|_| {
3592                                    Error::internal("a native close worker panicked")
3593                                })?;
3594                                let (job, bytes) = loop {
3595                                    let (jobs, busy) = &mut *held;
3596                                    if jobs.is_empty() {
3597                                        return Ok(mine);
3598                                    }
3599                                    let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3600                                        *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3601                                    });
3602                                    if let Some(at) = fits {
3603                                        let (job, bytes, _) = jobs.remove(at);
3604                                        *busy += bytes;
3605                                        break (job, bytes);
3606                                    }
3607                                    held = finished.wait(held).map_err(|_| {
3608                                        Error::internal("a native close worker panicked")
3609                                    })?;
3610                                };
3611                                drop(held);
3612                                // Given back on the way out whether the close worked, failed or
3613                                // panicked, so that a worker waiting for room is never left waiting.
3614                                let _room = Room { state: &state, finished: &finished, bytes };
3615                                mine.push(run(job, bytes)?);
3616                            }
3617                        })
3618                    })
3619                    .collect::<Vec<_>>()
3620                    .into_iter()
3621                    .map(|handle| {
3622                        handle
3623                            .join()
3624                            .map_err(|_| Error::internal("a native close worker panicked"))?
3625                    })
3626                    .collect::<Result<Vec<_>>>()
3627            })?
3628            .into_iter()
3629            .flatten()
3630            .collect()
3631        };
3632        for piece in pieces {
3633            match piece {
3634                Closed::Numeric(column, summary) => frequencies[column] = summary,
3635                Closed::Dictionary(index, one) => closed[index] = Some(one),
3636            }
3637        }
3638        Ok((frequencies, closed))
3639    }
3640
3641    /// One global dictionary's page and statistics, built from what is already in the file.
3642    ///
3643    /// Nothing is written here, so that [`Self::close`] can run this beside the numeric frequencies
3644    /// and put the pages down afterwards in column order, which is where they always went. The
3645    /// column's values are decoded in here and dropped before it returns, and
3646    /// [`Self::close_columns`] decides how many columns are in here at once.
3647    fn close_dictionary(
3648        &self,
3649        _index: usize,
3650        dictionary: &GlobalDictionary,
3651    ) -> Result<ClosedDictionary> {
3652        let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3653        // A code nothing counted is a code no non-null row of this column holds, which is the
3654        // empty string a null was written as and nothing else, because a code is only ever made by
3655        // a row asking for one. A demoted dictionary counted the stripes before its demotion and
3656        // none after, so it has no count or frequency of the column to give.
3657        let (distinct, frequencies, texts) = if dictionary.demoted {
3658            (None, None, Vec::new())
3659        } else {
3660            let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3661            let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3662            (Some(distinct), Some(frequencies), texts)
3663        };
3664        // Deriving a fixed SQL host expression at load time materializes its answer.
3665        let hosts = None;
3666        drop(flat);
3667        drop(bases);
3668        let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3669        let payload = dictionary
3670            .placed
3671            .iter()
3672            .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3673            .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3674        Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3675    }
3676
3677    /// Writes the statistics sections for the table being closed, as far as the budget reaches.
3678    ///
3679    /// Called from [`Self::close`] after the last stripe and after the dictionaries, which is the
3680    /// first moment the table's column bytes are final and the last moment before the directory is
3681    /// encoded. Both halves matter: the budget is a share of the column bytes, and a section that
3682    /// went in after the directory would be a section the directory does not name.
3683    ///
3684    /// Nothing here can fail the write. A column whose gather came back blind gets no sections, a
3685    /// column the budget could not reach gets none, and section 3.1 says both of those plan the way
3686    /// they planned before statistics existed. The two errors that are returned are an encode
3687    /// failure and a section count past the bound, and neither is a thing a column can cause.
3688    fn write_stats(&mut self) -> Result<()> {
3689        let gathers = std::mem::take(&mut self.gathers);
3690        let rows = self.table.rows as u64;
3691        let mut payloads = Vec::new();
3692        for (column, gather) in gathers.into_iter().enumerate() {
3693            let Some(gather) = gather else { continue };
3694            // A gather that saw a different number of rows than the table committed is a gather
3695            // that missed some, and a distinct count over some of a column is the one error an
3696            // estimator cannot see coming. This has no way of happening today, since a table is
3697            // written once and every chunk goes through `flush_pending`, and that is exactly why it
3698            // is worth a line: it stays true only while that stays true.
3699            if gather.rows() != rows {
3700                continue;
3701            }
3702            let Some(stats) = gather.finish() else { continue };
3703            let mut summary = Vec::new();
3704            stats.summary.encode(&mut summary)?;
3705            let mut sketches = Vec::new();
3706            stats.sketches.encode(&mut sketches)?;
3707            payloads.push((column, summary, sketches));
3708        }
3709        if payloads.is_empty() {
3710            return Ok(());
3711        }
3712        let summaries = payloads.iter().map(|(_, summary, _)| summary.len()).collect::<Vec<_>>();
3713        let sketches = payloads.iter().map(|(_, _, sketches)| sketches.len()).collect::<Vec<_>>();
3714        let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3715        // Nothing is spent yet. A table this writer is closing is one it wrote from nothing, so the
3716        // only statistics sections it can have are the ones about to go in.
3717        let keep = stats::kept(&summaries, &sketches, allowance, 0);
3718        for ((column, summary, sketches), &(built, sketched)) in payloads.iter().zip(&keep) {
3719            if !built {
3720                continue;
3721            }
3722            let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3723            let sections = [
3724                // A summary is a header the whole way down: there is nothing behind it a reader
3725                // could decide not to read.
3726                (*section::SUMMARY, summary, summary.len() as u32),
3727                (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3728            ];
3729            let wanted = 1 + usize::from(sketched);
3730            for (kind, bytes, header_bytes) in sections.into_iter().take(wanted) {
3731                let written = write_section(
3732                    &*self.file,
3733                    &mut self.at,
3734                    &section::Attachment { kind, id, flags: 0, header_bytes, bytes },
3735                    self.generation,
3736                )?;
3737                self.table.sections.push(written);
3738            }
3739        }
3740        if self.table.sections.len() > MAX_SECTIONS {
3741            return Err(invalid("the table would name more sections than the bound allows"));
3742        }
3743        Ok(())
3744    }
3745
3746    /// Commits every table this writer has written and syncs the file before publishing its header
3747    /// slot.
3748    ///
3749    /// The table handed back is the one the writer was on, which is the last of them. Callers that
3750    /// wrote several already know the others, since they named them.
3751    ///
3752    /// # Errors
3753    ///
3754    /// If directory encoding, writing, or syncing fails.
3755    pub fn finish(mut self) -> Result<Table> {
3756        let entry = self.close()?;
3757        let profile = self.profile.take();
3758        let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3759        let mut tables = std::mem::take(&mut self.closed);
3760        tables.push(entry);
3761        let catalog = encode_catalog(&tables, &self.views)?;
3762        if catalog.len() > MAX_DIRECTORY {
3763            return Err(invalid("catalog exceeds the configured bound"));
3764        }
3765        let offset = self.at;
3766        self.put(&catalog)?;
3767        if let Some(profile) = &profile {
3768            profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3769        }
3770        // Every page and every table directory is on the disk before anything points at them. The
3771        // slot write below is what makes this generation the one a reader picks, so the order of
3772        // these two syncs is the whole of the commit.
3773        synced(&*self.file, profile.as_deref())?;
3774        let slot = Slot {
3775            offset,
3776            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3777            generation: self.generation,
3778            hash: checksum(&catalog),
3779        };
3780        // The one write that is not an append, and the last one. It goes back over the slot in the
3781        // header, so it names its offset rather than going through `put`, and `at` does not move.
3782        // Which of the two slots it is alternates with the generation, so the one naming the
3783        // generation before this is still intact and still valid until this write lands.
3784        self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3785        synced(&*self.file, profile.as_deref())?;
3786        Ok(self.table)
3787    }
3788
3789    /// Commits a generation that changes the views and leaves every table exactly where it is.
3790    ///
3791    /// There was no way to do this before views existed, because everything that could change the
3792    /// catalog also wrote a table, so the only way to say something new about a file was to go
3793    /// through a table. A view is the first thing that can change on its own. Without this, adding
3794    /// a view to a database with eight tables in it would rewrite all eight, since the append path
3795    /// needs a table to append and the fallback is the whole file.
3796    ///
3797    /// It is the same commit as [`Writer::finish`] with nothing appended before it. The table
3798    /// entries are carried forward by directory pointer the way an append carries them, the new
3799    /// catalog goes on the end, and the slot write at the end is what publishes it.
3800    ///
3801    /// # Errors
3802    ///
3803    /// If the file has no valid committed directory, is not this build's format, or cannot be
3804    /// written.
3805    pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3806        let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3807        let size = file.len()?;
3808        let (slot, bytes, _) = committed_slot(&*file, size)?;
3809        let (closed, _) = decode_catalog(&bytes, size)?;
3810        let generation = slot
3811            .generation
3812            .checked_add(1)
3813            .ok_or_else(|| invalid("native file generation overflow"))?;
3814        let catalog = encode_catalog(&closed, views)?;
3815        if catalog.len() > MAX_DIRECTORY {
3816            return Err(invalid("catalog exceeds the configured bound"));
3817        }
3818        file.write_at(size, &catalog)?;
3819        file.sync()?;
3820        let slot = Slot {
3821            offset: size,
3822            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3823            generation,
3824            hash: checksum(&catalog),
3825        };
3826        file.write_at(slot_offset(generation), &slot.bytes())?;
3827        file.sync()?;
3828        Ok(())
3829    }
3830
3831    /// Adds exact count, sum, distinct, bound, and bounded frequency certificates to an older file without
3832    /// rewriting table pages. The old slot remains readable until the new catalog is synced.
3833    pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3834        let path = path.as_ref();
3835        let (_, size, slot, bytes, _) = slot_bytes(path)?;
3836        let (mut entries, views) = decode_catalog(&bytes, size)?;
3837        let native = Catalog::open(path)?;
3838        for entry in &mut entries {
3839            let reader = native.table(&entry.name)?;
3840            entry.nonzero.fill(None);
3841            entry.aggregates = reader_aggregate_sums(&reader)?;
3842            entry.distincts = (0..entry.fields.len())
3843                .map(|column| reader.distinct_values(column))
3844                .collect::<Result<Vec<_>>>()?;
3845            entry.extremes = reader_integer_extremes(&reader)?;
3846            entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3847        }
3848        let generation = slot
3849            .generation
3850            .checked_add(1)
3851            .ok_or_else(|| invalid("native file generation overflow"))?;
3852        let catalog = encode_catalog(&entries, &views)?;
3853        if catalog.len() > MAX_DIRECTORY {
3854            return Err(invalid("catalog exceeds the configured bound"));
3855        }
3856        let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
3857        file.write_at(size, &catalog)?;
3858        file.sync()?;
3859        let slot = Slot {
3860            offset: size,
3861            length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3862            generation,
3863            hash: checksum(&catalog),
3864        };
3865        file.write_at(slot_offset(generation), &slot.bytes())?;
3866        file.sync()?;
3867        Ok(())
3868    }
3869
3870    /// The earlier name for [`Self::certify_summaries`].
3871    pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3872        Self::certify_summaries(path)
3873    }
3874}
3875
3876/// Appends one run of bytes at `at` and moves it past them, answering where they went.
3877///
3878/// The append half of [`attach`], which cannot use [`Writer::put`] because it is not writing a
3879/// table. Every byte a section costs goes through here, so the offsets in an extent table come
3880/// from one place.
3881fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3882    let offset = *at;
3883    file.write_at(offset, bytes)?;
3884    *at =
3885        at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3886    Ok(offset)
3887}
3888
3889/// Writes one attachment's payload as extents and returns the entry that names it.
3890///
3891/// The split is by bytes, and the `first` of each extent is therefore a byte count. A section kind
3892/// whose extents should break on a row boundary instead will want to hand its extents over already
3893/// split; nothing needs that yet, and guessing at the shape of it now would be guessing.
3894fn write_section(
3895    file: &dyn rudb_io::File,
3896    at: &mut u64,
3897    one: &section::Attachment<'_>,
3898    generation: u64,
3899) -> Result<Section> {
3900    // A payload of nothing is the exception, and it is not a special case so much as a different
3901    // reading of the same field: an entry with no bytes has no header to be longer than them, and
3902    // `header_bytes` is what the structure would have cost. See `Section::refused`.
3903    if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3904        return Err(invalid("a section's header is longer than its payload"));
3905    }
3906    let mut extents = Vec::new();
3907    let mut first = 0_u64;
3908    let extent_size =
3909        if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
3910            1 << 19
3911        } else {
3912            section::MAX_EXTENT as usize
3913        };
3914    for chunk in one.bytes.chunks(extent_size) {
3915        let offset = append(file, at, chunk)?;
3916        extents.push(section::Extent {
3917            offset,
3918            length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3919            hash: checksum(chunk),
3920            first,
3921        });
3922        first += chunk.len() as u64;
3923    }
3924    let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3925    section::encode_extents(&extents, &mut table)?;
3926    // A payload of nothing is a section of no extents and no extent table, and its `extent_page`
3927    // is zero rather than the end of the file. Section 3.7 wants that entry to exist: it is how a
3928    // relationship that did not fit the budget is recorded as not built rather than forgotten.
3929    let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3930    Ok(Section {
3931        kind: one.kind,
3932        id: one.id,
3933        generation,
3934        extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3935        extent_page,
3936        extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3937        hash: checksum(&table),
3938        flags: one.flags,
3939        header_bytes: one.header_bytes,
3940    })
3941}
3942
3943/// Attaches graph sections to a table already committed in a file, without rewriting a page.
3944///
3945/// This is the second pass spec/graph/03-the-file-format.md section 3.8 asks for. A key map has to
3946/// exist before the link that uses it can be built, and it is built by reading the key column back,
3947/// so the structures of a table cannot be written during the load that wrote the table. They are
3948/// written afterwards, by this, and the file in between the two is a correct file that answers
3949/// every query more slowly.
3950///
3951/// Nothing is overwritten. The payloads, the extent tables, the new directory for this table and
3952/// the new catalog all go on the end of the file past the committed generation, and the last write
3953/// is the header slot, exactly as [`Writer::finish`] does it. So a crash anywhere in here leaves
3954/// the generation before it intact, and leaves unreferenced trailing bytes that the next commit
3955/// writes past.
3956///
3957/// An attachment replaces any section of the same kind and id, and every other section is carried
3958/// through untouched, including one whose kind this build does not know. The table's own generation
3959/// is carried through too, because attaching a section moves no row: see [`Table::generation`].
3960///
3961/// # Errors
3962///
3963/// If the file has no valid committed directory, is an older format than this build writes, holds
3964/// no table of that name, names a section whose payload cannot be written, or would end up naming
3965/// more sections than the format allows.
3966pub fn attach(
3967    path: impl AsRef<Path>,
3968    table: &str,
3969    attachments: &[section::Attachment<'_>],
3970) -> Result<Table> {
3971    let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3972    let file = &*file;
3973    let size = file.len()?;
3974    let (slot, bytes, _) = committed_slot(file, size)?;
3975    let (mut entries, views) = decode_catalog(&bytes, size)?;
3976    let at = entries
3977        .iter()
3978        .position(|entry| entry.name == table)
3979        .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3980    let mut version = [0; 4];
3981    read_at(file, 8, &mut version)?;
3982    let version = u32::from_le_bytes(version);
3983    // Readable is not the same as writable. A format 22 file has no section table, and giving its
3984    // directory one without moving the number in its header would leave a file that claims to be
3985    // format 22 and is not, which is worse than refusing. Rewriting it with this build is the
3986    // answer, and the format is at 0.3.x, so nobody has one of these that this project did not
3987    // just make.
3988    if version != FORMAT {
3989        return Err(invalid(&format!(
3990            "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3991             to be written again"
3992        )));
3993    }
3994    let mut directory = vec![0; entries[at].directory.length as usize];
3995    read_at(file, entries[at].directory.offset, &mut directory)?;
3996    if checksum(&directory) != entries[at].directory.hash {
3997        return Err(invalid(&format!("the directory of table {table} does not checksum")));
3998    }
3999    let mut held = decode_directory(&directory, size)?;
4000    let mut cursor = size;
4001    for one in attachments {
4002        let written = write_section(file, &mut cursor, one, held.generation)?;
4003        held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
4004        held.sections.push(written);
4005    }
4006    if held.sections.len() > MAX_SECTIONS {
4007        return Err(invalid("the table would name more sections than the bound allows"));
4008    }
4009    let encoded = encode_directory(&held)?;
4010    if encoded.len() > MAX_DIRECTORY {
4011        return Err(invalid("directory exceeds the configured bound"));
4012    }
4013    let offset = append(file, &mut cursor, &encoded)?;
4014    entries[at].directory = Page {
4015        offset,
4016        length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
4017        hash: checksum(&encoded),
4018    };
4019    // The views the file already had, written back unchanged. Attaching a section to a table says
4020    // nothing about a view and must not drop one.
4021    let catalog = encode_catalog(&entries, &views)?;
4022    if catalog.len() > MAX_DIRECTORY {
4023        return Err(invalid("catalog exceeds the configured bound"));
4024    }
4025    let offset = append(file, &mut cursor, &catalog)?;
4026    file.sync()?;
4027    let generation =
4028        slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
4029    let committed = Slot {
4030        offset,
4031        length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4032        generation,
4033        hash: checksum(&catalog),
4034    };
4035    file.write_at(slot_offset(generation), &committed.bytes())?;
4036    file.sync()?;
4037    Ok(held)
4038}
4039
4040/// One column's frequency synopsis as values with their row counts, shared by every clone of a
4041/// reader.
4042type Synopsis = Arc<Vec<(Value, u64)>>;
4043
4044/// Reads committed native column pages without holding the table in memory.
4045#[derive(Debug, Clone)]
4046pub struct Reader {
4047    file: Arc<File>,
4048    table: Arc<Table>,
4049    dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
4050    /// Held while a global dictionary is being opened, one per column.
4051    ///
4052    /// The [`OnceLock`] above says whether one has been opened, which is the question a reader that
4053    /// already has it needs answered and is free. It does not say whether one is being opened, and
4054    /// the difference matters because every worker of a scan wants the same dictionary at the same
4055    /// moment. Without this they all miss, all read the page, all verify it and all decode it, and
4056    /// all but one throw the answer away. ClickBench 38 reads the URL dictionary, which is 515,958
4057    /// entries, and was paying for it twice.
4058    loading: Arc<Vec<Mutex<()>>>,
4059    /// Each column's frequency synopsis as values, the first time anything asks for it. See
4060    /// [`Reader::decode_frequencies`].
4061    frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
4062    /// Stored frequency sections are decoded once per open table. A small directory can hold the
4063    /// summary inline, but a larger one otherwise rereads and decodes the same section on every
4064    /// plan and every summary-backed aggregate.
4065    frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4066    /// How many global dictionaries have been opened. A scan of a dictionary column should open its
4067    /// dictionary once however many workers it has, and the test that says so is the only thing
4068    /// keeping it that way.
4069    opened: Arc<AtomicUsize>,
4070    /// The membership sieves of one stripe of one column, by column and then by stripe, read the
4071    /// first time a probe asks about them. A query filters on one or two columns and never looks at
4072    /// the rest, so reading these at open would be the whole index for the sake of a fraction of it.
4073    sieves: Arc<Vec<Vec<SieveSlot>>>,
4074    /// The per part ranges of one stripe of one column, by column and then by stripe, read the
4075    /// first time something compares that column and kept after that.
4076    part_ranges: Arc<Vec<Vec<RangeSlot>>>,
4077    /// Which stripe and which part of it every part of the table is, by table wide part number.
4078    places: Arc<Vec<Place>>,
4079    cache: Arc<Shelf>,
4080    /// Where the pages above are counted against the database's budget. See [`PagePool`].
4081    pool: PagePool,
4082    /// How many whole stripe pages have been read, which is what the sharing above is judged on. A
4083    /// scan of a column should read each of its stripes once however many workers it has.
4084    pages: Arc<AtomicUsize>,
4085    /// How many index sections have been read. A scan of a column should read each of its stripes
4086    /// once here too, and the test that says so is the only thing keeping it that way.
4087    indexes: Arc<AtomicUsize>,
4088    /// The file's size when it was opened, for [`Reader::layout`].
4089    size: u64,
4090    /// The committed directory's size, for [`Reader::layout`].
4091    directory: u64,
4092    /// What opening the file cost, which is a number rather than a claim.
4093    opening: Opening,
4094}
4095
4096/// What [`Reader::open`] read before it returned.
4097///
4098/// `spec/stats/04-in-memory.md` section 4.2 says opening a table reads the header and the directory
4099/// and nothing else, and once that document's statistics are in the file the tempting change is to
4100/// load a column summary or two on the way past, because they are small and the next query will
4101/// want them. A hundred milliseconds of that is a hundred milliseconds nobody asked for, and an
4102/// embedded database is opened by processes that are about to run one trivial query.
4103///
4104/// So the claim gets a number. Both of these are fixed by the schema and the stripe count and are
4105/// independent of how many rows the file holds, and the test that says so is what stops the
4106/// tempting change from landing quietly.
4107#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4108pub struct Opening {
4109    /// How many times the file was read. The header, then each directory slot that looked valid
4110    /// enough to check, so three at the most.
4111    pub reads: u32,
4112    /// How many bytes those reads asked for.
4113    pub bytes: u64,
4114}
4115
4116/// What a reader has read, while it was being opened and since.
4117#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4118pub struct Reads {
4119    /// What opening cost, before any query had been planned.
4120    pub opening: Opening,
4121    /// Whole stripe pages read since.
4122    pub pages: usize,
4123    /// Index sections read since.
4124    pub indexes: usize,
4125    /// Global dictionaries opened since. One per dictionary column that a query touched, however
4126    /// many workers touched it, which is a claim only a test can keep true.
4127    pub dictionaries: usize,
4128}
4129
4130/// Where one table wide part number lands.
4131#[derive(Debug, Clone, Copy)]
4132struct Place {
4133    stripe: u32,
4134    part: u32,
4135    rows: u32,
4136}
4137
4138/// One part's bytes inside one column page.
4139#[derive(Debug, Clone, Copy)]
4140struct PartSpan {
4141    start: usize,
4142    length: usize,
4143    hash: u64,
4144}
4145
4146/// What a reader holds for one stripe of one column.
4147///
4148/// The index is small and is loaded whether the caller wants the whole page or one part of it. The
4149/// page is loaded only by a scan, because a sparse fetch that wants a thousand rows out of sixty
4150/// four thousand would be reading sixty four times what it uses.
4151#[derive(Debug, Clone)]
4152struct CachedColumn {
4153    stripe: usize,
4154    index: Arc<Vec<PartSpan>>,
4155    page: Option<Arc<HeldPage>>,
4156}
4157
4158/// One stripe's page of one column, with which of its parts have already matched their checksums.
4159///
4160/// The bytes never change once they are read, so a part that matched once matches for as long as
4161/// the page is held. Hashing it again on every read was 3.5% of a `GROUP BY CounterID` over the
4162/// held pages of the ClickBench sample, run seventy times in one process. A part read without its
4163/// page is still checked every time, since those bytes come fresh off the file.
4164#[derive(Debug)]
4165struct HeldPage {
4166    bytes: Vec<u8>,
4167    checked: Vec<AtomicBool>,
4168}
4169
4170impl HeldPage {
4171    /// The bytes of part `part`, checked against `span` the first time anyone asks for them.
4172    fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4173        let bytes = part_bytes(&self.bytes, span)?;
4174        let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4175        if !checked.load(Atomic::Relaxed) {
4176            verify_part(bytes, span)?;
4177            checked.store(true, Atomic::Relaxed);
4178        }
4179        Ok(bytes)
4180    }
4181}
4182
4183/// Checks one part's bytes against the hash its index carries for them.
4184fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4185    let got = checksum(bytes);
4186    if got != span.hash {
4187        return Err(invalid(&format!(
4188            "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4189            span.start, span.length, span.hash,
4190        )));
4191    }
4192    Ok(())
4193}
4194
4195/// One column's stripes a reader holds, and which of them somebody is reading right now.
4196///
4197/// The pages are one slot per stripe of the table rather than a list of the ones being kept, so
4198/// finding a page is an index and not a walk. That matters because the walk happened under the
4199/// lock, once per part per column, and a scan that gives a whole stripe to each of thirty two
4200/// workers keeps enough pages that walking them was the longest thing the lock was held for. The
4201/// slots cost a pointer per stripe per column, which on the ClickBench file is eight kilobytes
4202/// against the forty megabytes of pages they point at. `order` is which of them are filled, oldest
4203/// first, because that is the one thing the slots cannot say by themselves.
4204///
4205/// `loading` is what keeps a scan from reading the same page once per worker. It is a list and not
4206/// a set because it holds at most one stripe per worker on the column and is walked far less often
4207/// than a hash of it would be built.
4208///
4209/// `index` is every index this reader has ever read for the column, one slot per stripe, and it is
4210/// never evicted. An index is a few hundred bytes and a page is a quarter of a megabyte, so the two
4211/// do not belong under the same budget. Riding in the page cache meant a worker that came back to a
4212/// stripe after its page had been evicted read the index again with it, which on the full
4213/// ClickBench file was about thirteen hundred reads out of a hundred and fourteen thousand.
4214///
4215/// `seen` is which stripes have had their page read before, and `passing` is the pages read for the
4216/// first time that are still held, oldest first. A page goes into the pool the second time it is
4217/// read and not the first, which is the rule [`NativeText`] follows for its decoded blocks. A
4218/// process that runs one statement, which is how a script or a benchmark uses the engine, reads
4219/// each page once, and with no memory limit the pool kept every one of them to the end: ClickBench
4220/// q33 held all of `WatchID` and `ClientIP` at its peak for a second scan that never came. The
4221/// first read now keeps a page only while it is among the column's floor of newest ones, and a
4222/// session that scans the table again pays one more read of each page and keeps it from then on.
4223#[derive(Debug, Default)]
4224struct Cached {
4225    pages: Vec<Option<Resident>>,
4226    loading: Vec<usize>,
4227    index: Vec<Option<Arc<Vec<PartSpan>>>>,
4228    seen: Vec<bool>,
4229    passing: VecDeque<usize>,
4230}
4231
4232/// One page a reader holds, and whether anyone has read it since the pool last looked.
4233#[derive(Debug, Clone)]
4234struct Resident {
4235    page: Arc<HeldPage>,
4236    used: Arc<AtomicBool>,
4237}
4238
4239/// Every column's pages of one reader, with how many each column holds and the floor under that.
4240#[derive(Debug)]
4241struct Shelf {
4242    columns: Vec<Mutex<Cached>>,
4243    /// How many pages each column holds right now. Counted outside the column locks so that the
4244    /// pool can tell whether a column is at its floor without taking a lock it might be under.
4245    held: Vec<AtomicUsize>,
4246    /// How many stripes of one column are kept whatever the budget says. See
4247    /// [`CACHED_STRIPES_PER_COLUMN`] for what sets it and [`Reader::keep_stripes`] for who raises it.
4248    kept: AtomicUsize,
4249}
4250
4251/// The pages every reader of one database keeps, under one budget in bytes.
4252///
4253/// A reader lives as long as the database does, so the pages it holds are what the next query finds
4254/// already in memory. They used to be four stripes a column, oldest out first, which on TPC-H SF1
4255/// meant every query read every page of lineitem off the file again and paid the system call for
4256/// it. Keeping every page there costs 38 MB and took a third of the system time off the suite.
4257///
4258/// So the question is no longer how many stripes a column keeps but how many bytes the database
4259/// does, and one budget answers it for every reader at once. A table nobody queries gives its pages
4260/// up to one that is being queried, which a count per column cannot do.
4261///
4262/// Pages leave by the clock. Each has a bit a read sets, and when the pool is over budget it walks
4263/// from the oldest: a page with the bit set loses the bit and goes round again, and a page without
4264/// it goes. That keeps what is read over and over and lets a page one scan read once go first.
4265///
4266/// The old count is still a floor. A column never gives up a page while it holds four or fewer,
4267/// because a scan whose workers evict each other's pages reads a quarter of a megabyte for every
4268/// part it takes, and a budget of zero is the cache as it was before the pool existed.
4269#[derive(Debug, Clone, Default)]
4270pub struct PagePool {
4271    ring: Arc<Mutex<Ring>>,
4272    budget: Arc<AtomicUsize>,
4273}
4274
4275#[derive(Debug, Default)]
4276struct Ring {
4277    held: VecDeque<Held>,
4278    bytes: usize,
4279}
4280
4281/// One page in the pool, pointing back at the reader that holds it.
4282///
4283/// Weak, because a reader that has gone, which every reader does at a checkpoint, should take its
4284/// pages with it and not have them kept alive by the pool.
4285#[derive(Debug)]
4286struct Held {
4287    shelf: Weak<Shelf>,
4288    column: usize,
4289    stripe: usize,
4290    bytes: usize,
4291    used: Arc<AtomicBool>,
4292}
4293
4294impl PagePool {
4295    /// A pool that keeps up to `budget` bytes of pages beyond each column's floor.
4296    #[must_use]
4297    pub fn new(budget: usize) -> Self {
4298        let pool = Self::default();
4299        pool.budget.store(budget, Atomic::Relaxed);
4300        pool
4301    }
4302
4303    /// The bytes of pages the pool is counting now.
4304    ///
4305    /// # Panics
4306    ///
4307    /// If the pool's lock is poisoned, which takes a panic while it was held.
4308    #[must_use]
4309    pub fn bytes(&self) -> usize {
4310        self.ring.lock().map_or(0, |ring| ring.bytes)
4311    }
4312
4313    /// Counts a page a reader has just taken in, and lets pages go until the pool is back under its
4314    /// budget or it has looked at every page once.
4315    ///
4316    /// Called with no column lock held. The pages that go are chosen under the pool's lock and
4317    /// dropped under their column's lock afterwards, so no thread ever holds both.
4318    fn admit(&self, held: Held) {
4319        let budget = self.budget.load(Atomic::Relaxed);
4320        let mut gone = Vec::new();
4321        {
4322            let Ok(mut ring) = self.ring.lock() else { return };
4323            ring.bytes += held.bytes;
4324            ring.held.push_back(held);
4325            // One lap and no more. A page read since the last pass loses its bit on this one and
4326            // can only go on a later one, which is the second chance the clock is named for.
4327            let mut looked = 0;
4328            let limit = ring.held.len();
4329            while ring.bytes > budget && looked < limit {
4330                looked += 1;
4331                let Some(entry) = ring.held.pop_front() else { break };
4332                let Some(shelf) = entry.shelf.upgrade() else {
4333                    ring.bytes -= entry.bytes;
4334                    continue;
4335                };
4336                if entry.used.swap(false, Atomic::Relaxed) {
4337                    ring.held.push_back(entry);
4338                    continue;
4339                }
4340                let count = &shelf.held[entry.column];
4341                if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4342                    ring.held.push_back(entry);
4343                    continue;
4344                }
4345                count.fetch_sub(1, Atomic::Relaxed);
4346                ring.bytes -= entry.bytes;
4347                gone.push((shelf, entry));
4348            }
4349            // A reader that has gone leaves its entries behind, and with a budget nobody reaches
4350            // they would pile up one checkpoint after another. The front is where the oldest are.
4351            while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4352                if let Some(entry) = ring.held.pop_front() {
4353                    ring.bytes -= entry.bytes;
4354                }
4355            }
4356        }
4357        for (shelf, entry) in gone {
4358            let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4359            if let Some(slot) = cached.pages.get_mut(entry.stripe) {
4360                if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
4361                    *slot = None;
4362                }
4363            }
4364        }
4365    }
4366}
4367
4368/// Stripes of one column a reader keeps the bytes of, when nobody has asked for more.
4369///
4370/// This has to hold at least as many stripes as a column has workers in it at once, or the workers
4371/// evict each other's pages and read them again. Four is what a scan that hands parts out in order
4372/// needs, because then every worker is within a few parts of every other and at most a couple of
4373/// stripes are open at a time. A scan that hands a whole stripe to each worker has one stripe open
4374/// per worker for the length of that stripe, and it says so with [`Reader::keep_stripes`] rather
4375/// than paying for sixteen slots on every table that is read one part at a time.
4376///
4377/// It multiplies by the page size, which is a quarter of a megabyte for a four byte column, and by
4378/// the number of columns a query touches.
4379const CACHED_STRIPES_PER_COLUMN: usize = 4;
4380
4381/// The sieves of one stripe of one column, once somebody has asked for them.
4382type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4383
4384type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4385
4386#[derive(Debug)]
4387struct NativeText {
4388    file: Arc<File>,
4389    /// How many values the dictionary holds.
4390    values: usize,
4391    /// Where each value ends inside its payload block, packed at `offset_bits` in runs of
4392    /// [`TEXT_OFFSET_RUN`].
4393    ///
4394    /// Ends rather than starts, because then a block of 1,024 values is 1,024 numbers rather than
4395    /// 1,025: the start of a value is the end of the one before it, and the first value of a block
4396    /// starts at zero by construction. Relative to the block rather than to the payload, because a
4397    /// reader decodes a whole block and slices it, so an offset into the payload is a number it
4398    /// would have to subtract a base from anyway.
4399    ///
4400    /// The vector is the index as it was read, so the offsets start after the header, and
4401    /// [`Self::packed`] is where they are read from.
4402    offsets: Vec<u8>,
4403    /// Bits one offset is packed at, which is what the largest block of this column spans and is the
4404    /// same for every block of it.
4405    offset_bits: usize,
4406    /// The same ends unpacked, built once enough readers have asked for one at a time.
4407    ///
4408    /// Reading one offset out of the packed form costs about fifty instructions: a division to find
4409    /// the run, a bounds check to slice it, a shift to reach the bit the value starts at and a
4410    /// narrowing on the way out. That is the right price for a reader that wants a handful. It is
4411    /// the wrong price for `STRLEN` over a column, which asks for one per row and nothing else, and
4412    /// where a million of them was a third of ClickBench 28.
4413    ///
4414    /// With the ends unpacked every read is a load, and a vector of lengths is one loop over them.
4415    /// The table is built only once the reads say it will be used, which is what
4416    /// [`Self::ends_worth_unpacking`] decides and [`Self::ends_asked`] counts towards, because a
4417    /// table built for a reader that wanted three values is four bytes a value spent on nothing.
4418    value_ends: OnceLock<Option<Vec<u32>>>,
4419    /// The length of every value, worked out of [`Self::value_ends`] the first time a vector of
4420    /// lengths is asked for.
4421    ///
4422    /// A length out of the ends is two loads, a test for whether the value opens its block and a
4423    /// check that it does not end before it starts, which came to thirteen instructions a row on
4424    /// ClickBench 28. Out of this it is one load. The order is checked once for the whole table
4425    /// while it is built, and a column that fails it gets no table and goes on reading the ends,
4426    /// which is where the error is reported. Two bytes a value where every value is short enough,
4427    /// four otherwise, and only for a column something has asked the length of a vector at a time.
4428    value_lens: OnceLock<Option<Lengths>>,
4429    /// How many single offset reads have come in while the table is not built.
4430    ///
4431    /// Relaxed, and read only against a threshold, so two threads racing here means the table is
4432    /// built one read early or one read late. Counting stops the moment the table exists, because
4433    /// [`OnceLock::get`] settles it before this is touched.
4434    ends_asked: AtomicUsize,
4435    /// How many entries the sorted order has, which is the value count.
4436    ranks: usize,
4437    /// Where the sorted order starts in the file. It is read a block at a time and only when
4438    /// something searches it, so a query that never compares this column against a literal never
4439    /// touches it at all.
4440    rank_at: u64,
4441    /// Where each block of the sorted order ends, as a byte offset from `rank_at`. A block is packed
4442    /// at whatever width its own heads need, so unlike the entries it replaced its length is not
4443    /// arithmetic on the block number.
4444    rank_ends: Vec<u64>,
4445    rank_hashes: Vec<u64>,
4446    rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4447    /// Bits one code is packed at, which is what the value count needs and is the same for every
4448    /// block of the column.
4449    code_bits: usize,
4450    /// The sorted order turned round, built the first time a reader asks for it.
4451    ///
4452    /// Four bytes per value against the four the offsets already hold, so a column that has this is
4453    /// carrying half again what it carried before rather than something of a new order. It is built
4454    /// only when something asks, which is a grouped min or max over this column and nothing else,
4455    /// and that reader was going to read the payload of this column once per row otherwise.
4456    code_ranks: OnceLock<Option<Vec<u32>>>,
4457    /// Where each block of the payload starts in the file, and how many stored bytes it is.
4458    ///
4459    /// Absolute rather than an offset from a base the blocks share, because a block is written the
4460    /// moment it fills and what comes after it in the file is whatever the load wrote next. A file
4461    /// old enough to have them back to back is read into these same two lists by adding the base to
4462    /// the ends it carries, so nothing below here knows which kind of file it came from.
4463    starts: Vec<u64>,
4464    lengths: Vec<u64>,
4465    hashes: Vec<u64>,
4466    /// Conservative four-byte substring signatures, read only by a compatible LIKE filter.
4467    grams: Option<NativeGrams>,
4468    /// The payload, read and decoded a block at a time and kept after that.
4469    blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4470    /// The length in characters of every value of a block, worked out the first time `length` asks
4471    /// for a value in that block.
4472    ///
4473    /// Kept instead of the block it was counted out of. `length` reads every row of a column, and
4474    /// reading the bytes through [`Self::payload_block`] kept every block it touched, which is every
4475    /// distinct value of the column decoded: seven string columns of ClickBench held 13.9 GB to
4476    /// answer seven `max(length(...))`. The counts are four bytes a value, so the same scan keeps
4477    /// the counts and decodes each block once, the same number of times it did before.
4478    char_lens: Vec<OnceLock<Box<[u32]>>>,
4479    /// How many decoded payload bytes this column keeps before a sweep stops keeping what it reads.
4480    /// [`TEXT_KEEP_BUDGET`] everywhere but in the test of the ceiling.
4481    keep_budget: usize,
4482    /// Roughly how many decoded payload bytes are being kept, which is what [`TEXT_KEEP_BUDGET`]
4483    /// is measured against.
4484    ///
4485    /// Roughly, because two threads that keep the same block at the same time both add its length
4486    /// while [`OnceLock`] keeps one of the two. That makes the count read high and the budget bind
4487    /// a little early, which is the harmless direction, and it costs one relaxed add a block rather
4488    /// than a lock on the path every scan of a string column goes through.
4489    payload_kept: AtomicUsize,
4490    /// Which payload blocks a sweep has decoded before, one flag a block.
4491    ///
4492    /// A sweep keeps a block the second time it decodes it and not the first. A process that runs
4493    /// one statement, which is how a benchmark or a script uses the engine, sweeps each block once
4494    /// and so keeps nothing: on ten million rows a `URL LIKE` held 396 MB with every block kept and
4495    /// 97 MB with none, for the same processor time. A session that asks again pays the decode one
4496    /// more time and reads kept blocks from then on, under the same [`TEXT_KEEP_BUDGET`].
4497    swept: Vec<AtomicBool>,
4498    /// How many blocks [`TextSource::visit_at`] has decoded and dropped because the column was
4499    /// already holding its [`TEXT_KEEP_BUDGET`].
4500    ///
4501    /// A sweep reads the dictionary in order and touches a block once, so dropping what it reads
4502    /// past the budget costs one decode a block and bounds the column. A visit reads a vector of
4503    /// codes, and the codes of a scan land all over the dictionary: on ten million rows of
4504    /// ClickBench each vector of two thousand `URL`s touches about a hundred and forty of its two
4505    /// and a half thousand blocks, and so does the next one. A cache holding a tenth of the column
4506    /// still misses half of those, and dropping every block past the budget would decode the
4507    /// column hundreds of times over to answer one `lower(URL)`. So a visit drops past the budget
4508    /// only until it has dropped as many blocks as the column has, which is what a read whose codes
4509    /// are few or clustered never reaches, and keeps what it reads after that, the way a row at a
4510    /// time read always did. That bounds what a visit can cost over the old read at one more decode
4511    /// of the column.
4512    visit_dropped: AtomicUsize,
4513    /// The boundaries this dictionary has already been searched for, by the value searched for.
4514    ///
4515    /// A search is the expensive thing this type does. It settles a probe on the stored head where
4516    /// it can and reads a value where it cannot, and reading a value decodes the payload block it
4517    /// sits in, so one search can cost several blocks. The thing that makes remembering worth it is
4518    /// that the same search comes back: a top N asks once a chunk whether anything left can beat its
4519    /// worst candidate, and the worst candidate settles long before the chunks run out.
4520    ///
4521    /// Shared across the instances of a scan rather than kept per instance, because each of them has
4522    /// its own worst candidate and all of them are searching the same dictionary. One lock per chunk
4523    /// is nothing next to a probe of a file.
4524    ///
4525    /// Bounded by [`TEXT_SEARCH_MEMO`] and emptied rather than evicted when it is full. What fills
4526    /// it is a top N improving its bound, which happens a few dozen times and then stops, so the
4527    /// bound is there for the filter that searches for a different literal every chunk rather than
4528    /// for anything this is meant to help.
4529    searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4530}
4531
4532#[derive(Debug)]
4533struct NativeGrams {
4534    start: u64,
4535    length: usize,
4536    /// How long one block's signature is.
4537    width: usize,
4538    hash: u64,
4539    /// For each literal asked about lately, whether each block might hold it.
4540    ///
4541    /// The answer for every block at once, worked out by one pass over the signatures a window at a
4542    /// time, rather than the signatures read in and kept. On ClickBench `URL` they are 21 MB for
4543    /// ten million rows and a verdict is 2,650 flags, and a filter asks the same question of every
4544    /// block, so the pass is paid once and what stays resident is the flags.
4545    verdicts: Mutex<Vec<Verdict>>,
4546}
4547
4548/// A literal and whether each block might hold it.
4549type Verdict = (Vec<u8>, Arc<[bool]>);
4550
4551/// How many literals a column remembers the verdicts of.
4552const GRAM_VERDICTS: usize = 8;
4553
4554impl NativeGrams {
4555    /// Whether each block might hold `literal`, remembered or worked out now.
4556    ///
4557    /// The lock is held over the pass so that the threads of one scan, which all ask about the
4558    /// same literal at the start, read the signatures once between them.
4559    fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4560        let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4561        if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4562            return Ok(Arc::clone(verdict));
4563        }
4564        let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4565        let mut verdict = Vec::with_capacity(self.length / self.width);
4566        let window = GRAM_WINDOW / self.width * self.width;
4567        let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4568            verdict.extend(bytes.chunks(self.width).map(|bits| {
4569                wanted
4570                    .iter()
4571                    .flatten()
4572                    .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4573            }));
4574            Ok(())
4575        })?;
4576        if hash != self.hash {
4577            return Err(invalid("global dictionary substring signatures checksum differs"));
4578        }
4579        let verdict: Arc<[bool]> = verdict.into();
4580        if held.len() >= GRAM_VERDICTS {
4581            held.remove(0);
4582        }
4583        held.push((literal.to_vec(), Arc::clone(&verdict)));
4584        Ok(verdict)
4585    }
4586
4587    fn footprint(&self) -> usize {
4588        self.verdicts.lock().map_or(0, |held| {
4589            held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4590        })
4591    }
4592}
4593
4594/// How many searched for values a column's dictionary remembers the boundary of.
4595///
4596/// See [`NativeText::searched`]. Small because the case it is for repeats one value, not because a
4597/// larger one would be wrong.
4598const TEXT_SEARCH_MEMO: usize = 64;
4599
4600/// How many values of a dictionary go in one block of the payload.
4601///
4602/// The block is the unit the string cascade encodes, the unit a checksum covers, and the unit a
4603/// reader has to decode to get at a single value, so it is the one number the payload format turns
4604/// on. Blocking by values rather than by bytes is what keeps a value out of two blocks at once: the
4605/// block holding a code is `code / TEXT_PAYLOAD_VALUES` and nothing has to be stitched.
4606///
4607/// A probe on the five ClickBench columns that have a dictionary worth the name, written up on
4608/// #347, measured the ratio and the decode speed at 128, 256, 512, 1,024 and 4,096 values. Both get
4609/// better all the way up, because front coding and the LZ matcher have more to look back at and
4610/// because the per chunk setup is spread over more values. What stops it is the point read: a query
4611/// that wants ten values has to decode ten blocks, so the block is what a lookup costs. At 1,024
4612/// values a block is between 67 KB and 394 KB decoded across those five columns, and the ratios are
4613/// 2.3 to 4.5. Going up to 4,096 buys two to six percent more and makes a block as much as 1.5 MB.
4614/// Going down to 512 gives up five to nine percent.
4615const TEXT_PAYLOAD_VALUES: usize = 1024;
4616
4617/// Eight KiB per payload block, which is what makes a four-byte substring a useful negative test on
4618/// a column of URLs.
4619///
4620/// Two KiB was the first answer and on ClickBench `URL` it proved almost nothing. A block of 1,024
4621/// sorted URLs holds about seventeen thousand distinct four-byte grams, and at two bits each that
4622/// set nine in ten of the sixteen thousand bits there were, so `LIKE '%google%'` passed most blocks
4623/// it had no match in and decoded them. At eight KiB four bits in ten are set, and of the 2,650
4624/// blocks of `URL` in ten million rows a needle that is in none of them passes 36. The signatures
4625/// are not read into memory, see [`NativeGrams::verdicts`], so the width costs file and not
4626/// resident memory.
4627const TEXT_GRAM_BYTES: usize = 8192;
4628
4629/// The signature width of a format 28 file, which is still read.
4630const NARROW_GRAM_BYTES: usize = 2048;
4631
4632/// How much of a column's signatures a verdict reads at a time.
4633const GRAM_WINDOW: usize = 256 << 10;
4634
4635/// A fast mixing step for exactly four bytes, shared by load and query, into a signature of
4636/// `width` bytes.
4637fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4638    let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4639    let mut first = original ^ (original >> 16);
4640    first = first.wrapping_mul(0x7feb_352d);
4641    first ^= first >> 15;
4642    let mut second = original ^ (original >> 17);
4643    second = second.wrapping_mul(0x846c_a68b);
4644    second ^= second >> 16;
4645    let mask = width * 8 - 1;
4646    [(first as usize) & mask, (second as usize) & mask]
4647}
4648
4649/// How many decoded payload bytes one dictionary keeps before a sweep stops keeping what it reads.
4650///
4651/// A sweep of the whole dictionary decodes every block whatever it does, and the only question is
4652/// whether it hangs on to them. Keeping all of them is 4.2 GB on ClickBench `URL` at a hundred
4653/// million rows, which is what #997 was right to stop. Keeping none of them means the next query
4654/// asking the same thing decodes all of it again, and on the same column at a million rows that
4655/// took a `LIKE` from 2.7 ms to 16.2 ms, because the decode used to be paid once by a session and
4656/// is now paid by every statement in it. Neither end is the answer. A bound is.
4657///
4658/// So a sweep keeps what it decodes until the column is holding this much and decodes without
4659/// keeping after that. At a million rows the five ClickBench string columns decode to between 8 MB
4660/// and 85 MB, so they sit inside it and a repeated `LIKE` reads a decoded block rather than a
4661/// stored one. At a hundred million rows `URL` fills it and the rest of that column is read and
4662/// dropped, which is the old cost on the part that does not fit and none of the old footprint.
4663///
4664/// Two hundred and fifty six megabytes a column is a number and not a policy, and the policy is
4665/// what should replace it: this wants to be a buffer pool over the whole database, sized against
4666/// the memory limit the session was given, with the blocks of every column competing for it and the
4667/// least useful one evicted. That is F2 work. What is here is the part of it that can be written
4668/// without an eviction order, which is a ceiling.
4669const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4670
4671/// The length of every value of a column, as narrow as the longest of them allows.
4672///
4673/// The table is read at the codes a vector holds, which on a column the size of ClickBench `URL`
4674/// land all over it, so what a length costs is whether its line is in cache. Half a million URLs
4675/// are two megabytes at four bytes a length and one at two, which is the difference between the
4676/// table sitting in the second level cache or not.
4677#[derive(Debug)]
4678enum Lengths {
4679    /// Every length fits in sixteen bits.
4680    Narrow(Vec<u16>),
4681    /// Some value is longer than that.
4682    Wide(Vec<u32>),
4683}
4684
4685impl Lengths {
4686    /// The lengths at `indices`, appended to `into`, and zero for a position past the end, which
4687    /// is what a row at a time read says.
4688    fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4689        match self {
4690            Lengths::Narrow(lens) => into.extend(
4691                indices
4692                    .iter()
4693                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4694            ),
4695            Lengths::Wide(lens) => into.extend(
4696                indices
4697                    .iter()
4698                    .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4699            ),
4700        }
4701    }
4702
4703    /// The bytes the table holds on to.
4704    fn footprint(&self) -> usize {
4705        match self {
4706            Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4707            Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4708        }
4709    }
4710}
4711
4712/// The length of every value out of where each one ends inside its payload block, or `None` for
4713/// ends that go backwards somewhere inside a block.
4714///
4715/// A value that opens a block starts at zero and every other one starts where the value before it
4716/// ends, so a block is a run of differences.
4717///
4718/// Built at two bytes a length straight away, and built again at four only when some value turns
4719/// out too long for that, which is rare enough that the second pass is not worth avoiding.
4720fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4721    match lengths_as::<u16>(ends)? {
4722        Some(narrow) => Some(Lengths::Narrow(narrow)),
4723        None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4724    }
4725}
4726
4727/// [`lengths_of`] at one width: `None` for ends that go backwards, and `Some(None)` for a length
4728/// that does not fit in `T`.
4729fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4730    let mut lens = Vec::with_capacity(ends.len());
4731    for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4732        let mut start = 0;
4733        for &end in block {
4734            let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4735                return Some(None);
4736            };
4737            lens.push(len);
4738            start = end;
4739        }
4740    }
4741    Some(Some(lens))
4742}
4743
4744/// How many offsets go in one packed run.
4745///
4746/// A payload block holds 1,024 values and `bitpack::pack_tail` takes fewer than 1,024 at a time,
4747/// since a whole unit of that many belongs in the transposed layout instead. So the offsets of a
4748/// block go in two runs. Five hundred and twelve values at any width is a whole number of bytes, so
4749/// a run starts where a multiply says it does and nothing is padded.
4750const TEXT_OFFSET_RUN: usize = 512;
4751
4752/// Bytes at the front of a global dictionary index: the value count, the values a payload block
4753/// holds, the block count and the bits an offset is packed at.
4754const DICTIONARY_HEADER: usize = 16;
4755
4756/// Set beside the offset width in the fourth word of a global dictionary index, meaning each
4757/// payload block says where in the file it starts and how long it is, rather than sitting directly
4758/// behind the block before it.
4759///
4760/// In that word rather than in a word of its own because the width is at most 32 and lives in a
4761/// `u32`, so the top of it has never been anything. A build old enough not to know the flag reads
4762/// the file's format before it reads any of this and refuses it there, and if it somehow did get
4763/// here it would find an offset width of two billion and say so.
4764///
4765/// The point of the flag is that a block written the moment it fills does not know what will be
4766/// written after it, so the payload of a column cannot be one run of bytes unless the whole column
4767/// is held until the file is closed. That is the memory the load cannot afford. What it costs is
4768/// eight bytes a block, against the block being a thousand values.
4769const DICTIONARY_SCATTERED: u32 = 1 << 31;
4770/// The dictionary index carries one four-byte substring signature per payload block.
4771const DICTIONARY_GRAMS: u32 = 1 << 30;
4772/// Each signature is [`TEXT_GRAM_BYTES`] long rather than the [`NARROW_GRAM_BYTES`] a format 28
4773/// file wrote.
4774const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4775/// Every flag the width word of a dictionary can carry above the offset width.
4776const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4777
4778/// How many entries of a dictionary's sorted order sit in one block that is read and checked as a
4779/// unit.
4780///
4781/// Five hundred and twelve entries is between two and three kilobytes on the ClickBench string
4782/// columns, which is well under a page. A binary search over half a million entries makes nineteen
4783/// probes, and the first ten land in ten different blocks while the last nine land in the one block
4784/// that holds the answer, so the whole search reads about thirty kilobytes of a megabyte of order. A
4785/// smaller block would save a little on the early probes, cost a checksum and an end list four times
4786/// as long, and give the heads less to share a base with. A larger one would read more than it uses
4787/// on every probe.
4788const TEXT_RANK_BLOCK: usize = 512;
4789
4790/// Bytes at the front of a rank block, which is the base of its heads and the width they are packed
4791/// at.
4792///
4793/// An entry used to be twelve bytes flat, eight for the head and four for the code, and on the five
4794/// ClickBench columns that have a dictionary worth the name that was 744 MB of a 12.2 GB file. Both
4795/// halves of it are nearly empty. The heads are the first eight bytes of the values in sorted order,
4796/// so a block of five hundred and twelve of them spans a tiny slice of the column, and on a column of
4797/// URLs they are all `http://w` and the block holds one distinct head. The codes are positions in a
4798/// dictionary of eighteen million, which is twenty five bits and not thirty two.
4799///
4800/// So a block now writes the smallest head in it, the bits the largest is above that, and the heads
4801/// and the codes packed at the width each needs. A block where every head agrees costs nine bytes
4802/// and the codes.
4803const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4804
4805impl NativeText {
4806    /// One block of the payload, read and decoded the first time anything asks for a value in it.
4807    ///
4808    /// The bytes handed back are the values of the block laid end to end, which is what the offsets
4809    /// describe, so a caller slices it with the offsets it already has. Where the block sits in the
4810    /// file is the only thing the caller cannot work out for itself, because the stored form is
4811    /// shorter than the decoded one and by a different amount in every block.
4812    fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4813        let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4814        let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4815        Ok(Some(bytes.as_slice()))
4816    }
4817
4818    /// The character length of every value in one block, counted the first time it is asked for.
4819    ///
4820    /// The block is read out of [`Self::blocks`] where something already kept it and decoded and
4821    /// dropped where nothing did, so counting never adds a block to what this column holds. Two
4822    /// threads asking for the same block at once both count it and one of the two answers is kept,
4823    /// which costs a decode and is cheaper than a lock on every lookup.
4824    fn block_chars(&self, block: usize) -> Result<&[u32]> {
4825        let slot = self
4826            .char_lens
4827            .get(block)
4828            .ok_or_else(|| invalid("a block past the global dictionary"))?;
4829        if let Some(lens) = slot.get() {
4830            return Ok(lens);
4831        }
4832        let decoded;
4833        let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4834            Some(Ok(kept)) => kept,
4835            _ => {
4836                decoded = self.decode_block(block)?;
4837                &decoded
4838            }
4839        };
4840        let first = block * TEXT_PAYLOAD_VALUES;
4841        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4842        let ends = self.ends_within(first, last)?;
4843        if ends.len() != last - first {
4844            return Err(invalid("global dictionary offsets are short"));
4845        }
4846        let mut lens = Vec::with_capacity(ends.len());
4847        let mut start = u64::from(self.start_within(first)?);
4848        for &end in &ends {
4849            let value = usize::try_from(start)
4850                .ok()
4851                .zip(usize::try_from(end).ok())
4852                .and_then(|(from, to)| bytes.get(from..to))
4853                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4854            // A continuation byte of UTF-8 is `0b10xx_xxxx` and every other byte starts a
4855            // character, so the bytes that are not continuations are the characters.
4856            let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
4857            lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
4858            start = end;
4859        }
4860        Ok(slot.get_or_init(|| lens.into_boxed_slice()))
4861    }
4862
4863    /// Reads and decodes one block of the payload, without deciding who keeps it.
4864    ///
4865    /// [`Self::payload_block`] keeps it forever, which is what a point read wants and what a walk
4866    /// of the whole dictionary must not do. Both call this and they differ in nothing else.
4867    fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4868        let len = self.lengths[block];
4869        let mut stored = vec![
4870            0;
4871            usize::try_from(len).map_err(|_| invalid(
4872                "global dictionary block does not fit in memory"
4873            ))?
4874        ];
4875        read_at(&self.file, self.starts[block], &mut stored)?;
4876        if checksum(&stored) != self.hashes[block] {
4877            return Err(invalid("global dictionary payload checksum differs"));
4878        }
4879        let first = block * TEXT_PAYLOAD_VALUES;
4880        let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4881        let want = self.end_within(last - 1)? as usize;
4882        let values = string::decode_flat(&stored)?;
4883        if values.len() != last - first {
4884            return Err(invalid("global dictionary block holds the wrong value count"));
4885        }
4886        let bytes = values.into_bytes();
4887        if bytes.len() != want {
4888            return Err(invalid("global dictionary block decodes to the wrong length"));
4889        }
4890        Ok(bytes)
4891    }
4892
4893    /// The block holding a value that a read hands over on loan, kept or decoded for the call.
4894    ///
4895    /// A block something already kept is read where it is. One nothing kept is kept the second
4896    /// time a loaned read decodes it while the column is holding less than [`Self::keep_budget`],
4897    /// and decoded into `decoded` and dropped with it otherwise, which is the policy
4898    /// [`TextSource::sweep`] explains. `scattered` is a read by code rather than in order, which
4899    /// stops dropping once it has dropped a column's worth of blocks, for the reason
4900    /// [`Self::visit_dropped`] gives.
4901    fn loaned_block<'a>(
4902        &'a self,
4903        block: usize,
4904        decoded: &'a mut Vec<u8>,
4905        scattered: bool,
4906    ) -> Result<&'a [u8]> {
4907        let kept = self.blocks.get(block).and_then(OnceLock::get);
4908        if let Some(Ok(kept)) = kept {
4909            return Ok(kept);
4910        }
4911        let again = kept.is_none()
4912            && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4913        let keep = again
4914            && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
4915                || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
4916        if keep {
4917            let kept = self
4918                .payload_block(block)?
4919                .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4920            self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4921            return Ok(kept);
4922        }
4923        *decoded = self.decode_block(block)?;
4924        if scattered && again {
4925            self.visit_dropped.fetch_add(1, Atomic::Relaxed);
4926        }
4927        Ok(decoded)
4928    }
4929
4930    /// How many single offset reads make [`Self::value_ends`] worth building.
4931    ///
4932    /// As many reads as the dictionary has values. Building the table costs about thirty
4933    /// instructions a value once the fresh pages it lands in are counted, and a read out of it saves
4934    /// about thirty five, so it repays itself after roughly one read per value. The reads so far are
4935    /// the only guess there is at the reads to come, and waiting until they match the size of the
4936    /// dictionary is betting that a column read that much will be read that much again.
4937    ///
4938    /// A sixteenth was the first answer, from counting the unpacking alone at three instructions a
4939    /// value. ClickBench 38 showed what that missed: it reads about twenty thousand titles a
4940    /// statement out of a dictionary of three hundred and fifty thousand, crossed a sixteenth in its
4941    /// second statement and was two percent slower for a table it did not read enough to repay. A
4942    /// scan asking for the length of every row crosses it part way through its first statement on
4943    /// ClickBench, where a string column has about two rows for every value, and a filter that keeps
4944    /// a few thousand rows never does. The floor is there
4945    /// because a short dictionary would otherwise build a table for a handful of reads.
4946    fn ends_worth_unpacking(&self) -> usize {
4947        self.values.max(TEXT_PAYLOAD_VALUES)
4948    }
4949
4950    /// The unpacked ends, if they are built or if this read is the one that makes them worth it.
4951    fn value_ends(&self) -> Option<&[u32]> {
4952        if let Some(built) = self.value_ends.get() {
4953            return built.as_deref();
4954        }
4955        if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4956            return None;
4957        }
4958        self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4959    }
4960
4961    /// Every end of the column, a run at a time.
4962    ///
4963    /// `None` rather than an error on anything wrong, because this is a cache in front of a reader
4964    /// that answers the same question. A column whose offsets are short or whose ends do not fit in
4965    /// four bytes gets no table and the same error it would have got, from the read that wanted it.
4966    fn unpack_ends(&self) -> Option<Vec<u32>> {
4967        let mut ends = vec![0u32; self.values];
4968        for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4969            let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4970            bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4971                u32::try_from(bits).unwrap_or(u32::MAX)
4972            })
4973            .ok()?;
4974        }
4975        // An end that did not fit was stored as the sentinel, and a real one cannot reach it because
4976        // a payload block is far smaller than four gigabytes. So the column keeps the packed reader.
4977        if ends.contains(&u32::MAX) { None } else { Some(ends) }
4978    }
4979
4980    /// The packed offsets, which is the index past its header.
4981    fn packed(&self) -> &[u8] {
4982        self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
4983    }
4984
4985    /// Where the value at `index` ends inside its payload block.
4986    fn end_within(&self, index: usize) -> Result<u32> {
4987        if let Some(ends) = self.value_ends() {
4988            return ends
4989                .get(index)
4990                .copied()
4991                .ok_or_else(|| invalid("global dictionary offsets are short"));
4992        }
4993        let run = index / TEXT_OFFSET_RUN;
4994        let bytes = self
4995            .packed()
4996            .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4997            .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4998        let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4999            .map_err(|_| invalid("global dictionary offsets are short"))?;
5000        u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
5001    }
5002
5003    /// Where every value in `first..last` ends inside its payload block, in one pass over the runs.
5004    ///
5005    /// [`Self::end_within`] answers for one value and pays for it twice over: it shifts a window to
5006    /// the bit the value starts at, and the copy that fills that window is a length the compiler does
5007    /// not know, so it is a call to `memcpy` rather than a load. A sweep asked for two of those per
5008    /// value, one for the end and one for the start that is the end before it, and on the ClickBench
5009    /// `URL` dictionary of eighteen million that was most of the half second a `LIKE` over it took.
5010    ///
5011    /// [`bitpack::unpack_tail_into`] walks the run instead, which makes the window a fixed width and
5012    /// so an unaligned load, and reads the bit position off a counter. A run is five hundred and
5013    /// twelve values and a block is two of them, so a block of a thousand and twenty four values
5014    /// costs two calls here and nothing per value.
5015    ///
5016    /// The answer is written straight into the result. A run that is wanted from its first value,
5017    /// which is every run but the one the sweep starts in, unpacks into its own window of the result
5018    /// and is never copied. Only a run joined part way through needs the scratch buffer, and there is
5019    /// at most one of those per sweep, so the buffer is allocated the first time one turns up.
5020    fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
5021        let mut ends = vec![0u64; last.saturating_sub(first)];
5022        let mut scratch = Vec::new();
5023        let mut at = first;
5024        while at < last {
5025            let run = at / TEXT_OFFSET_RUN;
5026            let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
5027            let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
5028            let bytes = self
5029                .packed()
5030                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5031                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5032            let from = at % TEXT_OFFSET_RUN;
5033            let upto = stop - run * TEXT_OFFSET_RUN;
5034            if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
5035                return Err(invalid("global dictionary offsets are short"));
5036            }
5037            let into = &mut ends[at - first..stop - first];
5038            if from == 0 {
5039                bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
5040                    .map_err(|_| invalid("global dictionary offsets are short"))?;
5041            } else {
5042                scratch.resize(held, 0);
5043                bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
5044                    .map_err(|_| invalid("global dictionary offsets are short"))?;
5045                into.copy_from_slice(&scratch[from..upto]);
5046            }
5047            at = stop;
5048        }
5049        Ok(ends)
5050    }
5051
5052    /// Where the value at `index` starts inside its payload block, which is where the value before
5053    /// it ended unless it is the first of the block.
5054    fn start_within(&self, index: usize) -> Result<u32> {
5055        if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
5056    }
5057
5058    /// Where the value at `index` starts and ends inside its payload block.
5059    ///
5060    /// The two offsets sit next to each other in the same run unless the value opens one, and a run
5061    /// of seventeen bit offsets, which is what a block of a thousand strings needs, puts a pair of
5062    /// them inside one eight byte load. So the common case reads the packed bytes once rather than
5063    /// twice and does the bounds arithmetic once. This is asked once per string a text column hands
5064    /// out, and on ClickBench 27 the two reads together were a quarter of the query.
5065    fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5066        if let Some(ends) = self.value_ends() {
5067            let end =
5068                *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5069            // The value before it in the same block, and zero where there is no value before it.
5070            // `index` is inside the table, so the one under it is too.
5071            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5072            if start > end {
5073                return Err(invalid("global dictionary value ends before it starts"));
5074            }
5075            return Ok((start, end));
5076        }
5077        let within = index % TEXT_OFFSET_RUN;
5078        let (start, end) = if within == 0 {
5079            (self.start_within(index)?, self.end_within(index)?)
5080        } else {
5081            let run = index / TEXT_OFFSET_RUN;
5082            let bytes = self
5083                .packed()
5084                .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5085                .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5086            let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5087                .map_err(|_| invalid("global dictionary offsets are short"))?;
5088            let ends = u32::try_from(end)
5089                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5090            let starts = u32::try_from(start)
5091                .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5092            (starts, ends)
5093        };
5094        if start > end {
5095            return Err(invalid("global dictionary value ends before it starts"));
5096        }
5097        Ok((start, end))
5098    }
5099
5100    /// The block of the sorted order that holds `rank`, and where in it that rank sits.
5101    ///
5102    /// The block is read from the file and checked against the hash the index carries for it the
5103    /// first time anything asks, and kept after that, the same way a payload block is. A search
5104    /// makes about as many probes as the order has bits, so the whole search reads a handful of
5105    /// these and never the rest.
5106    fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5107        let slot = self
5108            .rank_blocks
5109            .get(rank / TEXT_RANK_BLOCK)
5110            .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5111        let block = slot
5112            .get_or_init(|| {
5113                let mut bytes = Vec::new();
5114                self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5115                Ok(bytes)
5116            })
5117            .as_ref()
5118            .map_err(Clone::clone)?;
5119        Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5120    }
5121
5122    /// Reads block `which` of the sorted order into `bytes`, checked against the hash the index
5123    /// carries for it.
5124    fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5125        let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5126        let end = self.rank_ends[which];
5127        bytes.clear();
5128        bytes.resize((end - start) as usize, 0);
5129        read_at(&self.file, self.rank_at + start, bytes)?;
5130        let expected = self
5131            .rank_hashes
5132            .get(which)
5133            .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5134        if checksum(bytes) != *expected {
5135            return Err(invalid("global dictionary rank checksum differs"));
5136        }
5137        Ok(())
5138    }
5139
5140    /// The first eight bytes of the value at `rank`, as the integer a comparison reads.
5141    fn head_at(&self, rank: usize) -> Result<u64> {
5142        let (block, within) = self.rank_parts(rank)?;
5143        let (base, width, packed) = rank_heads(block)?;
5144        let above = bitpack::tail_at(packed, width, within)
5145            .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5146        Ok(base.wrapping_add(above))
5147    }
5148
5149    /// The packed codes of one rank block, which follow the heads on the next byte boundary.
5150    fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5151        let (_, width, packed) = rank_heads(block)?;
5152        packed
5153            .get(bitpack::tail_len(count, width)..)
5154            .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5155    }
5156
5157    /// How many entries the block holding `rank` has, which is a full block except at the end.
5158    fn rank_block_len(&self, rank: usize) -> usize {
5159        let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5160        TEXT_RANK_BLOCK.min(self.ranks - first)
5161    }
5162}
5163
5164/// The base, the width and the packed bytes of one rank block's heads.
5165fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5166    let header = block
5167        .get(..RANK_BLOCK_HEADER)
5168        .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5169    let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5170    let width = header[8] as usize;
5171    if width > 64 {
5172        return Err(invalid("global dictionary rank block packs heads past a word"));
5173    }
5174    Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5175}
5176
5177/// Bits one offset of a dictionary takes, which is what its widest payload block spans.
5178///
5179/// One width for the whole column rather than one a block. A block is 1,024 values of the same
5180/// column, so the blocks of a column are within a factor of two of each other on every ClickBench
5181/// string column, and a width a block would save a fraction of a bit and cost a byte a block plus
5182/// the arithmetic that finds where a block starts.
5183fn offset_width(ends: &[u32]) -> usize {
5184    // The ends are already relative to the block the value is in, so the last end of a block is that
5185    // block's total and the largest end anywhere is the widest block. There is no subtraction left
5186    // to do and no need to walk the blocks to find where one starts.
5187    let span = ends.iter().copied().max().unwrap_or(0);
5188    (u32::BITS - span.leading_zeros()) as usize
5189}
5190
5191/// How many bytes `values` offsets take at `bits`, which is what the reader has to know before it
5192/// has read any of them.
5193fn offset_bytes(values: usize, bits: usize) -> usize {
5194    let full = values / TEXT_OFFSET_RUN;
5195    let rest = values % TEXT_OFFSET_RUN;
5196    full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5197}
5198
5199/// The end of every value within its payload block, packed a run at a time.
5200/// A run never straddles a block, because [`TEXT_OFFSET_RUN`] divides [`TEXT_PAYLOAD_VALUES`], which
5201/// is what lets this be a walk of the ends rather than arithmetic against a per block base.
5202fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5203    let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5204    for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5205        run.clear();
5206        run.extend(chunk.iter().map(|&end| u64::from(end)));
5207        bitpack::pack_tail(&run, bits, out)
5208            .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5209    }
5210    Ok(())
5211}
5212
5213/// How many bits a code of a dictionary of `values` entries takes.
5214fn code_width(values: usize) -> usize {
5215    match u64::try_from(values).unwrap_or(u64::MAX) {
5216        0 | 1 => 0,
5217        last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5218    }
5219}
5220
5221impl TextSource for NativeText {
5222    fn len(&self) -> usize {
5223        self.values
5224    }
5225
5226    fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5227        let Some(grams) = &self.grams else { return Ok(true) };
5228        if literal.len() < 4 || first >= self.values {
5229            return Ok(true);
5230        }
5231        let verdict = grams.verdicts(&self.file, literal)?;
5232        Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5233    }
5234
5235    fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5236        if index >= self.values {
5237            return Ok(None);
5238        }
5239        let (start, end) = self.span_within(index)?;
5240        if start == end {
5241            return Ok(Some(&[]));
5242        }
5243        // A block holds a fixed number of values rather than a fixed number of bytes, so the value
5244        // is in one block and the offsets already say where in it.
5245        let block = index / TEXT_PAYLOAD_VALUES;
5246        let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5247        Ok(bytes.get(start as usize..end as usize))
5248    }
5249
5250    fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5251        if index >= self.values {
5252            return Ok(None);
5253        }
5254        let (start, end) = self.span_within(index)?;
5255        Ok(Some((end - start) as usize))
5256    }
5257
5258    /// Every length out of the unpacked ends in one loop, which is the point of having them.
5259    ///
5260    /// The whole run of positions counts towards [`Self::ends_worth_unpacking`] at once, because a
5261    /// caller asking for a vector of lengths has said how many it wants, and a vector of them is
5262    /// usually enough on its own. Until the table is worth building this is the row at a time read,
5263    /// the same as the default.
5264    fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5265        self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5266        into.reserve(indices.len());
5267        let Some(ends) = self.value_ends() else {
5268            for &index in indices {
5269                into.push(
5270                    self.bytes_len_at(index as usize)?
5271                        .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5272                );
5273            }
5274            return Ok(());
5275        };
5276        if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5277            lens.extend_at(indices, into);
5278            return Ok(());
5279        }
5280        for &index in indices {
5281            let index = index as usize;
5282            // Past the end is no value and so no length, which is what a row at a time read says.
5283            let Some(&end) = ends.get(index) else {
5284                into.push(0);
5285                continue;
5286            };
5287            let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5288            if start > end {
5289                return Err(invalid("global dictionary value ends before it starts"));
5290            }
5291            into.push(i64::from(end - start));
5292        }
5293        Ok(())
5294    }
5295
5296    /// Every length in characters out of the counts kept a block at a time, which is what keeps a
5297    /// scan of `length` from holding the column decoded. See [`NativeText::char_lens`].
5298    fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5299        into.reserve(indices.len());
5300        for &index in indices {
5301            let index = index as usize;
5302            // Past the end is no value and so no length, which is what a row at a time read says.
5303            if index >= self.values {
5304                into.push(0);
5305                continue;
5306            }
5307            let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5308            let len = lens
5309                .get(index % TEXT_PAYLOAD_VALUES)
5310                .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5311            into.push(i64::from(*len));
5312        }
5313        Ok(())
5314    }
5315
5316    /// The rest of the block holding `first`, decoded into a buffer that may die with the call.
5317    ///
5318    /// A block is the unit this format decodes, so a walk that wants every value is going to decode
5319    /// every block whatever it does. The question is whether it keeps them, and both answers are
5320    /// wrong on their own. [`Self::payload_block`] keeps every block it is asked for, so a reader
5321    /// that walked the whole dictionary through `bytes_at` ended up holding the whole dictionary
5322    /// decoded, 4.2 GB on ClickBench `URL`. Keeping none of them makes the next statement asking
5323    /// the same question decode all of it again, which on the same column at a million rows is a
5324    /// `LIKE` going from 2.7 ms to 16.2 ms.
5325    ///
5326    /// So a sweep keeps what it decodes for the second time while the column is under
5327    /// [`TEXT_KEEP_BUDGET`] and drops it after that. A block already in hand is used where it is there and costs nothing either way.
5328    fn sweep(
5329        &self,
5330        first: usize,
5331        limit: usize,
5332        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5333    ) -> Result<usize> {
5334        let limit = limit.min(self.values);
5335        if first >= limit {
5336            return Ok(first);
5337        }
5338        let block = first / TEXT_PAYLOAD_VALUES;
5339        let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5340        let mut decoded = Vec::new();
5341        let bytes = self.loaned_block(block, &mut decoded, false)?;
5342        let ends = self.ends_within(first, last)?;
5343        if ends.len() != last - first {
5344            return Err(invalid("global dictionary offsets are short"));
5345        }
5346        let mut start = u64::from(self.start_within(first)?);
5347        // row at a time: the caller is handed one value after another, and what it does with one is
5348        // its own business, so there is no shape here for anything but a walk.
5349        for (index, &end) in (first..last).zip(&ends) {
5350            let value = usize::try_from(start)
5351                .ok()
5352                .zip(usize::try_from(end).ok())
5353                .and_then(|(from, to)| bytes.get(from..to))
5354                .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5355            body(index, value)?;
5356            start = end;
5357        }
5358        Ok(last)
5359    }
5360
5361    /// The values at `indices` a block at a time, each block read once for the call.
5362    ///
5363    /// The positions are put in code order first, because the codes of a vector are in row order
5364    /// and land all over the dictionary, and read in that order each block a vector touches would
5365    /// be looked up once for every row in it. Whether a block is kept is
5366    /// [`NativeText::loaned_block`]'s decision, which keeps at most the budget of this column
5367    /// until the reads have shown they come back to the same blocks too often for dropping them to
5368    /// be cheap.
5369    fn visit_at(
5370        &self,
5371        indices: &[u32],
5372        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5373    ) -> Result<()> {
5374        let mut order = (0..indices.len()).collect::<Vec<_>>();
5375        order.sort_unstable_by_key(|&at| indices[at]);
5376        let block_of = |at: usize| {
5377            let index = indices[at] as usize;
5378            (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5379        };
5380        let mut decoded = Vec::new();
5381        let mut run = 0;
5382        while run < order.len() {
5383            let Some(block) = block_of(order[run]) else {
5384                // Past the end is no value, and every position after this one is past it too.
5385                for &at in &order[run..] {
5386                    body(at, &[])?;
5387                }
5388                break;
5389            };
5390            let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5391            let bytes = self.loaned_block(block, &mut decoded, true)?;
5392            for &at in &order[run..upto] {
5393                let (start, end) = self.span_within(indices[at] as usize)?;
5394                let value = bytes
5395                    .get(start as usize..end as usize)
5396                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5397                body(at, value)?;
5398            }
5399            run = upto;
5400        }
5401        Ok(())
5402    }
5403
5404    /// Each block the indices land in, decoded once and dropped, or read where it is already kept.
5405    ///
5406    /// Never kept, unlike [`Self::sweep`] under its budget, because a scattered read is a one off:
5407    /// a synopsis turned into values is turned once and remembered by the reader as values, a few
5408    /// kilobytes, where the blocks it went through are megabytes nobody asks for again.
5409    fn visit(
5410        &self,
5411        indices: &[usize],
5412        body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5413    ) -> Result<()> {
5414        let mut at = 0;
5415        while at < indices.len() {
5416            let block = indices[at] / TEXT_PAYLOAD_VALUES;
5417            let upto =
5418                at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5419            let wanted = &indices[at..upto];
5420            if wanted.iter().any(|&index| index >= self.values) {
5421                return Err(invalid("a visited value is past the global dictionary"));
5422            }
5423            let decoded;
5424            let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5425                Some(Ok(kept)) => kept,
5426                _ => {
5427                    decoded = self.decode_block(block)?;
5428                    &decoded
5429                }
5430            };
5431            for (offset, &index) in wanted.iter().enumerate() {
5432                let (start, end) = self.span_within(index)?;
5433                let value = bytes
5434                    .get(start as usize..end as usize)
5435                    .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5436                body(at + offset, value)?;
5437            }
5438            at = upto;
5439        }
5440        Ok(())
5441    }
5442
5443    fn ranks(&self) -> Option<usize> {
5444        (self.ranks > 0).then_some(self.ranks)
5445    }
5446
5447    /// The boundary for `wanted`, out of [`Self::searched`] where it is there and put there where
5448    /// it is not.
5449    ///
5450    /// The lock is held over the search rather than dropped and taken again, so that two threads
5451    /// asking for the same value at the same time do the work once between them. That is the shape
5452    /// the scan actually arrives in: sixteen instances of a top N, all reading the same column, all
5453    /// improving their bound over the same early chunks.
5454    fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5455        let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5456        if let Some(&answer) = memo.get(wanted) {
5457            return Ok(answer);
5458        }
5459        let answer = search_below(self, ranks, wanted)?;
5460        if memo.len() >= TEXT_SEARCH_MEMO {
5461            memo.clear();
5462        }
5463        memo.insert(wanted.to_vec(), answer);
5464        Ok(answer)
5465    }
5466
5467    fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5468        // The head settles the probe unless the two values start with the same eight bytes, and
5469        // only then is a value read. On a column of URLs that is the difference between a search
5470        // that touches one block of the payload and a search that touches nineteen of them.
5471        let settled = self.head_at(rank)?.cmp(&head(wanted));
5472        if settled != Ordering::Equal {
5473            return Ok(settled);
5474        }
5475        let code = self.code_at_rank(rank)?;
5476        let bytes = self
5477            .bytes_at(code as usize)?
5478            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5479        Ok(bytes.cmp(wanted))
5480    }
5481
5482    fn code_at_rank(&self, rank: usize) -> Result<u32> {
5483        let (block, within) = self.rank_parts(rank)?;
5484        let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5485        let code = bitpack::tail_at(codes, self.code_bits, within)
5486            .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5487        let code = u32::try_from(code)
5488            .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5489        if code as usize >= self.len() {
5490            return Err(invalid("global dictionary order names a code it does not have"));
5491        }
5492        Ok(code)
5493    }
5494
5495    fn code_ranks(&self) -> Option<&[u32]> {
5496        // The order is a permutation of the positions, so inverting it needs every position to be
5497        // named exactly once. Anything else and the slice would have holes, and a caller indexing
5498        // it by a code would read a rank that belongs to nothing.
5499        if self.ranks == 0 || self.ranks != self.len() {
5500            return None;
5501        }
5502        self.code_ranks
5503            .get_or_init(|| {
5504                let mut ranks = vec![u32::MAX; self.ranks];
5505                // A block at a time rather than a rank at a time, because reading it per rank pays
5506                // for the bounds check, the division and the lock on every one of them.
5507                //
5508                // A block nothing has read yet is read into one buffer that is reused, rather than
5509                // through `rank_parts`, which would keep every block of the order once this is
5510                // done with it. The inverse is all anything wants after this, and on the `Referer`
5511                // column of the ClickBench file the blocks are tens of megabytes held for nothing.
5512                let mut scratch = Vec::new();
5513                let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5514                for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5515                    let which = first / TEXT_RANK_BLOCK;
5516                    let block = match self.rank_blocks.get(which)?.get() {
5517                        Some(kept) => kept.as_ref().ok()?.as_slice(),
5518                        None => {
5519                            self.read_rank_block(which, &mut scratch).ok()?;
5520                            scratch.as_slice()
5521                        }
5522                    };
5523                    let count = self.rank_block_len(first);
5524                    let packed = self.rank_codes(block, count).ok()?;
5525                    let codes = codes.get_mut(..count)?;
5526                    bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5527                    for (within, &code) in codes.iter().enumerate() {
5528                        let code = usize::try_from(code).ok()?;
5529                        *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5530                    }
5531                }
5532                if ranks.contains(&u32::MAX) {
5533                    return None;
5534                }
5535                Some(ranks)
5536            })
5537            .as_deref()
5538    }
5539
5540    fn footprint(&self) -> usize {
5541        self.offsets.capacity()
5542            + self
5543                .value_ends
5544                .get()
5545                .and_then(Option::as_ref)
5546                .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5547            + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5548            + self
5549                .code_ranks
5550                .get()
5551                .and_then(Option::as_ref)
5552                .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5553            + self.rank_hashes.capacity() * size_of::<u64>()
5554            + self.rank_ends.capacity() * size_of::<u64>()
5555            + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5556            + self
5557                .rank_blocks
5558                .iter()
5559                .filter_map(OnceLock::get)
5560                .filter_map(|result| result.as_ref().ok())
5561                .map(Vec::capacity)
5562                .sum::<usize>()
5563            + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5564            + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5565            + self
5566                .char_lens
5567                .iter()
5568                .filter_map(OnceLock::get)
5569                .map(|lens| lens.len() * size_of::<u32>())
5570                .sum::<usize>()
5571            + self.hashes.capacity() * size_of::<u64>()
5572            + self.starts.capacity() * size_of::<u64>()
5573            + self.lengths.capacity() * size_of::<u64>()
5574            + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5575            + self
5576                .blocks
5577                .iter()
5578                .filter_map(OnceLock::get)
5579                .filter_map(|result| result.as_ref().ok())
5580                .map(Vec::capacity)
5581                .sum::<usize>()
5582    }
5583}
5584
5585/// Every table wide part number in order, with the stripe it belongs to.
5586fn places(table: &Table) -> Result<Vec<Place>> {
5587    let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5588    for (at, stripe) in table.stripes.iter().enumerate() {
5589        let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5590        for (part, &rows) in stripe.parts.iter().enumerate() {
5591            places.push(Place {
5592                stripe: index,
5593                part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5594                rows,
5595            });
5596        }
5597    }
5598    Ok(places)
5599}
5600
5601/// Reads one column's section of a stripe's index page.
5602///
5603/// The section carries its own checksum, so a reader that wants one column out of a hundred and
5604/// five preads a few hundred bytes and still knows that what it got is what was written.
5605fn read_index<F: Positional + ?Sized>(
5606    file: &F,
5607    stripe: &Stripe,
5608    column: usize,
5609) -> Result<Vec<PartSpan>> {
5610    let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5611    read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5612}
5613
5614fn read_index_span<F: Positional + ?Sized>(
5615    file: &F,
5616    index: Span,
5617    page: Span,
5618    parts: usize,
5619    column: usize,
5620) -> Result<Vec<PartSpan>> {
5621    let section = index_section(parts)?;
5622    let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5623    let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5624    if end > index.length as usize {
5625        return Err(invalid("index page is shorter than its columns"));
5626    }
5627    let mut bytes = vec![0; section];
5628    let offset =
5629        index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5630    read_at(file, offset, &mut bytes)?;
5631    let entries = section - size_of::<u64>();
5632    let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5633    if checksum(&bytes[..entries]) != stored {
5634        // With where it was read from, because the two ways this fires look identical from the
5635        // message alone: a file somebody damaged, and a file we wrote to the wrong offset.
5636        return Err(invalid(&format!(
5637            "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5638             wanted {stored:016x} and got {:016x}",
5639            checksum(&bytes[..entries]),
5640        )));
5641    }
5642    let mut spans = Vec::with_capacity(parts);
5643    let mut start = 0_usize;
5644    for part in 0..parts {
5645        let at = part * INDEX_ENTRY;
5646        let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5647        let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5648        spans.push(PartSpan { start, length, hash });
5649        start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5650    }
5651    if start != page.length as usize {
5652        return Err(invalid("column page length differs from its index"));
5653    }
5654    Ok(spans)
5655}
5656
5657/// One part's bytes out of a whole column page.
5658fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5659    let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5660    page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5661}
5662
5663/// Puts one stripe of one column in the cache, and hands back the page for the pool to count when
5664/// it is a page the column did not already hold.
5665///
5666/// The index goes in its own slot and stays. Only the page is under the budget, and the pool is
5667/// what enforces it, once the caller has let go of the column's lock.
5668fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5669    if let Some(slot) = cached.index.get_mut(held.stripe) {
5670        if slot.is_none() {
5671            *slot = Some(Arc::clone(&held.index));
5672        }
5673    }
5674    let page = held.page.clone()?;
5675    let slot = cached.pages.get_mut(held.stripe)?;
5676    if slot.is_some() {
5677        return None;
5678    }
5679    let bytes = page.bytes.len();
5680    // Set, so that the page a worker has just paid to read is not the one the pass it pays for
5681    // lets go of before the worker has read a part out of it.
5682    let used = Arc::new(AtomicBool::new(true));
5683    *slot = Some(Resident { page, used: Arc::clone(&used) });
5684    Some((bytes, used))
5685}
5686
5687/// Every table a native file holds, without the directory of any of them.
5688///
5689/// This is what opening a database reads. It is the small level of the directory, so the cost is
5690/// proportional to how many tables there are rather than to how much data they hold, and a session
5691/// that touches two tables of eight decodes two table directories.
5692///
5693/// The file handle is shared with every reader this hands out. Eight tables in one file is one open
5694/// file descriptor, not eight, which is the other thing one file buys over a file per table.
5695#[derive(Debug, Clone)]
5696pub struct Catalog {
5697    file: Arc<File>,
5698    size: u64,
5699    entries: Arc<Vec<Entry>>,
5700    /// The views the file holds, whole, since a view has no second level to read later.
5701    views: Arc<Vec<ViewEntry>>,
5702    opening: Opening,
5703    /// Where every reader this hands out counts its pages.
5704    pool: PagePool,
5705}
5706
5707/// Signed integer sums and non-null counts for selected columns, plus total table rows.
5708#[derive(Debug, Clone, PartialEq, Eq)]
5709pub struct CertifiedSums {
5710    pub columns: Vec<(i128, u64)>,
5711    pub rows: u64,
5712}
5713
5714/// Exact ends of an integer or date column, including a certified all-null column.
5715#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5716pub enum IntegerExtremes {
5717    Null,
5718    Values { low: i128, high: i128 },
5719}
5720
5721/// A complete numeric value-to-row-count synopsis; `None` represents SQL NULL.
5722pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5723
5724impl Catalog {
5725    /// Reads the highest valid catalog slot and nothing under it.
5726    ///
5727    /// The readers it hands out keep pages in a pool of their own with no budget, so each column
5728    /// holds its floor of four stripes and no more. A database opens with [`Catalog::open_in`].
5729    ///
5730    /// # Errors
5731    ///
5732    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
5733    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5734        Self::open_in(path, &PagePool::default())
5735    }
5736
5737    /// The same, with every reader it hands out keeping its pages in `pool`.
5738    ///
5739    /// # Errors
5740    ///
5741    /// If the file has no valid committed catalog or a catalog pointer is out of bounds.
5742    pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
5743        let (file, size, _, bytes, opening) = slot_bytes(path)?;
5744        let (entries, views) = decode_catalog(&bytes, size)?;
5745        Ok(Self {
5746            file: Arc::new(file),
5747            size,
5748            entries: Arc::new(entries),
5749            views: Arc::new(views),
5750            opening,
5751            pool: pool.clone(),
5752        })
5753    }
5754
5755    /// The tables in the file, in the order they were written.
5756    pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
5757        self.entries.iter().map(|entry| entry.name.as_str())
5758    }
5759
5760    /// The same tables with how many rows each of them holds.
5761    ///
5762    /// The names alone answer which tables the file has, which is what a checkpoint needs to know.
5763    /// A load asks a second question: whether a table already in the file is really in the way of
5764    /// the one it wants to write. A table with no rows is not, because it has no pages the next
5765    /// generation would have to carry, so the count has to come out of the catalog beside the name.
5766    pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
5767        self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
5768    }
5769
5770    /// The views in the file, in the order they were written.
5771    ///
5772    /// Whole, unlike [`Catalog::names`], which hands back names and makes the caller ask for a table
5773    /// by one. A view is a few strings and a column list and it was all read at open, so there is
5774    /// nothing left to go and fetch and no reason to make the caller ask twice.
5775    pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
5776        self.views.iter()
5777    }
5778
5779    /// How many tables the file holds.
5780    #[must_use]
5781    pub fn len(&self) -> usize {
5782        self.entries.len()
5783    }
5784
5785    /// Whether the file holds no table at all, which is what [`Writer::empty`] writes and what a
5786    /// database somebody dropped the last table out of comes back as.
5787    #[must_use]
5788    pub fn is_empty(&self) -> bool {
5789        self.entries.is_empty()
5790    }
5791
5792    /// Opens one table by name, decoding its directory now.
5793    ///
5794    /// # Errors
5795    ///
5796    /// If there is no table by that name, or its directory is torn or points outside the file.
5797    pub fn table(&self, name: &str) -> Result<Reader> {
5798        let entry = self
5799            .entries
5800            .iter()
5801            .find(|entry| entry.name == name)
5802            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5803        // Checked and then decoded a window at a time, so that the directory's own bytes are never
5804        // all in memory beside the table they decode into. It is read twice, and the second read
5805        // comes out of the page cache the first one filled.
5806        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5807        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5808            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5809        }
5810        let mut opening = self.opening;
5811        opening.reads += 1;
5812        opening.bytes += u64::from(entry.directory.length);
5813        Reader::build(
5814            Arc::clone(&self.file),
5815            self.size,
5816            read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
5817            u64::from(entry.directory.length),
5818            opening,
5819            self.pool.clone(),
5820        )
5821    }
5822
5823    /// Counts one signed integer column from its encoded parts without building metadata for
5824    /// unrelated columns. The counts are computed from row encodings when this is called.
5825    /// Nullable and non-cascade parts use the ordinary decoder for that part.
5826    ///
5827    /// # Errors
5828    ///
5829    /// If the directory, selected page index, checksum, or encoded integer is invalid.
5830    pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
5831        let mut counts = BTreeMap::<i64, u64>::new();
5832        let Some(()) = self.integer_fold(name, column, |value, count| {
5833            let held = counts.entry(value).or_default();
5834            *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
5835            Ok(())
5836        })?
5837        else {
5838            return Ok(None);
5839        };
5840        Ok(Some(counts.into_iter().collect()))
5841    }
5842
5843    /// Visits a signed integer column's row values without building per-part or table-wide count
5844    /// maps. The caller combines the emitted counts for its query at runtime.
5845    ///
5846    /// # Errors
5847    ///
5848    /// If the selected file data is invalid or the callback rejects a count.
5849    pub fn integer_fold(
5850        &self,
5851        name: &str,
5852        column: usize,
5853        mut emit: impl FnMut(i64, u64) -> Result<()>,
5854    ) -> Result<Option<()>> {
5855        let entry = self
5856            .entries
5857            .iter()
5858            .find(|entry| entry.name == name)
5859            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5860        let field =
5861            entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
5862        if !signed_integer(&field.ty) {
5863            return Ok(None);
5864        }
5865        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5866        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5867            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5868        }
5869        quick_integer_fold(
5870            &self.file,
5871            Cursor::over(&self.file, offset, length),
5872            entry,
5873            self.size,
5874            column,
5875            &mut emit,
5876        )?;
5877        Ok(Some(()))
5878    }
5879
5880    /// Counts non-null, nonzero values from generic column frequencies when complete. For an
5881    /// older file or a partial catalog synopsis, reads the validated native directory without
5882    /// building a reader for every stripe. Returns `None` when the bounded frequency synopsis
5883    /// cannot prove the count, so callers can use the ordinary query path.
5884    pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5885        let entry = self
5886            .entries
5887            .iter()
5888            .find(|entry| entry.name == name)
5889            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5890        let Some(field) = entry.fields.get(column) else {
5891            return Err(invalid("frequency column index out of range"));
5892        };
5893        if !matches!(
5894            field.ty,
5895            LogicalType::TinyInt
5896                | LogicalType::SmallInt
5897                | LogicalType::Integer
5898                | LogicalType::BigInt
5899                | LogicalType::UTinyInt
5900                | LogicalType::USmallInt
5901                | LogicalType::UInteger
5902                | LogicalType::UBigInt
5903        ) {
5904            return Ok(None);
5905        }
5906        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5907        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5908            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5909        }
5910        if let Some(Some(frequencies)) = entry.frequencies.get(column) {
5911            return frequencies
5912                .iter()
5913                .filter(|(value, _)| value.is_some_and(|value| value != 0))
5914                .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
5915                .map(Some)
5916                .ok_or_else(|| invalid("numeric frequency count overflow"));
5917        }
5918        quick_nonzero(
5919            Cursor::over(&self.file, offset, length),
5920            &entry.name,
5921            &entry.fields,
5922            entry.rows,
5923            column,
5924        )
5925    }
5926
5927    /// Exact signed-integer sums and non-null counts from the small catalog. The table directory
5928    /// checksum is still checked once before any certificate can answer a query.
5929    pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
5930        let entry = self
5931            .entries
5932            .iter()
5933            .find(|entry| entry.name == name)
5934            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5935        let mut sums = Vec::with_capacity(columns.len());
5936        for &column in columns {
5937            let Some(field) = entry.fields.get(column) else {
5938                return Err(invalid("aggregate column index out of range"));
5939            };
5940            if !signed_integer(&field.ty) {
5941                return Ok(None);
5942            }
5943            let Some(sum) = entry.aggregates[column] else {
5944                return Ok(None);
5945            };
5946            sums.push(sum);
5947        }
5948        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5949        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5950            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5951        }
5952        Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5953    }
5954
5955    /// Exact non-null distinct count from the small catalog, after checking the table directory.
5956    pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5957        let entry = self
5958            .entries
5959            .iter()
5960            .find(|entry| entry.name == name)
5961            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5962        let Some(count) = entry.distincts.get(column).copied() else {
5963            return Err(invalid("distinct column index out of range"));
5964        };
5965        let Some(count) = count else { return Ok(None) };
5966        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5967        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5968            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5969        }
5970        Ok(Some(count))
5971    }
5972
5973    /// Exact integer or date ends from the small catalog after checking the table directory.
5974    pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5975        let entry = self
5976            .entries
5977            .iter()
5978            .find(|entry| entry.name == name)
5979            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5980        let Some(extremes) = entry.extremes.get(column).copied() else {
5981            return Err(invalid("extremes column index out of range"));
5982        };
5983        let Some(extremes) = extremes else { return Ok(None) };
5984        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5985        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5986            return Err(invalid(&format!("the directory of table {name} does not checksum")));
5987        }
5988        Ok(Some(match extremes {
5989            None => IntegerExtremes::Null,
5990            Some((low, high)) => IntegerExtremes::Values { low, high },
5991        }))
5992    }
5993
5994    /// Complete numeric frequencies from the small catalog, after checking the table directory.
5995    pub fn exact_numeric_frequencies(
5996        &self,
5997        name: &str,
5998        column: usize,
5999    ) -> Result<Option<NumericFrequencies>> {
6000        let entry = self
6001            .entries
6002            .iter()
6003            .find(|entry| entry.name == name)
6004            .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6005        let Some(frequencies) = entry.frequencies.get(column).cloned() else {
6006            return Err(invalid("numeric frequency column index out of range"));
6007        };
6008        let Some(frequencies) = frequencies else { return Ok(None) };
6009        let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6010        if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6011            return Err(invalid(&format!("the directory of table {name} does not checksum")));
6012        }
6013        Ok(Some(frequencies))
6014    }
6015
6016    /// The schema copied into the small file catalog, available without opening the table directory.
6017    pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
6018        self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
6019    }
6020}
6021
6022/// Where the slot naming `generation` goes, which is the one the generation before it did not use.
6023///
6024/// Generation 1 takes the slot at 16, so a file written once is byte for byte the file this wrote
6025/// before there was a second generation to write.
6026fn slot_offset(generation: u64) -> u64 {
6027    16 + (generation - 1) % 2 * SLOT_BYTES as u64
6028}
6029
6030/// The header and the bytes the highest valid slot points at.
6031///
6032/// Both levels of the directory are reached this way, so the magic check, the version check and the
6033/// choice between the two slots live here rather than being written out twice.
6034fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
6035    let file = File::open(path).map_err(io)?;
6036    let size = file.metadata().map_err(io)?.len();
6037    let (slot, bytes, opening) = committed_slot(&file, size)?;
6038    Ok((file, size, slot, bytes, opening))
6039}
6040
6041/// The committed slot of a file that is `size` bytes long, and the catalog it points at.
6042///
6043/// The half of [`slot_bytes`] that does not care how the file was opened. A reader comes here with
6044/// the `std::fs::File` it goes on to share between its threads, and a writer with the `rudb_io`
6045/// file it is about to append to.
6046fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
6047    if size < HEADER {
6048        return Err(invalid("file is shorter than its header"));
6049    }
6050    let mut header = [0; HEADER as usize];
6051    read_at(file, 0, &mut header)?;
6052    let mut opening = Opening { reads: 1, bytes: HEADER };
6053    let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
6054    // The two halves are worth telling apart. A wrong magic is a file that was never ours and
6055    // the answer is to look at the path. A wrong version is our own file from another build,
6056    // and the number this build wants is the only thing that tells the reader whether to
6057    // rebuild the file or to go back to the binary that wrote it.
6058    if &header[..8] != MAGIC {
6059        return Err(invalid("the header does not begin with a rudb native magic"));
6060    }
6061    if !READABLE.contains(&version) {
6062        return Err(invalid(&format!(
6063            "the file is format {version} and this build reads format {FORMAT}, so it has to \
6064                 be written again"
6065        )));
6066    }
6067    let mut selected = None;
6068    for start in [16, 16 + SLOT_BYTES] {
6069        let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6070        if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6071            continue;
6072        }
6073        let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6074        if slot.offset < HEADER || end > size {
6075            continue;
6076        }
6077        let mut bytes = vec![0; slot.length as usize];
6078        read_at(file, slot.offset, &mut bytes)?;
6079        opening.reads += 1;
6080        opening.bytes += u64::from(slot.length);
6081        if checksum(&bytes) == slot.hash
6082            && selected
6083                .as_ref()
6084                .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6085        {
6086            selected = Some((slot, bytes));
6087        }
6088    }
6089    let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6090    Ok((slot, bytes, opening))
6091}
6092
6093impl Reader {
6094    /// Opens a file that holds exactly one table.
6095    ///
6096    /// # Errors
6097    ///
6098    /// If the file has no valid committed directory, a directory pointer is out of bounds, or the
6099    /// file holds more than one table, which is a file that has to be opened by name.
6100    pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6101        let catalog = Catalog::open(path)?;
6102        let mut names = catalog.names();
6103        let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6104        if names.next().is_some() {
6105            return Err(invalid(
6106                "the file holds more than one table, so it has to be opened by name",
6107            ));
6108        }
6109        catalog.table(&name)
6110    }
6111
6112    /// Builds a reader over one decoded table directory.
6113    fn build(
6114        file: Arc<File>,
6115        size: u64,
6116        table: Table,
6117        directory: u64,
6118        opening: Opening,
6119        pool: PagePool,
6120    ) -> Result<Self> {
6121        let places = places(&table)?;
6122        let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6123        let table_fields = table.fields.len();
6124        let stripes = table.stripes.len();
6125        let columns = (0..table.fields.len())
6126            .map(|_| {
6127                Mutex::new(Cached {
6128                    pages: (0..stripes).map(|_| None).collect(),
6129                    index: (0..stripes).map(|_| None).collect(),
6130                    seen: vec![false; stripes],
6131                    ..Cached::default()
6132                })
6133            })
6134            .collect::<Vec<_>>();
6135        let cache = Shelf {
6136            columns,
6137            held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6138            kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6139        };
6140        let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
6141            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6142            .collect();
6143        let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
6144            .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6145            .collect();
6146        Ok(Self {
6147            file,
6148            table: Arc::new(table),
6149            dictionaries: Arc::new(dictionaries),
6150            loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6151            frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6152            frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6153            opened: Arc::new(AtomicUsize::new(0)),
6154            sieves: Arc::new(sieves),
6155            part_ranges: Arc::new(part_ranges),
6156            places: Arc::new(places),
6157            cache: Arc::new(cache),
6158            pool,
6159            pages: Arc::new(AtomicUsize::new(0)),
6160            indexes: Arc::new(AtomicUsize::new(0)),
6161            size,
6162            directory,
6163            opening,
6164        })
6165    }
6166
6167    /// What this reader has read so far, and what opening it cost.
6168    ///
6169    /// Public because the claim of `spec/stats/04-in-memory.md` section 4.2 is about this number
6170    /// and a claim nobody can check is a comment. A caller that wants to know whether opening a
6171    /// file touched the data asks here, and gets an answer that does not depend on what the page
6172    /// cache happened to hold.
6173    #[must_use]
6174    pub fn reads(&self) -> Reads {
6175        Reads {
6176            opening: self.opening,
6177            pages: self.pages.load(Atomic::Relaxed),
6178            indexes: self.indexes.load(Atomic::Relaxed),
6179            dictionaries: self.opened.load(Atomic::Relaxed),
6180        }
6181    }
6182
6183    /// Where the file's bytes went, from the directory alone.
6184    ///
6185    /// No page is read, so this costs the same on a 45 GB table as on an empty one. See [`Layout`]
6186    /// for what is charged where and for why the three things that are not columns stay separate.
6187    #[must_use]
6188    pub fn layout(&self) -> Layout {
6189        let table = &self.table;
6190        let stripes = table.stripes.as_slice();
6191        let columns = table
6192            .fields
6193            .iter()
6194            .enumerate()
6195            .map(|(at, field)| ColumnLayout {
6196                name: field.name.clone(),
6197                kind: field.ty.to_string(),
6198                pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6199                memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6200                sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6201                part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6202                dictionary: dictionary_bytes(table, at),
6203            })
6204            .collect();
6205        Layout {
6206            file: self.size,
6207            rows: table.rows,
6208            stripes: stripes.len(),
6209            parts: self.places.len(),
6210            columns,
6211            indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6212            directory: self.directory,
6213            header: HEADER,
6214        }
6215    }
6216
6217    /// What every part of one column is stored as, which is what `pragma_storage_info` reports.
6218    ///
6219    /// Unlike [`Self::layout`] this reads the data, because the encoder's choice is in the page and
6220    /// nowhere else. The directory says how many bytes a column took and says nothing about what
6221    /// shape they are in, and the shape is the question worth asking: the same rows in a different
6222    /// order come back bit packed on one file and plain on another, and that is the difference a
6223    /// clustered load makes to a scan.
6224    ///
6225    /// One read per stripe rather than one per part. A part is a few kilobytes out of a page that
6226    /// is a quarter of a megabyte, so asking part by part would read the same page sixty four
6227    /// times. Nothing is put in the page cache, because a caller asking what a file looks like is
6228    /// not about to scan it and evicting the pages a real query wants would be a poor trade.
6229    ///
6230    /// # Errors
6231    ///
6232    /// If the column is outside the schema, or a page, index section or checksum is invalid.
6233    pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6234        let field = self
6235            .table
6236            .fields
6237            .get(column)
6238            .ok_or_else(|| invalid("stored column index out of range"))?;
6239        let mut stored = Vec::with_capacity(self.places.len());
6240        let mut row = 0;
6241        for (at, stripe) in self.table.stripes.iter().enumerate() {
6242            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6243            let index = read_index(&self.file, stripe, column)?;
6244            let mut bytes = vec![0; page.length as usize];
6245            read_at(&self.file, page.offset, &mut bytes)?;
6246            let ranges = self.stripe_part_ranges(at, column);
6247            for (part, &rows) in stripe.parts.iter().enumerate() {
6248                let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6249                let held = part_bytes(&bytes, span)?;
6250                let range = ranges.and_then(|held| held.get(part));
6251                stored.push(StoredPart {
6252                    stripe: at,
6253                    part,
6254                    row,
6255                    rows: rows as usize,
6256                    encoding: page_encoding(&field.ty, rows as usize, held),
6257                    bytes: span.length as u64,
6258                    page: page.offset,
6259                    offset: span.start as u64,
6260                    low: range
6261                        .and_then(|range| range.low.clone())
6262                        .and_then(|bound| bound.into_value(&field.ty)),
6263                    high: range
6264                        .and_then(|range| range.high.clone())
6265                        .and_then(|bound| bound.into_value(&field.ty)),
6266                    nulls: range.map(|range| range.nulls),
6267                });
6268                row += rows as usize;
6269            }
6270        }
6271        Ok(stored)
6272    }
6273
6274    /// How many parts the table has, which is how many chunks a scan of it reads.
6275    #[must_use]
6276    pub fn parts(&self) -> usize {
6277        self.places.len()
6278    }
6279
6280    /// The parts of each stripe, in table wide part numbers.
6281    ///
6282    /// A scan that wants one worker to own the page it reads hands work out in these runs. The
6283    /// stripes are contiguous in part numbering and all but the last hold sixty four parts, but a
6284    /// stripe can be flushed early when rows arrive out of order, so the runs are read off the
6285    /// directory rather than worked out from a constant.
6286    #[must_use]
6287    pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6288        let mut runs = Vec::with_capacity(self.table.stripes.len());
6289        let mut start = 0;
6290        for stripe in &self.table.stripes {
6291            let end = start + stripe.parts.len();
6292            runs.push(start..end);
6293            start = end;
6294        }
6295        runs
6296    }
6297
6298    /// How many rows one stripe holds, in the numbering [`Self::stripe_parts`] hands back.
6299    ///
6300    /// Off the directory, which is already in memory, rather than by the caller asking for each
6301    /// part in turn through the catalog. Nothing past the end holds any rows.
6302    #[must_use]
6303    pub fn stripe_rows(&self, stripe: usize) -> usize {
6304        self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6305    }
6306
6307    /// Asks the page cache to keep `stripes` stripes of every column instead of the default.
6308    ///
6309    /// This only ever raises the number. A scan that gives each worker a whole stripe has one page
6310    /// per column per worker open at once, and a cache smaller than that is worse than no cache at
6311    /// all: every worker's page is evicted by the others before it has finished its stripe, so it
6312    /// reads a quarter of a megabyte for every part it takes out of it.
6313    pub fn keep_stripes(&self, stripes: usize) {
6314        self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6315    }
6316
6317    /// Rows in one part, or zero when the part number is past the table.
6318    #[must_use]
6319    pub fn part_rows(&self, at: usize) -> usize {
6320        self.places.get(at).map_or(0, |place| place.rows as usize)
6321    }
6322
6323    /// The committed table directory.
6324    #[must_use]
6325    pub fn table(&self) -> &Table {
6326        &self.table
6327    }
6328
6329    /// Exact leading frequencies when the stored synopsis proves a count-descending prefix.
6330    ///
6331    /// The returned list can be longer than `top`. Keeping the stored tail lets a later TopN apply
6332    /// additional ordering keys without losing a value tied with the requested boundary.
6333    ///
6334    /// # Errors
6335    ///
6336    /// If the column is outside the schema or a stored value does not fit its declared type.
6337    pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6338        let field = self
6339            .table
6340            .fields
6341            .get(column)
6342            .ok_or_else(|| invalid("frequency column index out of range"))?;
6343        let Some(summary) = self.frequency_summary(column)? else {
6344            return Ok(None);
6345        };
6346        if top == 0 || summary.entries.len() < top {
6347            return Ok(None);
6348        }
6349        let boundary = summary.entries[top - 1].count;
6350        if boundary <= summary.omitted_max {
6351            return Ok(None);
6352        }
6353        self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
6354    }
6355
6356    /// Exact leading counts for a numeric key paired with a stable-dictionary string key.
6357    ///
6358    /// Legacy pair summaries are parsed for file compatibility but never used as query output.
6359    ///
6360    /// # Errors
6361    ///
6362    /// If either column is outside the schema.
6363    pub fn top_pair_frequencies(
6364        &self,
6365        first: usize,
6366        second: usize,
6367        _top: usize,
6368    ) -> Result<Option<PairFrequencyCounts>> {
6369        if first >= self.table.fields.len() || second >= self.table.fields.len() {
6370            return Err(invalid("pair frequency column index out of range"));
6371        }
6372        Ok(None)
6373    }
6374
6375    /// Every value of one column with the number of rows holding it, when the synopsis is complete.
6376    ///
6377    /// The heavy hitter pass keeps a bounded set of candidates and decrements them all when it runs
6378    /// out of room, so what it usually ends with is the leading values and a bound on everything it
6379    /// dropped. `omitted_max` of zero says that never happened: no candidate was ever decremented and
6380    /// the entries did not overflow the stored budget, so the list is every distinct value of the
6381    /// column with an exact count, and a null counts as a value of its own rather than being skipped.
6382    ///
6383    /// That makes a whole class of question answerable without reading a row. How many rows hold a
6384    /// value, how many do not, and what a `GROUP BY` of that column with a count over it produces are
6385    /// all in here. It is only ever true of a column with few enough distinct values, which is the
6386    /// case worth having, because that is exactly the column a grouping or an equality filter would
6387    /// otherwise walk every row to answer.
6388    ///
6389    /// `None` when the column has no synopsis, or has one that dropped anything.
6390    ///
6391    /// # Errors
6392    ///
6393    /// If the column is outside the schema or a stored value does not fit its declared type.
6394    pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6395        let Some(prefix) = self.frequency_prefix(column)? else {
6396            return Ok(None);
6397        };
6398        Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6399    }
6400
6401    /// Every value the synopsis lists with the number of rows holding it, and a bound on the rest.
6402    ///
6403    /// The counts are exact whether or not the list is complete. The heavy hitter pass keeps a
6404    /// bounded candidate set and then recounts only the candidates that survived it, so a value that
6405    /// made it into the list carries the number of rows that really hold it rather than whatever the
6406    /// pass had left over. What the pass loses is values, not counts.
6407    ///
6408    /// `omitted_max` is how many rows the most common value left out can hold, and zero says nothing
6409    /// was left out at all, which is what [`exact_frequencies`] asks for. Above zero the list is the
6410    /// leading values of the column and everything else is somewhere between no rows and that bound.
6411    ///
6412    /// That prefix is worth reading on its own. A column with a value in half its rows and a long
6413    /// tail behind it has no complete synopsis and never will, and it is the column where dividing
6414    /// the rows by the distinct count is furthest from the truth.
6415    ///
6416    /// `None` when the column has no synopsis.
6417    ///
6418    /// # Errors
6419    ///
6420    /// If the column is outside the schema or a stored value does not fit its declared type.
6421    ///
6422    /// [`exact_frequencies`]: Self::exact_frequencies
6423    pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6424        let field = self
6425            .table
6426            .fields
6427            .get(column)
6428            .ok_or_else(|| invalid("frequency column index out of range"))?;
6429        let Some(summary) = self.frequency_summary(column)? else {
6430            return Ok(None);
6431        };
6432        let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6433        Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
6434    }
6435
6436    /// One column's synopsis, read back from the file when the directory left it there.
6437    fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6438        Ok(match self.table.frequencies.get(column) {
6439            None | Some(None) => None,
6440            Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6441            Some(Some(Frequencies::Stored { span, values })) => {
6442                let slot = self
6443                    .frequency_summaries
6444                    .get(column)
6445                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6446                if let Some(summary) = slot.get() {
6447                    return Ok(Some(Cow::Borrowed(summary.as_ref())));
6448                }
6449                let field = self
6450                    .table
6451                    .fields
6452                    .get(column)
6453                    .ok_or_else(|| invalid("frequency column index out of range"))?;
6454                let mut bytes = vec![0; span.length as usize];
6455                read_at(&self.file, span.offset, &mut bytes)?;
6456                let summary =
6457                    decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
6458                let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6459                let _ = slot.set(Arc::new(summary));
6460                Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6461            }
6462        })
6463    }
6464
6465    /// Turns stored frequency entries into values of the column's own type.
6466    ///
6467    /// Remembered per column, because the planner asks once for every estimate that touches the
6468    /// column and the executor asks again, and the answer is a few hundred values. The codes of a
6469    /// string column are read through [`Vector::try_values_visited`], which does not keep the blocks
6470    /// it decodes, so what a query answered out of the synopsis holds is those values and not the
6471    /// hundred or so dictionary blocks they are scattered over.
6472    fn decode_frequencies(
6473        &self,
6474        column: usize,
6475        ty: &LogicalType,
6476        entries: &[FrequencyEntry],
6477    ) -> Result<Vec<(Value, u64)>> {
6478        if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6479            return Ok(values.as_ref().clone());
6480        }
6481        let values = self.decode_frequencies_once(column, ty, entries)?;
6482        if let Some(slot) = self.frequency_values.get(column) {
6483            let _ = slot.set(Arc::new(values.clone()));
6484        }
6485        Ok(values)
6486    }
6487
6488    fn decode_frequencies_once(
6489        &self,
6490        column: usize,
6491        ty: &LogicalType,
6492        entries: &[FrequencyEntry],
6493    ) -> Result<Vec<(Value, u64)>> {
6494        let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6495        if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6496            return Err(invalid("frequency text count differs from its synopsis"));
6497        }
6498        let dictionary =
6499            if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6500        let mut codes = entries
6501            .iter()
6502            .filter_map(|entry| match entry.value {
6503                FrequencyValue::Code(code) => Some(code as usize),
6504                _ => None,
6505            })
6506            .collect::<Vec<_>>();
6507        codes.sort_unstable();
6508        codes.dedup();
6509        let texts = match &dictionary {
6510            Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6511            _ => Vec::new(),
6512        };
6513        let mut out = Vec::with_capacity(entries.len());
6514        for (entry_at, entry) in entries.iter().enumerate() {
6515            let value = match entry.value {
6516                FrequencyValue::Null => {
6517                    if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6518                        return Err(invalid("a null frequency entry has text"));
6519                    }
6520                    Value::Null
6521                }
6522                FrequencyValue::Integer(value) => match *ty {
6523                    LogicalType::TinyInt => Value::TinyInt(
6524                        i8::try_from(value)
6525                            .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6526                    ),
6527                    LogicalType::UTinyInt => Value::UTinyInt(
6528                        u8::try_from(value)
6529                            .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6530                    ),
6531                    LogicalType::USmallInt => Value::USmallInt(
6532                        u16::try_from(value)
6533                            .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6534                    ),
6535                    LogicalType::UInteger => Value::UInteger(
6536                        u32::try_from(value)
6537                            .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6538                    ),
6539                    LogicalType::UBigInt => Value::UBigInt(
6540                        u64::try_from(value)
6541                            .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6542                    ),
6543                    LogicalType::SmallInt => Value::SmallInt(
6544                        i16::try_from(value)
6545                            .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6546                    ),
6547                    LogicalType::Integer => Value::Integer(
6548                        i32::try_from(value)
6549                            .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6550                    ),
6551                    LogicalType::BigInt => Value::BigInt(
6552                        i64::try_from(value)
6553                            .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6554                    ),
6555                    LogicalType::Date => Value::Date(
6556                        i32::try_from(value)
6557                            .map_err(|_| invalid("frequency DATE is out of range"))?,
6558                    ),
6559                    LogicalType::Timestamp => Value::Timestamp(
6560                        i64::try_from(value)
6561                            .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6562                    ),
6563                    _ => return Err(invalid("integer frequency belongs to another type")),
6564                },
6565                FrequencyValue::Code(code) => {
6566                    if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6567                        if *ty == LogicalType::Blob {
6568                            Value::Blob(text.clone())
6569                        } else {
6570                            Value::Varchar(
6571                                String::from_utf8(text.clone())
6572                                    .map_err(|_| invalid("frequency text is not UTF-8"))?,
6573                            )
6574                        }
6575                    } else {
6576                        if dictionary.is_none() {
6577                            return Err(invalid("frequency code has no dictionary or stored text"));
6578                        }
6579                        let at = codes
6580                            .binary_search(&(code as usize))
6581                            .map_err(|_| invalid("frequency code was not among the codes read"))?;
6582                        texts[at].clone()
6583                    }
6584                }
6585            };
6586            out.push((value, entry.count));
6587        }
6588        Ok(out)
6589    }
6590
6591    /// Sparse rows belonging to the bounded numeric frequency candidate set.
6592    ///
6593    /// The list is omitted when collecting it would exceed the fixed storage budget. A composite
6594    /// aggregate may accept a result over these rows only when its requested boundary is strictly
6595    /// greater than `omitted_max`.
6596    ///
6597    /// # Errors
6598    ///
6599    /// If the column is outside the schema.
6600    pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6601        let field = self
6602            .table
6603            .fields
6604            .get(column)
6605            .ok_or_else(|| invalid("frequency column index out of range"))?;
6606        let Some(summary) = self.frequency_summary(column)? else {
6607            return Ok(None);
6608        };
6609        if summary.ordinals.is_empty() {
6610            return Ok(None);
6611        }
6612        let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6613            let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6614            (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
6615        } else {
6616            (Vec::new(), Vec::new())
6617        };
6618        Ok(Some(FrequencyOccurrences {
6619            omitted_max: summary.omitted_max,
6620            ordinals: summary.ordinals.clone(),
6621            anchors,
6622            anchor_indices,
6623        }))
6624    }
6625
6626    /// How many distinct values one column holds, counting a null as no value.
6627    ///
6628    /// A string column of this format is written against one dictionary that covers the whole table.
6629    /// A code is handed out the first time a value is seen and nothing ever removes one, so the
6630    /// number of codes is the number of distinct values exactly rather than an estimate. That makes
6631    /// `COUNT(DISTINCT column)` over a whole table a question the directory already knows the answer
6632    /// to, and the alternative is a hash table with a row per distinct value built from a pass over
6633    /// every row.
6634    ///
6635    /// A null in the column used to make this `None` and no longer does. A null row is written as
6636    /// the code for the empty string, so a nullable column's dictionary can hold an empty string
6637    /// that no row of it actually has, and the dictionary on its own does not say which case it is.
6638    /// The writer does know, because it counts the non-null rows that use each code on its way to
6639    /// the frequency summary, so it records how many codes any row holds and the directory carries
6640    /// that number. This reads it rather than the size of the dictionary, which also means the
6641    /// dictionary page is not opened to answer.
6642    ///
6643    /// An integer column has no dictionary, and its count comes from the set the writer keeps on its
6644    /// numeric frequency pass instead, which is exact up to a cap. `None` for a column past that cap
6645    /// and for every column that is neither, where a sketch would answer approximately and SQL asked
6646    /// for the exact number.
6647    ///
6648    /// # Errors
6649    ///
6650    /// If the column is outside the schema.
6651    pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
6652        self.table
6653            .distincts
6654            .get(column)
6655            .copied()
6656            .ok_or_else(|| invalid("distinct column index out of range"))
6657    }
6658
6659    /// How many rows of one column are null, added up over the stripes.
6660    ///
6661    /// Every stripe records this exactly when it is written, because a null count is not a bound
6662    /// that is allowed to be wide the way a minimum and a maximum are: a filter that reads one too
6663    /// many is slow and a `COUNT` that reads one too many is wrong. Adding up a few hundred numbers
6664    /// already in memory is what makes `COUNT(column)` over a whole table free.
6665    ///
6666    /// # Errors
6667    ///
6668    /// If the column is outside the schema.
6669    pub fn null_count(&self, column: usize) -> Result<u64> {
6670        if column >= self.table.fields.len() {
6671            return Err(invalid("null count column index out of range"));
6672        }
6673        let mut nulls = 0_u64;
6674        for stripe in &self.table.stripes {
6675            let range = stripe
6676                .zone
6677                .column(column)
6678                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6679            nulls = nulls
6680                .checked_add(range.nulls as u64)
6681                .ok_or_else(|| invalid("null count overflow"))?;
6682        }
6683        Ok(nulls)
6684    }
6685
6686    /// The smallest and the largest value of one string column, from the order beside its values.
6687    ///
6688    /// The dictionary holds exactly the values the column holds, so the first and the last of them
6689    /// in sorted order are the column's minimum and maximum. Two reads of a rank block settle what
6690    /// otherwise walks a million rows.
6691    ///
6692    /// `None` when the column is not a string, when the file was written before version 9 and so has
6693    /// no order, when the column has no values at all, or when it has a null in it, which is the
6694    /// placeholder again: the empty string a null is written as would sort ahead of every real
6695    /// value and be reported as the minimum.
6696    ///
6697    /// # Errors
6698    ///
6699    /// If the column is outside the schema, or a rank names a code the dictionary does not have.
6700    pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
6701        if self.null_count(column)? > 0 || self.demoted(column) {
6702            return Ok(None);
6703        }
6704        let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
6705        let Some(ranks) = dictionary.ranks() else { return Ok(None) };
6706        if ranks == 0 {
6707            return Ok(None);
6708        }
6709        let low = text_at_rank(&dictionary, 0)?;
6710        let high = text_at_rank(&dictionary, ranks - 1)?;
6711        Ok(Some((low, high)))
6712    }
6713
6714    /// The smallest and the largest value of one column, when every stripe wrote exact ends.
6715    ///
6716    /// A stripe's ends are allowed to be wider than the truth, because a bound that rules out a
6717    /// chunk that could not match is still correct when it rules out nothing. That is what makes
6718    /// them cheap to write for a bit packed or a dictionary column, and it is also what stops them
6719    /// answering a `MIN`. So each stripe says which of the two it wrote, and this answers only when
6720    /// all of them walked their rows.
6721    ///
6722    /// `None` for a column with no ends, for an empty table, and for a column any stripe of which
6723    /// guessed. Nulls need no special case, because the ends skip them the same way `MIN` does.
6724    ///
6725    /// One case is given up on that did not have to be. A stripe merges the ends of its sixty four
6726    /// parts, and a part with no ends at all erases the merged ones, because a part whose rows are
6727    /// not covered by the stripe's ends is a stripe that would skip rows it should keep. A part of
6728    /// nothing but nulls has no rows to cover and so did not need to erase anything, but the merge
6729    /// cannot tell that part from a part whose layout it could not read. So a column with a chunk
6730    /// of nothing but nulls in the middle of it goes and reads the rows. That is slow and right,
6731    /// and the fix is a row count per part rather than anything here.
6732    ///
6733    /// # Errors
6734    ///
6735    /// If the column is outside the schema.
6736    pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
6737        if column >= self.table.fields.len() {
6738            return Err(invalid("extremes column index out of range"));
6739        }
6740        let mut low: Option<Bound> = None;
6741        let mut high: Option<Bound> = None;
6742        for stripe in &self.table.stripes {
6743            let range = stripe
6744                .zone
6745                .column(column)
6746                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6747            if !range.exact {
6748                return Ok(None);
6749            }
6750            // A stripe of nothing but nulls has no ends and says nothing about the column's, which
6751            // is why this skips it rather than giving up on the whole column. A stripe that has
6752            // rows and still has no end is a layout whose values this cannot see, and skipping that
6753            // one would answer with an end taken from the other stripes, so it gives up instead.
6754            let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
6755                if stripe.rows > range.nulls {
6756                    return Ok(None);
6757                }
6758                continue;
6759            };
6760            low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
6761            high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
6762        }
6763        Ok(low.zip(high))
6764    }
6765
6766    /// The sum of one integer column and how many rows went into it, when every stripe wrote one.
6767    ///
6768    /// The count beside the sum is the non-null rows, because that is what a `SUM` adds up and what
6769    /// an `AVG` divides by, and a caller that had to work it out from the row count and the null
6770    /// count would be doing the same walk twice.
6771    ///
6772    /// `None` for anything that is not an integer column, for a file written by something that did
6773    /// not record it, and when adding the stripes together would overflow.
6774    ///
6775    /// # Errors
6776    ///
6777    /// If the column is outside the schema.
6778    pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
6779        if column >= self.table.fields.len() {
6780            return Err(invalid("sum column index out of range"));
6781        }
6782        let mut total = 0_i128;
6783        let mut rows = 0_u64;
6784        for stripe in &self.table.stripes {
6785            let range = stripe
6786                .zone
6787                .column(column)
6788                .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6789            let Some(part) = range.sum else { return Ok(None) };
6790            let Some(sum) = total.checked_add(part) else { return Ok(None) };
6791            total = sum;
6792            rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
6793        }
6794        Ok(Some((total, rows)))
6795    }
6796
6797    /// Legacy derived host groups are parsed for file compatibility but never used as query output.
6798    pub fn host_groups(
6799        &self,
6800        column: usize,
6801        _minimum_count: u64,
6802    ) -> Result<Option<Vec<host::HostEntry>>> {
6803        if column >= self.table.fields.len() {
6804            return Err(invalid("host group column index out of range"));
6805        }
6806        Ok(None)
6807    }
6808
6809    /// Whether the column's dictionary stopped taking values partway through the load, and so
6810    /// decodes the stripes written before that and says nothing about the column as a whole. See
6811    /// `DEMOTED`.
6812    #[must_use]
6813    pub fn demoted(&self, column: usize) -> bool {
6814        self.table.demoted.get(column).copied().unwrap_or(false)
6815    }
6816
6817    /// The global dictionary of a column, opened once however many workers ask for it at once.
6818    ///
6819    /// The unlocked look is first because it is the answer every time after the first and it costs a
6820    /// load. Everybody who misses it queues on [`Self::loading`] and looks again on the way in, so
6821    /// the one who arrived first does the reading and the rest take what it left. Waiting is the
6822    /// cheaper thing to do: the work behind the lock is a page read, a checksum and the decode of a
6823    /// dictionary that can hold half a million entries, and the alternative is every worker of the
6824    /// scan doing all of it and all but one dropping the result on the floor.
6825    fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
6826        let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
6827        if let Some(dictionary) = self.dictionaries[column].get() {
6828            return Ok(Some(Arc::clone(dictionary)));
6829        }
6830        let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
6831        if let Some(dictionary) = self.dictionaries[column].get() {
6832            return Ok(Some(Arc::clone(dictionary)));
6833        }
6834        self.opened.fetch_add(1, Atomic::Relaxed);
6835        let dictionary = Arc::new(open_global_dictionary(
6836            Arc::clone(&self.file),
6837            page,
6838            &self.table.fields[column].ty,
6839            TEXT_KEEP_BUDGET,
6840        )?);
6841        let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
6842        Ok(Some(dictionary))
6843    }
6844
6845    /// Reads one section's extent table and checks it against the entry that names it.
6846    ///
6847    /// # Errors
6848    ///
6849    /// If the entry points outside the file, the table does not checksum, or it does not decode as
6850    /// a run of extents in element order.
6851    pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
6852        if of.extent_bytes == 0 {
6853            return Ok(Vec::new());
6854        }
6855        let mut bytes = vec![0; of.extent_bytes as usize];
6856        read_at(&self.file, of.extent_page, &mut bytes)?;
6857        if checksum(&bytes) != of.hash {
6858            return Err(invalid("a section's extent table does not checksum"));
6859        }
6860        let extents = section::decode_extents(&bytes)?;
6861        if extents.len() != of.extents as usize {
6862            return Err(invalid("a section's extent table is not the length the entry says"));
6863        }
6864        Ok(extents)
6865    }
6866
6867    /// Reads and verifies one extent of a section.
6868    ///
6869    /// This is what section 3.2's second rule is for. A reduction that needs one extent of a two
6870    /// gigabyte forward link reads and checksums that extent and nothing else, which is the whole
6871    /// difference between a structure that works at SF100 and issue #745.
6872    ///
6873    /// # Errors
6874    ///
6875    /// If the extent points outside the file, or its bytes do not checksum.
6876    pub fn extent(&self, of: &section::Extent) -> Result<Vec<u8>> {
6877        let mut bytes = Vec::new();
6878        self.extent_into(of, &mut bytes)?;
6879        Ok(bytes)
6880    }
6881
6882    /// Read a verified extent into a caller-owned buffer so repeated extents can reuse its pages.
6883    fn extent_into(&self, of: &section::Extent, bytes: &mut Vec<u8>) -> Result<()> {
6884        let end = of
6885            .offset
6886            .checked_add(u64::from(of.length))
6887            .ok_or_else(|| invalid("an extent overflows the file"))?;
6888        if of.offset < HEADER || end > self.size {
6889            return Err(invalid("an extent is outside the file"));
6890        }
6891        bytes.resize(of.length as usize, 0);
6892        read_at(&self.file, of.offset, bytes)?;
6893        if checksum(bytes) != of.hash {
6894            return Err(invalid("an extent does not checksum"));
6895        }
6896        Ok(())
6897    }
6898
6899    /// Reads a whole section's payload, every extent of it, in order.
6900    ///
6901    /// For a structure that is resident anyway, which a key map is. Anything large enough that the
6902    /// split matters should be walking [`Reader::extents`] and taking the one it needs.
6903    ///
6904    /// # Errors
6905    ///
6906    /// If the extent table or any extent fails its check.
6907    pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6908        let extents = self.extents(of)?;
6909        let mut bytes =
6910            Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6911        for one in &extents {
6912            if one.first != bytes.len() as u64 {
6913                return Err(invalid("a section's extents do not join up"));
6914            }
6915            bytes.extend_from_slice(&self.extent(one)?);
6916        }
6917        // The same exception `write_section` makes: a budget record has no bytes, so its
6918        // `header_bytes` is a size rather than a header and there is nothing for it to run past.
6919        if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6920            return Err(invalid("a section's header is longer than its payload"));
6921        }
6922        Ok(bytes)
6923    }
6924
6925    /// Reads only the named columns from one part.
6926    ///
6927    /// The whole stripe page each column lives in is read and kept, because a scan asks for the
6928    /// parts of a stripe one after another and this is what turns sixty four reads into one.
6929    ///
6930    /// # Errors
6931    ///
6932    /// If a part, column, page, or checksum is invalid.
6933    pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6934        self.read_impl(part, columns, true, None)
6935    }
6936
6937    /// Reads named columns from one part without keeping the stripe page it came out of.
6938    ///
6939    /// This is for sparse row fetches after a selective TopN or filter, which reach a few parts of
6940    /// a stripe rather than all of them. A caller that will read most of a stripe should use
6941    /// [`Self::read`] instead, because this reads and discards the page index every time.
6942    ///
6943    /// # Errors
6944    ///
6945    /// If a part, column, page, or checksum is invalid.
6946    pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6947        self.read_impl(part, columns, false, None)
6948    }
6949
6950    /// Counts one signed integer part from its encoded row values when it uses an all-valid
6951    /// cascade. Sparse and run-length cascades are folded without expanding their rows. Other
6952    /// page forms return `None` so the caller can use the ordinary reader.
6953    ///
6954    /// # Errors
6955    ///
6956    /// If a part, column, page checksum, or encoded integer is invalid.
6957    pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6958        let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6959        let field =
6960            self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
6961        if !matches!(
6962            field.ty,
6963            LogicalType::TinyInt
6964                | LogicalType::SmallInt
6965                | LogicalType::Integer
6966                | LogicalType::BigInt
6967        ) {
6968            return Ok(None);
6969        }
6970        let (rows, counts) = match self.with_part(place, column, |bytes| {
6971            if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
6972                return Ok(None);
6973            }
6974            integer::tally(&bytes[2..]).map(Some)
6975        })? {
6976            Some(tallied) => tallied,
6977            None => return Ok(None),
6978        };
6979        if rows != place.rows as usize {
6980            return Err(invalid("encoded integer part holds the wrong number of rows"));
6981        }
6982        for &(value, _) in &counts {
6983            let fits = match field.ty {
6984                LogicalType::TinyInt => i8::try_from(value).is_ok(),
6985                LogicalType::SmallInt => i16::try_from(value).is_ok(),
6986                LogicalType::Integer => i32::try_from(value).is_ok(),
6987                LogicalType::BigInt => true,
6988                _ => false,
6989            };
6990            if !fits {
6991                return Err(invalid("encoded integer value is outside its column type"));
6992            }
6993        }
6994        Ok(Some(counts))
6995    }
6996
6997    /// The rows of one text part that hold `sequence`'s pieces in order, or with `negated` the rows
6998    /// that do not, answered on the compressed page without decompressing it. Nulls are in neither.
6999    /// `None` for a part that is not compressed text, which the caller reads the usual way.
7000    ///
7001    /// For a scan whose filter is the only thing that reads the column, which then never has the
7002    /// strings at all. In TPC-H q13 that is `o_comment NOT LIKE '%special%requests%'`, and
7003    /// decompressing the comments and searching them was most of the orders scan.
7004    ///
7005    /// # Errors
7006    ///
7007    /// If a part, column, page, or checksum is invalid.
7008    pub fn rows_holding(
7009        &self,
7010        part: usize,
7011        column: usize,
7012        sequence: &Sequence,
7013        negated: bool,
7014    ) -> Result<Option<Vec<u32>>> {
7015        let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7016        let field =
7017            self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7018        if field.ty != LogicalType::Varchar {
7019            return Ok(None);
7020        }
7021        let rows = place.rows as usize;
7022        self.with_part(place, column, |bytes| {
7023            if bytes.first() != Some(&6) {
7024                return Ok(None);
7025            }
7026            let mut cur = Cursor::new(bytes);
7027            cur.u8()?;
7028            let mask = match cur.u8()? {
7029                0 => None,
7030                1 => return Ok(Some(Vec::new())),
7031                2 => {
7032                    let from = cur.at;
7033                    cur.take(rows.div_ceil(8))?;
7034                    Some(&bytes[from..cur.at])
7035                }
7036                _ => return Err(invalid("page validity tag differs")),
7037            };
7038            let Some(held) = string::holds_in(&bytes[cur.at..], sequence)? else {
7039                return Ok(None);
7040            };
7041            if held.len() != rows {
7042                return Err(invalid("compressed text page holds the wrong number of rows"));
7043            }
7044            let valid = |row: usize| mask.is_none_or(|mask| mask[row / 8] >> (row % 8) & 1 == 1);
7045            Ok(Some(
7046                (0..rows)
7047                    .filter(|&row| held[row] != negated && valid(row))
7048                    .map(|row| row as u32)
7049                    .collect(),
7050            ))
7051        })
7052    }
7053
7054    /// Runs `read` over the stored bytes of one column of one part, out of the stripe's page when
7055    /// it is held and read off the file on their own when it is not.
7056    fn with_part<T>(
7057        &self,
7058        place: Place,
7059        column: usize,
7060        read: impl FnOnce(&[u8]) -> Result<T>,
7061    ) -> Result<T> {
7062        let stripe_index = place.stripe as usize;
7063        let stripe = self
7064            .table
7065            .stripes
7066            .get(stripe_index)
7067            .ok_or_else(|| invalid("stripe index out of range"))?;
7068        let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7069        let held = self.held(stripe_index, stripe, column, true)?;
7070        let span = *held
7071            .index
7072            .get(place.part as usize)
7073            .ok_or_else(|| invalid("part index out of range"))?;
7074        match &held.page {
7075            Some(page) => read(page.part(place.part as usize, span)?),
7076            None => {
7077                let offset = page
7078                    .offset
7079                    .checked_add(span.start as u64)
7080                    .ok_or_else(|| invalid("part range overflow"))?;
7081                let mut bytes = vec![0; span.length];
7082                read_at(&self.file, offset, &mut bytes)?;
7083                verify_part(&bytes, span)?;
7084                read(&bytes)
7085            }
7086        }
7087    }
7088
7089    /// Reads named columns from one part, only at the rows `positions` names.
7090    ///
7091    /// For a scan that already knows which rows of the part it keeps, from the columns it read
7092    /// first. A compressed string page decompresses only those rows, and every other page is
7093    /// decoded whole and gathered, which is what reading it and narrowing it costs anyway. With
7094    /// `whole` the stripe's pages are kept the way [`Self::read`] keeps them, and without it they
7095    /// are not, the way [`Self::read_sparse`] does.
7096    ///
7097    /// # Errors
7098    ///
7099    /// If a part, column, page, or checksum is invalid, or the positions do not rise or run past
7100    /// the end of the part.
7101    pub fn read_rows(
7102        &self,
7103        part: usize,
7104        columns: &[usize],
7105        positions: &[u32],
7106        whole: bool,
7107    ) -> Result<Chunk> {
7108        self.read_impl(part, columns, whole, Some(positions))
7109    }
7110
7111    /// Whether an exact global-code membership index proves that the stripe holding a part cannot
7112    /// contain any of the sorted candidate codes.
7113    ///
7114    /// # Errors
7115    ///
7116    /// If the part, column, index page, checksum, or delta stream is invalid.
7117    pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
7118        // A demoted column's later stripes hold values the dictionary never coded, so no list of
7119        // codes can prove a stripe of it holds none of a value.
7120        if self.demoted(column) {
7121            return Ok(false);
7122        }
7123        if candidates.is_empty() {
7124            return Ok(true);
7125        }
7126        if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
7127            return Err(Error::internal("native code candidates are not sorted and unique"));
7128        }
7129        let stripe = self.stripe_of(part)?;
7130        let Some(page) = stripe.memberships.get(column) else {
7131            return Ok(false);
7132        };
7133        let mut bytes = vec![0; page.length as usize];
7134        read_at(&self.file, page.offset, &mut bytes)?;
7135        if checksum(&bytes) != page.hash {
7136            return Err(invalid("membership page checksum differs"));
7137        }
7138        let codes = decode_membership(&bytes)?;
7139        let mut left = 0;
7140        let mut right = 0;
7141        while left < codes.len() && right < candidates.len() {
7142            match codes[left].cmp(&candidates[right]) {
7143                Ordering::Less => left += 1,
7144                Ordering::Greater => right += 1,
7145                Ordering::Equal => return Ok(false),
7146            }
7147        }
7148        Ok(true)
7149    }
7150
7151    fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7152        let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7153        self.table
7154            .stripes
7155            .get(place.stripe as usize)
7156            .ok_or_else(|| invalid("stripe index out of range"))
7157    }
7158
7159    /// The page index of one column of one stripe, and its page when the caller wants all of it.
7160    ///
7161    /// A scan hands parts out in order, so every worker on a column crosses into a new stripe within
7162    /// a few parts of the others and they all want the same page at the same moment. This used to
7163    /// let all of them read it, which cost the scan as many copies of every page as it had workers.
7164    /// On the full ClickBench file a `MIN(EventDate), MAX(EventDate)` moved 3.2 GB off the disk to
7165    /// look at 400 MB of column.
7166    ///
7167    /// A worker that finds the page it wants already being read neither waits for it nor reads it
7168    /// again. It comes back with the index alone, which sends [`Reader::read_impl`] down the path
7169    /// that reads the one part it came for, a few kilobytes against a quarter of a megabyte, and it
7170    /// picks the page up from the cache on its next part. Waiting would be the other way to avoid
7171    /// the duplicate read and it is worse: the pages that matter are the wide string ones, they take
7172    /// milliseconds to copy even warm, and every other worker would be stopped for all of it.
7173    ///
7174    /// The file is never read under the lock.
7175    fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
7176        let cache =
7177            self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7178        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7179        let known = cached.index.get(at).and_then(Clone::clone);
7180        let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7181            slot.used.store(true, Atomic::Relaxed);
7182            Arc::clone(&slot.page)
7183        });
7184        if let Some(index) = known.clone() {
7185            if !whole || page.is_some() {
7186                return Ok(CachedColumn { stripe: at, index, page });
7187            }
7188        }
7189        if cached.loading.contains(&at) {
7190            drop(cached);
7191            // The index is almost always already here, because somebody read this stripe to get
7192            // into the loading list in the first place, so this branch usually costs no read at
7193            // all and the one part read in `read_impl` is all the losing worker pays for.
7194            if let Some(index) = known {
7195                return Ok(CachedColumn { stripe: at, index, page: None });
7196            }
7197            let held = self.page_of(stripe, column, at, false, None)?;
7198            let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7199            remember(&mut cached, &held);
7200            return Ok(held);
7201        }
7202        cached.loading.push(at);
7203        drop(cached);
7204
7205        let read = self.page_of(stripe, column, at, whole, known);
7206
7207        // The stripe leaves the loading list and its page enters the cache under one lock. Doing
7208        // them separately would leave a moment where another worker sees neither and reads the
7209        // page a second time, which is the whole thing this is here to stop.
7210        let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7211        if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7212            cached.loading.remove(position);
7213        }
7214        let held = read?;
7215        let taken = remember(&mut cached, &held);
7216        let first = taken.is_some()
7217            && cached.seen.get_mut(at).is_some_and(|seen| !std::mem::replace(seen, true));
7218        if first {
7219            let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7220            cached.passing.push_back(at);
7221            while cached.passing.len() > floor {
7222                let Some(old) = cached.passing.pop_front() else { break };
7223                if let Some(slot) = cached.pages.get_mut(old) {
7224                    *slot = None;
7225                }
7226            }
7227            return Ok(held);
7228        }
7229        drop(cached);
7230        if let Some((bytes, used)) = taken {
7231            self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7232            self.pool.admit(Held {
7233                shelf: Arc::downgrade(&self.cache),
7234                column,
7235                stripe: at,
7236                bytes,
7237                used,
7238            });
7239        }
7240        Ok(held)
7241    }
7242
7243    /// Reads one stripe's index for a column, and its page when the caller wants all of it.
7244    ///
7245    /// `known` is the index when the reader has already read it, which after the first worker
7246    /// through a stripe it always has, because [`remember`] keeps every index for the life of the
7247    /// reader. Without that a scan reads the index again on every part that misses the page cache.
7248    fn page_of(
7249        &self,
7250        stripe: &Stripe,
7251        column: usize,
7252        at: usize,
7253        whole: bool,
7254        known: Option<Arc<Vec<PartSpan>>>,
7255    ) -> Result<CachedColumn> {
7256        let index = match known {
7257            Some(index) => index,
7258            None => {
7259                self.indexes.fetch_add(1, Atomic::Relaxed);
7260                Arc::new(read_index(&self.file, stripe, column)?)
7261            }
7262        };
7263        let page = if whole {
7264            self.pages.fetch_add(1, Atomic::Relaxed);
7265            let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7266            let mut bytes = vec![0; span.length as usize];
7267            read_at(&self.file, span.offset, &mut bytes)?;
7268            let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7269            Some(Arc::new(HeldPage { bytes, checked }))
7270        } else {
7271            None
7272        };
7273        Ok(CachedColumn { stripe: at, index, page })
7274    }
7275
7276    fn read_impl(
7277        &self,
7278        at: usize,
7279        columns: &[usize],
7280        whole: bool,
7281        positions: Option<&[u32]>,
7282    ) -> Result<Chunk> {
7283        let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7284        let index = place.stripe as usize;
7285        let stripe =
7286            self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7287        let rows = place.rows as usize;
7288        let mut picked = Vec::with_capacity(columns.len());
7289        for &column in columns {
7290            let field = self
7291                .table
7292                .fields
7293                .get(column)
7294                .ok_or_else(|| invalid("column index out of range"))?;
7295            let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7296            let held = self.held(index, stripe, column, whole)?;
7297            let span = *held
7298                .index
7299                .get(place.part as usize)
7300                .ok_or_else(|| invalid("part index out of range"))?;
7301            let owned;
7302            let bytes = match &held.page {
7303                Some(held) => held.part(place.part as usize, span),
7304                None => {
7305                    let offset = page
7306                        .offset
7307                        .checked_add(span.start as u64)
7308                        .ok_or_else(|| invalid("part range overflow"))?;
7309                    let mut bytes = vec![0; span.length];
7310                    read_at(&self.file, offset, &mut bytes)?;
7311                    owned = bytes;
7312                    verify_part(&owned, span).map(|()| owned.as_slice())
7313                }
7314            }
7315            .map_err(|error| {
7316                invalid(&format!(
7317                    "{}, column {column} part {} of the page at {}",
7318                    error.message(),
7319                    place.part,
7320                    page.offset,
7321                ))
7322            })?;
7323            let dictionary = self.dictionary(column)?;
7324            // Held as a page, because a column that came out of a file is handed out more than
7325            // once. A group by clones its key columns out of the chunk so the keys outlive it, a
7326            // projection of a bare column name does the same, and a cut of a flat run copies unless
7327            // the run is a page. One `Arc` per column per part buys all of those, and it moves the
7328            // run into the `Arc` without touching a value.
7329            let mut vector = match positions {
7330                None => decode(&field.ty, rows, bytes, dictionary)?,
7331                Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7332            };
7333            // A demoted column's codes are not the column's codes, only the codes of the stripes
7334            // written before the demotion, so they are not handed out as if they were. See
7335            // [`DEMOTED`].
7336            if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7337                vector = vector.flatten()?;
7338            }
7339            picked.push(vector.into_pages());
7340        }
7341        Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7342    }
7343
7344    /// Whether persisted statistics prove that a part cannot match the predicates.
7345    ///
7346    /// Three of them, asked cheapest first.
7347    ///
7348    /// The stripe's bounds are in memory already, so they are free, and they are also the coarsest:
7349    /// every part of a stripe gets the same answer and a scan that skips one part that way skips all
7350    /// sixty four. Then the part's own bounds, which are a read of one page per column per stripe
7351    /// and are sixty four times finer. Then the sieves, which are per part and answer equality, the
7352    /// test bounds are worst at: a column of identifiers has every stripe and nearly every part
7353    /// covering the whole of its type, so bounds keep them all and the sieve keeps the ones that
7354    /// really hold the value.
7355    ///
7356    /// The middle one is what an ordered comparison on a column the rows are not sorted by needs. On
7357    /// ClickBench 24 the stripe bounds leave eight stripes of sixteen alive, which is half the file,
7358    /// and the part bounds leave thirty parts of nine hundred and seventy four.
7359    #[must_use]
7360    pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7361        let Some(place) = self.places.get(part).copied() else { return false };
7362        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7363        if stripe.zone.skips(probes) {
7364            return true;
7365        }
7366        probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7367    }
7368
7369    /// Whether `rule` rules out a part from what the stored range of one column says about it.
7370    ///
7371    /// The same two steps as [`Self::skips`] without the sieve, for a test no [`Probe`] can write.
7372    /// A probe is one comparison against one constant, and the keys a join's build side holds are a
7373    /// set, which rules a part out when none of them falls inside the part's two ends. Handing the
7374    /// range to the caller is what lets the set stay with the join that knows what it is.
7375    #[must_use]
7376    pub fn ruled_by(&self, part: usize, column: usize, rule: impl Fn(&Range) -> bool) -> bool {
7377        let Some(place) = self.places.get(part).copied() else { return false };
7378        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7379        if stripe.zone.column(column).is_some_and(&rule) {
7380            return true;
7381        }
7382        self.stripe_part_ranges(place.stripe as usize, column)
7383            .and_then(|ranges| ranges.get(place.part as usize))
7384            .is_some_and(rule)
7385    }
7386
7387    /// The half of [`Self::ruled_by`] that reads nothing, asked about a whole stripe.
7388    #[must_use]
7389    pub fn stripe_ruled_by(
7390        &self,
7391        stripe: usize,
7392        column: usize,
7393        rule: impl Fn(&Range) -> bool,
7394    ) -> bool {
7395        self.table.stripes.get(stripe).and_then(|held| held.zone.column(column)).is_some_and(rule)
7396    }
7397
7398    /// Whether the bounds of one part rule out one probe.
7399    ///
7400    /// The part's own two ends, which are narrower than the stripe's and cost a page read the first
7401    /// time this is asked about a column. A column with no page here answers `false`, which is the
7402    /// answer a caller got before there were any.
7403    fn outside(&self, place: Place, probe: &Probe) -> bool {
7404        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7405            Some(ranges) => ranges
7406                .get(place.part as usize)
7407                .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7408            None => false,
7409        }
7410    }
7411
7412    /// The per part ranges of one stripe of one column, read once and kept.
7413    ///
7414    /// `None` when the column has no page in that stripe and when the page is damaged, on the same
7415    /// reasoning as the sieves: this is an index over data that is still there, so a caller that
7416    /// cannot read one reads the rows and gets the right answer slowly.
7417    fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7418        let slot = self.part_ranges.get(column)?.get(stripe)?;
7419        if let Some(held) = slot.get() {
7420            return Some(held);
7421        }
7422        let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7423        let mut bytes = vec![0; page.length as usize];
7424        read_at(&self.file, page.offset, &mut bytes).ok()?;
7425        if checksum(&bytes) != page.hash {
7426            return None;
7427        }
7428        let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7429        let _ = slot.set(ranges);
7430        slot.get().map(|held| held.as_slice())
7431    }
7432
7433    /// Whether persisted statistics prove that every row of a part matches the predicates.
7434    ///
7435    /// Only the bounds. The sieves say nothing here, because a sieve that holds a value is a sieve
7436    /// that may be holding somebody else's hash, so it can rule a part out and can never wave one
7437    /// through.
7438    ///
7439    /// The stripe first and the part after it, the same two steps and in the same order as
7440    /// [`Self::skips`]. The stripe's bounds are in memory already and its null count covers sixty
7441    /// four parts rather than one, so a stripe that answers is an answer for nothing, and the part's
7442    /// own bounds are only read for the probes it could not settle. Both directions are safe: a
7443    /// stretch where everything passes contains no narrower stretch where something fails, and a
7444    /// stripe with no nulls has no nulls in any of its parts.
7445    ///
7446    /// A string end a part recorded is cut down to its first few bytes, so a part's stretch can be
7447    /// wider than its rows really are as well. That is the same safe direction for the same reason,
7448    /// and it is why this asks the two ends rather than anything `exact` says.
7449    #[must_use]
7450    pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7451        let Some(place) = self.places.get(part).copied() else { return false };
7452        let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7453        if stripe.zone.certain(probes) {
7454            return true;
7455        }
7456        probes
7457            .iter()
7458            .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7459    }
7460
7461    /// Whether one part's own two ends prove that every row of it passes `probe`.
7462    ///
7463    /// The mirror of [`Self::outside`], reading the same page. `false` for a part whose stripe wrote
7464    /// no range page, which is a stripe of one part, because there the stripe's own bounds are the
7465    /// part's and the caller has already asked them.
7466    fn inside(&self, place: Place, probe: &Probe) -> bool {
7467        match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7468            Some(ranges) => ranges
7469                .get(place.part as usize)
7470                .is_some_and(|range| range.certain(probe.op, &probe.value)),
7471            None => false,
7472        }
7473    }
7474
7475    /// Whether the bounds of one stripe prove that none of its parts can match the predicates.
7476    ///
7477    /// The cheap half of [`Self::skips`], asked about a whole stripe at once. The bounds live in the
7478    /// directory and are already in memory, so this answers without touching the file, and that is
7479    /// the reason it is worth having on its own: a caller that wants to know roughly where the work
7480    /// is before it starts any workers can ask this about sixteen stripes for nothing, where asking
7481    /// [`Self::skips`] about nine hundred parts would read and decode a sieve page per stripe first.
7482    ///
7483    /// It keeps stripes that [`Self::skips`] would rule out part by part, which is the right way for
7484    /// it to be wrong: the parts are still checked when they are read.
7485    #[must_use]
7486    pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7487        self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7488    }
7489
7490    /// Whether the sieve of one part rules out one probe.
7491    ///
7492    /// Only equality. An ordered comparison is what the bounds are for and a sieve says nothing
7493    /// about it, and a read that cannot answer keeps the part, which is the answer a caller with no
7494    /// sieve gets anyway.
7495    fn sifted(&self, place: Place, probe: &Probe) -> bool {
7496        if probe.op != Op::Equal {
7497            return false;
7498        }
7499        match self.stripe_sieves(place.stripe as usize, probe.column) {
7500            Some(sieves) => sieves
7501                .get(place.part as usize)
7502                .and_then(Option::as_ref)
7503                .is_some_and(|sieve| sieve.excludes(&probe.value)),
7504            None => false,
7505        }
7506    }
7507
7508    /// The sieves of one stripe of one column, read once and kept.
7509    ///
7510    /// `None` when the column has no sieves in that stripe, when the page is damaged, and when the
7511    /// bytes are not a page this version can read. A sieve is an index over data that is still there
7512    /// and a caller that cannot read one reads the rows, so this is the one place in the file where
7513    /// a bad checksum is a slow query rather than an error.
7514    fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7515        let slot = self.sieves.get(column)?.get(stripe)?;
7516        if let Some(held) = slot.get() {
7517            return Some(held);
7518        }
7519        let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7520        let mut bytes = vec![0; page.length as usize];
7521        read_at(&self.file, page.offset, &mut bytes).ok()?;
7522        if checksum(&bytes) != page.hash {
7523            return None;
7524        }
7525        let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7526        let _ = slot.set(sieves);
7527        slot.get().map(|held| held.as_slice())
7528    }
7529}
7530
7531/// The value sitting at one position of a dictionary's sorted order.
7532fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7533    let code = dictionary.code_at_rank(rank)? as usize;
7534    if dictionary.logical_type() == &LogicalType::Blob {
7535        let bytes = dictionary
7536            .try_bytes_at(code)?
7537            .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7538        return Ok(Value::Blob(bytes.to_vec()));
7539    }
7540    let text = dictionary
7541        .try_text_at(code)?
7542        .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7543    Ok(Value::Varchar(text.into()))
7544}
7545
7546/// Reads one span of a file at an offset, without moving a cursor anybody else can see.
7547///
7548/// Every reader of a table shares one [`File`] behind an [`Arc`], and a grouped aggregate reads its
7549/// pages from several threads at once, so this has to be positional. Seeking and then reading is
7550/// two calls with a gap in the middle, and in that gap another thread's seek lands and the read
7551/// comes back with somebody else's bytes.
7552///
7553/// The writer reads back through here too, out of the `rudb_io` file it writes through, which is
7554/// why this takes anything [`Positional`] rather than a [`File`].
7555fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7556    file.fill_at(offset, bytes)
7557}
7558
7559/// Something a span of bytes can be read out of by offset.
7560///
7561/// There are two of these. The reader holds a `std::fs::File`, because it shares it between its
7562/// threads behind an [`Arc`] and every read it makes is on the hot path of a scan. The writer holds
7563/// an `rudb_io::File`, because everything it does to the file has to be something the simulated
7564/// filesystem can stop and crash. The few helpers both of them use, [`read_index`] and the choice
7565/// of committed slot, are written once over this rather than once for each.
7566trait Positional {
7567    /// Fills `bytes` from `offset`, or fails if the file ends first.
7568    ///
7569    /// Both kinds can come back short, so both loop. A read of zero bytes before the span is filled
7570    /// means the file stops earlier than the directory said it does.
7571    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
7572}
7573
7574impl<T: Positional + ?Sized> Positional for &T {
7575    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7576        (**self).fill_at(offset, bytes)
7577    }
7578}
7579
7580impl<T: Positional + ?Sized> Positional for Arc<T> {
7581    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7582        (**self).fill_at(offset, bytes)
7583    }
7584}
7585
7586impl<T: Positional + ?Sized> Positional for Box<T> {
7587    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7588        (**self).fill_at(offset, bytes)
7589    }
7590}
7591
7592impl Positional for dyn rudb_io::File + '_ {
7593    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7594        while !bytes.is_empty() {
7595            let read = self.read_at(offset, bytes)?;
7596            if read == 0 {
7597                return Err(invalid("column page ends before its declared length"));
7598            }
7599            offset += read as u64;
7600            bytes = &mut bytes[read..];
7601        }
7602        Ok(())
7603    }
7604}
7605
7606impl Positional for File {
7607    #[cfg(unix)]
7608    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7609        use std::os::unix::fs::FileExt;
7610        while !bytes.is_empty() {
7611            let read = self.read_at(bytes, offset).map_err(io)?;
7612            if read == 0 {
7613                return Err(invalid("column page ends before its declared length"));
7614            }
7615            offset += read as u64;
7616            bytes = &mut bytes[read..];
7617        }
7618        Ok(())
7619    }
7620
7621    /// The same read, on the call Windows spells differently.
7622    ///
7623    /// `seek_read` is one `ReadFile` carrying the offset with it, so two of them cannot interleave
7624    /// the way a seek and a read can. It does leave the shared cursor somewhere afterwards, which is
7625    /// why nothing in this file may read that cursor.
7626    #[cfg(windows)]
7627    fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7628        use std::os::windows::fs::FileExt;
7629        while !bytes.is_empty() {
7630            let read = self.seek_read(bytes, offset).map_err(io)?;
7631            if read == 0 {
7632                return Err(invalid("column page ends before its declared length"));
7633            }
7634            offset += read as u64;
7635            bytes = &mut bytes[read..];
7636        }
7637        Ok(())
7638    }
7639
7640    /// Somewhere that is neither, where the cursor is all there is.
7641    ///
7642    /// This one does race, and there is no way to write it so it does not. Nothing we build for
7643    /// runs here, so it exists to keep the crate compiling rather than to be correct under threads.
7644    #[cfg(not(any(unix, windows)))]
7645    fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7646        use std::io::{Read, Seek, SeekFrom};
7647        let mut file = self.try_clone().map_err(io)?;
7648        file.seek(SeekFrom::Start(offset)).map_err(io)?;
7649        file.read_exact(bytes).map_err(io)
7650    }
7651}
7652
7653/// Overwrites one span of a file in place, which is how the tests damage a file on purpose.
7654///
7655/// The writer does not come through here. It writes through `rudb_io`, and this is a
7656/// `std::fs::File` opened by a test beside it.
7657#[cfg(test)]
7658fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
7659    use std::io::{Seek, SeekFrom, Write};
7660    let mut file = file;
7661    file.seek(SeekFrom::Start(offset)).map_err(io)?;
7662    file.write_all(bytes).map_err(io)
7663}
7664
7665/// What a column type is called in the directory.
7666///
7667/// A tag is a number in a file somebody else wrote, so a tag that has been used is used forever and
7668/// the only thing that may happen to this list is that it grows. 1 to 13 are the tags the format
7669/// had when it could store thirteen types, and 14 to 27 are the rest, in the order they were added
7670/// rather than in an order that means anything.
7671fn type_tag(ty: &LogicalType) -> Result<u8> {
7672    match ty {
7673        LogicalType::SmallInt => Ok(1),
7674        LogicalType::Integer => Ok(2),
7675        LogicalType::BigInt => Ok(3),
7676        LogicalType::Varchar => Ok(4),
7677        LogicalType::Date => Ok(5),
7678        LogicalType::Timestamp => Ok(6),
7679        LogicalType::Boolean => Ok(7),
7680        LogicalType::TinyInt => Ok(8),
7681        LogicalType::UTinyInt => Ok(9),
7682        LogicalType::USmallInt => Ok(10),
7683        LogicalType::UInteger => Ok(11),
7684        LogicalType::UBigInt => Ok(12),
7685        LogicalType::Decimal { .. } => Ok(13),
7686        LogicalType::Float => Ok(14),
7687        LogicalType::Double => Ok(15),
7688        LogicalType::HugeInt => Ok(16),
7689        LogicalType::UHugeInt => Ok(17),
7690        LogicalType::Time => Ok(18),
7691        LogicalType::TimeTz => Ok(19),
7692        LogicalType::TimestampTz => Ok(20),
7693        LogicalType::Interval => Ok(21),
7694        LogicalType::Uuid => Ok(22),
7695        LogicalType::Blob => Ok(23),
7696        LogicalType::Bit => Ok(24),
7697        LogicalType::TimestampS => Ok(25),
7698        LogicalType::TimestampMs => Ok(26),
7699        LogicalType::TimestampNs => Ok(27),
7700        _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
7701    }
7702}
7703
7704/// The tag of a column type, and the parameters of the ones that have any.
7705///
7706/// Only `DECIMAL` has parameters today. Width and scale go after the tag rather than into it
7707/// because they are what says how wide a value is on disk, and a reader that guessed would read the
7708/// wrong number of bytes per row rather than the wrong number of digits.
7709fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
7710    out.push(type_tag(ty)?);
7711    if let LogicalType::Decimal { width, scale } = ty {
7712        out.push(*width);
7713        out.push(*scale);
7714    }
7715    Ok(())
7716}
7717
7718/// The other half of [`put_type`], reading the parameters the tag says are there.
7719fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
7720    let tag = cur.u8()?;
7721    if tag == 13 {
7722        let width = cur.u8()?;
7723        let scale = cur.u8()?;
7724        return LogicalType::decimal(width, scale)
7725            .map_err(|_| invalid("decimal column width and scale are not a decimal"));
7726    }
7727    tag_type(tag)
7728}
7729
7730fn tag_type(tag: u8) -> Result<LogicalType> {
7731    match tag {
7732        1 => Ok(LogicalType::SmallInt),
7733        2 => Ok(LogicalType::Integer),
7734        3 => Ok(LogicalType::BigInt),
7735        4 => Ok(LogicalType::Varchar),
7736        5 => Ok(LogicalType::Date),
7737        6 => Ok(LogicalType::Timestamp),
7738        7 => Ok(LogicalType::Boolean),
7739        8 => Ok(LogicalType::TinyInt),
7740        9 => Ok(LogicalType::UTinyInt),
7741        10 => Ok(LogicalType::USmallInt),
7742        11 => Ok(LogicalType::UInteger),
7743        12 => Ok(LogicalType::UBigInt),
7744        14 => Ok(LogicalType::Float),
7745        15 => Ok(LogicalType::Double),
7746        16 => Ok(LogicalType::HugeInt),
7747        17 => Ok(LogicalType::UHugeInt),
7748        18 => Ok(LogicalType::Time),
7749        19 => Ok(LogicalType::TimeTz),
7750        20 => Ok(LogicalType::TimestampTz),
7751        21 => Ok(LogicalType::Interval),
7752        22 => Ok(LogicalType::Uuid),
7753        23 => Ok(LogicalType::Blob),
7754        24 => Ok(LogicalType::Bit),
7755        25 => Ok(LogicalType::TimestampS),
7756        26 => Ok(LogicalType::TimestampMs),
7757        27 => Ok(LogicalType::TimestampNs),
7758        _ => Err(invalid("column type tag is unknown")),
7759    }
7760}
7761
7762fn put_u16(out: &mut Vec<u8>, value: u16) {
7763    out.extend_from_slice(&value.to_le_bytes());
7764}
7765fn put_u32(out: &mut Vec<u8>, value: u32) {
7766    out.extend_from_slice(&value.to_le_bytes());
7767}
7768fn put_u64(out: &mut Vec<u8>, value: u64) {
7769    out.extend_from_slice(&value.to_le_bytes());
7770}
7771fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
7772    while value >= 0x80 {
7773        out.push((value as u8 & 0x7f) | 0x80);
7774        value >>= 7;
7775    }
7776    out.push(value as u8);
7777}
7778
7779fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
7780    match (left, right) {
7781        (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
7782        (FrequencyValue::Null, _) => Ordering::Less,
7783        (_, FrequencyValue::Null) => Ordering::Greater,
7784        (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
7785        (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
7786        (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
7787        (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
7788    }
7789}
7790
7791/// Leaves the [`FREQUENCY_ENTRIES`] commonest entries in order and says what the next one counted.
7792///
7793/// There is one entry a distinct value, so on `URL` this is handed two and a quarter million of
7794/// them and keeps five hundred and twelve. Sorting all of them to throw almost all of them away is
7795/// the whole of what counting a dictionary column used to cost, 2.13 seconds of it on `URL` at eight
7796/// million rows against 11.93 for compressing the same column's values.
7797///
7798/// Partitioning answers both questions instead. It puts the five hundred and thirteenth entry where
7799/// it belongs and everything commoner in front of it, which is the entries to keep and the count to
7800/// report as the largest one omitted, and then only the part that survives is sorted. The order that
7801/// comes out is the order the sort gave, because the tie break makes the comparison total: two
7802/// entries never hold the same value.
7803fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
7804    let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
7805        right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
7806    };
7807    let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
7808        let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
7809        let omitted_max = next.count;
7810        entries.truncate(FREQUENCY_ENTRIES);
7811        omitted_max
7812    } else {
7813        0
7814    };
7815    entries.sort_unstable_by(order);
7816    omitted_max
7817}
7818
7819fn code_frequency(
7820    dictionary: &GlobalDictionary,
7821    flat: &[u8],
7822    bases: &[u64],
7823) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
7824    let mut entries = dictionary
7825        .counts
7826        .iter()
7827        .enumerate()
7828        .filter(|(_, count)| **count != 0)
7829        .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
7830        .collect::<Vec<_>>();
7831    if dictionary.nulls != 0 {
7832        entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
7833    }
7834    let omitted_max = keep_most_frequent(&mut entries);
7835    let mut spans = Vec::with_capacity(entries.len());
7836    let mut text_bytes = 0_usize;
7837    for entry in &entries {
7838        let span = match entry.value {
7839            FrequencyValue::Code(code) => {
7840                let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
7841                let bytes = flat
7842                    .get(span.0..span.1)
7843                    .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
7844                text_bytes = text_bytes.saturating_add(bytes.len());
7845                Some(span)
7846            }
7847            FrequencyValue::Null | FrequencyValue::Integer(_) => None,
7848        };
7849        spans.push(span);
7850    }
7851    let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
7852        Vec::new()
7853    } else {
7854        spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
7855    };
7856    Ok((
7857        FrequencySummary {
7858            entries,
7859            omitted_max,
7860            ordinals: Vec::new(),
7861            ordinal_entries: Vec::new(),
7862        },
7863        texts,
7864    ))
7865}
7866
7867fn encode_directory(table: &Table) -> Result<Vec<u8>> {
7868    let mut out = DIRECTORY.to_vec();
7869    let name = table.name.as_bytes();
7870    put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7871    out.extend_from_slice(name);
7872    put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
7873    for field in &table.fields {
7874        let name = field.name.as_bytes();
7875        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
7876        out.extend_from_slice(name);
7877        put_type(&mut out, &field.ty)?;
7878        out.push(u8::from(field.not_null));
7879    }
7880    for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
7881        match dictionary {
7882            None => out.push(0),
7883            Some(page) => {
7884                out.push(dictionary_tag(&field.ty));
7885                put_u64(&mut out, page.offset);
7886                put_u32(&mut out, page.length);
7887                put_u64(&mut out, page.hash);
7888            }
7889        }
7890    }
7891    for distinct in &table.distincts {
7892        match distinct {
7893            None => out.push(0),
7894            Some(count) => {
7895                out.push(1);
7896                put_u64(&mut out, *count);
7897            }
7898        }
7899    }
7900    put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
7901    put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
7902    for stripe in &table.stripes {
7903        put_u32(
7904            &mut out,
7905            u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
7906        );
7907        for &rows in &stripe.parts {
7908            put_u32(&mut out, rows);
7909        }
7910        put_u64(&mut out, stripe.index.offset);
7911        put_u32(&mut out, stripe.index.length);
7912        for page in &stripe.pages {
7913            put_u64(&mut out, page.offset);
7914            put_u32(&mut out, page.length);
7915        }
7916        // A membership index says which of a dictionary's codes a part holds, so a column the writer
7917        // decided against giving a dictionary has nothing for it to be about and writes none. Every
7918        // file written before that decision existed has a dictionary on every varchar column, so
7919        // this reads those files byte for byte the way it always did.
7920        for (column, ((field, dictionary), membership)) in
7921            table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
7922        {
7923            if !coded_type(&field.ty) || dictionary.is_none() {
7924                continue;
7925            }
7926            let page = match membership {
7927                Some(page) => page,
7928                None if table.demoted.get(column).copied().unwrap_or(false) => {
7929                    Page { offset: HEADER, length: 0, hash: 0 }
7930                }
7931                None => return Err(invalid("string page has no code membership index")),
7932            };
7933            put_u64(&mut out, page.offset);
7934            put_u32(&mut out, page.length);
7935            put_u64(&mut out, page.hash);
7936        }
7937        for sieve in stripe.sieves.slots() {
7938            match sieve {
7939                None => out.push(0),
7940                Some(page) => {
7941                    out.push(1);
7942                    put_u64(&mut out, page.offset);
7943                    put_u32(&mut out, page.length);
7944                    put_u64(&mut out, page.hash);
7945                }
7946            }
7947        }
7948        for held in stripe.part_ranges.slots() {
7949            match held {
7950                None => out.push(0),
7951                Some(page) => {
7952                    out.push(1);
7953                    put_u64(&mut out, page.offset);
7954                    put_u32(&mut out, page.length);
7955                    put_u64(&mut out, page.hash);
7956                }
7957            }
7958        }
7959        for range in stripe.zone.columns() {
7960            put_bound(&mut out, range.low.as_ref())?;
7961            put_bound(&mut out, range.high.as_ref())?;
7962            put_u32(
7963                &mut out,
7964                u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
7965            );
7966            out.push(u8::from(range.exact));
7967            match range.sum {
7968                None => out.push(0),
7969                Some(total) => {
7970                    out.push(1);
7971                    out.extend_from_slice(&total.to_le_bytes());
7972                }
7973            }
7974        }
7975    }
7976    out.extend_from_slice(FREQUENCIES);
7977    put_u16(
7978        &mut out,
7979        u16::try_from(table.frequencies.len())
7980            .map_err(|_| invalid("too many frequency columns"))?,
7981    );
7982    for summary in &table.frequencies {
7983        let summary = match summary {
7984            None => {
7985                out.push(0);
7986                continue;
7987            }
7988            Some(Frequencies::Held(summary)) => summary,
7989            // Only a reader leaves a synopsis in the file, and nothing writes a reader's table back.
7990            Some(Frequencies::Stored { .. }) => {
7991                return Err(invalid("a synopsis left in the file cannot be written back"));
7992            }
7993        };
7994        out.push(1);
7995        put_u64(&mut out, summary.omitted_max);
7996        put_u32(
7997            &mut out,
7998            u32::try_from(summary.entries.len())
7999                .map_err(|_| invalid("too many frequency entries"))?,
8000        );
8001        for entry in &summary.entries {
8002            match entry.value {
8003                FrequencyValue::Null => out.push(0),
8004                FrequencyValue::Integer(value) => {
8005                    out.push(1);
8006                    out.extend_from_slice(&value.to_le_bytes());
8007                }
8008                FrequencyValue::Code(value) => {
8009                    out.push(2);
8010                    put_u32(&mut out, value);
8011                }
8012            }
8013            put_u64(&mut out, entry.count);
8014        }
8015        put_u32(
8016            &mut out,
8017            u32::try_from(summary.ordinals.len())
8018                .map_err(|_| invalid("too many frequency ordinals"))?,
8019        );
8020        let mut previous = 0_u64;
8021        for (at, &ordinal) in summary.ordinals.iter().enumerate() {
8022            let delta = if at == 0 {
8023                ordinal
8024            } else {
8025                ordinal
8026                    .checked_sub(previous)
8027                    .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
8028            };
8029            if at != 0 && delta == 0 {
8030                return Err(invalid("frequency ordinals are not unique"));
8031            }
8032            put_var_u64(&mut out, delta);
8033            previous = ordinal;
8034        }
8035        if summary.ordinal_entries.len() != summary.ordinals.len() {
8036            return Err(invalid("frequency ordinal values have a different length"));
8037        }
8038        for &entry in &summary.ordinal_entries {
8039            if entry as usize >= summary.entries.len() {
8040                return Err(invalid("frequency ordinal value is outside its entries"));
8041            }
8042            put_u16(&mut out, entry);
8043        }
8044    }
8045    if !table.pair_frequencies.is_empty() {
8046        out.extend_from_slice(PAIR_FREQUENCIES);
8047        put_u16(
8048            &mut out,
8049            u16::try_from(table.pair_frequencies.len())
8050                .map_err(|_| invalid("too many pair frequency summaries"))?,
8051        );
8052        for summary in &table.pair_frequencies {
8053            put_u16(&mut out, summary.first);
8054            put_u16(&mut out, summary.second);
8055            put_u64(&mut out, summary.omitted_max);
8056            put_u16(
8057                &mut out,
8058                u16::try_from(summary.entries.len())
8059                    .map_err(|_| invalid("too many pair frequency entries"))?,
8060            );
8061            for entry in &summary.entries {
8062                put_u16(&mut out, entry.first_entry);
8063                match entry.second {
8064                    None => out.push(0),
8065                    Some(code) => {
8066                        out.push(1);
8067                        put_u32(&mut out, code);
8068                    }
8069                }
8070                put_u64(&mut out, entry.count);
8071            }
8072        }
8073    }
8074    let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
8075    if text_columns != 0 {
8076        out.extend_from_slice(FREQUENCY_TEXTS);
8077        put_u16(
8078            &mut out,
8079            u16::try_from(text_columns)
8080                .map_err(|_| invalid("too many string frequency columns"))?,
8081        );
8082        for (column, texts) in table.frequency_texts.iter().enumerate() {
8083            if texts.is_empty() {
8084                continue;
8085            }
8086            put_u16(
8087                &mut out,
8088                u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
8089            );
8090            put_u16(
8091                &mut out,
8092                u16::try_from(texts.len())
8093                    .map_err(|_| invalid("too many frequency text entries"))?,
8094            );
8095            for text in texts {
8096                match text {
8097                    None => out.push(0),
8098                    Some(text) => {
8099                        out.push(1);
8100                        put_u32(
8101                            &mut out,
8102                            u32::try_from(text.len())
8103                                .map_err(|_| invalid("frequency text is too long"))?,
8104                        );
8105                        out.extend_from_slice(text);
8106                    }
8107                }
8108            }
8109        }
8110    }
8111    if let Some(summary) = &table.host_groups {
8112        out.extend_from_slice(HOST_GROUPS);
8113        put_u16(
8114            &mut out,
8115            u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
8116        );
8117        put_u64(&mut out, summary.omitted_max);
8118        put_u16(
8119            &mut out,
8120            u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
8121        );
8122        for entry in &summary.entries {
8123            put_u32(
8124                &mut out,
8125                u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
8126            );
8127            out.extend_from_slice(entry.host.as_bytes());
8128            put_u64(&mut out, entry.count);
8129            out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
8130            put_u32(
8131                &mut out,
8132                u32::try_from(entry.minimum.len())
8133                    .map_err(|_| invalid("host minimum is too long"))?,
8134            );
8135            out.extend_from_slice(entry.minimum.as_bytes());
8136        }
8137    }
8138    // Written only when there is a declaration, so that the common file is the same bytes it was
8139    // and the section is not a byte of zero on every table in the world that never asked for one.
8140    if let Some(clustering) = &table.clustering {
8141        out.extend_from_slice(CLUSTERING);
8142        out.push(clustering.width().tag());
8143        put_u16(
8144            &mut out,
8145            u16::try_from(clustering.columns().len())
8146                .map_err(|_| invalid("too many clustering columns"))?,
8147        );
8148        for &column in clustering.columns() {
8149            put_u16(
8150                &mut out,
8151                u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
8152            );
8153        }
8154    }
8155    let demoted = (0..table.fields.len())
8156        .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
8157        .collect::<Vec<_>>();
8158    if !demoted.is_empty() {
8159        out.extend_from_slice(DEMOTED);
8160        put_u16(
8161            &mut out,
8162            u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8163        );
8164        for column in demoted {
8165            put_u16(
8166                &mut out,
8167                u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8168            );
8169        }
8170    }
8171    // The section table, last, behind its own magic, for the same reason the frequency block is
8172    // behind its own: a reader that stops before it gets a table with no sections, and a table with
8173    // no sections is a correct table. The one difference from the blocks before it is that this one
8174    // is written even when it is empty, so that a file written by this build always says which
8175    // sections it has rather than leaving a reader to infer it from where the bytes ran out.
8176    out.extend_from_slice(SECTIONS);
8177    put_u64(&mut out, table.generation);
8178    put_u16(
8179        &mut out,
8180        u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8181    );
8182    for held in &table.sections {
8183        held.encode(&mut out)?;
8184    }
8185    if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8186        out.extend_from_slice(DICTIONARY_PAYLOADS);
8187        put_u16(
8188            &mut out,
8189            u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8190        );
8191        for at in 0..table.fields.len() {
8192            put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8193        }
8194    }
8195    Ok(out)
8196}
8197
8198/// The small level of the directory, naming every table in the file.
8199///
8200/// This is what a footer slot points at. Each entry carries its own checksum over its table
8201/// directory, so a table whose directory is torn is found when that table is first touched rather
8202/// than being trusted because the catalog around it checksummed.
8203///
8204/// The views go after the tables and are whole here, since a view is text and a column list and has
8205/// no pages for a second level to point at.
8206fn signed_integer(ty: &LogicalType) -> bool {
8207    matches!(
8208        ty,
8209        LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8210    )
8211}
8212
8213fn integer_or_date(ty: &LogicalType) -> bool {
8214    matches!(
8215        ty,
8216        LogicalType::TinyInt
8217            | LogicalType::SmallInt
8218            | LogicalType::Integer
8219            | LogicalType::BigInt
8220            | LogicalType::UTinyInt
8221            | LogicalType::USmallInt
8222            | LogicalType::UInteger
8223            | LogicalType::UBigInt
8224            | LogicalType::Date
8225    )
8226}
8227
8228fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8229    table
8230        .fields
8231        .iter()
8232        .enumerate()
8233        .map(|(column, field)| {
8234            if !integer_or_date(&field.ty) {
8235                return None;
8236            }
8237            let mut low: Option<i128> = None;
8238            let mut high: Option<i128> = None;
8239            for stripe in &table.stripes {
8240                let range = stripe.zone.column(column)?;
8241                if !range.exact {
8242                    return None;
8243                }
8244                match (range.low.as_ref(), range.high.as_ref()) {
8245                    (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8246                        low = Some(low.map_or(*small, |held| held.min(*small)));
8247                        high = Some(high.map_or(*large, |held| held.max(*large)));
8248                    }
8249                    (None, None) if stripe.rows == range.nulls => {}
8250                    _ => return None,
8251                }
8252            }
8253            Some(low.zip(high))
8254        })
8255        .collect()
8256}
8257
8258fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8259    reader
8260        .table
8261        .fields
8262        .iter()
8263        .enumerate()
8264        .map(|(column, field)| {
8265            if !integer_or_date(&field.ty) {
8266                return Ok(None);
8267            }
8268            match reader.exact_extremes(column)? {
8269                Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8270                None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8271                _ => Ok(None),
8272            }
8273        })
8274        .collect()
8275}
8276
8277fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8278    table
8279        .fields
8280        .iter()
8281        .enumerate()
8282        .map(|(column, field)| {
8283            if !integer_or_date(&field.ty) {
8284                return None;
8285            }
8286            let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8287                return None;
8288            };
8289            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8290                return None;
8291            }
8292            let entries = summary
8293                .entries
8294                .iter()
8295                .map(|entry| {
8296                    let value = match entry.value {
8297                        FrequencyValue::Null => None,
8298                        FrequencyValue::Integer(value) => Some(value),
8299                        FrequencyValue::Code(_) => return None,
8300                    };
8301                    Some((value, entry.count))
8302                })
8303                .collect::<Option<Vec<_>>>()?;
8304            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8305            (rows == table.rows as u64).then_some(entries)
8306        })
8307        .collect()
8308}
8309
8310/// The sixty four bits the close keys a numeric column's frequencies by, for a value the writer's
8311/// tally held.
8312///
8313/// The same bits [`Writer::visit_numeric`] hands over: a signed value sign extended to `i64`, and an
8314/// unsigned one as it is.
8315/// The value a column's sixty four bits stand for, read as signed or unsigned the way the column is.
8316fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8317    if signed {
8318        FrequencyValue::Integer(i128::from(bits as i64))
8319    } else {
8320        FrequencyValue::Integer(i128::from(bits))
8321    }
8322}
8323
8324fn frequency_bits(value: &Value) -> Option<u64> {
8325    Some(match value {
8326        Value::TinyInt(value) => i64::from(*value) as u64,
8327        Value::SmallInt(value) => i64::from(*value) as u64,
8328        Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8329        Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8330        Value::UTinyInt(value) => u64::from(*value),
8331        Value::USmallInt(value) => u64::from(*value),
8332        Value::UInteger(value) => u64::from(*value),
8333        Value::UBigInt(value) => *value,
8334        _ => return None,
8335    })
8336}
8337
8338fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8339    Some(match value {
8340        Value::Null => None,
8341        Value::TinyInt(value) => Some(i128::from(*value)),
8342        Value::SmallInt(value) => Some(i128::from(*value)),
8343        Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8344        Value::BigInt(value) => Some(i128::from(*value)),
8345        Value::UTinyInt(value) => Some(i128::from(*value)),
8346        Value::USmallInt(value) => Some(i128::from(*value)),
8347        Value::UInteger(value) => Some(i128::from(*value)),
8348        Value::UBigInt(value) => Some(i128::from(*value)),
8349        _ => return None,
8350    })
8351}
8352
8353fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8354    reader
8355        .table
8356        .fields
8357        .iter()
8358        .enumerate()
8359        .map(|(column, field)| {
8360            if !integer_or_date(&field.ty) {
8361                return Ok(None);
8362            }
8363            let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
8364            if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8365                return Ok(None);
8366            }
8367            let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
8368            let Some(entries) = entries
8369                .iter()
8370                .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8371                .collect::<Option<Vec<_>>>()
8372            else {
8373                return Ok(None);
8374            };
8375            let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8376            Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8377        })
8378        .collect()
8379}
8380
8381fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8382    table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8383        let range = stripe.zone.column(column)?;
8384        let sum = sum.checked_add(range.sum?)?;
8385        let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8386        Some((sum, count.checked_add(nonnull)?))
8387    })
8388}
8389
8390fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8391    table
8392        .fields
8393        .iter()
8394        .enumerate()
8395        .map(|(column, field)| {
8396            signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8397        })
8398        .collect()
8399}
8400
8401fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8402    reader
8403        .table
8404        .fields
8405        .iter()
8406        .enumerate()
8407        .map(
8408            |(column, field)| {
8409                if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8410            },
8411        )
8412        .collect()
8413}
8414
8415fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
8416    let mut out = CATALOG.to_vec();
8417    put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8418    for entry in entries {
8419        let name = entry.name.as_bytes();
8420        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8421        out.extend_from_slice(name);
8422        put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8423        put_u16(
8424            &mut out,
8425            u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8426        );
8427        for field in &entry.fields {
8428            let name = field.name.as_bytes();
8429            put_u16(
8430                &mut out,
8431                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8432            );
8433            out.extend_from_slice(name);
8434            put_type(&mut out, &field.ty)?;
8435            out.push(u8::from(field.not_null));
8436        }
8437        put_u64(&mut out, entry.directory.offset);
8438        put_u32(&mut out, entry.directory.length);
8439        put_u64(&mut out, entry.directory.hash);
8440    }
8441    put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8442    for view in views {
8443        let name = view.name.as_bytes();
8444        put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8445        out.extend_from_slice(name);
8446        put_long_text(&mut out, &view.sql, "view body")?;
8447        put_long_text(&mut out, &view.statement, "view statement")?;
8448        put_u16(
8449            &mut out,
8450            u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8451        );
8452        for alias in &view.aliases {
8453            let alias = alias.as_bytes();
8454            put_u16(
8455                &mut out,
8456                u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8457            );
8458            out.extend_from_slice(alias);
8459        }
8460        put_u16(
8461            &mut out,
8462            u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8463        );
8464        for field in &view.columns {
8465            let name = field.name.as_bytes();
8466            put_u16(
8467                &mut out,
8468                u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8469            );
8470            out.extend_from_slice(name);
8471            put_type(&mut out, &field.ty)?;
8472            out.push(u8::from(field.not_null));
8473        }
8474    }
8475    out.extend_from_slice(NONZERO_COUNTS);
8476    for entry in entries {
8477        if entry.nonzero.len() != entry.fields.len() {
8478            return Err(invalid("nonzero count width differs from schema"));
8479        }
8480        for count in &entry.nonzero {
8481            match count {
8482                None => out.push(0),
8483                Some(count) => {
8484                    out.push(1);
8485                    put_u64(&mut out, *count);
8486                }
8487            }
8488        }
8489    }
8490    out.extend_from_slice(AGGREGATE_SUMS);
8491    for entry in entries {
8492        if entry.aggregates.len() != entry.fields.len() {
8493            return Err(invalid("aggregate sum width differs from schema"));
8494        }
8495        for summary in &entry.aggregates {
8496            match summary {
8497                None => out.push(0),
8498                Some((sum, count)) => {
8499                    out.push(1);
8500                    out.extend_from_slice(&sum.to_le_bytes());
8501                    put_u64(&mut out, *count);
8502                }
8503            }
8504        }
8505    }
8506    out.extend_from_slice(DISTINCT_COUNTS);
8507    for entry in entries {
8508        if entry.distincts.len() != entry.fields.len() {
8509            return Err(invalid("distinct count width differs from schema"));
8510        }
8511        for count in &entry.distincts {
8512            match count {
8513                None => out.push(0),
8514                Some(count) => {
8515                    if *count > entry.rows as u64 {
8516                        return Err(invalid("distinct count exceeds table rows"));
8517                    }
8518                    out.push(1);
8519                    put_u64(&mut out, *count);
8520                }
8521            }
8522        }
8523    }
8524    out.extend_from_slice(INTEGER_EXTREMES);
8525    for entry in entries {
8526        if entry.extremes.len() != entry.fields.len() {
8527            return Err(invalid("integer extremes width differs from schema"));
8528        }
8529        for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
8530            match extremes {
8531                None => out.push(0),
8532                Some(None) if integer_or_date(&field.ty) => out.push(1),
8533                Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
8534                    out.push(2);
8535                    out.extend_from_slice(&low.to_le_bytes());
8536                    out.extend_from_slice(&high.to_le_bytes());
8537                }
8538                _ => return Err(invalid("integer extremes type or range differs")),
8539            }
8540        }
8541    }
8542    out.extend_from_slice(COMPLETE_FREQUENCIES);
8543    for entry in entries {
8544        if entry.frequencies.len() != entry.fields.len() {
8545            return Err(invalid("numeric frequency width differs from schema"));
8546        }
8547        for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
8548            match frequencies {
8549                None => out.push(0),
8550                Some(entries)
8551                    if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
8552                {
8553                    let mut total = 0_u64;
8554                    for (at, (value, count)) in entries.iter().enumerate() {
8555                        if entries[..at].iter().any(|(held, _)| held == value) {
8556                            return Err(invalid("numeric frequency value repeats"));
8557                        }
8558                        total = total
8559                            .checked_add(*count)
8560                            .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8561                    }
8562                    if total != entry.rows as u64 {
8563                        return Err(invalid("numeric frequencies do not cover table rows"));
8564                    }
8565                    out.push(1);
8566                    out.push(entries.len() as u8);
8567                    for (value, count) in entries {
8568                        match value {
8569                            None => out.push(0),
8570                            Some(value) => {
8571                                out.push(1);
8572                                out.extend_from_slice(&value.to_le_bytes());
8573                            }
8574                        }
8575                        put_u64(&mut out, *count);
8576                    }
8577                }
8578                _ => return Err(invalid("numeric frequency type or width differs")),
8579            }
8580        }
8581    }
8582    Ok(out)
8583}
8584
8585/// A length and that many bytes, for text that is allowed to be longer than a name.
8586fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
8587    let bytes = text.as_bytes();
8588    put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
8589    out.extend_from_slice(bytes);
8590    Ok(())
8591}
8592
8593/// Reads the catalog directory back, checking every span against the file before anything is
8594/// allocated for it.
8595fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
8596    let mut cur = Cursor::new(bytes);
8597    if cur.take(8)? != CATALOG {
8598        return Err(invalid("catalog magic differs"));
8599    }
8600    let count = cur.u32()? as usize;
8601    let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
8602    for _ in 0..count {
8603        let name = cur.text()?;
8604        let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8605        let width = cur.u16()? as usize;
8606        let mut fields = Vec::with_capacity(width);
8607        for _ in 0..width {
8608            let name = cur.text()?;
8609            let ty = read_type(&mut cur)?;
8610            let not_null = match cur.u8()? {
8611                0 => false,
8612                1 => true,
8613                _ => return Err(invalid("nullability flag differs")),
8614            };
8615            fields.push(Field { name, ty, not_null });
8616        }
8617        let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8618        let end = directory
8619            .offset
8620            .checked_add(u64::from(directory.length))
8621            .ok_or_else(|| invalid("table directory offset overflow"))?;
8622        if directory.offset < HEADER
8623            || end > size
8624            || directory.length as usize > MAX_DIRECTORY
8625            || directory.length == 0
8626        {
8627            return Err(invalid("table directory range is outside the file"));
8628        }
8629        if entries.iter().any(|held| held.name == name) {
8630            return Err(invalid("two tables in the catalog have the same name"));
8631        }
8632        let nonzero = vec![None; fields.len()];
8633        let aggregates = vec![None; fields.len()];
8634        let distincts = vec![None; fields.len()];
8635        let extremes = vec![None; fields.len()];
8636        let frequencies = vec![None; fields.len()];
8637        entries.push(Entry {
8638            name,
8639            fields,
8640            rows,
8641            directory,
8642            nonzero,
8643            aggregates,
8644            distincts,
8645            extremes,
8646            frequencies,
8647        });
8648    }
8649    // A catalog that ends where the tables end is a catalog with no views in it, which is every
8650    // file written before format 25. That is why the count is allowed to be missing rather than
8651    // read as a zero that has to be there: an older file has nothing after the last table entry at
8652    // all, and [`READABLE`] says those files still open.
8653    let count = if cur.done() { 0 } else { cur.u32()? as usize };
8654    let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
8655    for _ in 0..count {
8656        let name = cur.text()?;
8657        let sql = cur.long_text()?;
8658        let statement = cur.long_text()?;
8659        let width = cur.u16()? as usize;
8660        let mut aliases = Vec::with_capacity(width);
8661        for _ in 0..width {
8662            aliases.push(cur.text()?);
8663        }
8664        let width = cur.u16()? as usize;
8665        let mut columns = Vec::with_capacity(width);
8666        for _ in 0..width {
8667            let name = cur.text()?;
8668            let ty = read_type(&mut cur)?;
8669            let not_null = match cur.u8()? {
8670                0 => false,
8671                1 => true,
8672                _ => return Err(invalid("nullability flag differs")),
8673            };
8674            columns.push(Field { name, ty, not_null });
8675        }
8676        // The same rule the tables above get, and for the same reason. Two entries under one name
8677        // is a catalog nothing can answer a lookup from, and finding that out here is better than
8678        // finding it out from whichever of the two a search happened to reach first.
8679        if views.iter().any(|held| held.name == name) {
8680            return Err(invalid("two views in the catalog have the same name"));
8681        }
8682        if entries.iter().any(|held| held.name == name) {
8683            return Err(invalid("a table and a view in the catalog have the same name"));
8684        }
8685        views.push(ViewEntry { name, sql, statement, aliases, columns });
8686    }
8687    if !cur.done() {
8688        if cur.take(8)? != NONZERO_COUNTS {
8689            return Err(invalid("catalog extension magic differs"));
8690        }
8691        for entry in &mut entries {
8692            for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
8693                *count = match cur.u8()? {
8694                    0 => None,
8695                    1 if matches!(
8696                        field.ty,
8697                        LogicalType::TinyInt
8698                            | LogicalType::SmallInt
8699                            | LogicalType::Integer
8700                            | LogicalType::BigInt
8701                            | LogicalType::UTinyInt
8702                            | LogicalType::USmallInt
8703                            | LogicalType::UInteger
8704                            | LogicalType::UBigInt
8705                    ) =>
8706                    {
8707                        let value = cur.u64()?;
8708                        if value > entry.rows as u64 {
8709                            return Err(invalid("nonzero count exceeds rows"));
8710                        }
8711                        Some(value)
8712                    }
8713                    _ => return Err(invalid("nonzero count tag or column type differs")),
8714                };
8715            }
8716        }
8717    }
8718    if !cur.done() {
8719        if cur.take(8)? != AGGREGATE_SUMS {
8720            return Err(invalid("aggregate catalog extension magic differs"));
8721        }
8722        for entry in &mut entries {
8723            for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
8724                *summary = match cur.u8()? {
8725                    0 => None,
8726                    1 if signed_integer(&field.ty) => {
8727                        let sum = i128::from_le_bytes(
8728                            cur.take(16)?
8729                                .try_into()
8730                                .map_err(|_| invalid("aggregate sum is truncated"))?,
8731                        );
8732                        let count = cur.u64()?;
8733                        if count > entry.rows as u64 {
8734                            return Err(invalid("aggregate count exceeds table rows"));
8735                        }
8736                        Some((sum, count))
8737                    }
8738                    _ => return Err(invalid("aggregate sum tag or column type differs")),
8739                };
8740            }
8741        }
8742    }
8743    if !cur.done() {
8744        if cur.take(8)? != DISTINCT_COUNTS {
8745            return Err(invalid("distinct catalog extension magic differs"));
8746        }
8747        for entry in &mut entries {
8748            for count in &mut entry.distincts {
8749                *count = match cur.u8()? {
8750                    0 => None,
8751                    1 => {
8752                        let value = cur.u64()?;
8753                        if value > entry.rows as u64 {
8754                            return Err(invalid("distinct count exceeds table rows"));
8755                        }
8756                        Some(value)
8757                    }
8758                    _ => return Err(invalid("distinct count tag differs")),
8759                };
8760            }
8761        }
8762    }
8763    if !cur.done() {
8764        if cur.take(8)? != INTEGER_EXTREMES {
8765            return Err(invalid("integer extremes catalog extension magic differs"));
8766        }
8767        for entry in &mut entries {
8768            for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
8769                *extremes = match cur.u8()? {
8770                    0 => None,
8771                    1 if integer_or_date(&field.ty) => Some(None),
8772                    2 if integer_or_date(&field.ty) => {
8773                        let low = i128::from_le_bytes(
8774                            cur.take(16)?
8775                                .try_into()
8776                                .map_err(|_| invalid("minimum is truncated"))?,
8777                        );
8778                        let high = i128::from_le_bytes(
8779                            cur.take(16)?
8780                                .try_into()
8781                                .map_err(|_| invalid("maximum is truncated"))?,
8782                        );
8783                        if low > high {
8784                            return Err(invalid("integer extremes are reversed"));
8785                        }
8786                        Some(Some((low, high)))
8787                    }
8788                    _ => return Err(invalid("integer extremes tag or type differs")),
8789                };
8790            }
8791        }
8792    }
8793    if !cur.done() {
8794        if cur.take(8)? != COMPLETE_FREQUENCIES {
8795            return Err(invalid("numeric frequency catalog extension magic differs"));
8796        }
8797        for entry in &mut entries {
8798            for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
8799                *frequencies = match cur.u8()? {
8800                    0 => None,
8801                    1 if integer_or_date(&field.ty) => {
8802                        let len = cur.u8()? as usize;
8803                        if len > MAX_CATALOG_FREQUENCIES {
8804                            return Err(invalid("too many catalog numeric frequencies"));
8805                        }
8806                        let mut values = Vec::with_capacity(len);
8807                        let mut total = 0_u64;
8808                        for _ in 0..len {
8809                            let value = match cur.u8()? {
8810                                0 => None,
8811                                1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
8812                                    |_| invalid("numeric frequency value is truncated"),
8813                                )?)),
8814                                _ => return Err(invalid("numeric frequency value tag differs")),
8815                            };
8816                            if values.iter().any(|(held, _)| *held == value) {
8817                                return Err(invalid("numeric frequency value repeats"));
8818                            }
8819                            let count = cur.u64()?;
8820                            total = total
8821                                .checked_add(count)
8822                                .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8823                            values.push((value, count));
8824                        }
8825                        if total != entry.rows as u64 {
8826                            return Err(invalid("numeric frequencies do not cover table rows"));
8827                        }
8828                        Some(values)
8829                    }
8830                    _ => return Err(invalid("numeric frequency tag or type differs")),
8831                };
8832            }
8833        }
8834    }
8835    if !cur.done() {
8836        return Err(invalid("catalog has trailing bytes"));
8837    }
8838    Ok((entries, views))
8839}
8840
8841/// Reads the fields of a directory or a catalog in order, off bytes in memory or out of the file.
8842///
8843/// A catalog is small and is read whole. A table directory is not: at ten million rows of `hits` it
8844/// is nearly a megabyte, and holding that buffer while the table it describes is built out of it
8845/// put both at the peak of every query. Out of the file, the cursor holds one window of
8846/// [`DIRECTORY_WINDOW`] bytes and moves it forward as the fields are read, so what a directory
8847/// costs at open is what it decodes into and not that plus its own bytes.
8848struct Cursor<'a> {
8849    bytes: &'a [u8],
8850    at: usize,
8851    window: Option<Window<'a>>,
8852}
8853
8854/// The part of a directory in the file that a [`Cursor`] has read in.
8855struct Window<'a> {
8856    file: &'a File,
8857    offset: u64,
8858    length: usize,
8859    /// Where `held` starts, counted from the start of the directory.
8860    start: usize,
8861    held: Vec<u8>,
8862    /// How much to read at once, which is [`DIRECTORY_WINDOW`] outside the tests.
8863    size: usize,
8864}
8865
8866/// How much of a directory a cursor reading one out of the file holds at once.
8867const DIRECTORY_WINDOW: usize = 64 << 10;
8868
8869impl<'a> Cursor<'a> {
8870    fn new(bytes: &'a [u8]) -> Self {
8871        Self { bytes, at: 0, window: None }
8872    }
8873
8874    /// A cursor over `length` bytes of `file` from `offset`, which it reads a window at a time.
8875    fn over(file: &'a File, offset: u64, length: usize) -> Self {
8876        let window =
8877            Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
8878        Self { bytes: &[], at: 0, window: Some(window) }
8879    }
8880
8881    /// How many bytes the cursor walks in all.
8882    fn len(&self) -> usize {
8883        self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
8884    }
8885
8886    /// Makes sure the next `len` bytes are in memory.
8887    fn ensure(&mut self, len: usize) -> Result<()> {
8888        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8889        if end > self.len() {
8890            return Err(invalid("directory is truncated"));
8891        }
8892        let Some(window) = &mut self.window else { return Ok(()) };
8893        if self.at < window.start || end > window.start + window.held.len() {
8894            let want = len.max(window.size).min(window.length - self.at);
8895            window.start = self.at;
8896            window.held.resize(want, 0);
8897            read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
8898        }
8899        Ok(())
8900    }
8901
8902    /// `len` bytes from `at`, which [`Self::ensure`] has already brought in.
8903    fn held(&self, at: usize, len: usize) -> &[u8] {
8904        match &self.window {
8905            Some(window) => &window.held[at - window.start..at - window.start + len],
8906            None => &self.bytes[at..at + len],
8907        }
8908    }
8909
8910    /// The next `len` bytes, without moving past them.
8911    #[inline]
8912    fn peek(&mut self, len: usize) -> Result<&[u8]> {
8913        if self.window.is_none() {
8914            let bytes = self.bytes;
8915            return Ok(&bytes[self.at..self.end(len)?]);
8916        }
8917        self.ensure(len)?;
8918        Ok(self.held(self.at, len))
8919    }
8920
8921    /// The next `len` bytes, moving past them.
8922    ///
8923    /// Every data page is decoded through this, a byte or a word at a time, so a cursor over bytes
8924    /// already in memory takes them here and never reaches [`Self::ensure`]. With the window check
8925    /// on every call, q06 on TPC-H spent a seventh of its instructions in it.
8926    #[inline]
8927    fn take(&mut self, len: usize) -> Result<&[u8]> {
8928        if self.window.is_none() {
8929            let bytes = self.bytes;
8930            let (at, end) = (self.at, self.end(len)?);
8931            self.at = end;
8932            return Ok(&bytes[at..end]);
8933        }
8934        self.take_windowed(len)
8935    }
8936
8937    /// Moves over a checked field without reading its payload from a windowed directory.
8938    fn skip(&mut self, len: usize) -> Result<()> {
8939        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8940        if end > self.len() {
8941            return Err(invalid("directory is truncated"));
8942        }
8943        self.at = end;
8944        Ok(())
8945    }
8946
8947    fn skip_bound(&mut self) -> Result<()> {
8948        match self.u8()? {
8949            0 => Ok(()),
8950            1 => self.skip(16),
8951            2 => self.skip(8),
8952            3 => {
8953                let length = self.u32()? as usize;
8954                self.skip(length)
8955            }
8956            4 => self.skip(17),
8957            _ => Err(invalid("a stored bound has an unknown tag")),
8958        }
8959    }
8960
8961    /// Where `len` bytes from here end, when they end inside the bytes.
8962    #[inline]
8963    fn end(&self, len: usize) -> Result<usize> {
8964        let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8965        if end > self.bytes.len() {
8966            return Err(invalid("directory is truncated"));
8967        }
8968        Ok(end)
8969    }
8970
8971    /// [`Self::take`] out of the file, a window at a time.
8972    #[inline(never)]
8973    fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
8974        self.ensure(len)?;
8975        self.at += len;
8976        Ok(self.held(self.at - len, len))
8977    }
8978    #[inline]
8979    fn u8(&mut self) -> Result<u8> {
8980        Ok(self.take(1)?[0])
8981    }
8982    #[inline]
8983    fn u16(&mut self) -> Result<u16> {
8984        Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
8985    }
8986    #[inline]
8987    fn u32(&mut self) -> Result<u32> {
8988        Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
8989    }
8990    #[inline]
8991    fn u64(&mut self) -> Result<u64> {
8992        Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
8993    }
8994    fn var_u64(&mut self) -> Result<u64> {
8995        let mut value = 0_u64;
8996        for shift in (0..=63).step_by(7) {
8997            let byte = self.u8()?;
8998            let part = u64::from(byte & 0x7f);
8999            if shift == 63 && part > 1 {
9000                return Err(invalid("frequency ordinal varint overflows"));
9001            }
9002            value |= part << shift;
9003            if byte & 0x80 == 0 {
9004                return Ok(value);
9005            }
9006        }
9007        Err(invalid("frequency ordinal varint is too long"))
9008    }
9009    /// A zone map's end, in the layout `rudb_common::bounds` defines.
9010    ///
9011    /// The bytes are the ones this directory has written since format 10 and the codec moved to
9012    /// rank zero rather than being copied, because a column summary now writes the same two ends
9013    /// and two encodings of one type is how the two quietly stop agreeing.
9014    ///
9015    /// A bound's length is in the bound, so out of the file the cursor offers the codec a few bytes
9016    /// and offers it twice as many whenever it runs out before the directory does.
9017    fn bound(&mut self) -> Result<Option<Bound>> {
9018        let rest = self.len().saturating_sub(self.at);
9019        let mut want = 32;
9020        loop {
9021            let offered = self.peek(want.min(rest))?;
9022            let mut used = 0;
9023            match bounds::get(offered, &mut used) {
9024                Ok(bound) => {
9025                    self.at += used;
9026                    return Ok(bound);
9027                }
9028                Err(_) if want < rest => want *= 2,
9029                Err(error) => return Err(error),
9030            }
9031        }
9032    }
9033    fn text(&mut self) -> Result<String> {
9034        let len = self.u16()? as usize;
9035        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
9036    }
9037    /// Whether everything has been read, which is how a section that an older file does not have at
9038    /// all is told from one that is there and empty.
9039    fn done(&self) -> bool {
9040        self.at >= self.len()
9041    }
9042    /// The same, for text that is a query rather than a name.
9043    ///
9044    /// A name fits in sixteen bits of length and a view body does not have to. Nobody writes a 64
9045    /// kilobyte identifier by accident and people do write generated queries that long, and a view
9046    /// that could not be written down because its body was too big would be a limit invented here
9047    /// rather than one anything else in the engine has.
9048    fn long_text(&mut self) -> Result<String> {
9049        let len = self.u32()? as usize;
9050        String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
9051    }
9052}
9053
9054/// One column's frequency synopsis, or `None` for a column that has none, checked against the column.
9055fn decode_summary(
9056    cur: &mut Cursor<'_>,
9057    field: &Field,
9058    rows: usize,
9059    values: bool,
9060) -> Result<Option<FrequencySummary>> {
9061    Ok(match cur.u8()? {
9062        0 => None,
9063        1 => {
9064            let omitted_max = cur.u64()?;
9065            let count = cur.u32()? as usize;
9066            if count > FREQUENCY_ENTRIES {
9067                return Err(invalid("frequency entry count exceeds its bound"));
9068            }
9069            let mut entries = Vec::with_capacity(count);
9070            // row at a time: directory decoding validates each persisted bounded frequency entry.
9071            for _ in 0..count {
9072                let value = match cur.u8()? {
9073                    0 => FrequencyValue::Null,
9074                    1 => FrequencyValue::Integer(i128::from_le_bytes(
9075                        cur.take(16)?.try_into().expect("sixteen bytes"),
9076                    )),
9077                    2 => FrequencyValue::Code(cur.u32()?),
9078                    _ => return Err(invalid("frequency value tag differs")),
9079                };
9080                let valid = matches!(
9081                    (&field.ty, value),
9082                    (_, FrequencyValue::Null)
9083                        | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
9084                        | (
9085                            LogicalType::TinyInt
9086                                | LogicalType::SmallInt
9087                                | LogicalType::Integer
9088                                | LogicalType::BigInt
9089                                | LogicalType::UTinyInt
9090                                | LogicalType::USmallInt
9091                                | LogicalType::UInteger
9092                                | LogicalType::UBigInt
9093                                | LogicalType::Date
9094                                | LogicalType::Timestamp,
9095                            FrequencyValue::Integer(_),
9096                        )
9097                );
9098                if !valid {
9099                    return Err(invalid("frequency value does not match its column"));
9100                }
9101                let count = cur.u64()?;
9102                if count == 0 || count > rows as u64 {
9103                    return Err(invalid("frequency count is outside the table"));
9104                }
9105                entries.push(FrequencyEntry { value, count });
9106            }
9107            if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9108                return Err(invalid("frequency entries are not descending"));
9109            }
9110            let ordinals = {
9111                let ordinal_count = cur.u32()? as usize;
9112                if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
9113                    return Err(invalid("frequency ordinal count exceeds its bound"));
9114                }
9115                let mut ordinals = Vec::with_capacity(ordinal_count);
9116                let mut previous = 0_u64;
9117                for at in 0..ordinal_count {
9118                    let delta = cur.var_u64()?;
9119                    if at != 0 && delta == 0 {
9120                        return Err(invalid("frequency ordinals are not increasing"));
9121                    }
9122                    let ordinal = if at == 0 {
9123                        delta
9124                    } else {
9125                        previous
9126                            .checked_add(delta)
9127                            .ok_or_else(|| invalid("frequency ordinal overflows"))?
9128                    };
9129                    if ordinal >= rows as u64 {
9130                        return Err(invalid("frequency ordinal is outside the table"));
9131                    }
9132                    ordinals.push(ordinal);
9133                    previous = ordinal;
9134                }
9135                ordinals
9136            };
9137            let ordinal_entries = if values {
9138                let mut ordinal_entries = Vec::with_capacity(ordinals.len());
9139                for _ in 0..ordinals.len() {
9140                    let entry = cur.u16()?;
9141                    if entry as usize >= entries.len() {
9142                        return Err(invalid("frequency ordinal value is outside its entries"));
9143                    }
9144                    ordinal_entries.push(entry);
9145                }
9146                ordinal_entries
9147            } else {
9148                Vec::new()
9149            };
9150            Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
9151        }
9152        _ => return Err(invalid("frequency summary tag differs")),
9153    })
9154}
9155
9156/// Skips a synopsis whose column the caller does not need. The directory checksum was checked
9157/// before this walk, and the fields still need their lengths and tags checked to find the next one.
9158fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
9159    match cur.u8()? {
9160        0 => Ok(()),
9161        1 => {
9162            cur.skip(8)?;
9163            let entries = cur.u32()? as usize;
9164            if entries > FREQUENCY_ENTRIES {
9165                return Err(invalid("frequency entry count exceeds its bound"));
9166            }
9167            for _ in 0..entries {
9168                match cur.u8()? {
9169                    0 => {}
9170                    1 => cur.skip(16)?,
9171                    2 => cur.skip(4)?,
9172                    _ => return Err(invalid("frequency value tag differs")),
9173                }
9174                cur.skip(8)?;
9175            }
9176            let ordinals = cur.u32()? as usize;
9177            if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9178                return Err(invalid("frequency ordinal count exceeds its bound"));
9179            }
9180            for _ in 0..ordinals {
9181                cur.var_u64()?;
9182            }
9183            if values {
9184                cur.skip(ordinals * 2)?;
9185            }
9186            Ok(())
9187        }
9188        _ => Err(invalid("frequency summary tag differs")),
9189    }
9190}
9191
9192/// Reads only the catalog, stripe null counts, and one frequency synopsis. This is the cold path
9193/// for a summary-backed count; constructing every page descriptor and zone map would make it cost
9194/// the size of the table directory even when no row is read.
9195fn quick_nonzero(
9196    mut cur: Cursor<'_>,
9197    name: &str,
9198    fields: &[Field],
9199    rows: usize,
9200    wanted: usize,
9201) -> Result<Option<u64>> {
9202    if cur.take(8)? != DIRECTORY || cur.text()? != name {
9203        return Err(invalid("table directory differs from the catalog"));
9204    }
9205    let width = cur.u16()? as usize;
9206    if width != fields.len() {
9207        return Err(invalid("table directory width differs from the catalog"));
9208    }
9209    for field in fields {
9210        let stored =
9211            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9212        if &stored != field {
9213            return Err(invalid("table directory schema differs from the catalog"));
9214        }
9215    }
9216    let mut dictionaries = Vec::with_capacity(width);
9217    for field in fields {
9218        let held = match cur.u8()? {
9219            0 => false,
9220            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9221                cur.skip(20)?;
9222                true
9223            }
9224            _ => return Err(invalid("dictionary page tag differs")),
9225        };
9226        dictionaries.push(held);
9227    }
9228    for _ in 0..width {
9229        match cur.u8()? {
9230            0 => {}
9231            1 => cur.skip(8)?,
9232            _ => return Err(invalid("distinct count tag differs")),
9233        }
9234    }
9235    if cur.u64()? != rows as u64 {
9236        return Err(invalid("table row count differs from the catalog"));
9237    }
9238    let stripes = cur.u32()? as usize;
9239    let mut total = 0_usize;
9240    let mut nulls = 0_u64;
9241    for _ in 0..stripes {
9242        let parts = cur.u32()? as usize;
9243        if parts == 0 || parts > STRIPE_PARTS {
9244            return Err(invalid("stripe part count is outside its bound"));
9245        }
9246        let mut stripe_rows = 0_usize;
9247        for _ in 0..parts {
9248            stripe_rows = stripe_rows
9249                .checked_add(cur.u32()? as usize)
9250                .ok_or_else(|| invalid("stripe row count overflow"))?;
9251        }
9252        total =
9253            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9254        cur.skip(12 + width * 12)?;
9255        for (field, held) in fields.iter().zip(&dictionaries) {
9256            if coded_type(&field.ty) && *held {
9257                cur.skip(20)?;
9258            }
9259        }
9260        for _ in 0..width * 2 {
9261            match cur.u8()? {
9262                0 => {}
9263                1 => cur.skip(20)?,
9264                _ => return Err(invalid("stripe page tag differs")),
9265            }
9266        }
9267        for column in 0..width {
9268            cur.skip_bound()?;
9269            cur.skip_bound()?;
9270            let count = cur.u32()? as u64;
9271            if count > stripe_rows as u64 {
9272                return Err(invalid("null count exceeds stripe rows"));
9273            }
9274            if column == wanted {
9275                nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9276            }
9277            cur.skip(1)?;
9278            match cur.u8()? {
9279                0 => {}
9280                1 => cur.skip(16)?,
9281                _ => return Err(invalid("a stripe sum has an unknown tag")),
9282            }
9283        }
9284    }
9285    if total != rows {
9286        return Err(invalid("table row count differs from stripes"));
9287    }
9288    if cur.done() {
9289        return Ok(None);
9290    }
9291    let magic = cur.take(8)?;
9292    let values = magic == FREQUENCIES;
9293    if !values && magic != FREQUENCIES_V2 {
9294        return Err(invalid("directory extension magic differs"));
9295    }
9296    if cur.u16()? as usize != width {
9297        return Err(invalid("frequency column count differs"));
9298    }
9299    for _ in 0..wanted {
9300        skip_summary(&mut cur, values, rows)?;
9301    }
9302    let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9303        return Ok(None);
9304    };
9305    let zero = summary
9306        .entries
9307        .iter()
9308        .find(|entry| entry.value == FrequencyValue::Integer(0))
9309        .map(|entry| entry.count)
9310        .or_else(|| (summary.omitted_max == 0).then_some(0));
9311    Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9312}
9313
9314/// Walks the row-oriented directory while retaining only one column's index and page spans.
9315/// The catalog supplies the schema and the caller checks the complete directory checksum first.
9316fn quick_integer_fold(
9317    file: &File,
9318    mut cur: Cursor<'_>,
9319    entry: &Entry,
9320    size: u64,
9321    wanted: usize,
9322    emit: &mut impl FnMut(i64, u64) -> Result<()>,
9323) -> Result<()> {
9324    let name = &entry.name;
9325    let fields = &entry.fields;
9326    let rows = entry.rows;
9327    if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9328        return Err(invalid("table directory differs from the catalog"));
9329    }
9330    let width = cur.u16()? as usize;
9331    if width != fields.len() {
9332        return Err(invalid("table directory width differs from the catalog"));
9333    }
9334    for field in fields {
9335        let stored =
9336            Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9337        if &stored != field {
9338            return Err(invalid("table directory schema differs from the catalog"));
9339        }
9340    }
9341    let mut dictionaries = Vec::with_capacity(width);
9342    for field in fields {
9343        dictionaries.push(match cur.u8()? {
9344            0 => false,
9345            tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9346                cur.skip(20)?;
9347                true
9348            }
9349            _ => return Err(invalid("dictionary page tag differs")),
9350        });
9351    }
9352    for _ in 0..width {
9353        match cur.u8()? {
9354            0 => {}
9355            1 => cur.skip(8)?,
9356            _ => return Err(invalid("distinct count tag differs")),
9357        }
9358    }
9359    if cur.u64()? != rows as u64 {
9360        return Err(invalid("table row count differs from the catalog"));
9361    }
9362    let stripes = cur.u32()? as usize;
9363    let mut total = 0_usize;
9364    let mut bytes = Vec::new();
9365    for _ in 0..stripes {
9366        let parts = cur.u32()? as usize;
9367        if parts == 0 || parts > STRIPE_PARTS {
9368            return Err(invalid("stripe part count is outside its bound"));
9369        }
9370        let mut part_rows = Vec::with_capacity(parts);
9371        for _ in 0..parts {
9372            let count = cur.u32()? as usize;
9373            if count == 0 {
9374                return Err(invalid("empty part"));
9375            }
9376            total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
9377            part_rows.push(count);
9378        }
9379        let index = Span { offset: cur.u64()?, length: cur.u32()? };
9380        let section = index_section(parts)?;
9381        let index_length =
9382            section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
9383        if index.offset < HEADER
9384            || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
9385            || index.length as usize != index_length
9386        {
9387            return Err(invalid("index page range is outside the file"));
9388        }
9389        cur.skip(wanted * 12)?;
9390        let page = Span { offset: cur.u64()?, length: cur.u32()? };
9391        if page.offset < HEADER
9392            || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
9393            || page.length as usize > MAX_PAGE
9394        {
9395            return Err(invalid("column page range is outside the file"));
9396        }
9397        cur.skip((width - wanted - 1) * 12)?;
9398        for (field, held) in fields.iter().zip(&dictionaries) {
9399            if coded_type(&field.ty) && *held {
9400                cur.skip(20)?;
9401            }
9402        }
9403        for _ in 0..width * 2 {
9404            match cur.u8()? {
9405                0 => {}
9406                1 => cur.skip(20)?,
9407                _ => return Err(invalid("stripe page tag differs")),
9408            }
9409        }
9410        for _ in 0..width {
9411            cur.skip_bound()?;
9412            cur.skip_bound()?;
9413            cur.skip(5)?;
9414            match cur.u8()? {
9415                0 => {}
9416                1 => cur.skip(16)?,
9417                _ => return Err(invalid("a stripe sum has an unknown tag")),
9418            }
9419        }
9420        let spans = read_index_span(file, index, page, parts, wanted)?;
9421        for (span, expected_rows) in spans.into_iter().zip(part_rows) {
9422            bytes.resize(span.length, 0);
9423            let at = page
9424                .offset
9425                .checked_add(span.start as u64)
9426                .ok_or_else(|| invalid("part range overflow"))?;
9427            read_at(file, at, &mut bytes)?;
9428            if checksum(&bytes) != span.hash {
9429                return Err(invalid("integer part checksum differs"));
9430            }
9431            if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
9432                let decoded_rows = integer::fold(&bytes[2..], |value, count| {
9433                    check_integer_tally_value(value, &fields[wanted].ty)?;
9434                    emit(value, count)
9435                })?;
9436                if decoded_rows != expected_rows {
9437                    return Err(invalid("encoded integer part holds the wrong number of rows"));
9438                }
9439            } else {
9440                let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
9441                if let Some(packed) = column.packed_parts() {
9442                    let validity = column.validity();
9443                    let all_valid = column.none_null();
9444                    let base = packed.base();
9445                    let mut codes = [0_u64; 64];
9446                    for from in (0..expected_rows).step_by(codes.len()) {
9447                        let count = (expected_rows - from).min(codes.len());
9448                        packed.unpack(from, &mut codes[..count]);
9449                        for (offset, &code) in codes[..count].iter().enumerate() {
9450                            if all_valid || validity.is_valid(from + offset) {
9451                                // Vector::packed checked that this entire range fits the type.
9452                                emit((base + i128::from(code)) as i64, 1)?;
9453                            }
9454                        }
9455                    }
9456                    continue;
9457                }
9458                let column = column.into_flat()?;
9459                let validity = column.validity();
9460                macro_rules! count_decoded {
9461                    ($values:expr) => {
9462                        for (row, &value) in $values.as_slice().iter().enumerate() {
9463                            if validity.is_valid(row) {
9464                                emit(i64::from(value), 1)?;
9465                            }
9466                        }
9467                    };
9468                }
9469                match column.data() {
9470                    Some(Data::Int8(values)) => count_decoded!(values),
9471                    Some(Data::Int16(values)) => count_decoded!(values),
9472                    Some(Data::Int32(values)) => count_decoded!(values),
9473                    Some(Data::Int64(values)) => count_decoded!(values),
9474                    _ => return Err(invalid("decoded integer part has the wrong type")),
9475                }
9476            }
9477        }
9478    }
9479    if total != rows {
9480        return Err(invalid("table row count differs from stripes"));
9481    }
9482    Ok(())
9483}
9484
9485fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
9486    let fits = match ty {
9487        LogicalType::TinyInt => i8::try_from(value).is_ok(),
9488        LogicalType::SmallInt => i16::try_from(value).is_ok(),
9489        LogicalType::Integer => i32::try_from(value).is_ok(),
9490        LogicalType::BigInt => true,
9491        _ => false,
9492    };
9493    if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
9494}
9495
9496fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
9497    read_directory(Cursor::new(bytes), size, None)
9498}
9499
9500/// A directory out of `cur`, which is a whole one in memory or one being read out of the file.
9501///
9502/// `stored_at` is where the directory starts in the file when it is being read out of it, and then
9503/// every frequency synopsis is checked and left there, as [`Frequencies::Stored`].
9504fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
9505    if cur.take(8)? != DIRECTORY {
9506        return Err(invalid("directory magic differs"));
9507    }
9508    let name = cur.text()?;
9509    let width = cur.u16()? as usize;
9510    let mut fields = Vec::with_capacity(width);
9511    for _ in 0..width {
9512        let name = cur.text()?;
9513        let ty = read_type(&mut cur)?;
9514        let not_null = match cur.u8()? {
9515            0 => false,
9516            1 => true,
9517            _ => return Err(invalid("nullability flag differs")),
9518        };
9519        fields.push(Field { name, ty, not_null });
9520    }
9521    let mut dictionaries = Vec::with_capacity(width);
9522    for field in &fields {
9523        dictionaries.push(match cur.u8()? {
9524            0 => None,
9525            tag if tag == dictionary_tag(&field.ty) => {
9526                let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9527                let end = page
9528                    .offset
9529                    .checked_add(u64::from(page.length))
9530                    .ok_or_else(|| invalid("dictionary page offset overflow"))?;
9531                // A global dictionary covers a whole column, not one bounded stripe. Its lazy
9532                // payload is intentionally allowed to grow past `MAX_PAGE`; only ordinary column
9533                // pages are capped there. `Writer::finish` has already bounded this length by the
9534                // on-disk `u32`, and the range check below keeps it inside the file.
9535                if page.offset < HEADER || end > size {
9536                    return Err(invalid("dictionary page range is outside the file"));
9537                }
9538                Some(page)
9539            }
9540            _ => return Err(invalid("dictionary page tag differs")),
9541        });
9542    }
9543    let mut distincts = Vec::with_capacity(width);
9544    for _ in 0..width {
9545        distincts.push(match cur.u8()? {
9546            0 => None,
9547            1 => Some(cur.u64()?),
9548            _ => return Err(invalid("distinct count tag differs")),
9549        });
9550    }
9551    let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9552    let count = cur.u32()? as usize;
9553    let mut stripes = Vec::with_capacity(count);
9554    let mut total = 0_usize;
9555    for _ in 0..count {
9556        let count = cur.u32()? as usize;
9557        if count == 0 || count > STRIPE_PARTS {
9558            return Err(invalid("stripe part count is outside its bound"));
9559        }
9560        let mut parts = Vec::with_capacity(count);
9561        let mut stripe_rows = 0_usize;
9562        for _ in 0..count {
9563            let rows = cur.u32()?;
9564            if rows == 0 {
9565                return Err(invalid("empty part"));
9566            }
9567            parts.push(rows);
9568            stripe_rows = stripe_rows
9569                .checked_add(rows as usize)
9570                .ok_or_else(|| invalid("stripe row count overflow"))?;
9571        }
9572        total =
9573            total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9574        let index = Span { offset: cur.u64()?, length: cur.u32()? };
9575        let section = index_section(count)?;
9576        let wanted = section
9577            .checked_mul(width)
9578            .and_then(|bytes| u32::try_from(bytes).ok())
9579            .ok_or_else(|| invalid("index page length overflow"))?;
9580        let end = index
9581            .offset
9582            .checked_add(u64::from(index.length))
9583            .ok_or_else(|| invalid("index page offset overflow"))?;
9584        if index.offset < HEADER || end > size || index.length != wanted {
9585            return Err(invalid("index page range is outside the file"));
9586        }
9587        let mut pages = Vec::with_capacity(width);
9588        for _ in 0..width {
9589            let offset = cur.u64()?;
9590            let length = cur.u32()?;
9591            let end = offset
9592                .checked_add(u64::from(length))
9593                .ok_or_else(|| invalid("page offset overflow"))?;
9594            if offset < HEADER || end > size || length as usize > MAX_PAGE {
9595                return Err(invalid("page range is outside the file"));
9596            }
9597            pages.push(Span { offset, length });
9598        }
9599        let mut memberships = vec![None; width];
9600        for (column, field) in fields.iter().enumerate() {
9601            if !coded_type(&field.ty) || dictionaries[column].is_none() {
9602                continue;
9603            }
9604            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9605            let end = page
9606                .offset
9607                .checked_add(u64::from(page.length))
9608                .ok_or_else(|| invalid("membership page offset overflow"))?;
9609            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9610                return Err(invalid("membership page range is outside the file"));
9611            }
9612            // No bytes is a stripe written after the column's dictionary was demoted, see
9613            // [`DEMOTED`], which is checked once the block that says so has been read.
9614            if page.length != 0 {
9615                memberships[column] = Some(page);
9616            }
9617        }
9618        let mut sieves = vec![None; width];
9619        for sieve in sieves.iter_mut().take(width) {
9620            match cur.u8()? {
9621                0 => continue,
9622                1 => {}
9623                _ => return Err(invalid("a sieve page has an unknown tag")),
9624            }
9625            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9626            let end = page
9627                .offset
9628                .checked_add(u64::from(page.length))
9629                .ok_or_else(|| invalid("sieve page offset overflow"))?;
9630            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9631                return Err(invalid("sieve page range is outside the file"));
9632            }
9633            *sieve = Some(page);
9634        }
9635        let mut part_ranges = vec![None; width];
9636        for held in part_ranges.iter_mut().take(width) {
9637            match cur.u8()? {
9638                0 => continue,
9639                1 => {}
9640                _ => return Err(invalid("a part range page has an unknown tag")),
9641            }
9642            let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9643            let end = page
9644                .offset
9645                .checked_add(u64::from(page.length))
9646                .ok_or_else(|| invalid("part range page offset overflow"))?;
9647            if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9648                return Err(invalid("part range page range is outside the file"));
9649            }
9650            *held = Some(page);
9651        }
9652        let mut ranges = Vec::with_capacity(width);
9653        for column in 0..width {
9654            let low = cur.bound()?;
9655            let high = cur.bound()?;
9656            let nulls = cur.u32()? as usize;
9657            if nulls > stripe_rows {
9658                return Err(invalid("null count exceeds stripe rows"));
9659            }
9660            let exact = cur.u8()? != 0;
9661            let sum = match cur.u8()? {
9662                0 => None,
9663                1 => Some(i128::from_le_bytes(
9664                    cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
9665                )),
9666                _ => return Err(invalid("a stripe sum has an unknown tag")),
9667            };
9668            // Files written before the ends of a decimal or a timestamp column carried their power
9669            // of ten hold a bare integer here, and that integer is the one the column holds, which
9670            // is what the power is over. So the type puts it back on the way in and an old file
9671            // prunes as well as a new one. A file that already wrote the power keeps it, because
9672            // this leaves anything that is not an integer alone.
9673            let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
9674            let low = low.map(|bound| scaled_as(bound, ty));
9675            let high = high.map(|bound| scaled_as(bound, ty));
9676            ranges.push(Range { low, high, nulls, exact, sum });
9677        }
9678        stripes.push(Stripe {
9679            rows: stripe_rows,
9680            parts,
9681            index,
9682            pages,
9683            memberships: Pages::from_slots(memberships)?,
9684            sieves: Pages::from_slots(sieves)?,
9685            part_ranges: Pages::from_slots(part_ranges)?,
9686            zone: Zone::from_ranges(ranges),
9687        });
9688    }
9689    if total != rows {
9690        return Err(invalid("table row count differs from stripes"));
9691    }
9692    // How many entries each column's synopsis lists, which is all a pair summary is checked against,
9693    // kept apart because the synopses themselves may be left in the file.
9694    let mut entry_counts = vec![0; width];
9695    let frequencies = if cur.done() {
9696        vec![None; width]
9697    } else {
9698        let frequency_magic = cur.take(8)?;
9699        let frequency_values = frequency_magic == FREQUENCIES;
9700        if !frequency_values && frequency_magic != FREQUENCIES_V2 {
9701            return Err(invalid("directory extension magic differs"));
9702        }
9703        if cur.u16()? as usize != width {
9704            return Err(invalid("frequency column count differs"));
9705        }
9706        let mut frequencies = Vec::with_capacity(width);
9707        for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
9708            let start = cur.at;
9709            let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
9710            *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
9711            frequencies.push(match (summary, stored_at) {
9712                (None, _) => None,
9713                (Some(summary), None) => Some(Frequencies::Held(summary)),
9714                (Some(_), Some(offset)) => Some(Frequencies::Stored {
9715                    span: Span {
9716                        offset: offset + start as u64,
9717                        length: u32::try_from(cur.at - start)
9718                            .map_err(|_| invalid("a frequency synopsis is too long"))?,
9719                    },
9720                    values: frequency_values,
9721                }),
9722            });
9723        }
9724        frequencies
9725    };
9726    // There are two optional trailing blocks now rather than one, so the reader dispatches on the
9727    // magic it finds rather than on where the bytes ran out. That is what lets the two arrive
9728    // independently: a format 22 directory ends here and has neither, a directory written before
9729    // the section table has only the clustering declaration, and each one still opens without a
9730    // rewrite. It is the G1 exit criterion, which is that a reader that knows about sections opens
9731    // a file that predates them and answers every query, only without the graph path.
9732    //
9733    // A repeated block is refused rather than allowed to win, because two clustering declarations
9734    // in one directory is a torn directory and the only question is which of them is the lie.
9735    let mut clustering = None;
9736    let mut sections = Vec::new();
9737    let mut pair_frequencies = Vec::new();
9738    let mut seen_pair_frequencies = false;
9739    let mut frequency_texts = vec![Vec::new(); width];
9740    let mut seen_frequency_texts = false;
9741    let mut host_groups = None;
9742    let mut demoted = Vec::new();
9743    let mut seen_sections = false;
9744    let mut dictionary_payloads = Vec::new();
9745    let mut seen_payloads = false;
9746    // Zero until a section table says otherwise, which is what a format 22 table gets and what
9747    // makes every section stamp fail to match on one, because real generations start at one.
9748    let mut generation = 0;
9749    while !cur.done() {
9750        let mut tag = [0u8; 8];
9751        tag.copy_from_slice(cur.take(8)?);
9752        if &tag == PAIR_FREQUENCIES {
9753            if seen_pair_frequencies {
9754                return Err(invalid("directory names two pair frequency blocks"));
9755            }
9756            seen_pair_frequencies = true;
9757            let count = cur.u16()? as usize;
9758            if count > MAX_PAIR_FREQUENCIES {
9759                return Err(invalid("pair frequency count exceeds its bound"));
9760            }
9761            pair_frequencies = Vec::with_capacity(count);
9762            for _ in 0..count {
9763                let first = cur.u16()?;
9764                let second = cur.u16()?;
9765                let first_at = first as usize;
9766                let second_at = second as usize;
9767                if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
9768                    return Err(invalid("pair frequency first column has no synopsis"));
9769                }
9770                let first_entries = entry_counts[first_at];
9771                if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
9772                    || dictionaries.get(second_at).copied().flatten().is_none()
9773                {
9774                    return Err(invalid("pair frequency second column has no stable dictionary"));
9775                }
9776                if pair_frequencies
9777                    .iter()
9778                    .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
9779                {
9780                    return Err(invalid("directory repeats a pair frequency summary"));
9781                }
9782                let omitted_max = cur.u64()?;
9783                if omitted_max > rows as u64 {
9784                    return Err(invalid("pair frequency omitted count exceeds the table"));
9785                }
9786                let entries_count = cur.u16()? as usize;
9787                if entries_count > FREQUENCY_ENTRIES {
9788                    return Err(invalid("pair frequency entry count exceeds its bound"));
9789                }
9790                let mut entries = Vec::with_capacity(entries_count);
9791                for _ in 0..entries_count {
9792                    let first_entry = cur.u16()?;
9793                    if first_entry as usize >= first_entries {
9794                        return Err(invalid("pair frequency anchor is outside its synopsis"));
9795                    }
9796                    let second = match cur.u8()? {
9797                        0 => None,
9798                        1 => Some(cur.u32()?),
9799                        _ => return Err(invalid("pair frequency string tag differs")),
9800                    };
9801                    let count = cur.u64()?;
9802                    if count == 0 || count > rows as u64 {
9803                        return Err(invalid("pair frequency count is outside the table"));
9804                    }
9805                    entries.push(PairFrequencyEntry { first_entry, second, count });
9806                }
9807                if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9808                    return Err(invalid("pair frequency entries are not descending"));
9809                }
9810                pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
9811            }
9812        } else if &tag == FREQUENCY_TEXTS {
9813            if seen_frequency_texts {
9814                return Err(invalid("directory names two frequency text blocks"));
9815            }
9816            seen_frequency_texts = true;
9817            let columns = cur.u16()? as usize;
9818            if columns > width {
9819                return Err(invalid("frequency text column count exceeds the schema"));
9820            }
9821            for _ in 0..columns {
9822                let column = cur.u16()? as usize;
9823                if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
9824                    return Err(invalid("frequency text column is repeated or out of range"));
9825                }
9826                if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
9827                    || dictionaries.get(column).copied().flatten().is_none()
9828                    || frequencies.get(column).and_then(Option::as_ref).is_none()
9829                {
9830                    return Err(invalid("frequency texts belong to a non-string synopsis"));
9831                }
9832                let count = cur.u16()? as usize;
9833                if count == 0 || count != entry_counts[column] {
9834                    return Err(invalid("frequency text count differs from its synopsis"));
9835                }
9836                let mut texts = Vec::with_capacity(count);
9837                for _ in 0..count {
9838                    texts.push(match cur.u8()? {
9839                        0 => None,
9840                        1 => {
9841                            let length = cur.u32()? as usize;
9842                            let bytes = cur.take(length)?.to_vec();
9843                            if fields[column].ty == LogicalType::Varchar {
9844                                std::str::from_utf8(&bytes)
9845                                    .map_err(|_| invalid("frequency text is not UTF-8"))?;
9846                            }
9847                            Some(bytes)
9848                        }
9849                        _ => return Err(invalid("frequency text tag differs")),
9850                    });
9851                }
9852                frequency_texts[column] = texts;
9853            }
9854        } else if &tag == HOST_GROUPS {
9855            if host_groups.is_some() {
9856                return Err(invalid("directory names two host group blocks"));
9857            }
9858            let column = cur.u16()? as usize;
9859            if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
9860                || dictionaries.get(column).copied().flatten().is_none()
9861            {
9862                return Err(invalid("host groups belong to a non-string dictionary"));
9863            }
9864            let omitted_max = cur.u64()?;
9865            if omitted_max > rows as u64 {
9866                return Err(invalid("host group bound exceeds the table"));
9867            }
9868            let count = cur.u16()? as usize;
9869            if count > host::CAPACITY {
9870                return Err(invalid("host group count exceeds its bound"));
9871            }
9872            let mut entries = Vec::with_capacity(count);
9873            let mut bytes = 0_usize;
9874            for _ in 0..count {
9875                let host_len = cur.u32()? as usize;
9876                bytes =
9877                    bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
9878                if bytes > host::BYTE_BUDGET {
9879                    return Err(invalid("host groups exceed their byte budget"));
9880                }
9881                let host = std::str::from_utf8(cur.take(host_len)?)
9882                    .map_err(|_| invalid("host is not UTF-8"))?
9883                    .to_owned();
9884                let count = cur.u64()?;
9885                if count == 0 || count > rows as u64 {
9886                    return Err(invalid("host group count exceeds the table"));
9887                }
9888                let bytes_sum = i128::from_le_bytes(
9889                    cur.take(16)?
9890                        .try_into()
9891                        .map_err(|_| invalid("host length sum is truncated"))?,
9892                );
9893                if bytes_sum < 0 {
9894                    return Err(invalid("host length sum is negative"));
9895                }
9896                let minimum_len = cur.u32()? as usize;
9897                bytes =
9898                    bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
9899                if bytes > host::BYTE_BUDGET {
9900                    return Err(invalid("host groups exceed their byte budget"));
9901                }
9902                let minimum = std::str::from_utf8(cur.take(minimum_len)?)
9903                    .map_err(|_| invalid("host minimum is not UTF-8"))?
9904                    .to_owned();
9905                entries.push(host::HostEntry { host, count, bytes_sum, minimum });
9906            }
9907            if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
9908                || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
9909            {
9910                return Err(invalid("host groups are not in certified order"));
9911            }
9912            host_groups = Some(host::HostSummary { column, omitted_max, entries });
9913        } else if &tag == CLUSTERING {
9914            if clustering.is_some() {
9915                return Err(invalid("directory names two clustering declarations"));
9916            }
9917            let bucket = Width::from_tag(cur.u8()?)
9918                .ok_or_else(|| invalid("clustering width tag differs"))?;
9919            let count = cur.u16()? as usize;
9920            let mut columns = Vec::with_capacity(count.min(fields.len()));
9921            for _ in 0..count {
9922                columns.push(u32::from(cur.u16()?));
9923            }
9924            // Through the constructor and not built by hand, so that a file claiming a column the
9925            // table does not have is caught at open rather than at the first scan that trusted it.
9926            clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
9927                invalid("stored clustering declaration does not match the table it is on")
9928            })?);
9929        } else if &tag == DEMOTED {
9930            if !demoted.is_empty() {
9931                return Err(invalid("directory names two demoted column blocks"));
9932            }
9933            let count = cur.u16()? as usize;
9934            if count == 0 || count > width {
9935                return Err(invalid("demoted column count is outside the schema"));
9936            }
9937            demoted = vec![false; width];
9938            for _ in 0..count {
9939                let column = cur.u16()? as usize;
9940                if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
9941                    return Err(invalid("a demoted column is repeated or has no dictionary"));
9942                }
9943                demoted[column] = true;
9944            }
9945        } else if &tag == SECTIONS {
9946            if seen_sections {
9947                return Err(invalid("directory names two section tables"));
9948            }
9949            seen_sections = true;
9950            generation = cur.u64()?;
9951            let count = cur.u16()? as usize;
9952            if count > MAX_SECTIONS {
9953                return Err(invalid("section count exceeds its bound"));
9954            }
9955            sections = Vec::with_capacity(count);
9956            // entry at a time: a malformed section entry is refused rather than turned into an
9957            // offset.
9958            for _ in 0..count {
9959                sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
9960            }
9961            for held in &sections {
9962                let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
9963                    return Err(invalid("a section's extent table overflows the file"));
9964                };
9965                // The bound check is here and not in `section`, because only the caller knows how
9966                // big the file is. A section pointing past the end is a torn directory, and reading
9967                // the payload it names would be reading whatever else is at that offset.
9968                if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
9969                    return Err(invalid("a section's extent table is outside the file"));
9970                }
9971                if held.extents == 0 && held.extent_bytes != 0 {
9972                    return Err(invalid("a section with no extents names an extent table"));
9973                }
9974            }
9975        } else if &tag == DICTIONARY_PAYLOADS {
9976            if seen_payloads {
9977                return Err(invalid("directory names two dictionary payload blocks"));
9978            }
9979            seen_payloads = true;
9980            let count = cur.u16()? as usize;
9981            if count != fields.len() {
9982                return Err(invalid("dictionary payload block does not match the table's columns"));
9983            }
9984            dictionary_payloads = Vec::with_capacity(count);
9985            for _ in 0..count {
9986                let bytes = cur.u64()?;
9987                if bytes > size {
9988                    return Err(invalid("a dictionary payload is larger than the file"));
9989                }
9990                dictionary_payloads.push(bytes);
9991            }
9992        } else {
9993            return Err(invalid("directory extension magic differs"));
9994        }
9995    }
9996    if !cur.done() {
9997        return Err(invalid("directory has trailing bytes"));
9998    }
9999    for stripe in &stripes {
10000        for (column, field) in fields.iter().enumerate() {
10001            if coded_type(&field.ty)
10002                && dictionaries[column].is_some()
10003                && stripe.memberships.get(column).is_none()
10004                && !demoted.get(column).copied().unwrap_or(false)
10005            {
10006                return Err(invalid("string page has no code membership index"));
10007            }
10008        }
10009    }
10010    Ok(Table {
10011        name,
10012        fields,
10013        stripes,
10014        rows,
10015        dictionaries,
10016        dictionary_payloads,
10017        demoted,
10018        distincts,
10019        frequencies,
10020        pair_frequencies,
10021        frequency_texts,
10022        host_groups,
10023        clustering,
10024        generation,
10025        sections,
10026    })
10027}
10028
10029/// A zone map's end, in the layout `rudb_common::bounds` defines. See [`Cursor::bound`].
10030fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
10031    bounds::put(out, bound)
10032}
10033
10034/// Which cascades are worth trying on a run of dictionary codes.
10035///
10036/// The exhaustive chooser encodes every candidate at every level of a cascade three deep and keeps
10037/// the smallest, which on a part of 1024 codes is around a hundred full encodes to decide something
10038/// three candidates were always going to win. It is the right default for a crate that does not
10039/// know what it is looking at. Here we do know. Codes are counted from zero in the order the values
10040/// were first seen, so a part of them is one value, or a narrow band, or a few long runs, and those
10041/// are constant, frame of reference and run length. Nothing else has ever come first on this data.
10042///
10043/// A dictionary of dictionary codes is the one candidate that can never pay, because the codes are
10044/// already the dictionary, and it is also the most expensive one to try. Below the top level the
10045/// streams are an RLE's run values and run lengths, which are integers in their own right with no
10046/// runs left in them, so only the two flat candidates go down there.
10047///
10048/// This is size given up for time on purpose, and the ablation is this chooser against
10049/// [`chooser::EXHAUSTIVE`] on the same file.
10050#[derive(Debug)]
10051struct Codes;
10052
10053impl chooser::Chooser for Codes {
10054    fn name(&self) -> &'static str {
10055        "codes"
10056    }
10057
10058    fn narrow_strings(
10059        &self,
10060        _values: &[&[u8]],
10061        offered: &[string::Kind],
10062        _depth: u8,
10063    ) -> Vec<string::Kind> {
10064        // Never reached, because nothing here encodes strings through the cascade. The trait asks
10065        // for it and the honest answer to a question we have no opinion on is the whole list.
10066        offered.to_vec()
10067    }
10068
10069    fn narrow_integers(
10070        &self,
10071        _values: &[i64],
10072        offered: &[integer::Kind],
10073        depth: u8,
10074    ) -> Vec<integer::Kind> {
10075        // The contract is a non empty subset, and a chunk that offers none of the three is a chunk
10076        // this has no opinion about rather than one that cannot be written.
10077        narrowed_to(Codes::keep(depth), offered)
10078    }
10079
10080    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10081        Codes::keep(depth).contains(&kind)
10082    }
10083}
10084
10085impl Codes {
10086    fn keep(depth: u8) -> &'static [integer::Kind] {
10087        if depth == 0 {
10088            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
10089        } else {
10090            &[integer::Kind::Constant, integer::Kind::Packed]
10091        }
10092    }
10093}
10094
10095/// The kinds of `offered` that are in `keep`, or all of `offered` when none of them are.
10096///
10097/// `Packed` applies to every chunk and both choosers keep it, so the fallback is never taken on a
10098/// chunk the cascade offers. It is there because the contract is a non empty subset and a chooser
10099/// that returned nothing would be a chunk that cannot be written. It is also why saying no to a kind
10100/// in `considers_integer` is safe: a kind that is never offered could only have been kept through
10101/// this fallback, and the fallback is never reached.
10102fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
10103    let narrowed: Vec<integer::Kind> =
10104        offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
10105    if narrowed.is_empty() { offered.to_vec() } else { narrowed }
10106}
10107
10108/// Which cascades are worth trying on a part of plain integers.
10109///
10110/// Wider than [`Codes`] because the values are not codes and carry whatever shape the column has.
10111/// A timestamp column climbs, so delta is the one that matters and is the reason this exists at
10112/// all: three timestamp columns in ClickBench were coming out at exactly eight bytes a row with
10113/// nothing asked of them. The same three columns are why the stride is here, since a timestamp
10114/// loaded from a source that recorded whole seconds is microseconds with twenty zero bits under
10115/// every value. A column that is one value with a handful of exceptions is sparse. What is still
10116/// left out is the dictionary, for the same reason as in [`Codes`]: it is the most
10117/// expensive candidate to try and this file already puts the columns that want one through a
10118/// dictionary of their own before they ever reach here.
10119#[derive(Debug)]
10120struct Fixed;
10121
10122impl chooser::Chooser for Fixed {
10123    fn name(&self) -> &'static str {
10124        "fixed"
10125    }
10126
10127    fn narrow_strings(
10128        &self,
10129        _values: &[&[u8]],
10130        offered: &[string::Kind],
10131        _depth: u8,
10132    ) -> Vec<string::Kind> {
10133        offered.to_vec()
10134    }
10135
10136    fn narrow_integers(
10137        &self,
10138        _values: &[i64],
10139        offered: &[integer::Kind],
10140        depth: u8,
10141    ) -> Vec<integer::Kind> {
10142        narrowed_to(Fixed::keep(depth), offered)
10143    }
10144
10145    fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10146        Fixed::keep(depth).contains(&kind)
10147    }
10148}
10149
10150impl Fixed {
10151    fn keep(depth: u8) -> &'static [integer::Kind] {
10152        if depth == 0 {
10153            &[
10154                integer::Kind::Constant,
10155                integer::Kind::Packed,
10156                integer::Kind::Delta,
10157                integer::Kind::Rle,
10158                integer::Kind::Sparse,
10159                integer::Kind::Strided,
10160            ]
10161        } else {
10162            &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
10163        }
10164    }
10165}
10166
10167/// Every value of an integer part as an `i64`, or `None` for a part this cannot widen without
10168/// losing one.
10169///
10170/// `UBIGINT` is the only integer type left out, because half its range does not fit and a page that
10171/// silently wrapped would be worse than a page that stays plain. Booleans and strings are not
10172/// integers and have their own ways of being small.
10173fn widened(data: &Data) -> Option<Vec<i64>> {
10174    match data {
10175        Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10176        Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10177        Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10178        Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10179        Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10180        Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10181        Data::Int64(values) => Some(values.to_vec()),
10182        _ => None,
10183    }
10184}
10185
10186/// A cascaded page decoded straight into the width the column is declared at.
10187///
10188/// A value that does not fit is a page that disagrees with the directory about what the column is,
10189/// which is a damaged file rather than a caller error, so it is refused rather than truncated. The
10190/// decoder does that check a block at a time where it can, see [`integer::decode_as`].
10191fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
10192    fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
10193        let values = integer::decode_as::<T>(bytes)
10194            .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
10195        if values.len() != rows {
10196            return Err(invalid("cascade page holds the wrong number of rows"));
10197        }
10198        Ok(values)
10199    }
10200    Ok(match ty {
10201        LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
10202        LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
10203        LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10204        LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
10205        LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10206        LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
10207        LogicalType::BigInt
10208        | LogicalType::Timestamp
10209        | LogicalType::Time
10210        | LogicalType::TimeTz
10211        | LogicalType::TimestampTz
10212        | LogicalType::TimestampS
10213        | LogicalType::TimestampMs
10214        | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10215        // A decimal is an integer of unscaled units, so the cascade reads back into whichever
10216        // integer the declared width says the column is stored as.
10217        LogicalType::Decimal { .. } => match ty.physical() {
10218            PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10219            PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10220            PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10221            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10222        },
10223        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10224    })
10225}
10226
10227/// How many bytes a part of this type costs written out plainly, which is what the cascade has to
10228/// beat before it is worth the decode.
10229fn plain_width(ty: &LogicalType) -> Option<usize> {
10230    Some(match ty {
10231        LogicalType::TinyInt | LogicalType::UTinyInt => 1,
10232        LogicalType::SmallInt | LogicalType::USmallInt => 2,
10233        LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
10234        LogicalType::BigInt
10235        | LogicalType::Timestamp
10236        | LogicalType::Time
10237        | LogicalType::TimeTz
10238        | LogicalType::TimestampTz
10239        | LogicalType::TimestampS
10240        | LogicalType::TimestampMs
10241        | LogicalType::TimestampNs => 8,
10242        LogicalType::Decimal { .. } => match ty.physical() {
10243            PhysicalType::Int16 => 2,
10244            PhysicalType::Int32 => 4,
10245            PhysicalType::Int64 => 8,
10246            // The widest decimals are stored as `i128`, which the cascade does not widen into, so
10247            // they take the plain path and there is nothing here to compare against.
10248            _ => return None,
10249        },
10250        _ => return None,
10251    })
10252}
10253
10254/// A part's plain integers through the cascade, or `None` when nothing it offers is worth it.
10255///
10256/// What it has to beat is whatever the page would otherwise have cost, which is the bit packed form
10257/// where there is one and the plain width where there is not. Both are cheaper to decode than a
10258/// cascade, so a tie goes to them.
10259fn cascaded(
10260    flat: &Vector,
10261    ty: &LogicalType,
10262    packed: Option<&Packed<'_>>,
10263    settling: &mut Settling,
10264) -> Result<Option<Vec<u8>>> {
10265    let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
10266    let Some(values) = widened(data) else { return Ok(None) };
10267    let plain = values.len().saturating_mul(width);
10268    let best = match packed {
10269        // The tag, the base, the word count and the words, which is what the codec 2 branch writes.
10270        Some(packed) => plain.min(21 + size_of_val(packed.words())),
10271        None => plain,
10272    };
10273    let out = settling.encode(&values)?;
10274    Ok((out.len() < best).then_some(out))
10275}
10276
10277/// How often the parts of one column in one stripe search the cascade again, in parts.
10278///
10279/// A stripe is 64 parts, so this is four searches a stripe where there were 64. The search is
10280/// what the cascade costs: on ClickBench `hits` the integer cascade was about a tenth of the load's
10281/// CPU and nearly all of it under `encode_pages`, trying six trees on every part to keep the one
10282/// the part before had kept.
10283const SEARCH_EVERY: usize = 16;
10284
10285/// What the parts of one column in one stripe have settled on in the integer cascade, and the
10286/// symbol table its text pages compress against.
10287///
10288/// One of these per column per stripe, used in part order, so what a part comes out as depends on
10289/// the stripe and not on which thread wrote it or on how many there were.
10290#[derive(Debug, Default)]
10291struct Settling {
10292    /// The shape of the last part that was searched, with what its top level offered, its length
10293    /// and its row count, which is the size a replay is held to.
10294    shape: Option<Shape>,
10295    /// Parts replayed since that search.
10296    since: usize,
10297    /// The FSST table of the last text page that trained one. See [`Settling::text`].
10298    symbols: Option<Symbols>,
10299}
10300
10301/// A table trained on one text page, with what that page came to and how many pages have used it
10302/// since.
10303#[derive(Debug)]
10304struct Symbols {
10305    shape: chooser::Settled,
10306    /// The trained page compressed and plain, in bytes, which is the ratio a later page is held to.
10307    /// Zero compressed when the table came out empty.
10308    len: usize,
10309    payload: usize,
10310    since: usize,
10311}
10312
10313impl Settling {
10314    /// A text page as one FSST chunk, against the table an earlier page of the stripe trained where
10315    /// there is one.
10316    ///
10317    /// The same rule as [`Self::encode`]: the table is used for [`SEARCH_EVERY`] pages and is kept
10318    /// while a page comes out no more than a quarter bigger a byte than the page it was trained on.
10319    /// Past that the page trains a table of its own and the pages after it use that one. Every page
10320    /// still carries the table it was compressed with, so nothing a reader does changes.
10321    fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
10322        if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
10323        {
10324            let out = string::encode_fsst(values, &symbols.shape)?;
10325            // An empty table stays empty for the pages after, which are the same kind of text.
10326            let held = match &out {
10327                None => symbols.len == 0,
10328                Some(out) => {
10329                    (out.len() as u128) * (symbols.payload as u128) * 4
10330                        <= (symbols.len as u128) * (payload as u128) * 5
10331                }
10332            };
10333            if held {
10334                symbols.since += 1;
10335                return Ok(out);
10336            }
10337        }
10338        let shape = string::fsst_shape(values);
10339        let out = string::encode_fsst(values, &shape)?;
10340        let len = out.as_ref().map_or(0, Vec::len);
10341        self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
10342        Ok(out)
10343    }
10344
10345    /// A part's integers through the cascade, replaying the settled shape where there is one.
10346    ///
10347    /// The replay is kept when it held and came out no more than a quarter bigger a row than the
10348    /// part the shape was searched on. Past that the column has changed under it and the part is
10349    /// searched. A replay that stopped fitting partway has already searched from where it stopped,
10350    /// so its shape is taken as the new one rather than searched a second time.
10351    fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
10352        if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
10353            let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
10354            let out = integer::encode_with(values, &replay)?;
10355            if !replay.held() {
10356                self.settle(&out, values.len(), replay.first_offered())?;
10357                return Ok(out);
10358            }
10359            let grown = (out.len() as u128) * (shape.rows as u128) * 4;
10360            if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
10361                self.since += 1;
10362                return Ok(out);
10363            }
10364        }
10365        // A replay of nothing is the search, and says what the top level offered on the way.
10366        let search = chooser::Replay::new(&[], &Fixed);
10367        let out = integer::encode_with(values, &search)?;
10368        self.settle(&out, values.len(), search.first_offered())?;
10369        Ok(out)
10370    }
10371
10372    fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
10373        let kinds = integer::shape(out)?;
10374        self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
10375        self.since = 0;
10376        Ok(())
10377    }
10378}
10379
10380/// A searched part's cascade, what its top level was offered, and what it came to.
10381#[derive(Debug)]
10382struct Shape {
10383    kinds: Vec<integer::Kind>,
10384    offered: Vec<integer::Kind>,
10385    len: usize,
10386    rows: usize,
10387}
10388
10389/// A part's dictionary codes through the integer cascade, or `None` when the cascade did not pay.
10390///
10391/// Until now this stream was a `u32` a row with nothing asked of it, and on ClickBench that was
10392/// 400,185,326 bytes for every one of the 28 varchar columns, the same count for `URL` as for a
10393/// column holding the empty string in nearly every row. Codes are dense integers counted from zero
10394/// and a part holds 1024 of them, which is the shape frame of reference is best at, and a column
10395/// with one value everywhere comes back a constant costing nothing per row rather than four bytes.
10396///
10397/// The result is taken only when it is smaller than the plain form. A cascade is allowed to come
10398/// out larger on a part whose codes are genuinely wide, `URL` has about sixty million distinct
10399/// values, and there is no reason to pay for the decode when it does.
10400/// A varchar page as one FSST layer, or `None` when it did not pay.
10401///
10402/// Until now a varchar page that neither the global dictionary nor the per page dictionary claimed
10403/// was written out raw: four bytes of offset a row and then the bytes. That is the right answer for
10404/// a page of values with nothing in common and the wrong one for a page of English, and a column of
10405/// comments is the case this exists for.
10406///
10407/// One layer and not the full string cascade, which is what the payload blocks of a global
10408/// dictionary go through. The cascade is a search: it encodes the page under every candidate it has
10409/// and recurses into the integer cascade for the lengths of each one, and on TPC-H `orders` that
10410/// took the write from 6.9 s to 48.3 s. It reads back no faster than the dictionary it replaced
10411/// either, 1.807 G instructions against 1.810 G for `select o_comment from orders`, because
10412/// unpicking a nest of layers a value at a time costs what the dictionary's payload block decode
10413/// cost. Raw pages of the same column read in 0.686 G, which says the whole of the difference is
10414/// what the page has to be put back together from.
10415///
10416/// FSST alone keeps most of what the cascade found and gives all of that back. Decoding it is one
10417/// pass over the payload into one buffer, the values are laid end to end in it the way the raw form
10418/// already lays them out, and what the reader hands a chunk is views over that buffer.
10419///
10420/// The page dictionary gets first refusal because it is cheaper still, and it wins on a page whose
10421/// values repeat. What is left for this is the page whose values mostly do not, which is exactly the
10422/// page that was being written raw.
10423///
10424/// Taken only when it comes out smaller than the raw form, so a page of incompressible values pays
10425/// nothing at read time for having been offered.
10426///
10427/// The symbol table is trained once for several pages of the stripe rather than once a page. See
10428/// [`Settling::text`].
10429fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
10430    let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
10431    let mut payload = 0_usize;
10432    for row in 0..flat.len() {
10433        // bytes_at: the rows were checked for UTF-8 on the way in, and checking them again here
10434        // was most of what the loop cost.
10435        let text = flat.bytes_at(row).unwrap_or(b"");
10436        payload = payload.saturating_add(text.len());
10437        values.push(text);
10438    }
10439    // What codec 0 writes for a varchar page: an offset a row and one more, then the payload.
10440    let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
10441    let Some(out) = settling.text(&values, payload)? else {
10442        return Ok(None);
10443    };
10444    Ok((out.len() < plain).then_some(out))
10445}
10446
10447fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
10448    let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
10449    let coded = integer::encode_with(&wide, &Codes)?;
10450    let plain = codes.len().saturating_mul(size_of::<u32>());
10451    Ok((coded.len() < plain).then_some(coded))
10452}
10453
10454/// The validity of a page, which is a flag and then, when some rows are null and some are not, a
10455/// bit a row with the valid ones set.
10456fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
10457    let flag = match flat.validity() {
10458        Validity::AllValid => 0,
10459        Validity::AllInvalid => 1,
10460        Validity::Mask(_) => 2,
10461    };
10462    out.push(flag);
10463    if flag == 2 {
10464        for group in (0..flat.len()).step_by(8) {
10465            let mut bits = 0_u8;
10466            for bit in 0..8 {
10467                if group + bit < flat.len() && !flat.is_null_at(group + bit) {
10468                    bits |= 1 << bit;
10469                }
10470            }
10471            out.push(bits);
10472        }
10473    }
10474}
10475
10476/// One part of a column coded against its global dictionary as a page, from the codes and the
10477/// validity [`push_validity`] wrote for it.
10478///
10479/// The codes go through the integer cascade when that comes out smaller than four bytes a code,
10480/// which on a column that repeats itself it nearly always does, and are written as they are when it
10481/// does not.
10482fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
10483    let coded = encoded_codes(codes)?;
10484    let mut out = Vec::with_capacity(
10485        1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
10486    );
10487    out.push(if coded.is_some() { 4 } else { 3 });
10488    out.extend_from_slice(validity);
10489    match coded {
10490        Some(coded) => out.extend_from_slice(&coded),
10491        None => {
10492            for &code in codes {
10493                put_u32(&mut out, code);
10494            }
10495        }
10496    }
10497    Ok(out)
10498}
10499
10500/// One part of one column as a page, for every column that is not coded against a global
10501/// dictionary. Those are built by [`coded_page`] from codes [`prepare`] handed out.
10502fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
10503    let ty = vector.logical_type();
10504    // flatten: the file writer needs a uniform scalar page and does it once per loaded chunk.
10505    let flat = vector.flatten()?;
10506    let mut out = Vec::new();
10507    let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
10508    let compressed_text = if dictionary.is_none() && coded_type(ty) {
10509        text_compressed(&flat, settling)?
10510    } else {
10511        None
10512    };
10513    let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
10514    let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
10515    // Only where nothing else has claimed the page, which is the plain integer case. A packed part
10516    // is still on the table because the cascade has to beat it too: the bit pack takes a part only
10517    // when it halves it, so a column that shrinks by a third was coming out whole.
10518    let cascade =
10519        if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
10520    out.push(if cascade.is_some() {
10521        5
10522    } else if dictionary.is_some() {
10523        1
10524    } else if compressed_text.is_some() {
10525        6
10526    } else if packed.is_some() {
10527        2
10528    } else {
10529        0
10530    });
10531    push_validity(&mut out, &flat);
10532    if let Some(cascade) = cascade {
10533        out.extend_from_slice(&cascade);
10534        return Ok(out);
10535    }
10536    if let Some(dictionary) = dictionary {
10537        out.extend_from_slice(&dictionary);
10538        return Ok(out);
10539    }
10540    if let Some(compressed_text) = compressed_text {
10541        out.extend_from_slice(&compressed_text);
10542        return Ok(out);
10543    }
10544    if let Some(packed) = packed {
10545        if packed.offset() != 0 {
10546            return Err(invalid("writer received a sliced packed vector"));
10547        }
10548        out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
10549        out.extend_from_slice(&packed.base().to_le_bytes());
10550        put_u32(
10551            &mut out,
10552            u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
10553        );
10554        for word in packed.words() {
10555            put_u64(&mut out, *word);
10556        }
10557        return Ok(out);
10558    }
10559    let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
10560    match (ty, data) {
10561        (LogicalType::TinyInt, Data::Int8(values)) => {
10562            for value in &**values {
10563                out.extend_from_slice(&value.to_le_bytes());
10564            }
10565        }
10566        (LogicalType::UTinyInt, Data::UInt8(values)) => {
10567            for value in &**values {
10568                out.extend_from_slice(&value.to_le_bytes());
10569            }
10570        }
10571        (LogicalType::SmallInt, Data::Int16(values)) => {
10572            for value in &**values {
10573                out.extend_from_slice(&value.to_le_bytes());
10574            }
10575        }
10576        (LogicalType::USmallInt, Data::UInt16(values)) => {
10577            for value in &**values {
10578                out.extend_from_slice(&value.to_le_bytes());
10579            }
10580        }
10581        (LogicalType::UInteger, Data::UInt32(values)) => {
10582            for value in &**values {
10583                out.extend_from_slice(&value.to_le_bytes());
10584            }
10585        }
10586        (LogicalType::UBigInt, Data::UInt64(values)) => {
10587            for value in &**values {
10588                out.extend_from_slice(&value.to_le_bytes());
10589            }
10590        }
10591        (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
10592            for value in &**values {
10593                out.extend_from_slice(&value.to_le_bytes());
10594            }
10595        }
10596        (
10597            LogicalType::BigInt
10598            | LogicalType::Timestamp
10599            | LogicalType::Time
10600            | LogicalType::TimeTz
10601            | LogicalType::TimestampTz
10602            | LogicalType::TimestampS
10603            | LogicalType::TimestampMs
10604            | LogicalType::TimestampNs,
10605            Data::Int64(values),
10606        ) => {
10607            for value in &**values {
10608                out.extend_from_slice(&value.to_le_bytes());
10609            }
10610        }
10611        // A hugeint and a uuid are both the 128 bit lane, and a uuid's bits are the ones the rest of
10612        // the engine already carries it in, so nothing about the value changes on the way down.
10613        (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
10614            for value in &**values {
10615                out.extend_from_slice(&value.to_le_bytes());
10616            }
10617        }
10618        (LogicalType::UHugeInt, Data::UInt128(values)) => {
10619            for value in &**values {
10620                out.extend_from_slice(&value.to_le_bytes());
10621            }
10622        }
10623        // Plainly, in the IEEE bytes. The integer encodings do not apply to a float and none of the
10624        // float codecs is worth having before somebody has measured a corpus of them.
10625        (LogicalType::Float, Data::Float32(values)) => {
10626            for value in &**values {
10627                out.extend_from_slice(&value.to_le_bytes());
10628            }
10629        }
10630        (LogicalType::Double, Data::Float64(values)) => {
10631            for value in &**values {
10632                out.extend_from_slice(&value.to_le_bytes());
10633            }
10634        }
10635        // Three counts and not one number. Months, days and microseconds stay apart on disk because
10636        // they are apart in the value: a month is not a fixed number of days and a day is not a
10637        // fixed number of microseconds, which is the whole reason the type has three fields.
10638        (LogicalType::Interval, Data::Interval(values)) => {
10639            for (months, days, micros) in &**values {
10640                out.extend_from_slice(&months.to_le_bytes());
10641                out.extend_from_slice(&days.to_le_bytes());
10642                out.extend_from_slice(&micros.to_le_bytes());
10643            }
10644        }
10645        (LogicalType::Boolean, Data::Bool(values)) => {
10646            for value in &**values {
10647                out.push(u8::from(*value));
10648            }
10649        }
10650        // The unscaled integer and nothing else. Scale is a property of the column and it is in the
10651        // directory already, so writing it a value at a time would be paying for it twice.
10652        (LogicalType::Decimal { .. }, Data::Int16(values)) => {
10653            for value in &**values {
10654                out.extend_from_slice(&value.to_le_bytes());
10655            }
10656        }
10657        (LogicalType::Decimal { .. }, Data::Int32(values)) => {
10658            for value in &**values {
10659                out.extend_from_slice(&value.to_le_bytes());
10660            }
10661        }
10662        (LogicalType::Decimal { .. }, Data::Int64(values)) => {
10663            for value in &**values {
10664                out.extend_from_slice(&value.to_le_bytes());
10665            }
10666        }
10667        (LogicalType::Decimal { .. }, Data::Int128(values)) => {
10668            for value in &**values {
10669                out.extend_from_slice(&value.to_le_bytes());
10670            }
10671        }
10672        // A blob and a bit string go down the way a varchar does, because the layout is the same
10673        // one: an offset a value and then the bytes. What is not the same is that nothing here may
10674        // read the payload as text, which is why this arm asks the column for bytes rather than for
10675        // a string, and why the codecs above that do read text are all asked of a varchar by name.
10676        (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
10677            let mut bytes = Vec::new();
10678            put_u32(&mut out, 0);
10679            for row in 0..vector.len() {
10680                let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
10681                bytes.extend_from_slice(value);
10682                put_u32(
10683                    &mut out,
10684                    u32::try_from(bytes.len())
10685                        .map_err(|_| invalid("string payload exceeds 4GiB"))?,
10686                );
10687            }
10688            out.extend_from_slice(&bytes);
10689        }
10690        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10691    }
10692    Ok(out)
10693}
10694
10695fn put_varint(out: &mut Vec<u8>, mut value: u32) {
10696    while value >= 0x80 {
10697        out.push((value as u8 & 0x7f) | 0x80);
10698        value >>= 7;
10699    }
10700    out.push(value as u8);
10701}
10702
10703/// The distinct codes of one part, which is what a stripe's membership index is merged from.
10704fn unique_codes(codes: &[u32]) -> Vec<u32> {
10705    let mut unique = codes.to_vec();
10706    unique.sort_unstable();
10707    unique.dedup();
10708    unique
10709}
10710
10711/// The union of the sorted distinct codes of every part in a stripe.
10712///
10713/// Pairwise up a tree rather than one long list concatenated and sorted. Both are the same order of
10714/// work on paper and the tree is the one that does not sort what is already in order: sixty four
10715/// sorted lists become one in six passes over the values.
10716fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
10717    let mut lists = lists;
10718    while lists.len() > 1 {
10719        let mut next = Vec::with_capacity(lists.len().div_ceil(2));
10720        for pair in lists.chunks(2) {
10721            match pair {
10722                [left, right] => next.push(merged_pair(left, right)),
10723                [only] => next.push(only.clone()),
10724                _ => {}
10725            }
10726        }
10727        lists = next;
10728    }
10729    lists.pop().unwrap_or_default()
10730}
10731
10732fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
10733    let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
10734    let mut at = 0;
10735    let mut to = 0;
10736    while at < left.len() && to < right.len() {
10737        match left[at].cmp(&right[to]) {
10738            Ordering::Less => {
10739                out.push(left[at]);
10740                at += 1;
10741            }
10742            Ordering::Greater => {
10743                out.push(right[to]);
10744                to += 1;
10745            }
10746            Ordering::Equal => {
10747                out.push(left[at]);
10748                at += 1;
10749                to += 1;
10750            }
10751        }
10752    }
10753    out.extend_from_slice(&left[at..]);
10754    out.extend_from_slice(&right[to..]);
10755    out
10756}
10757
10758/// The widest bounds and the total null count of a stripe, from the bounds of its parts.
10759///
10760/// A bound that is missing from any part is missing from the stripe, because a missing bound means
10761/// nothing is known and a stripe that holds an unknown cannot claim one.
10762fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
10763    let mut merged = Range::default();
10764    let mut first = true;
10765    for range in ranges {
10766        merged.nulls = merged.nulls.saturating_add(range.nulls);
10767        // Both of these have to survive every part, so one part that could not say anything makes
10768        // the stripe unable to say it either. A sum is dropped on overflow rather than wrapped,
10769        // which leaves the stripe with exact ends and no total, which is a true thing to say.
10770        merged.sum = match (merged.sum.take(), range.sum) {
10771            (Some(held), Some(next)) if !first => held.checked_add(next),
10772            (_, next) if first => next,
10773            _ => None,
10774        };
10775        merged.exact = if first { range.exact } else { merged.exact && range.exact };
10776        if first {
10777            merged.low = range.low;
10778            merged.high = range.high;
10779            first = false;
10780            continue;
10781        }
10782        merged.low = match (merged.low.take(), range.low) {
10783            (Some(held), Some(next)) => Some(held.smaller(next)),
10784            _ => None,
10785        };
10786        merged.high = match (merged.high.take(), range.high) {
10787            (Some(held), Some(next)) => Some(held.larger(next)),
10788            _ => None,
10789        };
10790    }
10791    merged
10792}
10793
10794/// One stripe's sieves for one column: the part count, a length for each part, then their bytes.
10795///
10796/// One page for the whole stripe rather than one per part, because a part's sieve is a few hundred
10797/// bytes and sixty four of those are sixty four directory entries and sixty four reads for something
10798/// a scan walks straight through. A part with no sieve writes a length of zero and costs four bytes.
10799/// `bound` cut down to [`PART_BOUND_BYTES`], still a bound of the side it was.
10800///
10801/// A prefix of a string sorts at or before the string, so cutting one down leaves a low end that is
10802/// still a low end. A high end has to go the other way, so the cut prefix is stepped up at the last
10803/// byte that can carry it, and a prefix of nothing but `0xFF` has no such byte and gives up the
10804/// bound rather than claiming one that is too small. Anything that is not a string is already a
10805/// fixed width and is left alone.
10806fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
10807    match bound {
10808        Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
10809            value.truncate(PART_BOUND_BYTES);
10810            if !high {
10811                return Some(Bound::Bytes(value));
10812            }
10813            while let Some(last) = value.pop() {
10814                if last < u8::MAX {
10815                    value.push(last + 1);
10816                    return Some(Bound::Bytes(value));
10817                }
10818            }
10819            None
10820        }
10821        other => other,
10822    }
10823}
10824
10825/// The ranges of one column's parts of one stripe, as a page.
10826///
10827/// The two ends and the null count, and not `exact` or the total. Those two answer a `MIN` or a
10828/// `SUM` out of the directory, and the directory already answers those per stripe, where the same
10829/// number costs sixty times less to keep. What a part range is for is skipping the part, and
10830/// skipping needs the ends. So a range read back from here says it is not exact, which is true of a
10831/// string end that was cut down anyway.
10832fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
10833    let mut out = Vec::new();
10834    put_u32(
10835        &mut out,
10836        u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10837    );
10838    for range in ranges {
10839        put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
10840        put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
10841        put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
10842    }
10843    Ok(out)
10844}
10845
10846/// The ranges one encoded page holds, one entry per part of the stripe.
10847fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
10848    let mut cur = Cursor::new(bytes);
10849    let parts = cur.u32()? as usize;
10850    let mut out = Vec::new();
10851    for _ in 0..parts {
10852        let low = cur.bound()?;
10853        let high = cur.bound()?;
10854        let nulls = cur.u32()? as usize;
10855        out.push(Range { low, high, nulls, exact: false, sum: None });
10856    }
10857    Ok(out)
10858}
10859
10860fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
10861    let held: Vec<&Option<Sieve>> = sieves.collect();
10862    let mut out = Vec::new();
10863    put_u32(
10864        &mut out,
10865        u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10866    );
10867    for sieve in &held {
10868        let length = sieve.as_ref().map_or(0, Sieve::len);
10869        put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
10870    }
10871    // flatten: a part with no sieve wrote a length of zero above and contributes no bytes here.
10872    for sieve in held.into_iter().flatten() {
10873        out.extend_from_slice(&sieve.to_bytes());
10874    }
10875    Ok(out)
10876}
10877
10878/// The sieves one encoded page holds, one entry per part of the stripe.
10879///
10880/// A part whose bytes are not a sieve this version understands comes back as `None`, which is a part
10881/// that gets read. That is how a file written by a later version of the sieve stays readable rather
10882/// than being a corrupt page.
10883fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
10884    let parts = u32::from_le_bytes(
10885        bytes
10886            .get(..4)
10887            .ok_or_else(|| invalid("sieve page is truncated"))?
10888            .try_into()
10889            .map_err(|_| invalid("sieve page is truncated"))?,
10890    ) as usize;
10891    let mut lengths = Vec::with_capacity(parts);
10892    for part in 0..parts {
10893        let at = 4 + part * 4;
10894        let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
10895        lengths.push(u32::from_le_bytes(
10896            field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
10897        ) as usize);
10898    }
10899    let mut at = 4 + parts * 4;
10900    let mut out = Vec::with_capacity(parts);
10901    for length in lengths {
10902        if length == 0 {
10903            out.push(None);
10904            continue;
10905        }
10906        let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
10907        let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
10908        out.push(Sieve::from_bytes(field));
10909        at = end;
10910    }
10911    if at != bytes.len() {
10912        return Err(invalid("sieve page has trailing bytes"));
10913    }
10914    Ok(out)
10915}
10916
10917/// One stripe's membership index: the code count and then the codes as ascending deltas.
10918///
10919/// The codes have to be sorted and distinct already, which is what [`unique_codes`] and
10920/// [`merged_codes`] hand over. Anything else decodes as different codes, so neither of those two is
10921/// a step a caller can skip.
10922fn encode_membership(unique: &[u32]) -> Vec<u8> {
10923    let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
10924    put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
10925    let mut previous = 0;
10926    for (at, &code) in unique.iter().enumerate() {
10927        put_varint(&mut out, if at == 0 { code } else { code - previous });
10928        previous = code;
10929    }
10930    out
10931}
10932
10933fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
10934    let mut value = 0_u32;
10935    for shift in (0..35).step_by(7) {
10936        let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
10937        *at += 1;
10938        let part = u32::from(byte & 0x7f);
10939        if shift == 28 && part > 0x0f {
10940            return Err(invalid("membership varint overflow"));
10941        }
10942        value = value
10943            .checked_add(
10944                part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
10945            )
10946            .ok_or_else(|| invalid("membership varint overflow"))?;
10947        if byte & 0x80 == 0 {
10948            return Ok(value);
10949        }
10950    }
10951    Err(invalid("membership varint is too long"))
10952}
10953
10954fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
10955    let mut at = 0;
10956    let count = take_varint(bytes, &mut at)? as usize;
10957    let mut codes = Vec::with_capacity(count);
10958    let mut previous = 0_u32;
10959    for index in 0..count {
10960        let delta = take_varint(bytes, &mut at)?;
10961        let code = if index == 0 {
10962            delta
10963        } else {
10964            previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
10965        };
10966        if index > 0 && code <= previous {
10967            return Err(invalid("membership codes are not increasing"));
10968        }
10969        codes.push(code);
10970        previous = code;
10971    }
10972    if at != bytes.len() {
10973        return Err(invalid("membership page has trailing bytes"));
10974    }
10975    Ok(codes)
10976}
10977
10978/// A varchar page as a dictionary of its distinct values and a code a row, or `None` when that does
10979/// not come out smaller than the raw form.
10980///
10981/// Every text page that no global dictionary claims asks this first, including the page of
10982/// comments that never has a repeat, so the map is hashed with [`Spread`] rather than SipHash and
10983/// sized for the page up front. With the default hasher and growth it was 4% of the instructions of
10984/// a `lineitem` load from CSV, all of it on `l_comment` pages this then refused.
10985fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
10986    let mut by_text: HashMap<&[u8], u32, Spread> =
10987        HashMap::with_capacity_and_hasher(vector.len(), Spread);
10988    let mut values = Vec::new();
10989    let mut codes = Vec::with_capacity(vector.len());
10990    let mut plain_bytes = 0_usize;
10991    for row in 0..vector.len() {
10992        let text = vector.bytes_at(row).unwrap_or(b"");
10993        plain_bytes = plain_bytes.saturating_add(text.len());
10994        let code = match by_text.get(text) {
10995            Some(&code) => code,
10996            None => {
10997                let code = u32::try_from(values.len())
10998                    .map_err(|_| invalid("too many dictionary values"))?;
10999                by_text.insert(text, code);
11000                values.push(text);
11001                code
11002            }
11003        };
11004        codes.push(code);
11005    }
11006    let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
11007    let encoded = 8_usize
11008        .saturating_add((values.len() + 1).saturating_mul(4))
11009        .saturating_add(dictionary_bytes)
11010        .saturating_add(codes.len().saturating_mul(4));
11011    let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
11012    if encoded >= plain {
11013        return Ok(None);
11014    }
11015    let mut out = Vec::with_capacity(encoded);
11016    put_u32(
11017        &mut out,
11018        u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
11019    );
11020    put_u32(
11021        &mut out,
11022        u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
11023    );
11024    let mut offset = 0_u32;
11025    put_u32(&mut out, offset);
11026    for value in &values {
11027        offset = offset
11028            .checked_add(
11029                u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
11030            )
11031            .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
11032        put_u32(&mut out, offset);
11033    }
11034    for value in values {
11035        out.extend_from_slice(value);
11036    }
11037    for code in codes {
11038        put_u32(&mut out, code);
11039    }
11040    Ok(Some(out))
11041}
11042
11043/// The room one closing column takes under [`CLOSE_BYTES`], given back when dropped.
11044struct Room<'a, T> {
11045    state: &'a Mutex<(T, usize)>,
11046    finished: &'a Condvar,
11047    bytes: usize,
11048}
11049
11050impl<T> Drop for Room<'_, T> {
11051    fn drop(&mut self) {
11052        let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
11053        held.1 -= self.bytes;
11054        drop(held);
11055        self.finished.notify_all();
11056    }
11057}
11058
11059/// One column's work at the end of a load, as [`Writer::close_columns`] schedules it.
11060enum Closing<'a> {
11061    /// A numeric column's frequencies, whether to count its distinct values exactly, and the
11062    /// range to count them in a flat array when it is short enough.
11063    Numeric {
11064        column: usize,
11065        counted: bool,
11066        dense: Option<(u64, usize)>,
11067    },
11068    Dictionary {
11069        index: usize,
11070        dictionary: &'a GlobalDictionary,
11071    },
11072}
11073
11074/// What one [`Closing`] came back with, by column.
11075enum Closed {
11076    Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
11077    Dictionary(usize, ClosedDictionary),
11078}
11079
11080/// What [`Writer::close_dictionary`] builds for one column and [`Writer::close`] writes.
11081struct ClosedDictionary {
11082    /// `None` for a demoted dictionary, which holds only some of the column. See [`DEMOTED`].
11083    distinct: Option<u64>,
11084    frequencies: Option<FrequencySummary>,
11085    texts: Vec<Option<Vec<u8>>>,
11086    hosts: Option<host::HostSummary>,
11087    encoded: EncodedDictionary,
11088    /// The bytes of the column's payload blocks, which are already in the file.
11089    payload: u64,
11090}
11091
11092struct EncodedDictionary {
11093    index: Vec<u8>,
11094    ranks: Vec<u8>,
11095    grams: Vec<u8>,
11096}
11097
11098/// Sorts codes into the byte order of the values they name, eight bytes of depth at a time.
11099///
11100/// # What the shape of the data does to a comparison sort
11101///
11102/// Distinct values against distinct prefixes, on the eight million row `hits`:
11103///
11104/// ```text
11105///   distinct   first 8   first 16   first 32   column
11106///  2,266,417        50      8,892    232,630   URL
11107///  2,346,025        49      8,534    204,060   Referer
11108///  1,357,764    81,362    348,340    861,579   Title
11109/// ```
11110///
11111/// Two and a quarter million URLs have fifty distinct first eight bytes between them, because they
11112/// all begin `http://` and then a host and there are not many hosts. So a sort that leads with
11113/// those eight bytes settles almost nothing on `URL` and `Referer`, whatever the comment on it used
11114/// to say, and almost every pair falls through to a comparison of whole values that agree for most
11115/// of their length. `Title` is free text and separates at eight bytes, which is why the design
11116/// looked right when it was written.
11117///
11118/// # What is done about it
11119///
11120/// Sort on eight bytes of the value at the current depth, held beside the code, and then take each
11121/// run that those eight bytes leave tied and sort it again on the next eight. A value is fetched
11122/// from the payload once per eight bytes of depth rather than once per comparison, and the sort
11123/// itself runs over an array of integers that is in cache rather than over pointers into a payload
11124/// that is hundreds of megabytes.
11125///
11126/// That is the whole trick, and it matters because the payload touch is the expensive part. The
11127/// bytes themselves are nearly free once the line is in cache, so reading eight at a time and
11128/// throwing away the ones that were not needed beats going back for each one.
11129///
11130/// # Why the length has to be carried
11131///
11132/// The eight bytes are padded with zero when the value has fewer than eight left, and a zero byte
11133/// can appear in a value, so equal keys do not mean equal bytes. What is true is that a value which
11134/// ran out inside the window is a prefix of any other value with the same key, and a prefix sorts
11135/// first, so how many of the eight bytes were real is the tie break and nothing further is needed.
11136/// A run is only worth another pass when all eight were real, because otherwise the run is one
11137/// value: a dictionary holds a value once.
11138fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
11139    let mut work = vec![(0, codes.len(), 0)];
11140    let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
11141    while let Some((from, to, depth)) = work.pop() {
11142        let part = &mut codes[from..to];
11143        keyed.clear();
11144        keyed.extend(part.iter().map(|&code| {
11145            let value = values(code);
11146            let rest = value.get(depth..).unwrap_or_default();
11147            (head(rest), rest.len().min(8) as u8, code)
11148        }));
11149        keyed.sort_unstable();
11150        for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
11151            *slot = entry.2;
11152        }
11153        let mut start = 0;
11154        while start < keyed.len() {
11155            let (key, taken, _) = keyed[start];
11156            let mut end = start + 1;
11157            while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
11158                end += 1;
11159            }
11160            if taken == 8 && end - start > 1 {
11161                work.push((from + start, from + end, depth + 8));
11162            }
11163            start = end;
11164        }
11165    }
11166}
11167
11168/// How few codes are worth sorting on more than one thread.
11169const PARALLEL_SORT_MIN: usize = 1 << 16;
11170
11171/// How many buckets a thread gets in [`sort_by_value_across`], so that a thread that drew a slow
11172/// bucket is not what the others wait for.
11173const BUCKETS_PER_WORKER: usize = 4;
11174
11175/// How many sampled codes stand for each bucket when the splitters are picked.
11176const SAMPLES_PER_BUCKET: usize = 32;
11177
11178/// [`sort_by_value`] over `workers` threads, with the same answer.
11179///
11180/// A sample sort. A sample of the codes is sorted and cut into as many equal runs as there are
11181/// buckets, and the values at the cuts are the splitters. Every code goes to the bucket its value
11182/// falls in by a binary search of the splitters, the buckets are laid end to end in splitter order,
11183/// and each bucket is then sorted on its own by whichever thread takes it. Every value in a bucket
11184/// sorts after every value in the bucket before, so the buckets sorted one by one are the codes
11185/// sorted.
11186///
11187/// The answer is the one [`sort_by_value`] gives down to the order of equal values, not only the
11188/// order of different ones. A global dictionary holds each value once, so there are none, but the
11189/// sort does not rely on it: equal values land in the same bucket in code order, which is the order
11190/// [`sort_by_value`] leaves them in, since the code is the last thing it sorts on.
11191///
11192/// On the 10m ClickBench sample the close sorts five columns of one to three and a half million
11193/// distinct values, one column at a time, and until this each sort ran on one thread while the
11194/// other thirty one waited for it.
11195fn sort_by_value_across<'a>(
11196    codes: &mut [u32],
11197    values: impl Fn(u32) -> &'a [u8] + Sync,
11198    workers: usize,
11199) {
11200    if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
11201        sort_by_value(codes, values);
11202        return;
11203    }
11204    let buckets = workers * BUCKETS_PER_WORKER;
11205    let wanted = buckets * SAMPLES_PER_BUCKET;
11206    let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
11207    sort_by_value(&mut sample, &values);
11208    let splitters =
11209        (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
11210    let values = &values;
11211    let splitters = &splitters;
11212    let per = codes.len().div_ceil(workers);
11213    // Which bucket each code goes to, a run of the codes per thread.
11214    let places = std::thread::scope(|scope| {
11215        codes
11216            .chunks(per)
11217            .map(|run| {
11218                scope.spawn(move || {
11219                    run.iter()
11220                        .map(|&code| {
11221                            let value = values(code);
11222                            splitters.partition_point(|splitter| *splitter <= value) as u32
11223                        })
11224                        .collect::<Vec<_>>()
11225                })
11226            })
11227            .collect::<Vec<_>>()
11228            .into_iter()
11229            .flat_map(|handle| {
11230                handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
11231            })
11232            .collect::<Vec<_>>()
11233    });
11234    let mut starts = vec![0_usize; buckets + 1];
11235    for &place in &places {
11236        starts[place as usize + 1] += 1;
11237    }
11238    for bucket in 0..buckets {
11239        starts[bucket + 1] += starts[bucket];
11240    }
11241    let mut laid = vec![0_u32; codes.len()];
11242    let mut next = starts.clone();
11243    for (&code, &place) in codes.iter().zip(&places) {
11244        laid[next[place as usize]] = code;
11245        next[place as usize] += 1;
11246    }
11247    drop(places);
11248    let mut runs = Vec::with_capacity(buckets);
11249    let mut rest = laid.as_mut_slice();
11250    for bucket in 0..buckets {
11251        let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
11252        runs.push(run);
11253        rest = after;
11254    }
11255    // The largest buckets first, since they are taken from the back.
11256    runs.sort_by_key(|run| run.len());
11257    let queue = Mutex::new(runs);
11258    std::thread::scope(|scope| {
11259        for _ in 0..workers {
11260            scope.spawn(|| {
11261                loop {
11262                    let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
11263                    let Some(run) = taken else { break };
11264                    sort_by_value(run, values);
11265                }
11266            });
11267        }
11268    });
11269    codes.copy_from_slice(&laid);
11270}
11271
11272/// The first eight bytes of a value as an integer that sorts the way the bytes sort.
11273fn head(bytes: &[u8]) -> u64 {
11274    let mut word = [0; 8];
11275    let take = bytes.len().min(8);
11276    word[..take].copy_from_slice(&bytes[..take]);
11277    u64::from_be_bytes(word)
11278}
11279
11280/// One column's dictionary page, which is its index and its sorted order.
11281///
11282/// The payload is not in it. Its blocks are in the file already, written as each was encoded, and
11283/// `places` says where, in block order. With `scattered` set the index records each block's start
11284/// and length, so a reader can find one wherever it went.
11285///
11286/// `scattered` false lays the blocks out the way a file written before format 26 has them, one
11287/// behind the next with only the ends recorded. Nothing in the writer asks for that any more. It is
11288/// kept because [`open_global_dictionary`] still reads those files and a reading path that nothing
11289/// can produce is a reading path nothing tests.
11290fn encode_global_dictionary(
11291    dictionary: &GlobalDictionary,
11292    order: &[(u64, u32)],
11293    places: &[Placed],
11294    scattered: bool,
11295) -> Result<EncodedDictionary> {
11296    let values = dictionary.values();
11297    if order.len() != values {
11298        return Err(invalid("global dictionary order does not cover its values"));
11299    }
11300    let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
11301    if places.len() != blocks {
11302        return Err(invalid("global dictionary payload is not the blocks it says it is"));
11303    }
11304    if dictionary.grams.len() != blocks {
11305        return Err(invalid("global dictionary signatures do not cover its blocks"));
11306    }
11307    let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
11308    let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
11309    let offset_bits = offset_width(&dictionary.ends);
11310    let payload_words = if scattered { 3 } else { 2 };
11311    let index_len = DICTIONARY_HEADER
11312        .checked_add(offset_bytes(values, offset_bits))
11313        .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
11314        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11315        .and_then(|len| len.checked_add(8))
11316        .ok_or_else(|| invalid("global dictionary index length overflow"))?;
11317    let mut index = Vec::with_capacity(index_len);
11318    put_u32(
11319        &mut index,
11320        u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
11321    );
11322    put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
11323    put_u32(
11324        &mut index,
11325        u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
11326    );
11327    let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
11328        | DICTIONARY_GRAMS
11329        | DICTIONARY_WIDE_GRAMS;
11330    put_u32(&mut index, offset_bits as u32 | flag);
11331    encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
11332    // Where each block is and how long it is, so a reader can find one. The stored blocks are
11333    // shorter than the decoded ones and by a different amount each, so their lengths are the one
11334    // thing the offsets above no longer say, and where they start is no longer arithmetic on the
11335    // block before once a block is written the moment it is encoded.
11336    let mut end = 0_u64;
11337    for place in places {
11338        if scattered {
11339            put_u64(&mut index, place.start);
11340            put_u64(&mut index, place.length);
11341        } else {
11342            end = end
11343                .checked_add(place.length)
11344                .ok_or_else(|| invalid("global dictionary payload overflow"))?;
11345            put_u64(&mut index, end);
11346        }
11347    }
11348    for place in places {
11349        put_u64(&mut index, place.hash);
11350    }
11351    // The same two lists for the sorted order. A rank block is packed at whatever width its own
11352    // heads need, so where one ends is no longer arithmetic on the block number.
11353    if rank_ends.len() != rank_blocks {
11354        return Err(invalid("global dictionary order is not the blocks it says it is"));
11355    }
11356    for end in &rank_ends {
11357        put_u64(&mut index, *end);
11358    }
11359    let mut at = 0_usize;
11360    for end in &rank_ends {
11361        let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
11362        put_u64(&mut index, checksum(&ranks[at..end]));
11363        at = end;
11364    }
11365    let gram_len = blocks
11366        .checked_mul(TEXT_GRAM_BYTES)
11367        .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
11368    let mut grams = Vec::with_capacity(gram_len);
11369    for block in &dictionary.grams {
11370        grams.extend_from_slice(block);
11371    }
11372    put_u64(&mut index, checksum(&grams));
11373    if index.len() != index_len {
11374        return Err(invalid("global dictionary index is not the length it was laid out for"));
11375    }
11376    Ok(EncodedDictionary { index, ranks, grams })
11377}
11378
11379/// How many blocks of the payload the shape is settled on.
11380///
11381/// Eight blocks is 8,192 values, which is the sample `chooser::Sampled` draws and is that size for
11382/// the same reason. They are spread across the dictionary rather than taken off the front, because
11383/// a dictionary is in the order values were first seen and the front of it is the first morsel of
11384/// the load.
11385const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
11386
11387/// The shapes the payload encoder picks between.
11388///
11389/// Narrow on purpose. The exhaustive search encodes every candidate at every level and runs at two
11390/// to six megabytes a second on this data, which over the twelve gigabytes of dictionary `hits`
11391/// carries is about an hour of processor time, so it cannot be what a load does. Each of these
11392/// settles the outer level and the one below it, which is where almost all of that hour goes, and
11393/// leaves the levels under them to the exhaustive search where the chunks are small enough for it
11394/// to cost nothing.
11395///
11396/// Measured on the five ClickBench columns that have a dictionary worth the name, at 1,024 values a
11397/// block, against the exhaustive search over the same blocks:
11398///
11399/// | column | exhaustive | FRONT then LZ | LZ then FSST | LZ then PLAIN |
11400/// |---|---|---|---|---|
11401/// | 2 | 2.923 at 4.3 MB/s | 2.587 at 21.2 | 2.593 at 36.1 | 2.538 at 53.6 |
11402/// | 13 | 3.093 at 3.1 | 3.029 at 36.4 | 2.921 at 35.7 | 2.770 at 82.9 |
11403/// | 14 | 2.330 at 2.1 | 2.283 at 24.3 | 2.213 at 23.5 | 2.113 at 67.6 |
11404/// | 39 | 2.459 at 5.3 | 2.147 at 10.6 | 2.145 at 29.3 | 2.088 at 43.1 |
11405/// | 56 | 4.694 at 6.3 | 4.381 at 51.0 | 4.172 at 50.6 | 3.983 at 86.8 |
11406///
11407/// The best of the three per column is 98 percent of the exhaustive ratio for a tenth of the time.
11408/// `FSST` and `PLAIN` on their own are in the list as a floor rather than to win. `FSST` is the
11409/// right answer for text that does not share prefixes with its neighbours, and `PLAIN` is there so
11410/// that a column nothing compresses is found out in the sample and written at a gigabyte a second
11411/// rather than searched for an answer that does not exist.
11412fn payload_shapes() -> Vec<chooser::Settled> {
11413    let integers = vec![integer::Kind::Packed];
11414    [
11415        vec![string::Kind::Front, string::Kind::Lz],
11416        vec![string::Kind::Lz, string::Kind::Fsst],
11417        vec![string::Kind::Lz, string::Kind::Plain],
11418        vec![string::Kind::Fsst],
11419        vec![string::Kind::Plain],
11420    ]
11421    .into_iter()
11422    .map(|strings| chooser::Settled::new(strings, integers.clone()))
11423    .collect()
11424}
11425
11426/// Syncs the file, and counts the sync and how long it took as a publish wait when a load is being
11427/// profiled.
11428///
11429/// A wait rather than time, because the time is already in the publish span around it. What the
11430/// wait columns add is how much of publish was the device, which on the WSL2 disk of the gaming PC
11431/// is most of it: a sync there costs about two milliseconds (see `rudb_device_card`).
11432fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
11433    let started = profile.map(|_| std::time::Instant::now());
11434    file.sync()?;
11435    if let (Some(profile), Some(started)) = (profile, started) {
11436        profile.waited(
11437            Stage::Publish,
11438            u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
11439        );
11440    }
11441    Ok(())
11442}
11443
11444/// One sealed dictionary block on its way to being encoded outside the writer's lock.
11445///
11446/// Handed out by the merge that sealed it and encoded with the pages of the same stripe. See
11447/// [`GlobalDictionary::hand_out`].
11448#[derive(Debug)]
11449pub(crate) struct Unencoded {
11450    column: usize,
11451    at: usize,
11452    ends: Vec<u32>,
11453    bytes: Vec<u8>,
11454    shape: chooser::Settled,
11455}
11456
11457impl Unencoded {
11458    /// The encoded block and its signature.
11459    pub(crate) fn encode(&self) -> Result<EncodedBlock> {
11460        let values = block_values(&self.ends, &self.bytes);
11461        Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
11462    }
11463
11464    /// The column and the block number the encoded block goes back to.
11465    pub(crate) fn place(&self) -> (usize, usize) {
11466        (self.column, self.at)
11467    }
11468}
11469
11470/// One encoded dictionary block and the signature of the values in it.
11471///
11472/// Boxed because it is carried around in things that are otherwise small.
11473pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
11474
11475/// The conservative four-byte substring signature of one block's values.
11476fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
11477    let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
11478    for value in values {
11479        for gram in value.windows(4) {
11480            for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
11481                grams[bit / 8] |= 1 << (bit % 8);
11482            }
11483        }
11484    }
11485    grams
11486}
11487
11488/// The values of one block, given where each of them ends relative to the block.
11489fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
11490    let mut out = Vec::with_capacity(ends.len());
11491    let mut from = 0;
11492    for &to in ends {
11493        out.push(&bytes[from..to as usize]);
11494        from = to as usize;
11495    }
11496    out
11497}
11498
11499/// Encodes every block still raw at the end of a load: the part block each column ends on and,
11500/// for a column too small to have settled a shape, every block it has.
11501///
11502/// Across threads, the way [`encode_ready`] does it. This ran one column at a time on the thread
11503/// closing the table, and a column that never settled a shape encodes each block by trying every
11504/// candidate, so on a million rows of `hits` it was most of the load's CPU on one core.
11505fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11506    for dictionary in dictionaries.iter_mut().flatten() {
11507        if !dictionary.early.is_empty() {
11508            return Err(Error::internal("a dictionary block handed out never came back"));
11509        }
11510        dictionary.seal_rest();
11511        dictionary.settle_rest()?;
11512    }
11513    encode_waiting(dictionaries)?;
11514    // A block handed out and never given back leaves a gap nothing above would notice when it was
11515    // the last one, so the count is checked against the values as well.
11516    if dictionaries
11517        .iter()
11518        .flatten()
11519        .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
11520    {
11521        return Err(Error::internal("a dictionary block handed out never came back"));
11522    }
11523    Ok(())
11524}
11525
11526/// Encodes the waiting blocks of every dictionary across threads, and appends them to their columns
11527/// in order.
11528fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11529    let jobs = dictionaries
11530        .iter()
11531        .enumerate()
11532        .flat_map(|(column, held)| {
11533            (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
11534        })
11535        .collect::<Vec<_>>();
11536    if jobs.is_empty() {
11537        return Ok(());
11538    }
11539    let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
11540        let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
11541        Ok((column, at, held.encode_waiting(at)?))
11542    };
11543    let workers = std::thread::available_parallelism()
11544        .map_or(1, usize::from)
11545        .min(MAX_FREQUENCY_WORKERS)
11546        .min(jobs.len());
11547    let made = if workers <= 1 {
11548        jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
11549    } else {
11550        let next = AtomicUsize::new(0);
11551        let jobs = &jobs;
11552        let pieces = std::thread::scope(|scope| {
11553            (0..workers)
11554                .map(|_| {
11555                    scope.spawn(|| {
11556                        let mut mine = Vec::new();
11557                        loop {
11558                            let job = next.fetch_add(1, Atomic::Relaxed);
11559                            let Some(&(column, at)) = jobs.get(job) else { break };
11560                            mine.push(one(column, at)?);
11561                        }
11562                        Ok(mine)
11563                    })
11564                })
11565                .collect::<Vec<_>>()
11566                .into_iter()
11567                .map(|handle| {
11568                    handle
11569                        .join()
11570                        .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
11571                })
11572                .collect::<Result<Vec<_>>>()
11573        })?;
11574        pieces.into_iter().flatten().collect()
11575    };
11576    let mut done: Vec<Vec<(usize, EncodedBlock)>> =
11577        (0..dictionaries.len()).map(|_| Vec::new()).collect();
11578    for (column, at, bytes) in made {
11579        done[column].push((at, bytes));
11580    }
11581    for (column, mut made) in done.into_iter().enumerate() {
11582        if made.is_empty() {
11583            continue;
11584        }
11585        let Some(held) = dictionaries[column].as_mut() else { continue };
11586        made.sort_by_key(|(at, _)| *at);
11587        let waiting = std::mem::take(&mut held.waiting);
11588        for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
11589            if held.encoded() != at {
11590                return Err(Error::internal("a dictionary block was encoded out of order"));
11591            }
11592            held.push_block(block);
11593        }
11594    }
11595    Ok(())
11596}
11597
11598/// Which of [`payload_shapes`] comes out smallest over a sample of the blocks.
11599///
11600/// Every shape is encoded over the same sample and the smallest wins, which is the exhaustive
11601/// search moved up a level: over shapes of a column rather than over candidates of a chunk. The
11602/// sample is spread across the dictionary so that the first and last blocks are both in it, because
11603/// a dictionary written in first seen order has its common values at the front and its long tail at
11604/// the back, and those do not compress alike. Which blocks those are is
11605/// [`GlobalDictionary::seal`]'s to decide, because by the time this is called the rest of them have
11606/// been encoded and the raw bytes are gone.
11607fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
11608    let mut best: Option<(chooser::Settled, usize)> = None;
11609    for shape in payload_shapes() {
11610        let mut size = 0;
11611        for block in sample {
11612            size += string::encode_with(block, &shape)?.len();
11613        }
11614        if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
11615            best = Some((shape, size));
11616        }
11617    }
11618    best.map(|(shape, _)| shape)
11619        .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
11620}
11621
11622/// The sorted order laid out the way a reader reads it, in blocks of [`TEXT_RANK_BLOCK`] entries.
11623///
11624/// Each block holds its heads first and then its codes, rather than pairing them, because a search
11625/// asks for a head at every probe and for a code about once a search. Keeping the heads together
11626/// means a probe touches eight bytes of a block rather than twelve spread over it, and the last few
11627/// probes of a search, which are the ones that land in the same block, touch the same cache line.
11628fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
11629    let mut out = Vec::with_capacity(order.len() * 4);
11630    let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
11631    let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
11632    let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
11633    for block in order.chunks(TEXT_RANK_BLOCK) {
11634        // The order is sorted by value and a head is a prefix of a value, so the heads of a block
11635        // rise, the smallest is the first and the largest is the last.
11636        let base = block.first().map_or(0, |&(head, _)| head);
11637        let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
11638        let width = (u64::BITS - span.leading_zeros()) as usize;
11639        heads.clear();
11640        codes.clear();
11641        for &(head, code) in block {
11642            heads.push(head.wrapping_sub(base));
11643            codes.push(u64::from(code));
11644        }
11645        put_u64(&mut out, base);
11646        out.push(width as u8);
11647        bitpack::pack_tail(&heads, width, &mut out)
11648            .map_err(|_| invalid("global dictionary heads do not pack"))?;
11649        bitpack::pack_tail(&codes, code_bits, &mut out)
11650            .map_err(|_| invalid("global dictionary codes do not pack"))?;
11651        ends.push(out.len() as u64);
11652    }
11653    Ok((out, ends))
11654}
11655
11656/// Opens a column's global dictionary, which reads its index and none of its payload.
11657///
11658/// `keep_budget` is how many decoded payload bytes this dictionary may hold on to, and every
11659/// caller bar the test of the ceiling passes [`TEXT_KEEP_BUDGET`]. It is a parameter rather than
11660/// the constant read where it is used because a test of a ceiling that cannot be moved has to build
11661/// a quarter of a gigabyte of dictionary to reach it.
11662fn open_global_dictionary(
11663    file: Arc<File>,
11664    page: Page,
11665    ty: &LogicalType,
11666    keep_budget: usize,
11667) -> Result<Vector> {
11668    if !coded_type(ty) {
11669        return Err(invalid("global dictionary belongs to a non-string column"));
11670    }
11671    let mut header = [0; DICTIONARY_HEADER];
11672    read_at(&file, page.offset, &mut header)?;
11673    let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
11674    let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
11675    let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
11676    let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11677    let scattered = width & DICTIONARY_SCATTERED != 0;
11678    let has_grams = width & DICTIONARY_GRAMS != 0;
11679    let gram_width =
11680        if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
11681    let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
11682    if per_block != TEXT_PAYLOAD_VALUES {
11683        return Err(invalid("global dictionary block width differs"));
11684    }
11685    if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
11686        return Err(invalid("global dictionary block count differs from its value count"));
11687    }
11688    if offset_bits > u32::BITS as usize {
11689        return Err(invalid("global dictionary packs offsets past a payload"));
11690    }
11691    let offset_len = offset_bytes(count, offset_bits);
11692    // The sorted order is kept out of the index on purpose. The index is read and checksummed in
11693    // full the moment the column is first touched, and the order is half again the size of the
11694    // offsets, so putting it there would make every query that reads a string column pay for a
11695    // search that most of them never make.
11696    let ranks = count;
11697    let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
11698    // Three words a payload block, for where it starts, how long it is and what it hashes to, or
11699    // two of them on a file that has the blocks back to back and needs no start. Two a rank block
11700    // either way, since those are still one run.
11701    let payload_words = if scattered { 3 } else { 2 };
11702    let hash_len = blocks
11703        .checked_mul(payload_words * 8)
11704        .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11705        .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
11706        .ok_or_else(|| invalid("global dictionary block count overflow"))?;
11707    let gram_len = if has_grams {
11708        blocks
11709            .checked_mul(gram_width)
11710            .ok_or_else(|| invalid("global dictionary signature count overflow"))?
11711    } else {
11712        0
11713    };
11714    let index_len = DICTIONARY_HEADER
11715        .checked_add(offset_len)
11716        .and_then(|len| len.checked_add(hash_len))
11717        .ok_or_else(|| invalid("global dictionary header overflow"))?;
11718    if index_len > page.length as usize {
11719        return Err(invalid("global dictionary offset index exceeds its page"));
11720    }
11721    let mut index = vec![0; index_len];
11722    index[..DICTIONARY_HEADER].copy_from_slice(&header);
11723    read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
11724    if checksum(&index) != page.hash {
11725        return Err(invalid("global dictionary index checksum differs"));
11726    }
11727    let word_end = index_len - usize::from(has_grams) * 8;
11728    let gram_hash = has_grams
11729        .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
11730    let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
11731        .chunks_exact(8)
11732        .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
11733        .collect::<Vec<_>>();
11734    let mut rest = words.split_off(blocks * payload_words);
11735    let rank_hashes = rest.split_off(rank_blocks);
11736    let rank_ends = rest;
11737    // A rank block packs its heads at whatever width its own values need, so its length is no longer
11738    // arithmetic on the block number and the reader has to be told where each one ends.
11739    if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
11740        return Err(invalid("global dictionary order blocks do not rise"));
11741    }
11742    let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
11743        .map_err(|_| invalid("global dictionary rank overflow"))?;
11744    let body_len = index_len
11745        .checked_add(rank_len)
11746        .ok_or_else(|| invalid("global dictionary header overflow"))?;
11747    if body_len > page.length as usize {
11748        return Err(invalid("global dictionary order exceeds its page"));
11749    }
11750    let gram_end = body_len
11751        .checked_add(gram_len)
11752        .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
11753    if gram_end > page.length as usize {
11754        return Err(invalid("global dictionary signatures exceed their page"));
11755    }
11756    let grams = gram_hash.map(|hash| NativeGrams {
11757        start: page.offset + body_len as u64,
11758        length: gram_len,
11759        width: gram_width,
11760        hash,
11761        verdicts: Mutex::new(Vec::new()),
11762    });
11763    // The offsets stay where they were read, behind the header, rather than being copied out. On a
11764    // dictionary of millions of values they are megabytes, and a copy is as many fresh pages to
11765    // fault in again on a query that may want a handful of strings.
11766    let mut offsets = index;
11767    offsets.truncate(DICTIONARY_HEADER + offset_len);
11768    let hashes = words.split_off(blocks * (payload_words - 1));
11769    let (starts, lengths) = if scattered {
11770        let mut starts = Vec::with_capacity(blocks);
11771        let mut lengths = Vec::with_capacity(blocks);
11772        for pair in words.chunks_exact(2) {
11773            starts.push(pair[0]);
11774            lengths.push(pair[1]);
11775        }
11776        (starts, lengths)
11777    } else {
11778        // A file written before the blocks said where they were has them behind one another at the
11779        // end of the page, so the base is where the sorted order stops and each end is the start of
11780        // the one after it. Turning them round here is what lets everything below take one shape.
11781        let base = page.offset + gram_end as u64;
11782        let mut starts = Vec::with_capacity(blocks);
11783        let mut lengths = Vec::with_capacity(blocks);
11784        let mut at = 0_u64;
11785        for &end in &words {
11786            let len = end
11787                .checked_sub(at)
11788                .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
11789            starts.push(base + at);
11790            lengths.push(len);
11791            at = end;
11792        }
11793        (starts, lengths)
11794    };
11795    // What the offsets bound is the decoded payload, and what the page length counts is the stored
11796    // one, so on a format 26 file the block lengths adding up to the rest of the page is the one
11797    // thing that ties the index to the page. From format 27 the blocks are written during the load
11798    // and the page is only the index and the order, so there the most that can be said is that
11799    // every block is somewhere in the file past its header.
11800    let stored_len = page.length as u64 - gram_end as u64;
11801    if scattered && stored_len == 0 {
11802        let size = file.metadata().map_err(io)?.len();
11803        let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
11804            start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
11805        });
11806        if !inside {
11807            return Err(invalid("global dictionary block lies outside the file"));
11808        }
11809    } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
11810        return Err(invalid("global dictionary blocks do not bound the payload"));
11811    }
11812    Vector::external_text(
11813        ty.clone(),
11814        Arc::new(NativeText {
11815            file,
11816            values: count,
11817            offsets,
11818            offset_bits,
11819            value_ends: OnceLock::new(),
11820            value_lens: OnceLock::new(),
11821            ends_asked: AtomicUsize::new(0),
11822            ranks,
11823            rank_at: page.offset + index_len as u64,
11824            rank_ends,
11825            rank_hashes,
11826            rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
11827            code_bits: code_width(count),
11828            code_ranks: OnceLock::new(),
11829            starts,
11830            lengths,
11831            hashes,
11832            grams,
11833            blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
11834            char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
11835            keep_budget,
11836            payload_kept: AtomicUsize::new(0),
11837            swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
11838            visit_dropped: AtomicUsize::new(0),
11839            searched: Mutex::new(HashMap::new()),
11840        }),
11841    )
11842}
11843
11844/// What a stored page is, without decoding a value out of it.
11845///
11846/// Two layers, and both of them belong in the answer. The codec byte at the front of every page is
11847/// the format's own choice, and it is what says whether the column came back as codes into a table
11848/// wide dictionary, as a bit packed page, as an encoding cascade or as the bytes themselves. Under
11849/// the cascade codecs there is a second choice the encoder made per chunk, and that is what
11850/// [`integer::describe`] and [`string::describe`] already write out as `DICT(PACKED, PACKED)`.
11851///
11852/// This mirrors the tags [`decode`] reads and has to be kept beside it. A page whose header this
11853/// cannot walk comes back as text rather than as an error, because a caller asking what a file
11854/// looks like is usually asking because something is wrong with it, and a report that stops at the
11855/// first bad page is a report that says nothing about the other nine hundred.
11856fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
11857    /// The page header is the codec, the validity tag and, for a page that stores a mask, the mask.
11858    fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
11859        let mut cur = Cursor::new(bytes);
11860        let codec = cur.u8()?;
11861        if cur.u8()? == 2 {
11862            cur.take(rows.div_ceil(8))?;
11863        }
11864        Ok((codec, cur.at))
11865    }
11866    let Ok((codec, at)) = cascade_at(rows, bytes) else {
11867        return "UNREADABLE".to_string();
11868    };
11869    let tail = &bytes[at..];
11870    let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
11871    match codec {
11872        0 => match ty {
11873            LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
11874            _ => "FIXED".to_string(),
11875        },
11876        1 => "DICT(PLAIN)".to_string(),
11877        2 => "FOR+BITPACK".to_string(),
11878        3 => "TABLE DICT".to_string(),
11879        4 => format!("TABLE DICT({})", described(integer::describe(tail))),
11880        5 => described(integer::describe(tail)),
11881        6 => described(string::describe(tail)),
11882        other => format!("CODEC {other}"),
11883    }
11884}
11885
11886/// Selected stable dictionary codes from one page.
11887///
11888/// Pair-frequency construction needs at most the bounded heavy-hitter rows. Reading those code
11889/// positions directly avoids materializing every code in each part that contains a candidate.
11890fn decode_selected_stable_codes(
11891    rows: usize,
11892    bytes: &[u8],
11893    positions: &[usize],
11894    out: &mut Vec<Option<u32>>,
11895) -> Result<bool> {
11896    if positions.windows(2).any(|pair| pair[0] >= pair[1])
11897        || positions.last().is_some_and(|&position| position >= rows)
11898    {
11899        return Err(invalid("selected code positions are not sorted and in range"));
11900    }
11901    let mut cur = Cursor::new(bytes);
11902    let codec = cur.u8()?;
11903    if codec != 3 && codec != 4 {
11904        return Ok(false);
11905    }
11906    let flag = cur.u8()?;
11907    let mask = match flag {
11908        0 | 1 => None,
11909        2 => {
11910            let at = cur.at;
11911            let len = rows.div_ceil(8);
11912            cur.take(len)?;
11913            Some((at, len))
11914        }
11915        _ => return Err(invalid("page validity tag differs")),
11916    };
11917    let valid = |row: usize| match flag {
11918        0 => true,
11919        1 => false,
11920        2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
11921        _ => unreachable!("the validity tag was checked"),
11922    };
11923    if codec == 4 {
11924        let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
11925        for (&row, code) in positions.iter().zip(wide) {
11926            let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
11927            out.push(valid(row).then_some(code));
11928        }
11929        return Ok(true);
11930    }
11931    let codes_at = cur.at;
11932    let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
11933    cur.take(codes_len)?;
11934    if cur.at != bytes.len() {
11935        return Err(invalid("global code page has trailing bytes"));
11936    }
11937    let codes = &bytes[codes_at..codes_at + codes_len];
11938    for &row in positions {
11939        let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
11940        let code = u32::from_le_bytes(
11941            codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
11942        );
11943        out.push(valid(row).then_some(code));
11944    }
11945    Ok(true)
11946}
11947
11948/// [`decode`] of only the rows at `positions`, which rise.
11949///
11950/// A compressed text page decompresses only those rows, see [`string::decode_flat_at`], and checks
11951/// only those rows are text. Every other page is decoded whole and gathered, since its values are
11952/// fixed width or its strings are shared through a dictionary, and there picking comes after.
11953fn decode_at(
11954    ty: &LogicalType,
11955    rows: usize,
11956    bytes: &[u8],
11957    global: Option<Arc<Vector>>,
11958    positions: &[u32],
11959) -> Result<Vector> {
11960    if positions.last().is_some_and(|&last| last as usize >= rows) {
11961        return Err(invalid("a position is past the end of the part"));
11962    }
11963    // Past about one row in eight, unpacking the whole part and picking the rows out is the cheaper
11964    // of the two, since a unit unpacks at a fraction of what a row unpacked on its own costs.
11965    if bytes.first() == Some(&5)
11966        && positions.len().saturating_mul(8) <= rows
11967        // Past the codec, the validity flag and the mask a flag of 2 has.
11968        && bytes
11969            .get(2 + if bytes.get(1) == Some(&2) { rows.div_ceil(8) } else { 0 }..)
11970            .is_some_and(integer::pointed)
11971    {
11972        return cascade_at(ty, rows, bytes, positions);
11973    }
11974    if bytes.first() != Some(&6) {
11975        return decode(ty, rows, bytes, global)?.gather(positions);
11976    }
11977    if !coded_type(ty) {
11978        return Err(invalid("compressed text codec belongs to a non-string page"));
11979    }
11980    let mut cur = Cursor::new(bytes);
11981    cur.u8()?;
11982    let validity = match cur.u8()? {
11983        0 => Validity::AllValid,
11984        1 => Validity::AllInvalid,
11985        2 => {
11986            let mask = cur.take(rows.div_ceil(8))?;
11987            Validity::from_iter(positions.len(), |at| {
11988                let row = positions[at] as usize;
11989                mask[row / 8] >> (row % 8) & 1 == 1
11990            })
11991        }
11992        _ => return Err(invalid("page validity tag differs")),
11993    };
11994    let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
11995    let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11996    push_values(&mut values, ty, &ends)?;
11997    Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
11998}
11999
12000/// The rows `positions` names of an integer cascade page, unpacked at those rows alone.
12001///
12002/// A scan whose join keeps a few rows in a thousand reads its other columns only at those rows, and
12003/// decoding the whole part to pick them out afterwards was most of what it cost. In TPC-H q17 the
12004/// bitmap over the parts of one brand and container keeps about one `lineitem` row in a thousand.
12005fn cascade_at(ty: &LogicalType, rows: usize, bytes: &[u8], positions: &[u32]) -> Result<Vector> {
12006    fn wanted<T: integer::Lane>(values: &[i64]) -> Result<Vec<T>> {
12007        values
12008            .iter()
12009            .map(|&value| T::fit(value).ok_or_else(|| invalid("page value is not of its type")))
12010            .collect()
12011    }
12012    let mut cur = Cursor::new(bytes);
12013    cur.u8()?;
12014    let validity = match cur.u8()? {
12015        0 => Validity::AllValid,
12016        1 => Validity::AllInvalid,
12017        2 => {
12018            let mask = cur.take(rows.div_ceil(8))?;
12019            Validity::from_iter(positions.len(), |at| {
12020                let row = positions[at] as usize;
12021                mask[row / 8] >> (row % 8) & 1 == 1
12022            })
12023        }
12024        _ => return Err(invalid("page validity tag differs")),
12025    };
12026    let at: Vec<usize> = positions.iter().map(|&row| row as usize).collect();
12027    let values = integer::decode_selected(&bytes[cur.at..], &at)
12028        .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
12029    if values.len() != positions.len() {
12030        return Err(invalid("cascade page holds the wrong number of rows"));
12031    }
12032    let data = match ty {
12033        LogicalType::TinyInt => Data::Int8(wanted::<i8>(&values)?.into()),
12034        LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(&values)?.into()),
12035        LogicalType::SmallInt => Data::Int16(wanted::<i16>(&values)?.into()),
12036        LogicalType::USmallInt => Data::UInt16(wanted::<u16>(&values)?.into()),
12037        LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(&values)?.into()),
12038        LogicalType::UInteger => Data::UInt32(wanted::<u32>(&values)?.into()),
12039        LogicalType::BigInt
12040        | LogicalType::Timestamp
12041        | LogicalType::Time
12042        | LogicalType::TimeTz
12043        | LogicalType::TimestampTz
12044        | LogicalType::TimestampS
12045        | LogicalType::TimestampMs
12046        | LogicalType::TimestampNs => Data::Int64(values.into()),
12047        LogicalType::Decimal { .. } => match ty.physical() {
12048            PhysicalType::Int16 => Data::Int16(wanted::<i16>(&values)?.into()),
12049            PhysicalType::Int32 => Data::Int32(wanted::<i32>(&values)?.into()),
12050            PhysicalType::Int64 => Data::Int64(values.into()),
12051            _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
12052        },
12053        _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
12054    };
12055    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12056}
12057
12058/// The values of a string or blob page, laid end to end in the page's payload from its start, each
12059/// ending where `ends` says. A varchar is checked for text on the way in, once over the whole run,
12060/// and a blob or a bit string is not, since neither ever claimed to hold any.
12061fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
12062    if ty == &LogicalType::Varchar {
12063        return values.push_run_in_place(0, ends);
12064    }
12065    let mut start = 0;
12066    for &end in ends {
12067        let len = end
12068            .checked_sub(start)
12069            .ok_or_else(|| invalid("a string value ends before it starts"))?;
12070        values.push_bytes_in_place(start, len)?;
12071        start = end;
12072    }
12073    Ok(())
12074}
12075
12076fn decode(
12077    ty: &LogicalType,
12078    rows: usize,
12079    bytes: &[u8],
12080    global: Option<Arc<Vector>>,
12081) -> Result<Vector> {
12082    let mut cur = Cursor::new(bytes);
12083    let codec = cur.u8()?;
12084    let flag = cur.u8()?;
12085    let validity = match flag {
12086        0 => Validity::AllValid,
12087        1 => Validity::AllInvalid,
12088        2 => {
12089            let mask = cur.take(rows.div_ceil(8))?;
12090            Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
12091        }
12092        _ => return Err(invalid("page validity tag differs")),
12093    };
12094    if codec == 1 {
12095        if !coded_type(ty) {
12096            return Err(invalid("dictionary codec belongs to a non-string page"));
12097        }
12098        let count = cur.u32()? as usize;
12099        let payload_len = cur.u32()? as usize;
12100        let offset_bytes = cur.take(
12101            (count + 1)
12102                .checked_mul(4)
12103                .ok_or_else(|| invalid("dictionary offset count overflow"))?,
12104        )?;
12105        let offsets = offset_bytes
12106            .chunks_exact(4)
12107            .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12108            .collect::<Vec<_>>();
12109        let payload = cur.take(payload_len)?.to_vec();
12110        if offsets.first() != Some(&0)
12111            || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12112            || offsets.windows(2).any(|pair| pair[0] > pair[1])
12113        {
12114            return Err(invalid("dictionary offsets do not bound the payload"));
12115        }
12116        // A page, because every chunk cut out of this dictionary points at the same payload and a
12117        // page is what lets a cut be the views and nothing else.
12118        let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
12119        let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12120        push_values(&mut strings, ty, &ends)?;
12121        let mut codes = Vec::with_capacity(rows);
12122        for _ in 0..rows {
12123            codes.push(cur.u32()?);
12124        }
12125        if codes.iter().any(|code| *code as usize >= count) {
12126            return Err(invalid("dictionary code is out of range"));
12127        }
12128        if cur.at != bytes.len() {
12129            return Err(invalid("dictionary page has trailing bytes"));
12130        }
12131        let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
12132        return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
12133    }
12134    if codec == 3 || codec == 4 {
12135        let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
12136        let codes = if codec == 4 {
12137            // The cascade holds the whole tail of the page and says how long it is itself, so the
12138            // check that nothing is left over is the one the decoder already makes.
12139            // Straight into `u32`, which is also the check that every code is one: a code outside
12140            // it is a corrupt file and the decoder refuses it, a block at a time where it can.
12141            let codes = integer::decode_as::<u32>(&bytes[cur.at..])
12142                .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
12143            if codes.len() != rows {
12144                return Err(invalid("encoded code page holds the wrong number of rows"));
12145            }
12146            codes
12147        } else {
12148            let mut codes = Vec::with_capacity(rows);
12149            for _ in 0..rows {
12150                codes.push(cur.u32()?);
12151            }
12152            if cur.at != bytes.len() {
12153                return Err(invalid("global code page has trailing bytes"));
12154            }
12155            codes
12156        };
12157        let highest = codes.iter().copied().max();
12158        return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
12159            .with_validity(validity));
12160    }
12161    if codec == 6 {
12162        if !coded_type(ty) {
12163            return Err(invalid("compressed text codec belongs to a non-string page"));
12164        }
12165        // As codec 5, the layer holds the whole tail of the page and says how long it is itself.
12166        // It comes back as one buffer with the values laid end to end and where each one ends, which
12167        // is the raw form's layout, so what is left to do here is what codec 0 does.
12168        let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
12169        if ends.len() != rows {
12170            return Err(invalid("compressed text page holds the wrong number of rows"));
12171        }
12172        // A page, because this is read once and handed out a chunk at a time, and a cut of a paged
12173        // payload moves views rather than bytes.
12174        let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12175        push_values(&mut values, ty, &ends)?;
12176        return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
12177    }
12178    if codec == 5 {
12179        // The cascade holds the whole tail of the page and says how long it is itself.
12180        let data = cascade(ty, &bytes[cur.at..], rows)?;
12181        return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
12182    }
12183    if codec == 2 {
12184        let width = u32::from(cur.u8()?);
12185        let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
12186        let count = cur.u32()? as usize;
12187        let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
12188        let words: Vec<u64> = cur
12189            .take(length)?
12190            .chunks_exact(8)
12191            .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
12192            .collect();
12193        if cur.at != bytes.len() {
12194            return Err(invalid("packed page has trailing bytes"));
12195        }
12196        return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
12197    }
12198    if codec != 0 {
12199        return Err(invalid("page codec is unknown"));
12200    }
12201    let data = match ty {
12202        LogicalType::TinyInt => {
12203            let values = cur.take(rows)?;
12204            Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
12205        }
12206        LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
12207        LogicalType::SmallInt => {
12208            let values =
12209                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12210            Data::Int16(
12211                values
12212                    .chunks_exact(2)
12213                    .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12214                    .collect::<Vec<_>>()
12215                    .into(),
12216            )
12217        }
12218        LogicalType::USmallInt => {
12219            let values =
12220                cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12221            Data::UInt16(
12222                values
12223                    .chunks_exact(2)
12224                    .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
12225                    .collect::<Vec<_>>()
12226                    .into(),
12227            )
12228        }
12229        LogicalType::UInteger => {
12230            let values =
12231                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12232            Data::UInt32(
12233                values
12234                    .chunks_exact(4)
12235                    .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
12236                    .collect::<Vec<_>>()
12237                    .into(),
12238            )
12239        }
12240        LogicalType::UBigInt => {
12241            let values =
12242                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12243            Data::UInt64(
12244                values
12245                    .chunks_exact(8)
12246                    .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
12247                    .collect::<Vec<_>>()
12248                    .into(),
12249            )
12250        }
12251        LogicalType::Integer | LogicalType::Date => {
12252            let values =
12253                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12254            Data::Int32(
12255                values
12256                    .chunks_exact(4)
12257                    .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12258                    .collect::<Vec<_>>()
12259                    .into(),
12260            )
12261        }
12262        LogicalType::BigInt
12263        | LogicalType::Timestamp
12264        | LogicalType::Time
12265        | LogicalType::TimeTz
12266        | LogicalType::TimestampTz
12267        | LogicalType::TimestampS
12268        | LogicalType::TimestampMs
12269        | LogicalType::TimestampNs => {
12270            let values =
12271                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12272            Data::Int64(
12273                values
12274                    .chunks_exact(8)
12275                    .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12276                    .collect::<Vec<_>>()
12277                    .into(),
12278            )
12279        }
12280        LogicalType::HugeInt | LogicalType::Uuid => {
12281            let values =
12282                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12283            Data::Int128(
12284                values
12285                    .chunks_exact(16)
12286                    .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12287                    .collect::<Vec<_>>()
12288                    .into(),
12289            )
12290        }
12291        LogicalType::UHugeInt => {
12292            let values =
12293                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12294            Data::UInt128(
12295                values
12296                    .chunks_exact(16)
12297                    .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12298                    .collect::<Vec<_>>()
12299                    .into(),
12300            )
12301        }
12302        LogicalType::Float => {
12303            let values =
12304                cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12305            Data::Float32(
12306                values
12307                    .chunks_exact(4)
12308                    .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
12309                    .collect::<Vec<_>>()
12310                    .into(),
12311            )
12312        }
12313        LogicalType::Double => {
12314            let values =
12315                cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12316            Data::Float64(
12317                values
12318                    .chunks_exact(8)
12319                    .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
12320                    .collect::<Vec<_>>()
12321                    .into(),
12322            )
12323        }
12324        LogicalType::Interval => {
12325            let values =
12326                cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12327            Data::Interval(
12328                values
12329                    .chunks_exact(16)
12330                    .map(|item| {
12331                        (
12332                            i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
12333                            i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
12334                            i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
12335                        )
12336                    })
12337                    .collect::<Vec<_>>()
12338                    .into(),
12339            )
12340        }
12341        LogicalType::Boolean => {
12342            let values = cur.take(rows)?;
12343            if values.iter().any(|value| *value > 1) {
12344                return Err(invalid("boolean page has another value"));
12345            }
12346            Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
12347        }
12348        // Whichever integer the declared width says, which is the mapping the rest of the engine
12349        // already uses for a decimal in memory.
12350        LogicalType::Decimal { .. } => match ty.physical() {
12351            PhysicalType::Int16 => {
12352                let values =
12353                    cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12354                Data::Int16(
12355                    values
12356                        .chunks_exact(2)
12357                        .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12358                        .collect::<Vec<_>>()
12359                        .into(),
12360                )
12361            }
12362            PhysicalType::Int32 => {
12363                let values =
12364                    cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12365                Data::Int32(
12366                    values
12367                        .chunks_exact(4)
12368                        .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12369                        .collect::<Vec<_>>()
12370                        .into(),
12371                )
12372            }
12373            PhysicalType::Int64 => {
12374                let values =
12375                    cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12376                Data::Int64(
12377                    values
12378                        .chunks_exact(8)
12379                        .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12380                        .collect::<Vec<_>>()
12381                        .into(),
12382                )
12383            }
12384            _ => {
12385                let values =
12386                    cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12387                Data::Int128(
12388                    values
12389                        .chunks_exact(16)
12390                        .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12391                        .collect::<Vec<_>>()
12392                        .into(),
12393                )
12394            }
12395        },
12396        LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
12397            let offset_bytes = cur
12398                .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
12399            let offsets = offset_bytes
12400                .chunks_exact(4)
12401                .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12402                .collect::<Vec<_>>();
12403            let payload = cur.take(bytes.len() - cur.at)?.to_vec();
12404            if offsets.first() != Some(&0)
12405                || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12406                || offsets.windows(2).any(|pair| pair[0] > pair[1])
12407            {
12408                return Err(invalid("string offsets do not bound the payload"));
12409            }
12410            // A page for the reason the dictionary payload above is one: the page is read once and
12411            // handed out a chunk at a time, and a cut of a paged payload moves views rather than
12412            // bytes.
12413            //
12414            // A varchar is checked for text on the way in and a blob and a bit string are not,
12415            // because the second pair never claimed to hold any. Reading them through the checking
12416            // seam would refuse a column for holding exactly what it was told to hold.
12417            let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12418            let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12419            push_values(&mut values, ty, &ends)?;
12420            Data::Varlen(values)
12421        }
12422        _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
12423    };
12424    if cur.at != bytes.len() {
12425        return Err(invalid("page has trailing bytes"));
12426    }
12427    Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12428}
12429
12430#[cfg(test)]
12431mod tests {
12432    use std::fs::{self, OpenOptions};
12433    use std::io::{Seek, SeekFrom, Write};
12434    use std::path::PathBuf;
12435    use std::time::{SystemTime, UNIX_EPOCH};
12436
12437    use rudb_common::Stat;
12438    use rudb_common::Value;
12439    use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
12440    use rudb_common::stat::Provenance;
12441
12442    use super::*;
12443
12444    #[test]
12445    fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
12446        let bytes: Vec<u8> =
12447            (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
12448        for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
12449            let whole = content_name(&bytes[..length]);
12450            for step in [1, 3, 8, 31, 32, 33, 64, 301] {
12451                let mut namer = ContentNamer::default();
12452                bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
12453                assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
12454            }
12455        }
12456    }
12457
12458    /// The chooser as it was before it could rule kinds out up front: the same narrowing, with every
12459    /// kind tested for. What it writes is what the file used to hold.
12460    #[derive(Debug)]
12461    struct TestsEverything<'a>(&'a dyn chooser::Chooser);
12462
12463    impl chooser::Chooser for TestsEverything<'_> {
12464        fn name(&self) -> &'static str {
12465            "tests everything"
12466        }
12467
12468        fn narrow_strings(
12469            &self,
12470            values: &[&[u8]],
12471            offered: &[string::Kind],
12472            depth: u8,
12473        ) -> Vec<string::Kind> {
12474            self.0.narrow_strings(values, offered, depth)
12475        }
12476
12477        fn narrow_integers(
12478            &self,
12479            values: &[i64],
12480            offered: &[integer::Kind],
12481            depth: u8,
12482        ) -> Vec<integer::Kind> {
12483            self.0.narrow_integers(values, offered, depth)
12484        }
12485    }
12486
12487    #[test]
12488    fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
12489        let columns: Vec<Vec<i64>> = vec![
12490            vec![],
12491            vec![5; 1000],
12492            (0..1000).collect(),
12493            (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
12494            (0..1000).map(|row| row / 50).collect(),
12495            (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
12496            (0..1000).map(|row| (row * 7919) % 13).collect(),
12497            (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
12498            (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
12499            (0..1000).map(|row| i64::MIN + row % 3).collect(),
12500        ];
12501        let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
12502        for column in &columns {
12503            for chooser in choosers {
12504                let quick = integer::encode_with(column, chooser).unwrap();
12505                let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
12506                assert_eq!(
12507                    quick,
12508                    full,
12509                    "{} on {:?}",
12510                    chooser.name(),
12511                    &column[..column.len().min(8)]
12512                );
12513            }
12514        }
12515    }
12516
12517    /// Parts of a column that all look alike come out of a settled shape byte for byte as they
12518    /// come out of a search, because the search would have kept the same tree on every one.
12519    #[test]
12520    fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
12521        let mut settling = Settling::default();
12522        for part in 0..STRIPE_PARTS as i64 {
12523            let values: Vec<i64> = (0..2048)
12524                .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
12525                .collect();
12526            let searched = integer::encode_with(&values, &Fixed).unwrap();
12527            assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
12528        }
12529    }
12530
12531    /// Text pages compressed against a table an earlier page trained read back as they went in, and
12532    /// a page of different text trains a table of its own rather than coming out as big as the
12533    /// earlier table would make it.
12534    #[test]
12535    fn text_pages_share_a_table_until_the_text_changes() {
12536        let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
12537        let english: Vec<Vec<u8>> = (0..1024)
12538            .map(|row: usize| {
12539                let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
12540                format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
12541            })
12542            .collect();
12543        let digits: Vec<Vec<u8>> =
12544            (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
12545        let mut settling = Settling::default();
12546        for page in 0..8 {
12547            let values: Vec<&[u8]> =
12548                if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
12549            let payload = values.iter().map(|value| value.len()).sum();
12550            let out = settling.text(&values, payload).unwrap().unwrap();
12551            assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
12552            let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
12553            assert!(
12554                out.len() * 4 <= alone.len() * 5,
12555                "page {page}: {} against {}",
12556                out.len(),
12557                alone.len()
12558            );
12559            let since = settling.symbols.as_ref().unwrap().since;
12560            assert_eq!(since, page % 4, "page {page}");
12561        }
12562    }
12563
12564    /// A column that changes shape partway through a stripe still reads back, and no part comes
12565    /// out much bigger than a search would have made it, because a replay that stops fitting or
12566    /// grows past a quarter a row is searched.
12567    #[test]
12568    fn a_column_that_changes_under_the_shape_is_searched_again() {
12569        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12570        let mut noise = move || {
12571            state ^= state << 13;
12572            state ^= state >> 7;
12573            state ^= state << 17;
12574            (state % 1_000_000) as i64
12575        };
12576        let mut settling = Settling::default();
12577        for part in 0..STRIPE_PARTS as i64 {
12578            let values: Vec<i64> = match part / 16 {
12579                0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
12580                1 => (0..2048).map(|_| noise()).collect(),
12581                2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
12582                _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
12583            };
12584            let settled = settling.encode(&values).unwrap();
12585            assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
12586            let searched = integer::encode_with(&values, &Fixed).unwrap();
12587            assert!(
12588                settled.len() * 4 <= searched.len() * 5,
12589                "part {part}: {} settled against {} searched, {} against {}",
12590                settled.len(),
12591                searched.len(),
12592                integer::describe(&settled).unwrap(),
12593                integer::describe(&searched).unwrap(),
12594            );
12595        }
12596    }
12597
12598    #[test]
12599    fn checksum_matches_fixed_vectors() {
12600        assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
12601        assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
12602        assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
12603    }
12604
12605    #[test]
12606    fn sorting_across_threads_matches_sorting_on_one() {
12607        let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12608        let mut next = move || {
12609            state ^= state << 13;
12610            state ^= state >> 7;
12611            state ^= state << 17;
12612            state
12613        };
12614        let mut values = Vec::new();
12615        for at in 0..150_000_u64 {
12616            let value = match next() % 6 {
12617                0 => Vec::new(),
12618                1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
12619                2 => format!("https://example.com/path/{at}").into_bytes(),
12620                3 => b"same".to_vec(),
12621                4 => vec![0xff; (next() % 12) as usize],
12622                _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
12623            };
12624            values.push(value);
12625        }
12626        let value = |code: u32| values[code as usize].as_slice();
12627        for workers in [1, 2, 3, 8, 32] {
12628            let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
12629            let mut across = one.clone();
12630            sort_by_value(&mut one, value);
12631            sort_by_value_across(&mut across, value, workers);
12632            assert_eq!(one, across, "{workers} workers");
12633        }
12634        let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
12635        sort_by_value_across(&mut sorted, value, 8);
12636        assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
12637    }
12638
12639    fn path(label: &str) -> PathBuf {
12640        let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
12641        std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
12642    }
12643
12644    /// Every value of a dictionary in code order, which the tests have no other way to ask for now
12645    /// that a dictionary does not keep the bytes of the values it has seen.
12646    ///
12647    /// Only valid once `finish_blocks` has run, because until then the last part block is still raw.
12648    fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
12649        let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
12650        (0..dictionary.values())
12651            .map(|code| {
12652                let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
12653                flat[from..to].to_vec()
12654            })
12655            .collect()
12656    }
12657
12658    /// The sections a test put in the table, which is every one the writer did not.
12659    ///
12660    /// A table now carries a summary and a sketch per column out of the write itself, and a test
12661    /// about the section table is not about those. Filtering by kind rather than by count, so a
12662    /// table that turns out to have no room for its summaries does not quietly change what these
12663    /// tests are asserting over.
12664    fn attached(table: &Table) -> Vec<&Section> {
12665        table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
12666    }
12667
12668    /// A read names the offset it wants, so a cursor somebody else moved cannot reach it.
12669    #[test]
12670    fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
12671        const SPANS: usize = 64;
12672        const SPAN: usize = 512;
12673        let path = path("positional");
12674        let content: Vec<u8> =
12675            (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
12676        fs::write(&path, &content).expect("the file is written");
12677        let file = Arc::new(File::open(&path).expect("the file opens"));
12678        std::thread::scope(|scope| {
12679            for _ in 0..8 {
12680                let file = Arc::clone(&file);
12681                scope.spawn(move || {
12682                    for _ in 0..64 {
12683                        for span in 0..SPANS {
12684                            let mut bytes = [0_u8; SPAN];
12685                            read_at(&file, (span * SPAN) as u64, &mut bytes)
12686                                .expect("the span reads");
12687                            assert!(
12688                                bytes.iter().all(|byte| *byte == span as u8),
12689                                "span {span} came back as {}",
12690                                bytes[0],
12691                            );
12692                        }
12693                    }
12694                });
12695            }
12696        });
12697        let mut past = [0_u8; SPAN];
12698        let end = (SPANS * SPAN) as u64;
12699        let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
12700        assert!(error.message().contains("ends before its declared length"), "{error}");
12701        drop(file);
12702        let _ = fs::remove_file(&path);
12703    }
12704
12705    /// The writer records where it put a page and puts it there.
12706    ///
12707    /// This used to move the file's cursor between the steps that record an offset, which is what
12708    /// reading the pages back to build the frequencies did on a platform with no `pread`, and the
12709    /// directory landed on top of a page. The writer's file is an `rudb_io` file now and has no
12710    /// cursor to move, so what is left is the check that every page is where the directory says.
12711    #[test]
12712    fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
12713        let path = path("cursor");
12714        let mut writer = Writer::create(
12715            &path,
12716            "items",
12717            vec![
12718                Field::required("id", LogicalType::Integer),
12719                Field::new("text", LogicalType::Varchar),
12720            ],
12721        )
12722        .expect("new file");
12723        writer.append(&sample()).expect("first part");
12724        writer.append(&sample()).expect("second part");
12725        writer.finish().expect("commit");
12726        let reader = Reader::open(&path).expect("reopen from disk");
12727        assert_eq!(reader.table().rows(), 6);
12728        let ids = reader.read(0, &[0]).expect("the integer page reads back");
12729        assert_eq!(ids.value_at(0, 0), Value::Integer(4));
12730        assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
12731        let text = reader.read(1, &[1]).expect("the text page reads back");
12732        assert_eq!(text.value_at(1, 0), Value::Null);
12733        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12734        // Nothing the directory points at may run past the end of the file, which is the shape the
12735        // failure took: a page recorded at an offset the directory had already been written over.
12736        let end = reader.table().stripes().iter().flat_map(|stripe| {
12737            stripe
12738                .pages
12739                .iter()
12740                .map(|page| page.offset + u64::from(page.length))
12741                .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
12742        });
12743        let last = end.fold(HEADER, u64::max);
12744        let directory = fs::metadata(&path).expect("the file is there").len();
12745        assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
12746        fs::remove_file(path).expect("remove scratch file");
12747    }
12748
12749    /// How long a global dictionary index is, read out of the page's own header.
12750    ///
12751    /// The tests below damage a byte of the order or of the payload, so they need to know where each
12752    /// one starts, and working it out here rather than writing a number down means adding something
12753    /// to the index does not quietly turn one of them into a test that damages the index instead.
12754    fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
12755        let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
12756        let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
12757        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12758        let bits = (width & !DICTIONARY_FLAGS) as usize;
12759        let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
12760        let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
12761        DICTIONARY_HEADER as u64
12762            + offset_bytes(count as usize, bits) as u64
12763            + blocks * payload_words * 8
12764            + rank_blocks * 16
12765            + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
12766    }
12767
12768    fn sample() -> Chunk {
12769        Chunk::new(vec![
12770            Vector::from_values(
12771                LogicalType::Integer,
12772                &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
12773            )
12774            .expect("integers"),
12775            Vector::from_values(
12776                LogicalType::Varchar,
12777                &[
12778                    Value::Varchar("alpha".into()),
12779                    Value::Null,
12780                    Value::Varchar("long text after a slash".into()),
12781                ],
12782            )
12783            .expect("strings"),
12784        ])
12785        .expect("matching rows")
12786    }
12787
12788    fn sample_ids() -> Chunk {
12789        Chunk::new(vec![
12790            Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
12791                .expect("integers"),
12792        ])
12793        .expect("one column")
12794    }
12795
12796    #[test]
12797    fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
12798        // Six rows, two of them null. `IS NULL` used to get the same fifth any unreadable
12799        // condition gets, and the number was in the stripe entry next to the bounds all along.
12800        let path = path("nulls_for_the_planner");
12801        let mut writer =
12802            Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
12803                .expect("new file");
12804        let rows = Chunk::new(vec![
12805            Vector::from_values(
12806                LogicalType::Integer,
12807                &[
12808                    Value::Integer(4),
12809                    Value::Null,
12810                    Value::Integer(9),
12811                    Value::Null,
12812                    Value::Integer(1),
12813                    Value::Integer(2),
12814                ],
12815            )
12816            .expect("integers"),
12817        ])
12818        .expect("one column");
12819        writer.append(&rows).expect("the only part");
12820        writer.finish().expect("commit");
12821        let reader = Reader::open(&path).expect("reopen from disk");
12822        let stripes = Stripes::new(reader);
12823        let column = stripes.column("a").expect("the file has that column");
12824        assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
12825        // A column the file does not have. Zero here would be a fact about a column that is not
12826        // there, which the planner would then divide by.
12827        assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
12828        fs::remove_file(&path).expect("clean up");
12829    }
12830
12831    #[test]
12832    fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
12833        // Six rows hold three values. The two leading counts help equality planning, while the
12834        // omitted value keeps the directory from being a complete grouped-count result.
12835        let path = path("frequencies_for_the_planner");
12836        let mut writer =
12837            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12838                .expect("new file");
12839        let rows = Chunk::new(vec![
12840            Vector::from_values(
12841                LogicalType::Integer,
12842                &[
12843                    Value::Integer(4),
12844                    Value::Integer(4),
12845                    Value::Integer(4),
12846                    Value::Integer(9),
12847                    Value::Integer(9),
12848                    Value::Integer(1),
12849                ],
12850            )
12851            .expect("integers"),
12852        ])
12853        .expect("one column");
12854        writer.append(&rows).expect("the only part");
12855        writer.finish().expect("commit");
12856        let reader = Reader::open(&path).expect("reopen from disk");
12857        let common = Common::new(reader);
12858        assert_eq!(common.rows(), 6);
12859        let column = common.column("id").expect("the file has that column");
12860        assert_eq!(common.column("nothing"), None);
12861        assert_eq!(
12862            common.rows_with(column, &Bound::Int(4)),
12863            Stat::exact(3, Provenance::FrequencySynopsis)
12864        );
12865        // An absent value cannot be distinguished from the omitted one by the synopsis.
12866        assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
12867        // A constant of another domain against an integer column. Nothing in the list compares
12868        // with it, so the zero above would be an artefact of the mismatch rather than a fact.
12869        assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
12870        assert!(common.remainder(column).is_some());
12871        fs::remove_file(&path).expect("clean up");
12872    }
12873
12874    #[test]
12875    fn string_frequency_estimates_do_not_open_the_global_dictionary() {
12876        let path = path("string_frequencies_for_the_planner");
12877        let mut writer =
12878            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12879                .expect("new file");
12880        let rows = Chunk::new(vec![
12881            Vector::from_values(
12882                LogicalType::Varchar,
12883                &[
12884                    Value::Varchar(String::new()),
12885                    Value::Varchar("alpha".into()),
12886                    Value::Varchar(String::new()),
12887                    Value::Varchar("beta".into()),
12888                    Value::Varchar(String::new()),
12889                ],
12890            )
12891            .expect("strings"),
12892        ])
12893        .expect("one column");
12894        writer.append(&rows).expect("the only part");
12895        writer.finish().expect("commit");
12896
12897        let reader = Reader::open(&path).expect("reopen from disk");
12898        assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
12899        let common = Common::new(reader.clone());
12900        let column = common.column("text").expect("the file has that column");
12901        assert_eq!(
12902            common.rows_with(column, &Bound::Bytes(Vec::new())),
12903            Stat::exact(3, Provenance::FrequencySynopsis)
12904        );
12905        assert_eq!(
12906            common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
12907            Stat::exact(0, Provenance::FrequencySynopsis)
12908        );
12909        assert_eq!(
12910            reader.reads().dictionaries,
12911            0,
12912            "the bounded spellings answer without opening the dictionary index"
12913        );
12914        fs::remove_file(&path).expect("clean up");
12915    }
12916
12917    #[test]
12918    fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
12919        let path = path("certified_host_groups");
12920        let mut writer =
12921            Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
12922                .expect("new file");
12923        let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
12924        values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
12925        values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
12926        values.push(Value::Varchar(String::new()));
12927        for part in values.chunks(512) {
12928            writer
12929                .append(
12930                    &Chunk::new(vec![
12931                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
12932                    ])
12933                    .expect("one column"),
12934                )
12935                .expect("part written");
12936        }
12937        writer.finish().expect("commit");
12938        let reader = Reader::open(&path).expect("reopen");
12939        assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
12940        fs::remove_file(&path).expect("clean up");
12941    }
12942
12943    /// A table directory with nothing in it but a name and one column, for the section tests.
12944    ///
12945    /// The section table is orthogonal to everything else in a directory, so the tests that pin it
12946    /// say so by starting from the emptiest table that encodes.
12947    fn bare_table(sections: Vec<Section>) -> Table {
12948        Table {
12949            name: "linked".to_owned(),
12950            fields: vec![Field::required("id", LogicalType::Integer)],
12951            stripes: Vec::new(),
12952            rows: 0,
12953            dictionaries: vec![None],
12954            dictionary_payloads: Vec::new(),
12955            demoted: Vec::new(),
12956            distincts: vec![None],
12957            frequencies: vec![None],
12958            pair_frequencies: Vec::new(),
12959            frequency_texts: Vec::new(),
12960            host_groups: None,
12961            clustering: None,
12962            generation: 1,
12963            sections,
12964        }
12965    }
12966
12967    fn a_key_map_section() -> Section {
12968        Section {
12969            kind: *section::KEY_MAP,
12970            id: 1,
12971            generation: 3,
12972            extents: 1,
12973            extent_page: HEADER,
12974            extent_bytes: section::EXTENT_BYTES as u32,
12975            hash: 0x1234_5678_9abc_def0,
12976            flags: 0,
12977            header_bytes: 24,
12978        }
12979    }
12980
12981    #[test]
12982    fn a_section_table_round_trips_through_a_directory() {
12983        let mut later = a_key_map_section();
12984        later.kind = *b"RUDBZZ9\0";
12985        later.id = 2;
12986        let table = bare_table(vec![a_key_map_section(), later]);
12987        let directory = encode_directory(&table).expect("directory");
12988        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12989        assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
12990        // The second is a kind this build has no name for, and it survived the round trip anyway.
12991        // That is what keeps an old build from silently discarding a newer build's work when it
12992        // rewrites a directory.
12993        assert!(decoded.sections()[0].known());
12994        assert!(!decoded.sections()[1].known());
12995    }
12996
12997    #[test]
12998    fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
12999        // The G1 exit criterion, at the directory level. A format 22 directory is exactly this
13000        // build's directory with the trailing section block cut off, so cutting it off is the
13001        // honest way to make one: no fixture to go stale, and no separate encoder to drift.
13002        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13003        let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
13004        let older = &directory[..directory.len() - block];
13005        let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
13006        assert!(decoded.sections().is_empty());
13007        assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
13008        assert_eq!(decoded.name(), "linked");
13009        assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
13010    }
13011
13012    #[test]
13013    fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
13014        // The same criterion end to end, which is the one the milestone actually asks for: a build
13015        // that knows about sections opens a file written by a build that did not, with no rewrite
13016        // and no repair, and answers from it. The version field is patched rather than a file
13017        // committed by an old binary because the bytes either side of it are identical: format 22
13018        // and format 23 differ only in a trailing directory block, and a reader that stops before
13019        // that block gets a table with no sections.
13020        let path = path("format_twenty_two");
13021        let mut writer =
13022            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13023                .expect("new file");
13024        let rows = Chunk::new(vec![
13025            Vector::from_values(
13026                LogicalType::Integer,
13027                &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
13028            )
13029            .expect("integers"),
13030        ])
13031        .expect("one column");
13032        writer.append(&rows).expect("the only part");
13033        writer.finish().expect("commit");
13034
13035        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13036        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13037        drop(file);
13038
13039        let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
13040        assert_eq!(reader.table().rows(), 3);
13041        // The rows and not the section table, because the section block is found by the magic at
13042        // the end of the directory rather than by the number in the header, so stamping the header
13043        // back does not take away the summaries this writer put there. What the test is about is
13044        // that the version check accepts 22, and the rows coming back is what says it did.
13045        assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
13046
13047        // And a format this build has never written is still refused, so the accept set is a list
13048        // and not an absence of a check.
13049        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13050        write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
13051        drop(file);
13052        let error = Reader::open(&path).expect_err("format 21 is not readable");
13053        assert!(error.to_string().contains("format 21"), "{error}");
13054
13055        fs::remove_file(&path).expect("clean up");
13056    }
13057
13058    #[test]
13059    fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
13060        // The bound the format has to check and `section` cannot, because only the reader knows how
13061        // big the file is. Reading the payload a section like this names would be reading whatever
13062        // else happens to be at that offset, which is the one way a graph section could turn into a
13063        // wrong answer rather than a slow one.
13064        let mut past = a_key_map_section();
13065        past.extent_page = 1 << 30;
13066        let directory = encode_directory(&bare_table(vec![past])).expect("directory");
13067        let error = decode_directory(&directory, 1 << 20).expect_err("refused");
13068        assert!(error.to_string().contains("outside the file"), "{error}");
13069
13070        let mut inside_the_header = a_key_map_section();
13071        inside_the_header.extent_page = 8;
13072        let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
13073        assert!(
13074            decode_directory(&directory, 1 << 20).is_err(),
13075            "a section may not overlap a header"
13076        );
13077    }
13078
13079    #[test]
13080    fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
13081        // Section 3.7: a relationship that does not fit the budget is recorded with its size so
13082        // that `rudb_links()` can report what a larger budget would buy. That record is a section
13083        // entry with no extents, so it has to survive a round trip while naming nothing.
13084        let not_built = Section {
13085            kind: *section::FORWARD_LINK,
13086            id: 9,
13087            generation: 3,
13088            extents: 0,
13089            extent_page: 0,
13090            extent_bytes: 0,
13091            hash: 0,
13092            flags: 0,
13093            header_bytes: 0,
13094        };
13095        let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
13096        let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13097        assert_eq!(decoded.sections(), &[not_built]);
13098
13099        // But a section with no extents that still names an extent table is incoherent, and an
13100        // incoherent entry is a torn directory rather than a relationship that was skipped.
13101        let mut incoherent = not_built;
13102        incoherent.extent_bytes = 28;
13103        incoherent.extent_page = HEADER;
13104        let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
13105        assert!(decode_directory(&directory, 1 << 20).is_err());
13106    }
13107
13108    #[test]
13109    fn a_directory_naming_more_sections_than_the_bound_is_refused() {
13110        let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13111        let mut torn = directory.clone();
13112        let count_at = torn.len() - size_of::<u16>();
13113        torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
13114        // Not an allocation of sixty five thousand entries off a torn count: either the bound
13115        // refuses it or the bytes run out, and both are errors rather than a read past the end.
13116        assert!(decode_directory(&torn, 1 << 20).is_err());
13117    }
13118
13119    /// A committed one column file of `rows` integers, for the attach tests.
13120    fn linked_file(label: &str, rows: i32) -> PathBuf {
13121        let path = path(label);
13122        let mut writer =
13123            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13124                .expect("new file");
13125        let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
13126        let chunk =
13127            Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
13128                .expect("one column");
13129        writer.append(&chunk).expect("the only part");
13130        writer.finish().expect("commit");
13131        path
13132    }
13133
13134    fn a_key_map_payload() -> Vec<u8> {
13135        // Shaped like one without being one: this crate never reads a payload, so what matters here
13136        // is that every byte comes back and that the header the entry measures is at the front.
13137        (0..512_u32).flat_map(u32::to_le_bytes).collect()
13138    }
13139
13140    #[test]
13141    fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
13142        let path = linked_file("attach", 64);
13143        let payload = a_key_map_payload();
13144        let table = attach(
13145            &path,
13146            "items",
13147            &[section::Attachment {
13148                kind: *section::KEY_MAP,
13149                id: 0,
13150                flags: 2,
13151                header_bytes: 40,
13152                bytes: &payload,
13153            }],
13154        )
13155        .expect("attach a key map");
13156        assert_eq!(attached(&table).len(), 1);
13157
13158        let reader = Reader::open(&path).expect("reopen after the attach");
13159        let held = attached(reader.table());
13160        assert_eq!(held.len(), 1);
13161        assert_eq!(held[0].kind, *section::KEY_MAP);
13162        assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
13163        assert_eq!(held[0].header_bytes, 40);
13164        // The generation is the one the pages were written at, not the one the attach committed at.
13165        // Attaching a section moved no row, so a section written by it is current, and a second
13166        // table added to this file later would not make it stale.
13167        assert_eq!(held[0].generation, 1);
13168        assert!(held[0].usable(reader.table().generation()));
13169        assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
13170        assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
13171
13172        fs::remove_file(&path).expect("clean up");
13173    }
13174
13175    #[test]
13176    fn attaching_a_section_answers_every_row_exactly_as_before() {
13177        // Section 3.1 end to end, and the reason the whole layer is safe to build incrementally. A
13178        // file with a section in it and the same file without one have to agree row for row, so the
13179        // comparison is made against the answers taken before the attach rather than against a
13180        // constant somebody typed.
13181        let path = linked_file("attach_changes_nothing", 300);
13182        let before = Reader::open(&path).expect("open before");
13183        let rows = before.table().rows();
13184        let first = before.read(0, &[0]).expect("read before");
13185        let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
13186        let layout = before.layout().columns_total();
13187        drop(before);
13188
13189        let payload = a_key_map_payload();
13190        attach(
13191            &path,
13192            "items",
13193            &[section::Attachment {
13194                kind: *section::KEY_MAP,
13195                id: 0,
13196                flags: 0,
13197                header_bytes: 0,
13198                bytes: &payload,
13199            }],
13200        )
13201        .expect("attach");
13202
13203        let after = Reader::open(&path).expect("open after");
13204        assert_eq!(after.table().rows(), rows);
13205        let read = after.read(0, &[0]).expect("read after");
13206        for (at, value) in values.iter().enumerate() {
13207            assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
13208        }
13209        assert_eq!(
13210            after.layout().columns_total(),
13211            layout,
13212            "an attach appends and does not rewrite a column page"
13213        );
13214
13215        fs::remove_file(&path).expect("clean up");
13216    }
13217
13218    #[test]
13219    fn a_rebuilt_section_replaces_the_one_it_supersedes() {
13220        // Rebuilding a key map has to be a write and not a question. If an attach added rather than
13221        // replaced, a table rebuilt a few times would name several maps for one column and a reader
13222        // would have to pick, which is a decision with no right answer in it.
13223        let path = linked_file("attach_twice", 32);
13224        let one = a_key_map_payload();
13225        let two = vec![7_u8; 1024];
13226        let entry = |bytes| section::Attachment {
13227            kind: *section::KEY_MAP,
13228            id: 4,
13229            flags: 1,
13230            header_bytes: 0,
13231            bytes,
13232        };
13233        attach(&path, "items", &[entry(&one)]).expect("first build");
13234        attach(&path, "items", &[entry(&two)]).expect("rebuild");
13235
13236        let reader = Reader::open(&path).expect("reopen");
13237        let held = attached(reader.table());
13238        assert_eq!(held.len(), 1, "one map per column and not one per build");
13239        assert_eq!(reader.payload(held[0]).expect("payload"), two);
13240
13241        fs::remove_file(&path).expect("clean up");
13242    }
13243
13244    #[test]
13245    fn an_attach_carries_through_a_kind_it_does_not_know() {
13246        // The first of section 3.2's three rules, at the point where it is easiest to break: a build
13247        // that rewrites a directory has to carry an entry it has no name for, or opening a file with
13248        // an older binary and attaching one section quietly deletes the work of a newer one.
13249        let path = linked_file("attach_unknown", 16);
13250        let payload = vec![3_u8; 96];
13251        attach(
13252            &path,
13253            "items",
13254            &[section::Attachment {
13255                kind: *b"RUDBZZ9\0",
13256                id: 1,
13257                flags: 0,
13258                header_bytes: 0,
13259                bytes: &payload,
13260            }],
13261        )
13262        .expect("a kind this build does not know still writes");
13263        let key_map = a_key_map_payload();
13264        attach(
13265            &path,
13266            "items",
13267            &[section::Attachment {
13268                kind: *section::KEY_MAP,
13269                id: 0,
13270                flags: 0,
13271                header_bytes: 0,
13272                bytes: &key_map,
13273            }],
13274        )
13275        .expect("attach beside it");
13276
13277        let reader = Reader::open(&path).expect("reopen");
13278        let held = attached(reader.table());
13279        assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
13280        let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
13281        assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
13282
13283        fs::remove_file(&path).expect("clean up");
13284    }
13285
13286    #[test]
13287    fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
13288        let path = linked_file("attach_not_built", 8);
13289        attach(
13290            &path,
13291            "items",
13292            &[section::Attachment {
13293                kind: *section::FORWARD_LINK,
13294                id: 2,
13295                flags: 0,
13296                header_bytes: 0,
13297                bytes: &[],
13298            }],
13299        )
13300        .expect("record a link that did not fit the budget");
13301
13302        let reader = Reader::open(&path).expect("reopen");
13303        let held = attached(reader.table());
13304        assert_eq!(held.len(), 1);
13305        assert_eq!(held[0].extents, 0);
13306        assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
13307        assert!(reader.extents(held[0]).expect("no extent table").is_empty());
13308        assert!(reader.payload(held[0]).expect("no payload").is_empty());
13309
13310        fs::remove_file(&path).expect("clean up");
13311    }
13312
13313    #[test]
13314    fn a_payload_past_one_extent_is_split_and_joined_back() {
13315        // Issue #745's rule, exercised rather than argued. One byte past the bound is the smallest
13316        // payload that has to be two extents, and it is the case a split written for the common
13317        // size gets wrong.
13318        let path = linked_file("attach_two_extents", 8);
13319        let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
13320        attach(
13321            &path,
13322            "items",
13323            &[section::Attachment {
13324                kind: *section::KEY_MAP,
13325                id: 0,
13326                flags: 0,
13327                header_bytes: 0,
13328                bytes: &payload,
13329            }],
13330        )
13331        .expect("attach a payload past the bound");
13332
13333        let reader = Reader::open(&path).expect("reopen");
13334        let held = attached(reader.table());
13335        let extents = reader.extents(held[0]).expect("extent table");
13336        assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
13337        assert_eq!(extents[0].length, section::MAX_EXTENT);
13338        assert_eq!(extents[1].length, 1);
13339        assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
13340        // And the extent the caller wants is readable on its own, which is the point of the split.
13341        assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
13342        assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
13343
13344        fs::remove_file(&path).expect("clean up");
13345    }
13346
13347    #[test]
13348    fn a_torn_extent_is_refused_rather_than_decoded() {
13349        let path = linked_file("attach_torn", 8);
13350        let payload = a_key_map_payload();
13351        attach(
13352            &path,
13353            "items",
13354            &[section::Attachment {
13355                kind: *section::KEY_MAP,
13356                id: 0,
13357                flags: 0,
13358                header_bytes: 0,
13359                bytes: &payload,
13360            }],
13361        )
13362        .expect("attach");
13363
13364        let reader = Reader::open(&path).expect("reopen");
13365        let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
13366        let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
13367        write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
13368        drop(file);
13369
13370        let reader = Reader::open(&path).expect("the table still opens");
13371        let error = reader
13372            .payload(&reader.table().sections()[0])
13373            .expect_err("a corrupt payload is not handed out");
13374        assert!(error.to_string().contains("checksum"), "{error}");
13375        // And the table is still readable, which is section 3.1: a section that cannot be trusted
13376        // costs the query its shortcut and nothing else.
13377        assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
13378
13379        fs::remove_file(&path).expect("clean up");
13380    }
13381
13382    #[test]
13383    fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
13384        // Readable is not writable. A format 22 directory has no section block, and adding one
13385        // without moving the number in the header would leave a file claiming a format it is not.
13386        let path = linked_file("attach_old_format", 8);
13387        let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13388        write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13389        drop(file);
13390
13391        let payload = a_key_map_payload();
13392        let error = attach(
13393            &path,
13394            "items",
13395            &[section::Attachment {
13396                kind: *section::KEY_MAP,
13397                id: 0,
13398                flags: 0,
13399                header_bytes: 0,
13400                bytes: &payload,
13401            }],
13402        )
13403        .expect_err("format 22 cannot gain a section");
13404        assert!(error.to_string().contains("format 22"), "{error}");
13405        assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
13406
13407        fs::remove_file(&path).expect("clean up");
13408    }
13409
13410    #[test]
13411    fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
13412        let path = linked_file("attach_bad_header", 8);
13413        let error = attach(
13414            &path,
13415            "items",
13416            &[section::Attachment {
13417                kind: *section::KEY_MAP,
13418                id: 0,
13419                flags: 0,
13420                header_bytes: 40,
13421                bytes: &[1, 2, 3],
13422            }],
13423        )
13424        .expect_err("a writer's bug stops at the write");
13425        assert!(error.to_string().contains("header is longer"), "{error}");
13426
13427        fs::remove_file(&path).expect("clean up");
13428    }
13429
13430    #[test]
13431    fn attaching_to_a_name_the_file_does_not_hold_says_so() {
13432        let path = linked_file("attach_wrong_name", 8);
13433        let error = attach(&path, "orders", &[]).expect_err("no such table");
13434        assert!(error.to_string().contains("orders"), "{error}");
13435        fs::remove_file(&path).expect("clean up");
13436    }
13437
13438    #[test]
13439    fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
13440        // The case a complete synopsis does not cover, and the one worth the most. 16,000 rows over
13441        // 601 distinct values, 10,000 of them holding a single value and the rest spread ten apiece
13442        // over six hundred more. The writer holds 512 values, so the list is a prefix and most of
13443        // the tail is outside it. The counts inside it are still exact, because the pass recounts
13444        // the candidates that survived it, so `id = 1` is ten thousand rows rather than the
13445        // twenty six a distinct count of 601 would divide its way to.
13446        let path = path("frequency_prefix_for_the_planner");
13447        let mut writer =
13448            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13449                .expect("new file");
13450        let mut values = vec![Value::Integer(1); 10_000];
13451        for _ in 0..10 {
13452            values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
13453        }
13454        // A vector holds 8,192 rows, so this goes in as several parts. The pass that takes the
13455        // synopsis walks the whole column rather than a part, so the counts are the same either way.
13456        for part in values.chunks(8_000) {
13457            let rows = Chunk::new(vec![
13458                Vector::from_values(LogicalType::Integer, part).expect("integers"),
13459            ])
13460            .expect("one column");
13461            writer.append(&rows).expect("a part");
13462        }
13463        writer.finish().expect("commit");
13464        let reader = Reader::open(&path).expect("reopen from disk");
13465        let prefix =
13466            reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
13467        // A prefix and not the whole column, and the writer said how many rows anything left out of
13468        // it can hold.
13469        assert_eq!(prefix.entries.len(), 512);
13470        assert_eq!(prefix.omitted_max, 10);
13471        let common = Common::new(reader);
13472        assert_eq!(common.rows(), 16_000);
13473        let column = common.column("id").expect("the file has that column");
13474        assert_eq!(
13475            common.rows_with(column, &Bound::Int(1)),
13476            Stat::exact(10_000, Provenance::FrequencySynopsis)
13477        );
13478        // In the prefix, because ties go to the smaller value and the prefix reaches 1,510.
13479        assert_eq!(
13480            common.rows_with(column, &Bound::Int(1_100)),
13481            Stat::exact(10, Provenance::FrequencySynopsis)
13482        );
13483        // Outside it, and a prefix says nothing about a value it does not list. Not zero, which is
13484        // what a complete list would say, and the file holds ten rows of this one.
13485        assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
13486        // Not in the file at all, and still nothing rather than a zero. A prefix cannot tell the
13487        // two apart, which is the whole of what it gives up.
13488        assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
13489        // What the prefix left out, which is what turns the unknown above into a number. The 512
13490        // entries account for 15,110 rows, so 890 are left for the 89 values the writer dropped,
13491        // and 890 over 89 is the ten rows each of them really holds.
13492        let remainder = common.remainder(column).expect("the list is a prefix");
13493        assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
13494        assert_eq!(remainder.rows / (601 - remainder.listed), 10);
13495        fs::remove_file(&path).expect("clean up");
13496    }
13497
13498    /// A file with no table in it is a file, and opening it says so rather than failing.
13499    #[test]
13500    fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
13501        let path = path("empty");
13502        Writer::empty(&path, &[]).expect("a file with nothing in it");
13503        let catalog = Catalog::open(&path).expect("the empty file opens");
13504        assert_eq!(catalog.len(), 0);
13505        assert!(catalog.is_empty());
13506        assert_eq!(catalog.names().count(), 0);
13507        // The next generation goes over the top of it the way it goes over any other, which is what
13508        // says this is a committed file and not a special case somebody has to know about.
13509        let mut writer =
13510            Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13511                .expect("a table goes into the empty file");
13512        writer.append(&sample_ids()).expect("rows");
13513        writer.finish().expect("commit");
13514        let catalog = Catalog::open(&path).expect("the file opens again");
13515        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13516        fs::remove_file(&path).expect("clean up");
13517    }
13518
13519    /// A committed table with no rows is a name the next generation takes over, and one with rows
13520    /// is a name it refuses.
13521    ///
13522    /// The refusal is what it always was and it is load bearing: carrying a table that holds rows
13523    /// forward means reading and rewriting its pages, and a writer that quietly wrote a second
13524    /// entry under the same name would leave a file with two tables a reader cannot tell apart. An
13525    /// empty one has no pages and no reader, so there is nothing to carry and nothing to lose, and
13526    /// taking its place is what lets a schema committed by an earlier session be loaded by a stream
13527    /// instead of through memory.
13528    #[test]
13529    fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
13530        let path = path("empty-name");
13531        let field = || vec![Field::required("id", LogicalType::Integer)];
13532        Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
13533        let catalog = Catalog::open(&path).expect("the file opens");
13534        assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
13535
13536        let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
13537        writer.append(&sample_ids()).expect("rows");
13538        writer.finish().expect("commit");
13539        let catalog = Catalog::open(&path).expect("the file opens again");
13540        // One entry and not two. The generation replaced the empty table rather than joining it.
13541        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13542        let held = catalog.rows().collect::<Vec<_>>();
13543        assert_eq!(held.len(), 1);
13544        assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
13545
13546        // The same call against the same name now that it holds rows, which is still refused.
13547        let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
13548        assert!(error.to_string().contains("same name"), "{error}");
13549        fs::remove_file(&path).expect("clean up");
13550    }
13551
13552    /// A view, with everything about it that a reopened catalog has to be able to answer from.
13553    fn sample_view(name: &str) -> ViewEntry {
13554        ViewEntry {
13555            name: name.to_string(),
13556            sql: "SELECT id FROM items WHERE id > 0".to_string(),
13557            statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
13558            aliases: vec!["n".to_string()],
13559            columns: vec![Field::new("n", LogicalType::Integer)],
13560        }
13561    }
13562
13563    #[test]
13564    fn a_view_written_into_the_catalog_comes_back_whole() {
13565        let path = path("views");
13566        let mut writer =
13567            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13568                .expect("new file");
13569        writer.append(&sample_ids()).expect("rows");
13570        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13571        let catalog = Catalog::open(&path).expect("reopen");
13572        assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
13573        // The tables are still there and are still read the same way, so the section on the end did
13574        // not move anything in front of it.
13575        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13576        fs::remove_file(&path).expect("clean up");
13577    }
13578
13579    /// A writer opened to append a table says nothing about views and must not lose them.
13580    #[test]
13581    fn appending_a_table_carries_the_views_forward() {
13582        let path = path("viewscarry");
13583        let mut writer =
13584            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13585                .expect("new file");
13586        writer.append(&sample_ids()).expect("rows");
13587        writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13588        let mut writer =
13589            Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
13590                .expect("a second table");
13591        writer.append(&sample_ids()).expect("rows");
13592        writer.finish().expect("commit");
13593        let catalog = Catalog::open(&path).expect("reopen");
13594        assert_eq!(catalog.views().count(), 1);
13595        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
13596        fs::remove_file(&path).expect("clean up");
13597    }
13598
13599    /// The whole point of [`Writer::restate`]: the views change and the pages do not move.
13600    #[test]
13601    fn restating_the_views_leaves_every_table_where_it_was() {
13602        let path = path("restate");
13603        let mut writer =
13604            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13605                .expect("new file");
13606        writer.append(&sample_ids()).expect("rows");
13607        writer.finish().expect("commit");
13608        let before = fs::metadata(&path).expect("the file is there").len();
13609        Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
13610        let catalog = Catalog::open(&path).expect("reopen");
13611        assert_eq!(catalog.views().count(), 2);
13612        assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13613        // A catalog on the end and nothing else, so what it grew by is the size of a catalog rather
13614        // than the size of the table.
13615        let after = fs::metadata(&path).expect("the file is there").len();
13616        assert!(after > before, "a generation was written");
13617        assert!(after - before < before, "the table was not written again");
13618        // The rows are still readable through the new generation, which is the part that would go
13619        // wrong if the catalog carried the wrong directory pointers forward.
13620        let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
13621        assert_eq!(reader.table().rows, 3);
13622        // And a restate over a restate keeps working, because each one reads the slot that
13623        // checksummed rather than the highest number in the header.
13624        Writer::restate(&path, &[]).expect("no views at all");
13625        assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
13626        fs::remove_file(&path).expect("clean up");
13627    }
13628
13629    /// Two entries under one name is a catalog no lookup can answer, whichever two they are.
13630    #[test]
13631    fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
13632        let bytes = encode_catalog(
13633            &[Entry {
13634                name: "items".to_string(),
13635                fields: vec![Field::required("id", LogicalType::Integer)],
13636                rows: 1,
13637                directory: Page { offset: HEADER, length: 8, hash: 0 },
13638                nonzero: vec![None],
13639                aggregates: vec![None],
13640                distincts: vec![None],
13641                extremes: vec![None],
13642                frequencies: vec![None],
13643            }],
13644            &[sample_view("items")],
13645        )
13646        .expect("it encodes, because encoding does not look");
13647        let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
13648        assert!(error.to_string().contains("same name"), "{error}");
13649    }
13650
13651    /// A compressed text page read at some rows is those rows of the page read whole, nulls and
13652    /// all, and a row past the end or rows out of order are refused rather than guessed at.
13653    #[test]
13654    fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
13655        let rows: usize = 300;
13656        let text: Vec<String> =
13657            (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
13658        let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
13659        let mut page = vec![6, 2];
13660        page.extend((0..rows.div_ceil(8)).map(|byte| {
13661            (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
13662        }));
13663        let compressed = string::encode_only(string::Kind::Fsst, &values)
13664            .expect("encoded")
13665            .expect("text this repetitive compresses");
13666        page.extend_from_slice(&compressed);
13667        let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
13668        let positions = [0_u32, 3, 8, 13, 200, 299];
13669        let some =
13670            decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
13671        assert_eq!(some.len(), positions.len());
13672        for (at, &row) in positions.iter().enumerate() {
13673            assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
13674        }
13675        assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
13676        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
13677        assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
13678    }
13679
13680    /// Every column of a part read at some rows is the part read whole and gathered, whatever the
13681    /// page holds.
13682    #[test]
13683    fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
13684        let path = path("rows");
13685        let mut writer = Writer::create(
13686            &path,
13687            "items",
13688            vec![
13689                Field::required("id", LogicalType::Integer),
13690                Field::new("text", LogicalType::Varchar),
13691            ],
13692        )
13693        .expect("new file");
13694        let rows = 2_000;
13695        let chunk = Chunk::new(vec![
13696            Vector::from_values(
13697                LogicalType::Integer,
13698                &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
13699            )
13700            .expect("integers"),
13701            Vector::from_values(
13702                LogicalType::Varchar,
13703                &(0..rows)
13704                    .map(|row| {
13705                        if row % 7 == 2 {
13706                            Value::Null
13707                        } else {
13708                            Value::Varchar(format!("a comment about order {}", row * 13))
13709                        }
13710                    })
13711                    .collect::<Vec<_>>(),
13712            )
13713            .expect("strings"),
13714        ])
13715        .expect("matching rows");
13716        writer.append(&chunk).expect("one part");
13717        writer.finish().expect("commit");
13718        let reader = Reader::open(&path).expect("reopen from disk");
13719        let positions = [1_u32, 2, 9, 1_000, 1_999];
13720        for whole in [true, false] {
13721            let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
13722            let all = reader.read(0, &[0, 1]).expect("the whole part");
13723            assert_eq!(some.len(), positions.len());
13724            for column in 0..2 {
13725                for (at, &row) in positions.iter().enumerate() {
13726                    assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
13727                }
13728            }
13729        }
13730        assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
13731    }
13732
13733    #[test]
13734    fn committed_file_reopens_and_reads_only_requested_columns() {
13735        let path = path("reopen");
13736        let mut writer = Writer::create(
13737            &path,
13738            "items",
13739            vec![
13740                Field::required("id", LogicalType::Integer),
13741                Field::new("text", LogicalType::Varchar),
13742            ],
13743        )
13744        .expect("new file");
13745        writer.append(&sample()).expect("first part");
13746        writer.append(&sample()).expect("second part");
13747        writer.finish().expect("commit");
13748        let reader = Reader::open(&path).expect("reopen from disk");
13749        assert_eq!(reader.table().rows(), 6);
13750        // Two appends below the stripe bound are two parts of one stripe, which is the whole point
13751        // of the split: the directory describes the stripe and the scan still reads a part.
13752        assert_eq!(reader.table().stripes().len(), 1);
13753        assert_eq!(reader.parts(), 2);
13754        assert_eq!(reader.part_rows(0), 3);
13755        assert_eq!(reader.part_rows(1), 3);
13756        let text = reader.read(1, &[1]).expect("only text page");
13757        assert_eq!(text.width(), 1);
13758        assert_eq!(text.value_at(1, 0), Value::Null);
13759        assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13760        let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
13761        assert_eq!(sparse.width(), 1);
13762        assert_eq!(sparse.value_at(1, 0), Value::Null);
13763        assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13764        assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
13765        assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
13766        assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
13767        let count = reader.read(0, &[]).expect("no page is needed for count");
13768        assert_eq!(count.len(), 3);
13769        assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
13770        assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
13771        assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
13772        let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
13773        assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
13774        assert_eq!(integers.omitted_max, 2);
13775        let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
13776        assert_eq!(strings.len(), 3);
13777        assert!(strings.contains(&(Value::Null, 2)));
13778        assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
13779        assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
13780        fs::remove_file(path).expect("remove scratch file");
13781    }
13782
13783    /// Two pipeline instances handing over whole runs, which is what makes the native sink safe to
13784    /// instance.
13785    ///
13786    /// The runs arrive in the order the instances finished reading them rather than in source
13787    /// order, and the second one to finish is the one that read the earlier rows. Each run is still
13788    /// a stripe of its own and the table still reads back in source order, which is the whole of
13789    /// what the writer promises about ordering.
13790    #[test]
13791    fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
13792        let path = path("interleaved-runs");
13793        let mut writer =
13794            Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
13795                .expect("new file");
13796        for morsel in [2_u64, 0, 3, 1] {
13797            let parts = (0..4_u64)
13798                .map(|chunk| {
13799                    let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
13800                    let values =
13801                        (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
13802                    let column =
13803                        Vector::from_values(LogicalType::BigInt, &values).expect("a column");
13804                    ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
13805                })
13806                .collect::<Vec<_>>();
13807            writer.append_stripe(parts).expect("a stripe");
13808        }
13809        writer.finish().expect("commit");
13810
13811        let reader = Reader::open(&path).expect("valid directory");
13812        assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
13813        assert_eq!(reader.table().rows(), 128);
13814        for part in 0..16_usize {
13815            let read = reader.read(part, &[0]).expect("a part back");
13816            for row in 0..8_usize {
13817                let want = i64::try_from(part * 8 + row).expect("small");
13818                assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
13819            }
13820        }
13821        fs::remove_file(path).expect("remove scratch file");
13822    }
13823
13824    /// Runs from different callers may interleave and may not overlap, and the commit is what
13825    /// catches an overlap.
13826    #[test]
13827    fn runs_that_overlap_each_other_are_refused_at_commit() {
13828        let path = path("overlapping-runs");
13829        let mut writer =
13830            Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
13831                .expect("new file");
13832        let one = |order: (u64, u64)| {
13833            let column =
13834                Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
13835            (order, Chunk::new(vec![column]).expect("one column"))
13836        };
13837        // The second run sits inside the first rather than after it, which is a thing no instance
13838        // holding its own contiguous run can produce and a thing the file cannot represent.
13839        writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
13840        writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
13841        let error = writer.finish().expect_err("the runs overlap");
13842        assert!(error.message().contains("source order"), "{error}");
13843        fs::remove_file(path).expect("remove scratch file");
13844    }
13845
13846    /// A stripe holds [`STRIPE_PARTS`] parts, so a run longer than that is a caller bug rather than
13847    /// something to split, and the writer says so at the door instead of quietly cutting it in two.
13848    #[test]
13849    fn a_run_longer_than_a_stripe_is_refused() {
13850        let path = path("overlong-run");
13851        let mut writer =
13852            Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
13853                .expect("new file");
13854        let parts = (0..=STRIPE_PARTS)
13855            .map(|at| {
13856                let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
13857                    .expect("a column");
13858                let chunk = Chunk::new(vec![column]).expect("one column");
13859                ((0, u64::try_from(at).expect("small")), chunk)
13860            })
13861            .collect::<Vec<_>>();
13862        let error = writer.append_stripe(parts).expect_err("one part too many");
13863        assert!(error.message().contains("more parts than it holds"), "{error}");
13864        fs::remove_file(path).expect("remove scratch file");
13865    }
13866
13867    /// Parts past the stripe bound start a new stripe, and every part stays addressable on its own.
13868    ///
13869    /// This is the shape the format exists for, so both ends of the split are checked here. The
13870    /// directory holds three stripes rather than a hundred and thirty one, and a read of any one
13871    /// part still answers with that part's rows rather than with its whole stripe's.
13872    #[test]
13873    fn parts_past_the_stripe_bound_start_a_new_stripe() {
13874        let path = path("stripe-bound");
13875        let mut writer = Writer::create(
13876            &path,
13877            "items",
13878            vec![
13879                Field::required("id", LogicalType::Integer),
13880                Field::new("text", LogicalType::Varchar),
13881            ],
13882        )
13883        .expect("new file");
13884        let parts = STRIPE_PARTS * 2 + 3;
13885        for part in 0..parts {
13886            let id = part as i32;
13887            let chunk = Chunk::new(vec![
13888                Vector::from_values(
13889                    LogicalType::Integer,
13890                    &[Value::Integer(id), Value::Integer(-id)],
13891                )
13892                .expect("integers"),
13893                Vector::from_values(
13894                    LogicalType::Varchar,
13895                    &[Value::Varchar(format!("value {part}")), Value::Null],
13896                )
13897                .expect("strings"),
13898            ])
13899            .expect("matching rows");
13900            writer.append(&chunk).expect("one part");
13901        }
13902        writer.finish().expect("commit");
13903
13904        let reader = Reader::open(&path).expect("reopen from disk");
13905        assert_eq!(reader.parts(), parts);
13906        assert_eq!(reader.table().rows(), parts * 2);
13907        assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
13908        assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
13909        assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
13910        assert_eq!(reader.table().stripes()[2].parts(), 3);
13911        // Backwards on purpose. The reader keeps four stripes a column, so a scan that walks the
13912        // table the other way is what catches a cache that only ever holds what it just read.
13913        for part in (0..parts).rev() {
13914            let dense = reader.read(part, &[0, 1]).expect("a whole page read");
13915            let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
13916            for chunk in [&dense, &sparse] {
13917                assert_eq!(chunk.len(), 2, "part {part} has its own row count");
13918                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13919                assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
13920                assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
13921                assert_eq!(chunk.value_at(1, 1), Value::Null);
13922            }
13923        }
13924        // The bounds are merged over the stripe, so they answer for the range the whole stripe
13925        // covers and not for the part that was asked about.
13926        let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
13927        assert!(reader.skips(0, &above), "the first stripe stops at 63");
13928        assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
13929        fs::remove_file(path).expect("remove scratch file");
13930    }
13931
13932    /// A scattered value in the column that decides `WHERE UserID = ?`.
13933    fn scattered(n: i64) -> i64 {
13934        n.wrapping_mul(-7_046_029_254_386_353_131)
13935    }
13936
13937    /// A part whose sieve does not hold the constant is skipped, and a range would skip none of them.
13938    ///
13939    /// This is ClickBench query 19 in miniature. The values are spread over the whole of `BIGINT`, so
13940    /// every stripe's bounds cover nearly all of it and rule out nothing, and the part that really
13941    /// holds the value is the only one a scan has to read.
13942    #[test]
13943    fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
13944        let path = path("sieve-skip");
13945        let mut writer =
13946            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
13947                .expect("new file");
13948        let parts = STRIPE_PARTS + 3;
13949        // Big enough that the filter is worth its bytes. A part of eight numbers packs to under a
13950        // hundred bytes and the smallest filter there is is sixty nine, so a filter over a part
13951        // that small costs about as much to read as the rows do and is no longer written.
13952        let per_part = 128;
13953        for part in 0..parts {
13954            let held: Vec<Value> = (0..per_part)
13955                .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
13956                .collect();
13957            let chunk =
13958                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13959                    .expect("one column");
13960            writer.append(&chunk).expect("one part");
13961        }
13962        writer.finish().expect("commit");
13963
13964        let reader = Reader::open(&path).expect("reopen from disk");
13965        let probe = |value: i64| Probe {
13966            column: 0,
13967            op: Op::Equal,
13968            value: Bound::Int(i128::from(scattered(value))),
13969        };
13970        for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
13971            let tests = [probe(wanted)];
13972            let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
13973            let home = wanted as usize / per_part;
13974            assert!(kept.contains(&home), "the part holding {wanted} is read");
13975            // A filter answers maybe, so a part it keeps need not hold the value. Sixty seven parts
13976            // of a hundred and twenty eight numbers each, at a dozen bits a value, is about one
13977            // stray part across the whole file and that is what this leaves room for.
13978            assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
13979        }
13980        let absent = [probe((parts * per_part) as i64 + 1)];
13981        let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
13982        assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
13983        // The same probes against the bounds alone, which is what this replaces. A column of
13984        // scattered numbers has a range per stripe that covers nearly the whole type.
13985        let tests = [probe(0)];
13986        assert!(
13987            reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
13988            "the bounds rule out no stripe at all"
13989        );
13990        fs::remove_file(path).expect("remove scratch file");
13991    }
13992
13993    /// A part whose own bounds rule out an ordered comparison is skipped where the stripe's keep it.
13994    ///
13995    /// This is the shape of ClickBench 24. Each part covers a narrow stretch of the column and the
13996    /// stripe covers all sixty four of them at once, so a comparison that lands inside the stripe
13997    /// rules out none of it and rules out all but a few parts.
13998    #[test]
13999    fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
14000        let path = path("part-range-skip");
14001        let mut writer =
14002            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14003                .expect("new file");
14004        let parts = STRIPE_PARTS + 3;
14005        let per_part = 128;
14006        for part in 0..parts {
14007            // Scattered inside the part's own band rather than a run, because a run of
14008            // consecutive numbers encodes to a stride of a few bytes and then the page of ranges
14009            // costs more than reading the column it indexes, which is the case the writer declines.
14010            let held: Vec<Value> = (0..per_part)
14011                .map(|row| {
14012                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14013                })
14014                .collect();
14015            let chunk =
14016                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14017                    .expect("one column");
14018            writer.append(&chunk).expect("one part");
14019        }
14020        writer.finish().expect("commit");
14021
14022        let reader = Reader::open(&path).expect("reopen from disk");
14023        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14024        let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
14025        assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
14026        // The same question asked of the stripe alone, which is what this replaces.
14027        assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
14028        fs::remove_file(path).expect("remove scratch file");
14029    }
14030
14031    /// The other half of the same page. A part whose own bounds put every row of it inside the
14032    /// filter is waved through, so the comparison never runs on it, where the stripe's bounds reach
14033    /// across every part and can prove nothing.
14034    #[test]
14035    fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
14036        let path = path("part-range-certain");
14037        let mut writer =
14038            Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14039                .expect("new file");
14040        let parts = STRIPE_PARTS + 3;
14041        let per_part = 128;
14042        for part in 0..parts {
14043            let held: Vec<Value> = (0..per_part)
14044                .map(|row| {
14045                    Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14046                })
14047                .collect();
14048            let chunk =
14049                Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14050                    .expect("one column");
14051            writer.append(&chunk).expect("one part");
14052        }
14053        writer.finish().expect("commit");
14054
14055        let reader = Reader::open(&path).expect("reopen from disk");
14056        let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14057        let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
14058        assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
14059        // The first stripe reaches from zero to past sixty thousand, so it straddles three thousand
14060        // and settles nothing either way. The three yeses above are the parts' own ends talking.
14061        assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
14062        fs::remove_file(path).expect("remove scratch file");
14063    }
14064
14065    /// The page is worth its bytes on a column with parts to tell apart and is not written on one
14066    /// that has a single part, where the stripe bounds already are the part's.
14067    #[test]
14068    fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
14069        for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
14070            let path = path("part-range-page");
14071            let mut writer =
14072                Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14073                    .expect("new file");
14074            for part in 0..parts {
14075                let held: Vec<Value> = (0..128)
14076                    .map(|row| {
14077                        Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
14078                    })
14079                    .collect();
14080                let chunk = Chunk::new(vec![
14081                    Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
14082                ])
14083                .expect("one column");
14084                writer.append(&chunk).expect("one part");
14085            }
14086            writer.finish().expect("commit");
14087            let reader = Reader::open(&path).expect("reopen from disk");
14088            let bytes = reader.layout().columns[0].part_ranges;
14089            assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
14090            fs::remove_file(path).expect("remove scratch file");
14091        }
14092    }
14093
14094    /// A cut down string end is still an end on the side it was, which is the only thing that keeps
14095    /// a shortened bound from turning a skip into a wrong answer.
14096    #[test]
14097    fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
14098        let long = vec![b'a'; PART_BOUND_BYTES * 2];
14099        let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
14100        let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
14101        let Bound::Bytes(low) = low else { panic!("a string stays a string") };
14102        let Bound::Bytes(high) = high else { panic!("a string stays a string") };
14103        assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
14104        assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
14105        assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
14106    }
14107
14108    /// A string of nothing but the largest byte has no prefix that can be stepped up, so the high
14109    /// end is given up rather than claimed too small. No end keeps the part, which is always safe.
14110    #[test]
14111    fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
14112        let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
14113        assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
14114        let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
14115        assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
14116    }
14117
14118    /// What a column is stored as, asked of two files holding the same rows in a different order.
14119    ///
14120    /// This is the question the report exists to answer and it is the one the directory cannot. The
14121    /// two files have the same rows, the same schema and the same number of parts, and the column
14122    /// comes out four times smaller in one of them, because ascending keys delta encode to a few
14123    /// bits a row and shuffled ones do not. Nothing about the file's shape says so. The page header
14124    /// says so, and reading it is what this does.
14125    ///
14126    /// It is q18 on TPC-H in miniature: clustering lineitem by ship date leaves `l_orderkey`
14127    /// ascending inside a partition but sparse, its deltas go from six bits to twelve, and the scan
14128    /// pays for the wider ones.
14129    #[test]
14130    fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
14131        let parts = 4;
14132        let per_part = 1024;
14133        let rows = parts * per_part;
14134        let written = |name: &str, keys: &[i64]| {
14135            let path = path(name);
14136            let fields = vec![Field::required("key", LogicalType::BigInt)];
14137            let mut writer = Writer::create(&path, "keys", fields).expect("new file");
14138            for part in 0..parts {
14139                let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
14140                    .iter()
14141                    .map(|key| Value::BigInt(*key))
14142                    .collect();
14143                let chunk = Chunk::new(vec![
14144                    Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
14145                ])
14146                .expect("one column");
14147                writer.append(&chunk).expect("one part");
14148            }
14149            writer.finish().expect("commit");
14150            path
14151        };
14152        // Ascending with a small irregular step, which is what a key column in arrival order looks
14153        // like: an order has one to seven line items, so the key repeats and then moves on by one.
14154        let climbing = |step: &dyn Fn(usize) -> i64| {
14155            let mut key = 0;
14156            (0..rows)
14157                .map(|row| {
14158                    key += step(row);
14159                    key
14160                })
14161                .collect::<Vec<i64>>()
14162        };
14163        let ascending = climbing(&|row| (row % 3) as i64);
14164        // The same rows in the same direction over a range a thousand times wider, which is what a
14165        // partition of a clustered table holds: still ascending, and far enough apart that the
14166        // deltas no longer fit in a handful of bits.
14167        let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
14168        let near_path = written("stored-near", &ascending);
14169        let far_path = written("stored-far", &sparse);
14170
14171        let one = Reader::open(&near_path).expect("reopen from disk");
14172        let other = Reader::open(&far_path).expect("reopen from disk");
14173        let near = one.stored(0).expect("the column is stored");
14174        let far = other.stored(0).expect("the column is stored");
14175        assert_eq!(near.len(), parts, "one row per part");
14176        assert_eq!(far.len(), parts);
14177        // The bytes are the same bytes the directory totals, which is the check that this is
14178        // reading the pages the file really holds rather than some other pages.
14179        let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
14180        assert_eq!(total(&near), one.layout().columns[0].pages);
14181        assert_eq!(total(&far), other.layout().columns[0].pages);
14182        assert!(
14183            total(&near) * 2 < total(&far),
14184            "the sparse keys cost more, {} against {}",
14185            total(&far),
14186            total(&near)
14187        );
14188        // Every part accounted for, in order, with the row it starts at following the one before.
14189        for (at, part) in near.iter().enumerate() {
14190            assert_eq!(part.part, at);
14191            assert_eq!(part.row, at * per_part);
14192            assert_eq!(part.rows, per_part);
14193            let held = &ascending[at * per_part..(at + 1) * per_part];
14194            assert_eq!(part.low, Some(Value::BigInt(held[0])));
14195            assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
14196            assert_eq!(part.nulls, Some(0));
14197        }
14198        // And the encoding is a line of text that names what the encoder chose, which is the whole
14199        // point. Both are a cascade over deltas and the widths inside them are what differ.
14200        assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
14201        assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
14202        assert_ne!(near[0].encoding, far[0].encoding);
14203        fs::remove_file(near_path).expect("remove scratch file");
14204        fs::remove_file(far_path).expect("remove scratch file");
14205    }
14206
14207    /// A sieve bigger than the part it indexes is not written, and one smaller than it still is.
14208    ///
14209    /// Both columns hold values spread over the whole of `BIGINT`, so neither gets a bitmap and both
14210    /// reach the filter. They differ in what the part costs to read. `spread` is a thousand distinct
14211    /// numbers and packs to eight kilobytes, so a filter of about thirteen hundred bytes is a good
14212    /// trade. `repeated` is the same thousand rows over four numbers in runs and encodes to
14213    /// almost nothing, but the filter is sized for the rows rather than the values it turns out to
14214    /// hold, so it comes out larger than the data. Reading it to decide whether to read the part spends more than
14215    /// the part, every time, and that is the case this drops.
14216    #[test]
14217    fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
14218        let path = path("sieve-pays");
14219        let fields = vec![
14220            Field::required("spread", LogicalType::BigInt),
14221            Field::required("repeated", LogicalType::BigInt),
14222        ];
14223        let mut writer = Writer::create(&path, "hits", fields).expect("new file");
14224        let parts = 3;
14225        let per_part = 1024;
14226        for part in 0..parts {
14227            let base = (part * per_part) as i64;
14228            let spread: Vec<Value> =
14229                (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
14230            let repeated: Vec<Value> =
14231                (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
14232            let chunk = Chunk::new(vec![
14233                Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
14234                Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
14235            ])
14236            .expect("two columns");
14237            writer.append(&chunk).expect("one part");
14238        }
14239        writer.finish().expect("commit");
14240
14241        let reader = Reader::open(&path).expect("reopen from disk");
14242        let layout = reader.layout();
14243        let spread = &layout.columns[0];
14244        let repeated = &layout.columns[1];
14245        assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
14246        assert_eq!(
14247            repeated.sieves, 0,
14248            "a column whose filter costs more than its parts keeps none"
14249        );
14250        // Per part this is the rule itself, so it holds over the column as well: a part without a
14251        // sieve adds to one side of this and to nothing on the other.
14252        for column in &layout.columns {
14253            assert!(
14254                column.sieves < column.pages,
14255                "{} spends {} on sieves over {} of data",
14256                column.name,
14257                column.sieves,
14258                column.pages
14259            );
14260        }
14261        // The filter that was kept still does what it is for.
14262        let absent = [Probe {
14263            column: 0,
14264            op: Op::Equal,
14265            value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
14266        }];
14267        assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
14268        fs::remove_file(path).expect("remove scratch file");
14269    }
14270
14271    /// A damaged sieve page is a part that gets read, not a query that fails.
14272    ///
14273    /// A sieve is an index over rows that are still there and still correct, so losing one costs
14274    /// time and costs no answers. That is the opposite of the membership index beside it, which is
14275    /// the only thing standing between a string page and a wrong answer.
14276    #[test]
14277    fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
14278        let path = path("sieve-damaged");
14279        let mut writer =
14280            Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14281                .expect("new file");
14282        let rows = 128;
14283        let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
14284        let chunk =
14285            Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14286                .expect("one column");
14287        writer.append(&chunk).expect("one part");
14288        writer.finish().expect("commit");
14289
14290        let page = Reader::open(&path).expect("reopen").table.stripes[0]
14291            .sieves
14292            .get(0)
14293            .expect("a sieve page");
14294        let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
14295        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
14296        file.write_all(&[0xff]).expect("damage one byte");
14297        drop(file);
14298
14299        let reader = Reader::open(&path).expect("reopen the damaged file");
14300        let absent =
14301            [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
14302        assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
14303        assert_eq!(
14304            reader.read(0, &[0]).expect("the rows are untouched").len(),
14305            usize::try_from(rows).expect("a small count")
14306        );
14307        fs::remove_file(path).expect("remove scratch file");
14308    }
14309
14310    /// Eight workers over one stripe read it once between them.
14311    ///
14312    /// This is the shape a scan actually has. Parts are handed out in order, so every worker on a
14313    /// column crosses into a stripe within a few parts of the others, and before [`Reader::held`]
14314    /// started sharing the read every one of them read the whole page. On the full ClickBench file
14315    /// that was a `MIN(EventDate), MAX(EventDate)` moving 3.2 GB off the disk to look at 400 MB of
14316    /// column, which is most of what a first touch costs.
14317    ///
14318    /// The workers that lose the race still answer, out of the part reads they do instead, which is
14319    /// what the values below are checking.
14320    #[test]
14321    fn workers_that_want_the_same_stripe_read_it_once() {
14322        let path = path("single-flight");
14323        let mut writer =
14324            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14325                .expect("new file");
14326        for part in 0..STRIPE_PARTS {
14327            let id = part as i32;
14328            let chunk = Chunk::new(vec![
14329                Vector::from_values(
14330                    LogicalType::Integer,
14331                    &[Value::Integer(id), Value::Integer(-id)],
14332                )
14333                .expect("integers"),
14334            ])
14335            .expect("matching rows");
14336            writer.append(&chunk).expect("one part");
14337        }
14338        writer.finish().expect("commit");
14339
14340        let reader = Reader::open(&path).expect("reopen from disk");
14341        assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
14342        let barrier = std::sync::Barrier::new(8);
14343        std::thread::scope(|scope| {
14344            for worker in 0..8 {
14345                let reader = &reader;
14346                let barrier = &barrier;
14347                scope.spawn(move || {
14348                    barrier.wait();
14349                    for part in (worker..STRIPE_PARTS).step_by(8) {
14350                        let chunk = reader.read(part, &[0]).expect("a whole page read");
14351                        assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14352                        assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14353                    }
14354                });
14355            }
14356        });
14357        assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
14358        fs::remove_file(path).expect("remove scratch file");
14359    }
14360
14361    /// Opening a file reads the header and the directory, and nothing that depends on the rows.
14362    ///
14363    /// `spec/stats/04-in-memory.md` section 4.2. There are no statistics in the file yet, so this
14364    /// holds today by not having anything to load, and that is exactly why it is worth pinning now.
14365    /// The change that breaks it is the reasonable looking one: summaries are a few hundred bytes,
14366    /// the next query will want them, so read them on the way past. A process that opened the
14367    /// database to run one trivial query pays for all of it and gets nothing.
14368    ///
14369    /// Two files of the same shape and a thousand times the rows in one of them, opened, and the
14370    /// two openings cost the same. The stripe count is held equal so that the directory is the same
14371    /// size in both, which leaves the rows as the only thing that changed. Anything read out of the
14372    /// data would show up here.
14373    #[test]
14374    fn opening_costs_the_same_over_a_thousand_times_the_rows() {
14375        let opened = |label: &str, rows_per_part: i32| {
14376            let path = path(label);
14377            let mut writer =
14378                Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14379                    .expect("new file");
14380            for part in 0..STRIPE_PARTS * 3 {
14381                // Scrambled rather than sequential, so that the fat file is actually fatter. A run
14382                // of consecutive integers encodes to almost nothing and would leave the two files
14383                // the same size, which would make this test pass for the wrong reason.
14384                let values = (0..rows_per_part)
14385                    .map(|row| {
14386                        Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
14387                    })
14388                    .collect::<Vec<_>>();
14389                let chunk = Chunk::new(vec![
14390                    Vector::from_values(LogicalType::Integer, &values).expect("integers"),
14391                ])
14392                .expect("matching rows");
14393                writer.append(&chunk).expect("one part");
14394            }
14395            writer.finish().expect("commit");
14396            let reader = Reader::open(&path).expect("reopen from disk");
14397            let size = fs::metadata(&path).expect("the file is there").len();
14398            let out = (reader.reads(), reader.table().stripes().len(), size);
14399            fs::remove_file(path).expect("remove scratch file");
14400            out
14401        };
14402
14403        let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
14404        let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
14405        assert_eq!(
14406            thin_stripes, fat_stripes,
14407            "the same stripe count is what makes this a fair ask"
14408        );
14409        assert!(
14410            fat_size > thin_size * 50,
14411            "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
14412        );
14413
14414        assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
14415        assert_eq!(thin.pages, 0, "opening read a page");
14416        assert_eq!(fat.pages, 0, "opening read a page");
14417        assert_eq!(thin.indexes, 0, "opening read an index");
14418        assert_eq!(fat.indexes, 0, "opening read an index");
14419        // Not exactly equal, because a directory holds offsets and a larger file has larger ones,
14420        // and a handful of bytes of varint is not somebody loading statistics. A factor is.
14421        assert!(
14422            fat.opening.bytes < thin.opening.bytes * 2,
14423            "opening the thin file read {} bytes and the fat one read {}",
14424            thin.opening.bytes,
14425            fat.opening.bytes
14426        );
14427    }
14428
14429    /// The reads a file costs to open are fixed by its shape and not by what ran before.
14430    ///
14431    /// `spec/stats/04-in-memory.md` section 4.3, which is the rule that keeps a plan reproducible:
14432    /// the plan is a function of the data, the generation and the settings, and never of what
14433    /// happened to be in cache. Opening the same file twice in the same process has to cost the
14434    /// same, because a second open that read less would be an open that was about to plan
14435    /// differently.
14436    #[test]
14437    fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
14438        let path = path("open-twice");
14439        let mut writer =
14440            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14441                .expect("new file");
14442        for part in 0..STRIPE_PARTS * 3 {
14443            let chunk = Chunk::new(vec![
14444                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14445                    .expect("integers"),
14446            ])
14447            .expect("matching rows");
14448            writer.append(&chunk).expect("one part");
14449        }
14450        writer.finish().expect("commit");
14451
14452        let first = Reader::open(&path).expect("open");
14453        // A whole scan in between, so the operating system's page cache is as warm as it gets and
14454        // anything that consulted it would show up in the second open.
14455        for part in 0..first.parts() {
14456            first.read(part, &[0]).expect("a part");
14457        }
14458        assert!(first.reads().pages > 0, "the scan has to have read something");
14459        let second = Reader::open(&path).expect("open again");
14460
14461        assert_eq!(first.reads().opening, second.reads().opening);
14462        assert_eq!(
14463            second.reads().pages,
14464            0,
14465            "the second open read a page off the back of the first"
14466        );
14467        assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
14468        fs::remove_file(path).expect("remove scratch file");
14469    }
14470
14471    /// A scan reads a stripe's index once for the whole scan, not once per part that misses.
14472    ///
14473    /// The page cache holds four stripes and an index used to ride inside it, so a table with more
14474    /// stripes than that read the index again every time a stripe came back around. The index is a
14475    /// few hundred bytes and the page is a quarter of a megabyte, which is why they are now under
14476    /// different budgets. This is the test that keeps them there, since the saving is small enough
14477    /// that nothing in a benchmark would notice it going away again.
14478    #[test]
14479    fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
14480        let path = path("index-cache");
14481        let mut writer =
14482            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14483                .expect("new file");
14484        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14485        for part in 0..parts {
14486            let id = part as i32;
14487            let chunk = Chunk::new(vec![
14488                Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
14489            ])
14490            .expect("matching rows");
14491            writer.append(&chunk).expect("one part");
14492        }
14493        writer.finish().expect("commit");
14494
14495        let reader = Reader::open(&path).expect("reopen from disk");
14496        let stripes = reader.table().stripes().len();
14497        assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
14498        // Twice over, so that the second pass finds every page evicted and every index kept.
14499        for _ in 0..2 {
14500            for part in 0..parts {
14501                let chunk = reader.read(part, &[0]).expect("a part");
14502                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14503            }
14504        }
14505        assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
14506        assert!(
14507            reader.pages.load(Atomic::Relaxed) > stripes,
14508            "the pages are the ones that get read again, which is what makes the index count mean \
14509             something"
14510        );
14511        fs::remove_file(path).expect("remove scratch file");
14512    }
14513
14514    /// A page stays in memory from one scan to the next while the pool has room for it, and a
14515    /// table that is being read takes room from one that is not, down to the floor and no further.
14516    ///
14517    /// This is what the pool is for. Each reader lives as long as its database, so a second query
14518    /// over the same table should find every page it read the first time, and before the pool it
14519    /// found four stripes a column and read the rest off the file again.
14520    #[test]
14521    fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
14522        let path = path("page-pool");
14523        let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
14524        let fields = || vec![Field::required("id", LogicalType::Integer)];
14525        let mut writer = Writer::create(&path, "a", fields()).expect("new file");
14526        for table in ["a", "b"] {
14527            if table == "b" {
14528                writer = writer.next("b".to_string(), fields()).expect("a second table");
14529            }
14530            for part in 0..parts {
14531                let chunk = Chunk::new(vec![
14532                    Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14533                        .expect("integers"),
14534                ])
14535                .expect("matching rows");
14536                writer.append(&chunk).expect("one part");
14537            }
14538        }
14539        writer.finish().expect("commit");
14540
14541        let pool = PagePool::new(usize::MAX);
14542        let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
14543        let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
14544        let stripes = a.table().stripes().len();
14545        assert!(
14546            stripes > CACHED_STRIPES_PER_COLUMN * 2,
14547            "the floor has to be smaller than a table"
14548        );
14549        let scan = |reader: &Reader| {
14550            for part in 0..parts {
14551                let chunk = reader.read(part, &[0]).expect("a part");
14552                assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14553            }
14554        };
14555        // The first scan keeps the newest pages of the floor and no more, so the second reads the
14556        // rest again and keeps them, and the third reads nothing.
14557        scan(&a);
14558        assert_eq!(pool.bytes(), 0, "a page read once is not the pool's");
14559        scan(&a);
14560        let twice = stripes * 2 - CACHED_STRIPES_PER_COLUMN;
14561        assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the second scan reads the rest again");
14562        scan(&a);
14563        assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the third scan reads nothing");
14564        let one = pool.bytes();
14565        assert!(one > 0, "the pool counts what the reader holds");
14566
14567        // Room for one table. Reading the other takes the first one's pages down to its floor.
14568        pool.budget.store(one, Atomic::Relaxed);
14569        scan(&b);
14570        scan(&b);
14571        assert_eq!(b.pages.load(Atomic::Relaxed), twice, "a page is never let go while in use");
14572        assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
14573        let column = a.cache.columns[0].lock().expect("the column");
14574        let held = column.pages.iter().flatten().count();
14575        assert_eq!(
14576            held,
14577            CACHED_STRIPES_PER_COLUMN + column.passing.len(),
14578            "the count and the slots agree"
14579        );
14580        drop(column);
14581
14582        // A reader that goes takes its pages out of the count with it.
14583        drop((a, b, catalog));
14584        let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
14585        scan(&c);
14586        scan(&c);
14587        assert!(pool.bytes() <= one, "only what the live reader holds is counted");
14588        fs::remove_file(path).expect("remove scratch file");
14589    }
14590
14591    /// A worker per stripe reads its stripe once, once the cache has been told how many there are.
14592    ///
14593    /// This is the shape a scan has when it hands out a whole stripe per morsel rather than a part.
14594    /// Nobody races for a page any more, but every worker holds a different one for the length of a
14595    /// stripe, so a cache that keeps four pages while eight workers are in eight stripes evicts
14596    /// every one of them before its owner has finished with it, and the owner reads a quarter of a
14597    /// megabyte again for the next part. The barrier is what makes that certain rather than likely:
14598    /// without it a worker can run a whole stripe before the next one starts and never collide.
14599    #[test]
14600    fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
14601        let workers = CACHED_STRIPES_PER_COLUMN + 4;
14602        let path = path("stripe-per-worker");
14603        let mut writer =
14604            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14605                .expect("new file");
14606        for part in 0..STRIPE_PARTS * workers {
14607            let chunk = Chunk::new(vec![
14608                Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14609                    .expect("integers"),
14610            ])
14611            .expect("matching rows");
14612            writer.append(&chunk).expect("one part");
14613        }
14614        writer.finish().expect("commit");
14615
14616        let read = |told: bool| {
14617            let reader = Reader::open(&path).expect("reopen from disk");
14618            assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
14619            if told {
14620                reader.keep_stripes(workers);
14621            }
14622            let barrier = std::sync::Barrier::new(workers);
14623            std::thread::scope(|scope| {
14624                for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
14625                    let reader = &reader;
14626                    let barrier = &barrier;
14627                    scope.spawn(move || {
14628                        for part in run {
14629                            barrier.wait();
14630                            let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
14631                            assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14632                        }
14633                        assert!(worker < workers);
14634                    });
14635                }
14636            });
14637            reader.pages.load(Atomic::Relaxed)
14638        };
14639
14640        assert_eq!(read(true), workers, "one page read per stripe and no more");
14641        assert!(read(false) > workers, "a cache that small is read again on every part");
14642        fs::remove_file(path).expect("remove scratch file");
14643    }
14644
14645    /// A damaged index page is caught before anything decodes a part out of it.
14646    ///
14647    /// The index is the one structure a reader trusts to find bytes with, so it carries a checksum
14648    /// per column section rather than one for the page, and this is what says that check runs.
14649    #[test]
14650    fn a_damaged_index_page_is_an_error() {
14651        let path = path("damaged-index");
14652        let mut writer =
14653            Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14654                .expect("new file");
14655        writer.append(&sample_ids()).expect("first part");
14656        writer.append(&sample_ids()).expect("second part");
14657        writer.finish().expect("commit");
14658
14659        let reader = Reader::open(&path).expect("valid directory");
14660        let index = reader.table.stripes[0].index;
14661        let mut byte = [0; 1];
14662        read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
14663        let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
14664        file.seek(SeekFrom::Start(index.offset)).expect("index start");
14665        file.write_all(&[!byte[0]]).expect("damage the first part length");
14666        let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
14667        assert!(error.message().contains("index page section checksum differs"), "{error}");
14668        fs::remove_file(path).expect("remove scratch file");
14669    }
14670
14671    /// Every integer width the format knows about, written and read back.
14672    ///
14673    /// The unsigned ones are the reason ClickBench can be stored at all: `hits` types `EventDate`
14674    /// as `USMALLINT`, and one unsupported column meant the whole table was refused. The extremes
14675    /// are in here on purpose, because a width that round trips through the wrong signedness only
14676    /// goes wrong at the end of its range.
14677    #[test]
14678    fn every_integer_width_round_trips_through_a_page() {
14679        let path = path("integer-widths");
14680        let columns = [
14681            (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
14682            (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
14683            (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
14684            (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
14685            (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
14686            (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
14687            (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
14688            (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
14689        ];
14690        let fields = columns
14691            .iter()
14692            .enumerate()
14693            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14694            .collect::<Vec<_>>();
14695        let vectors = columns
14696            .iter()
14697            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14698            .collect::<Vec<_>>();
14699        let mut writer = Writer::create(&path, "widths", fields).expect("new file");
14700        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14701        writer.finish().expect("commit");
14702
14703        let reader = Reader::open(&path).expect("reopen from disk");
14704        let wanted = (0..columns.len()).collect::<Vec<_>>();
14705        let read = reader.read(0, &wanted).expect("every column");
14706        assert_eq!(read.len(), 2);
14707        // row at a time: each column has its own type and its own pair of extremes.
14708        for (at, (ty, values)) in columns.iter().enumerate() {
14709            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14710            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14711        }
14712        fs::remove_file(path).expect("remove scratch file");
14713    }
14714
14715    /// The rest of the fixed width types, and the byte strings, written and read back.
14716    ///
14717    /// The extremes again, and for a float that means more than the ends of the range. Negative
14718    /// zero and a NaN are the two values that go through an encoder unnoticed and come back
14719    /// different, so they are here on purpose, and the NaN is compared by its bits rather than by
14720    /// `==`, which a NaN fails against itself.
14721    ///
14722    /// A blob is here beside them because it is the same round trip asked of bytes that are not
14723    /// text. The value in it is not UTF-8, so a path that reads a payload as a string on the way
14724    /// past turns this test red rather than turning a user's column into nulls.
14725    #[test]
14726    fn every_other_type_the_format_knows_round_trips_through_a_page() {
14727        let path = path("other-types");
14728        let columns = [
14729            (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
14730            (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
14731            (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
14732            (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
14733            (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
14734            (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
14735            (
14736                LogicalType::TimestampTz,
14737                vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
14738            ),
14739            (
14740                LogicalType::Interval,
14741                vec![
14742                    Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
14743                    Value::Interval { months: 13, days: -1, micros: 1 },
14744                ],
14745            ),
14746            (
14747                LogicalType::Blob,
14748                vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
14749            ),
14750        ];
14751        let fields = columns
14752            .iter()
14753            .enumerate()
14754            .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14755            .collect::<Vec<_>>();
14756        let vectors = columns
14757            .iter()
14758            .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14759            .collect::<Vec<_>>();
14760        let mut writer = Writer::create(&path, "others", fields).expect("new file");
14761        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14762        writer.finish().expect("commit");
14763
14764        let reader = Reader::open(&path).expect("reopen from disk");
14765        let wanted = (0..columns.len()).collect::<Vec<_>>();
14766        let read = reader.read(0, &wanted).expect("every column");
14767        assert_eq!(read.len(), 2);
14768        for (at, (ty, values)) in columns.iter().enumerate() {
14769            assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14770            assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14771        }
14772        // A float keeps its sign through a zero, which `==` says nothing about because negative
14773        // zero and zero compare equal.
14774        let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
14775        assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
14776
14777        fs::remove_file(path).expect("remove scratch file");
14778    }
14779
14780    /// A NaN is still a NaN after a trip through a page.
14781    ///
14782    /// Apart from the other floats because it cannot be asserted the same way. A NaN is not equal
14783    /// to itself, so a comparison against the value that was written passes for every NaN and for
14784    /// nothing else, which is the one assertion that would not catch a page that lost it.
14785    #[test]
14786    fn a_nan_survives_being_written_down() {
14787        let path = path("nan");
14788        let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
14789            .expect("a NaN vector");
14790        let mut writer =
14791            Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
14792                .expect("new file");
14793        writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
14794        writer.finish().expect("commit");
14795        let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
14796        let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
14797        assert!(back.is_nan(), "a NaN came back as {back}");
14798        fs::remove_file(path).expect("remove scratch file");
14799    }
14800
14801    /// A uuid and a bit string, which have no `Value` arm of their own and are checked as bits.
14802    ///
14803    /// A uuid is the 128 bit lane and a bit string is bytes, and neither of them reads back as
14804    /// anything in `Value` today, so asking for a value here would compare two nulls and pass
14805    /// whatever the file held. The data underneath is what the storage promise is about, so that is
14806    /// what this reads.
14807    #[test]
14808    fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
14809        let path = path("uuid-and-bit");
14810        let uuids = vec![0_i128, i128::MIN, -1];
14811        let mut bits = StringColumn::new();
14812        for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
14813            bits.push_bytes(value);
14814        }
14815        let expected = bits.clone();
14816        let fields =
14817            vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
14818        let vectors = vec![
14819            Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
14820            Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
14821        ];
14822        let mut writer = Writer::create(&path, "ids", fields).expect("new file");
14823        writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14824        writer.finish().expect("commit");
14825
14826        let reader = Reader::open(&path).expect("reopen from disk");
14827        let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
14828        let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
14829            panic!("a uuid column is the 128 bit lane")
14830        };
14831        assert_eq!(back.as_slice(), uuids.as_slice());
14832        let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
14833            panic!("a bit column is bytes")
14834        };
14835        for row in 0..expected.len() {
14836            assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
14837        }
14838        fs::remove_file(path).expect("remove scratch file");
14839    }
14840
14841    /// Counting a run at once has to leave the candidate table exactly where counting its rows one
14842    /// at a time would, including once the table is full and a run is turned away row by row.
14843    #[test]
14844    fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
14845        let mut rows: Vec<Option<u64>> = Vec::new();
14846        let mut state = 0x2545_f491_4f6c_dd1d_u64;
14847        for index in 0..400_000_u64 {
14848            state ^= state << 13;
14849            state ^= state >> 7;
14850            state ^= state << 17;
14851            let times = 1 + (state % 7) as usize;
14852            let bits = match state % 11 {
14853                0 => None,
14854                1..=3 => Some(state % 16),
14855                _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
14856            };
14857            rows.extend(std::iter::repeat_n(bits, times));
14858        }
14859        let mut by_row = Candidates::default();
14860        for &bits in &rows {
14861            by_row.add(bits, 1);
14862        }
14863        let mut by_run = Candidates::default();
14864        let mut run = Run::default();
14865        let mut runs = 0_usize;
14866        for &bits in &rows {
14867            if let Some((bits, times)) = run.push(bits) {
14868                by_run.add(bits, times);
14869                runs += 1;
14870            }
14871        }
14872        if let Some((bits, times)) = run.take() {
14873            by_run.add(bits, times);
14874        }
14875        assert!(runs < rows.len() / 2, "the rows came in runs");
14876        assert!(by_row.decrements > 0, "the table filled and turned values away");
14877        assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
14878        assert_eq!(by_run.nulls, by_row.nulls);
14879        assert_eq!(by_run.decrements, by_row.decrements);
14880    }
14881
14882    fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
14883        let mut pairs = candidates.pairs().collect::<Vec<_>>();
14884        pairs.sort_unstable();
14885        assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
14886        pairs
14887    }
14888
14889    /// The Misra-Gries table as it was written over a `HashMap`, kept as the oracle the open
14890    /// addressed one has to agree with.
14891    #[derive(Default)]
14892    struct MapCandidates {
14893        counts: HashMap<u64, u32>,
14894        nulls: u32,
14895        decrements: u64,
14896    }
14897
14898    impl MapCandidates {
14899        fn add(&mut self, bits: Option<u64>, mut times: u32) {
14900            while times > 0 {
14901                let held = match bits {
14902                    Some(bits) => self.counts.get_mut(&bits),
14903                    None if self.nulls != 0 => Some(&mut self.nulls),
14904                    None => None,
14905                };
14906                if let Some(count) = held {
14907                    *count = count.saturating_add(times);
14908                    return;
14909                }
14910                if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
14911                    match bits {
14912                        Some(bits) => {
14913                            self.counts.insert(bits, times);
14914                        }
14915                        None => self.nulls = times,
14916                    }
14917                    return;
14918                }
14919                self.counts.retain(|_, count| {
14920                    *count -= 1;
14921                    *count != 0
14922                });
14923                self.nulls = self.nulls.saturating_sub(1);
14924                self.decrements = self.decrements.saturating_add(1);
14925                times -= 1;
14926            }
14927        }
14928    }
14929
14930    /// Near unique values, a few heavy ones, nulls, and runs, through enough rows that the table
14931    /// fills, grows through every size and is decremented many times over. Both tables have to hold
14932    /// the same candidates with the same counts at the end, and at points along the way.
14933    #[test]
14934    fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
14935        for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
14936            let mut table = Candidates::default();
14937            let mut oracle = MapCandidates::default();
14938            let mut state = seed;
14939            for index in 0..300_000_u64 {
14940                state ^= state << 13;
14941                state ^= state >> 7;
14942                state ^= state << 17;
14943                let bits = match state % 13 {
14944                    0 => None,
14945                    1..=4 => Some(state % 40),
14946                    5 => Some((index % 1000) * 1_000_000),
14947                    _ => Some(state),
14948                };
14949                let times = 1 + (state >> 60) as u32 % 3;
14950                table.add(bits, times);
14951                oracle.add(bits, times);
14952                if index % 50_000 == 0 {
14953                    let mut expected =
14954                        oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14955                    expected.sort_unstable();
14956                    assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
14957                }
14958            }
14959            let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14960            expected.sort_unstable();
14961            assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
14962            assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
14963            assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
14964            assert!(table.decrements > 0, "seed {seed} never filled the table");
14965            for &(bits, _) in &expected {
14966                assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
14967            }
14968        }
14969    }
14970
14971    #[test]
14972    fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
14973        let path = path("frequency-ordinals");
14974        let mut writer =
14975            Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
14976                .expect("new file");
14977        let mut values = Vec::new();
14978        for leader in 0..10_i64 {
14979            values.extend(std::iter::repeat_n(leader, 100));
14980        }
14981        values.extend(1_000_i64..41_000);
14982        for part in values.chunks(1_024) {
14983            let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
14984                .expect("big integers");
14985            writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
14986        }
14987        writer.finish().expect("commit");
14988
14989        let reader = Reader::open(&path).expect("reopen from disk");
14990        let occurrences =
14991            reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
14992        assert!(occurrences.omitted_max < 100);
14993        assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
14994        assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
14995        assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
14996        assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
14997        assert_eq!(
14998            &occurrences.anchor_indices[..1_000]
14999                .iter()
15000                .map(|&entry| occurrences.anchors[entry as usize].clone())
15001                .collect::<Vec<_>>(),
15002            &(0_i64..10)
15003                .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
15004                .collect::<Vec<_>>()
15005        );
15006        fs::remove_file(path).expect("remove scratch file");
15007    }
15008
15009    #[test]
15010    fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
15011        // Ten leaders, then more unique values than the candidate table holds, so the first pass
15012        // has to decrement and the counts come from the recount. The unsigned leaders sit above
15013        // `i64::MAX`, where reading the bits as signed would give a different value, and the signed
15014        // ones are negative, where reading them as unsigned would.
15015        let path = path("frequency-bits");
15016        let mut writer = Writer::create(
15017            &path,
15018            "items",
15019            vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
15020        )
15021        .expect("new file");
15022        let mut rows = Vec::new();
15023        let mut leaders = Vec::new();
15024        for leader in 0..10_u64 {
15025            let count = 300 - leader * 10;
15026            let (unsigned, signed) = if leader == 0 {
15027                (Value::Null, Value::Null)
15028            } else {
15029                (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
15030            };
15031            rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
15032            leaders.push(((unsigned, count), (signed, count)));
15033        }
15034        rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
15035        for part in rows.chunks(1_024) {
15036            let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
15037            let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
15038            let chunk = Chunk::new(vec![
15039                Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
15040                Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
15041            ])
15042            .expect("matching columns");
15043            writer.append(&chunk).expect("rows");
15044        }
15045        writer.finish().expect("commit");
15046
15047        let reader = Reader::open(&path).expect("reopen from disk");
15048        for column in 0..2 {
15049            let prefix =
15050                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
15051            let wanted = leaders
15052                .iter()
15053                .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
15054                .cloned()
15055                .collect::<Vec<_>>();
15056            assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
15057            assert!(prefix.omitted_max < 210, "column {column}");
15058            assert_eq!(
15059                reader.distinct_values(column).expect("valid metadata"),
15060                Some(9 + 40_000),
15061                "column {column}"
15062            );
15063        }
15064        fs::remove_file(path).expect("remove scratch file");
15065    }
15066
15067    #[test]
15068    fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
15069        // Every column here has fewer distinct values than the tally holds, so the close takes its
15070        // counts from the gather rather than reading the pages back. The types are the ones whose
15071        // bits could come out wrong on that road: a negative tiny integer that has to be sign
15072        // extended, an unsigned one past the top of `INTEGER`, a date and a timestamp. A null every
15073        // thirteenth row checks that the nulls come from the pass and not from the list.
15074        let path = path("frequency-tally");
15075        let types = [
15076            LogicalType::TinyInt,
15077            LogicalType::UInteger,
15078            LogicalType::Date,
15079            LogicalType::Timestamp,
15080        ];
15081        let value = |ty: &LogicalType, at: i64| match ty {
15082            LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
15083            LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
15084            LogicalType::Date => Value::Date(19_000 - at as i32),
15085            _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
15086        };
15087        let fields = types
15088            .iter()
15089            .enumerate()
15090            .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
15091            .collect::<Vec<_>>();
15092        let mut writer = Writer::create(&path, "items", fields).expect("new file");
15093        let mut rows = Vec::new();
15094        for at in 0..250_i64 {
15095            for _ in 0..=(at % 37) {
15096                rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
15097            }
15098        }
15099        for part in rows.chunks(1_000) {
15100            let columns = types
15101                .iter()
15102                .map(|ty| {
15103                    let values = part
15104                        .iter()
15105                        .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
15106                        .collect::<Vec<_>>();
15107                    Vector::from_values(ty.clone(), &values).expect("a column")
15108                })
15109                .collect();
15110            writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
15111        }
15112        writer.finish().expect("commit");
15113
15114        let reader = Reader::open(&path).expect("reopen from disk");
15115        for (column, ty) in types.iter().enumerate() {
15116            let mut counts = HashMap::<Option<i64>, u64>::new();
15117            for row in &rows {
15118                *counts.entry(*row).or_default() += 1;
15119            }
15120            let wanted = counts
15121                .into_iter()
15122                .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
15123                .collect::<Vec<_>>();
15124            let prefix =
15125                reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
15126            assert_eq!(prefix.entries.len(), 2, "column {column}");
15127            assert!(prefix.omitted_max > 0, "column {column}");
15128            for (value, count) in &prefix.entries {
15129                let held =
15130                    wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
15131                assert_eq!(held, Some(count), "column {column} value {value:?}");
15132            }
15133            assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
15134            assert_eq!(
15135                reader.distinct_values(column).expect("valid metadata"),
15136                Some(wanted.len() as u64 - 1),
15137                "column {column}"
15138            );
15139        }
15140        fs::remove_file(path).expect("remove scratch file");
15141    }
15142
15143    #[test]
15144    fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
15145        // The count comes from the candidate table while it has room and from the set once it
15146        // fills, so the sizes around the fill, with and without a null taking a place, are where a
15147        // value could be counted twice or missed. Zero is in every column because the set keeps it
15148        // apart from the other values, and every value comes back later to be counted again.
15149        let edge = FREQUENCY_CANDIDATES as i64;
15150        for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
15151            for with_null in [false, true] {
15152                let path = path("distinct-edge");
15153                let mut writer =
15154                    Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
15155                        .expect("new file");
15156                let mut values = Vec::new();
15157                for round in 0..2 {
15158                    for value in 0..distinct {
15159                        let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
15160                        values.extend(std::iter::repeat_n(
15161                            Value::BigInt(value * 7_919 % distinct),
15162                            repeat,
15163                        ));
15164                        if with_null && value % 1_000 == 0 {
15165                            values.push(Value::Null);
15166                        }
15167                    }
15168                }
15169                if with_null {
15170                    values.push(Value::Null);
15171                }
15172                for part in values.chunks(1_024) {
15173                    let chunk = Chunk::new(vec![
15174                        Vector::from_values(LogicalType::BigInt, part).expect("ids"),
15175                    ])
15176                    .expect("one column");
15177                    writer.append(&chunk).expect("rows");
15178                }
15179                writer.finish().expect("commit");
15180                let reader = Reader::open(&path).expect("reopen from disk");
15181                assert_eq!(
15182                    reader.distinct_values(0).expect("valid metadata"),
15183                    Some(distinct as u64),
15184                    "{distinct} values, null {with_null}"
15185                );
15186                fs::remove_file(path).expect("remove scratch file");
15187            }
15188        }
15189    }
15190
15191    #[test]
15192    fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
15193        let path = path("quick-nonzero");
15194        let mut writer = Writer::create(
15195            &path,
15196            "items",
15197            vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
15198        )
15199        .expect("create");
15200        for ids in [
15201            &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
15202            &[Value::Integer(0), Value::Integer(7), Value::Null][..],
15203        ] {
15204            let labels = vec![Value::Varchar("same".into()); ids.len()];
15205            writer
15206                .append(
15207                    &Chunk::new(vec![
15208                        Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
15209                        Vector::from_values(LogicalType::Integer, ids).expect("ids"),
15210                    ])
15211                    .expect("chunk"),
15212                )
15213                .expect("append");
15214        }
15215        writer.finish().expect("finish");
15216        let catalog = Catalog::open(&path).expect("catalog");
15217        assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
15218        assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
15219        assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
15220        assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
15221        let prefix = catalog
15222            .table("items")
15223            .expect("reader")
15224            .frequency_prefix(1)
15225            .expect("valid metadata")
15226            .expect("partial frequencies");
15227        assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
15228        assert_eq!(prefix.omitted_max, 1);
15229        assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
15230        assert_eq!(
15231            catalog.integer_extremes("items", 1).expect("extremes"),
15232            Some(IntegerExtremes::Values { low: 0, high: 7 })
15233        );
15234        assert_eq!(
15235            catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
15236            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15237        );
15238        assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
15239        let mut legacy = catalog.clone();
15240        Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
15241        assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
15242        Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
15243        assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
15244        Writer::certify_counts(&path).expect("recertify");
15245        assert_eq!(
15246            Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
15247            Some(2)
15248        );
15249        assert_eq!(
15250            Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
15251            Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15252        );
15253        assert_eq!(
15254            Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
15255            Some(3)
15256        );
15257        assert_eq!(
15258            Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
15259            Some(IntegerExtremes::Values { low: 0, high: 7 })
15260        );
15261        assert_eq!(
15262            Catalog::open(&path)
15263                .expect("reopen")
15264                .exact_numeric_frequencies("items", 1)
15265                .expect("frequencies"),
15266            None
15267        );
15268        assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
15269        fs::remove_file(path).expect("remove scratch file");
15270    }
15271
15272    #[test]
15273    fn numeric_string_pair_leaders_are_certified_in_the_directory() {
15274        let path = path("pair-frequencies");
15275        let mut pairs = Vec::new();
15276        pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
15277        pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
15278        pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
15279        pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
15280        let mut writer = Writer::create(
15281            &path,
15282            "items",
15283            vec![
15284                Field::required("id", LogicalType::BigInt),
15285                Field::required("phrase", LogicalType::Varchar),
15286            ],
15287        )
15288        .expect("new file");
15289        for part in pairs.chunks(1_024) {
15290            let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
15291            let phrases =
15292                part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
15293            writer
15294                .append(
15295                    &Chunk::new(vec![
15296                        Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
15297                        Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
15298                    ])
15299                    .expect("matching columns"),
15300                )
15301                .expect("rows");
15302        }
15303        writer.finish().expect("commit");
15304
15305        let reader = Reader::open(&path).expect("reopen from disk");
15306        assert!(
15307            reader.table.pair_frequencies.is_empty(),
15308            "no query-specific pair result is stored"
15309        );
15310        fs::remove_file(path).expect("remove scratch file");
15311    }
15312
15313    #[test]
15314    fn legacy_group_answers_are_ignored() {
15315        let path = path("legacy-group-answers");
15316        let mut writer = Writer::create(
15317            &path,
15318            "items",
15319            vec![
15320                Field::required("id", LogicalType::BigInt),
15321                Field::required("text", LogicalType::Varchar),
15322            ],
15323        )
15324        .expect("new file");
15325        writer
15326            .append(
15327                &Chunk::new(vec![
15328                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
15329                    Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
15330                        .expect("text"),
15331                ])
15332                .expect("row"),
15333            )
15334            .expect("append");
15335        writer.finish().expect("commit");
15336        let mut reader = Reader::open(&path).expect("reopen");
15337        let table = Arc::make_mut(&mut reader.table);
15338        table.pair_frequencies.push(PairFrequencySummary {
15339            first: 0,
15340            second: 1,
15341            entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
15342            omitted_max: 0,
15343        });
15344        table.host_groups = Some(host::HostSummary {
15345            column: 1,
15346            omitted_max: 0,
15347            entries: vec![host::HostEntry {
15348                host: "fake.test".into(),
15349                count: 999,
15350                bytes_sum: 999,
15351                minimum: "x".into(),
15352            }],
15353        });
15354        assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
15355        assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
15356        fs::remove_file(path).expect("remove scratch file");
15357    }
15358
15359    /// The bug this is here for cost a 43 GB ClickBench table and an hour of reloading it. The
15360    /// format went from 11 to 12, every binary built after that said "magic or major version is
15361    /// unsupported" about the file, and there was no way to tell from the message whether the path
15362    /// was wrong, the file was truncated, or it was ours and simply older. The number this build
15363    /// wants is the whole answer and it was the one thing the message did not carry.
15364    #[test]
15365    fn a_file_from_another_format_says_which_format_it_is() {
15366        let older = path("older-format");
15367        let mut writer =
15368            Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
15369                .expect("new file");
15370        let chunk = Chunk::new(vec![
15371            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15372                .expect("integers"),
15373        ])
15374        .expect("chunk");
15375        writer.append(&chunk).expect("page written");
15376        writer.finish().expect("commit");
15377
15378        // A format below the whole readable set, rather than `FORMAT - 1`, because the set has
15379        // more than one member now: format 22 is deliberately still readable, so the version that
15380        // has to be refused is the one under the oldest one accepted.
15381        let unreadable =
15382            READABLE.iter().copied().min().expect("at least one format is readable") - 1;
15383        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15384        file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
15385        file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
15386        drop(file);
15387        let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
15388        assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
15389        assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
15390
15391        let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15392        file.seek(SeekFrom::Start(0)).expect("the magic is first");
15393        file.write_all(b"NOTRUDB!").expect("write another engine's magic");
15394        drop(file);
15395        let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
15396        assert!(complaint.contains("magic"), "{complaint}");
15397        assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
15398        fs::remove_file(older).expect("remove scratch file");
15399    }
15400
15401    #[test]
15402    fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
15403        let unfinished = path("unfinished");
15404        let mut writer =
15405            Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
15406                .expect("new file");
15407        let chunk = Chunk::new(vec![
15408            Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15409                .expect("integers"),
15410        ])
15411        .expect("chunk");
15412        writer.append(&chunk).expect("page written");
15413        drop(writer);
15414        assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
15415        fs::remove_file(unfinished).expect("remove scratch file");
15416
15417        let damaged = path("damaged");
15418        let mut writer =
15419            Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
15420                .expect("new file");
15421        writer.append(&chunk).expect("page written");
15422        writer.finish().expect("commit");
15423        let reader = Reader::open(&damaged).expect("valid directory");
15424        let mut file =
15425            OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
15426        file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
15427        file.write_all(&[255]).expect("damage one byte");
15428        assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
15429        fs::remove_file(damaged).expect("remove scratch file");
15430    }
15431
15432    #[test]
15433    fn damaged_lazy_dictionary_payload_is_an_error() {
15434        let path = path("damaged-dictionary");
15435        let mut writer = Writer::create(
15436            &path,
15437            "items",
15438            vec![
15439                Field::required("id", LogicalType::Integer),
15440                Field::new("text", LogicalType::Varchar),
15441            ],
15442        )
15443        .expect("new file");
15444        writer.append(&sample()).expect("stripe written");
15445        writer.finish().expect("commit");
15446
15447        let reader = Reader::open(&path).expect("valid directory");
15448        let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
15449        // Read the count out of the page rather than writing it here, so that adding something
15450        // else to the index does not silently turn this into a test that damages the index.
15451        let mut header = [0; DICTIONARY_HEADER];
15452        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15453        // The first block's start is the first word after the offsets, since the blocks are written
15454        // during the load and are wherever the writer was when each was encoded.
15455        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15456        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15457        assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
15458        let bits = (width & !DICTIONARY_FLAGS) as usize;
15459        let mut start = [0; 8];
15460        let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
15461        read_at(&reader.file, at, &mut start).expect("the first block's start");
15462        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15463        file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
15464        file.write_all(&[255]).expect("damage dictionary payload");
15465
15466        let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
15467        let error =
15468            chunk.validate_external().expect_err("payload corruption must reach the caller");
15469        assert!(error.message().contains("payload checksum differs"), "{error}");
15470        fs::remove_file(path).expect("remove scratch file");
15471    }
15472
15473    /// A column whose values are all different is written without a dictionary, and one whose
15474    /// values repeat keeps it.
15475    ///
15476    /// The two columns go in the same table and hold the same number of rows, so the only thing
15477    /// separating them is how much of the first stripe was a value it had not seen before. Both have
15478    /// to read back the values that were written, because the decision is about cost and nothing
15479    /// else. The file size is the other half of it: a column written without a dictionary goes
15480    /// through the string cascade instead, so dropping the dictionary must not turn into storing the
15481    /// column raw.
15482    #[test]
15483    fn a_column_of_all_different_values_is_written_without_a_dictionary() {
15484        let path = path("dictionary-decide");
15485        let rows = 20_000;
15486        // Long enough that storing it raw would show, and different in every row.
15487        let unique =
15488            |row: usize| format!("{row:09} a value that appears exactly once in the table");
15489        // The same values in the same shape, each one used forty times over.
15490        let repeated = |row: usize| unique(row / 40);
15491        let mut writer = Writer::create(
15492            &path,
15493            "items",
15494            vec![
15495                Field::required("unique", LogicalType::Varchar),
15496                Field::required("repeated", LogicalType::Varchar),
15497            ],
15498        )
15499        .expect("new file");
15500        for part in (0..rows).step_by(1_000) {
15501            let span = part..(part + 1_000).min(rows);
15502            let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
15503            let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
15504            writer
15505                .append(
15506                    &Chunk::new(vec![
15507                        Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
15508                        Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
15509                    ])
15510                    .expect("two columns"),
15511                )
15512                .expect("a part");
15513        }
15514        writer.finish().expect("commit");
15515
15516        let reader = Reader::open(&path).expect("reopen from disk");
15517        assert!(
15518            reader.table.dictionaries[0].is_none(),
15519            "a column with no repeats has nothing to say twice"
15520        );
15521        assert!(
15522            reader.table.dictionaries[1].is_some(),
15523            "a column whose values come round again keeps its dictionary"
15524        );
15525        let mut first = 0;
15526        for part in 0..reader.parts() {
15527            let chunk = reader.read(part, &[0, 1]).expect("a part");
15528            for row in 0..chunk.len() {
15529                assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
15530                assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
15531            }
15532            first += chunk.len();
15533        }
15534        assert_eq!(first, rows, "every row was read back");
15535        let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
15536        let size = fs::metadata(&path).expect("the file is there").len() as usize;
15537        assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
15538        fs::remove_file(path).expect("remove scratch file");
15539    }
15540
15541    /// A payload of many blocks reads and checks every block of it.
15542    ///
15543    /// The test above has a dictionary of three values, which is one block, so it says nothing
15544    /// about a reader finding the right block among many. This one has thirty two thousand values,
15545    /// which is thirty two blocks, and it reads a value out of the first block and a value out of
15546    /// the last and then damages the last and asks for it again.
15547    ///
15548    /// Forty thousand rows over those thirty two thousand values, because a column the writer finds
15549    /// to be all distinct does not get a dictionary at all and there would be nothing here to test.
15550    /// Four rows in five holding a value the stripe has not seen before is a column that keeps one.
15551    /// The repeats are put at the front so that the values still arrive in order after them, which
15552    /// is what keeps the last part of the table on the last block of the payload.
15553    #[test]
15554    fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
15555        let path = path("dictionary-blocks");
15556        let value = |row: usize| {
15557            let row = row.saturating_sub(8_000);
15558            format!("{row:07} a value long enough to be worth a payload block")
15559        };
15560        let parts = 40;
15561        let per_part = 1000;
15562        let mut writer =
15563            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15564                .expect("new file");
15565        for part in 0..parts {
15566            let values = (0..per_part)
15567                .map(|row| Value::Varchar(value(part * per_part + row)))
15568                .collect::<Vec<_>>();
15569            let chunk = Chunk::new(vec![
15570                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15571            ])
15572            .expect("matching rows");
15573            writer.append(&chunk).expect("a part");
15574        }
15575        writer.finish().expect("commit");
15576
15577        let reader = Reader::open(&path).expect("reopen from disk");
15578        let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
15579        assert!(
15580            parts * per_part > TEXT_PAYLOAD_VALUES * 4,
15581            "the dictionary has to be several blocks for this to be testing anything"
15582        );
15583        for part in [0, parts - 1] {
15584            let chunk = reader.read(part, &[0]).expect("a part");
15585            chunk.validate_external().expect("every payload block checks out");
15586            assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
15587        }
15588
15589        // The last block is wherever the writer was when it was encoded, which the index says.
15590        let mut header = [0; DICTIONARY_HEADER];
15591        read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15592        let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15593        let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
15594        let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15595        let bits = (width & !DICTIONARY_FLAGS) as usize;
15596        let mut place = [0; 16];
15597        let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
15598        read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
15599        let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
15600        let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
15601        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15602        file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
15603        file.write_all(&[255]).expect("damage the last payload block");
15604        let reader = Reader::open(&path).expect("the directory and the index are untouched");
15605        let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
15606        let error = chunk.validate_external().expect_err("the damage must reach the caller");
15607        assert!(error.message().contains("payload checksum differs"), "{error}");
15608        fs::remove_file(path).expect("remove scratch file");
15609    }
15610
15611    /// Values of different lengths read back where the offsets say they do.
15612    ///
15613    /// The offsets are packed at one width for the column, they are relative to the payload block a
15614    /// value lands in, and they go in runs of half a block, so there are two boundaries where the
15615    /// arithmetic could be off by one and neither shows up on values that are all the same length.
15616    /// This writes 5,000 values whose lengths cycle through a wide range and reads every one back,
15617    /// so the first value of a block, the last value of a run and the last value of a block are all
15618    /// covered several times over. An empty value is in the cycle because a zero length span is the
15619    /// case the reader short circuits.
15620    ///
15621    /// Six thousand rows over those 5,000 values, because a column the writer finds to be all
15622    /// distinct is written without a dictionary and then there are no packed offsets to be off by
15623    /// one in.
15624    #[test]
15625    fn values_of_different_lengths_read_back_out_of_packed_offsets() {
15626        let path = path("dictionary-offsets");
15627        let value = |row: usize| {
15628            let row = row % 5_000;
15629            if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
15630        };
15631        let rows = 6_000;
15632        let mut writer =
15633            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15634                .expect("new file");
15635        let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
15636        for part in values.chunks(1_000) {
15637            let chunk =
15638                Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
15639                    .expect("matching rows");
15640            writer.append(&chunk).expect("a part");
15641        }
15642        writer.finish().expect("commit");
15643
15644        let reader = Reader::open(&path).expect("reopen from disk");
15645        assert!(
15646            rows > TEXT_PAYLOAD_VALUES * 4,
15647            "the dictionary has to be several blocks for this to be testing anything"
15648        );
15649        for part in 0..rows / 1_000 {
15650            let chunk = reader.read(part, &[0]).expect("a part");
15651            for row in 0..1_000 {
15652                let row = part * 1_000 + row;
15653                assert_eq!(
15654                    chunk.value_at(row % 1_000, 0),
15655                    Value::Varchar(value(row)),
15656                    "value {row}"
15657                );
15658            }
15659        }
15660        // The lengths a vector at a time, twice over, because the first pass is what makes the
15661        // table of ends worth building and the second is read out of the lengths worked out of it.
15662        for _ in 0..2 {
15663            for part in 0..rows / 1_000 {
15664                let chunk = reader.read(part, &[0]).expect("a part");
15665                let mut lens = vec![0_i64; 1_000];
15666                let column = chunk.column(0).expect("one column");
15667                assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
15668                for (row, &len) in lens.iter().enumerate() {
15669                    let row = part * 1_000 + row;
15670                    assert_eq!(len as usize, value(row).len(), "the length of value {row}");
15671                }
15672            }
15673        }
15674        fs::remove_file(path).expect("remove scratch file");
15675    }
15676
15677    /// Lengths start again at every block, and ends that go backwards inside one give no table.
15678    #[test]
15679    fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
15680        let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
15681        ends.extend([3, 3, 10]);
15682        let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
15683        assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
15684        assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
15685        // One value longer than sixteen bits keeps every length at four bytes.
15686        let long = [5, 70_005, 70_006];
15687        let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
15688        assert_eq!(lens, [5, 70_000, 1]);
15689        let mut read = Vec::new();
15690        Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
15691        assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
15692        ends.push(9);
15693        assert!(lengths_of(&ends).is_none());
15694    }
15695
15696    /// Every worker of a scan wants the dictionary at the same moment and one of them fetches it.
15697    ///
15698    /// Asking a `OnceLock` whether it holds something answers the question a worker that already has
15699    /// the dictionary is asking and not the one a worker without it is asking, which is whether
15700    /// somebody is already on their way with it. Sixteen workers that all miss will all read the
15701    /// page, all verify it and all decode it, and fifteen will drop the result. Nothing about that
15702    /// is incorrect, which is why it went unnoticed, and it showed up as ClickBench 38 getting
15703    /// slower when the scan in front of it got faster and stopped staggering the arrivals.
15704    ///
15705    /// The barrier is what makes the test about that rather than about luck. Without it the first
15706    /// thread is usually finished before the last one starts and the count is one either way.
15707    #[test]
15708    fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
15709        let path = path("dictionary-once");
15710        let parts = 8;
15711        let per_part = 500;
15712        let value =
15713            |row: usize| format!("{row:07} a value long enough to be worth a payload block");
15714        let mut writer =
15715            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15716                .expect("new file");
15717        for part in 0..parts {
15718            let values = (0..per_part)
15719                .map(|row| Value::Varchar(value(part * per_part + row)))
15720                .collect::<Vec<_>>();
15721            let chunk = Chunk::new(vec![
15722                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15723            ])
15724            .expect("matching rows");
15725            writer.append(&chunk).expect("a part");
15726        }
15727        writer.finish().expect("commit");
15728
15729        let reader = Reader::open(&path).expect("reopen from disk");
15730        assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
15731        assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
15732
15733        let workers = 16;
15734        let gate = std::sync::Barrier::new(workers);
15735        std::thread::scope(|scope| {
15736            for worker in 0..workers {
15737                let reader = reader.clone();
15738                let gate = &gate;
15739                scope.spawn(move || {
15740                    gate.wait();
15741                    let chunk = reader.read(worker % parts, &[0]).expect("a part");
15742                    assert_eq!(
15743                        chunk.value_at(0, 0),
15744                        Value::Varchar(value((worker % parts) * per_part))
15745                    );
15746                });
15747            }
15748        });
15749
15750        assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
15751        fs::remove_file(path).expect("remove scratch file");
15752    }
15753
15754    /// The sorted order sits outside the index the page checksum covers, because a query that
15755    /// never searches a dictionary should not read it, so it carries its own checksums and this is
15756    /// what says they are checked. A search that trusted a damaged order would give a wrong answer
15757    /// rather than a slow one.
15758    #[test]
15759    fn a_damaged_sorted_order_is_an_error() {
15760        let path = path("damaged-order");
15761        let mut writer = Writer::create(
15762            &path,
15763            "items",
15764            vec![
15765                Field::required("id", LogicalType::Integer),
15766                Field::new("text", LogicalType::Varchar),
15767            ],
15768        )
15769        .expect("new file");
15770        writer.append(&sample()).expect("stripe written");
15771        writer.finish().expect("commit");
15772
15773        let reader = Reader::open(&path).expect("valid directory");
15774        let page = reader.table.dictionaries[1].expect("string dictionary page");
15775        let mut header = [0; DICTIONARY_HEADER];
15776        read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
15777        let index_len = dictionary_index_len(&header);
15778        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15779        file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
15780        file.write_all(&[255]).expect("damage the order");
15781
15782        let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
15783        let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
15784        assert!(error.message().contains("rank checksum differs"), "{error}");
15785        fs::remove_file(path).expect("remove scratch file");
15786    }
15787
15788    /// Codes stay in first appearance order and the sorted order is written beside them, so a
15789    /// reader can put the values back in order without the writer having had to know them all
15790    /// before it handed out the first code.
15791    #[test]
15792    fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
15793        // Chosen so the sort cannot be decided on the first eight bytes alone. Three values share
15794        // a nine byte prefix, one is a prefix of another, and one is empty.
15795        let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
15796        let path = path("dictionary-order");
15797        let mut writer =
15798            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15799                .expect("new file");
15800        writer
15801            .append(
15802                &Chunk::new(vec![
15803                    Vector::from_values(
15804                        LogicalType::Varchar,
15805                        &spellings.map(|text| Value::Varchar(text.into())),
15806                    )
15807                    .expect("strings"),
15808                ])
15809                .expect("one column"),
15810            )
15811            .expect("stripe written");
15812        writer.finish().expect("commit");
15813
15814        let reader = Reader::open(&path).expect("valid directory");
15815        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15816        let count = dictionary.ranks().expect("a v10 file stores one");
15817        assert_eq!(count, spellings.len(), "every distinct value has a rank");
15818        let order = (0..count)
15819            .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
15820            .collect::<Vec<_>>();
15821        let mut seen = order.clone();
15822        seen.sort_unstable();
15823        assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
15824
15825        let ranked = order
15826            .iter()
15827            .map(|&code| {
15828                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15829            })
15830            .collect::<Vec<_>>();
15831        let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
15832        expected.sort();
15833        assert_eq!(ranked, expected, "rank order is value order");
15834
15835        // What a search asks, on the values themselves rather than through a kernel, so that a
15836        // file whose heads disagree with its bytes is caught here rather than as a wrong answer.
15837        for (rank, value) in expected.iter().enumerate() {
15838            assert_eq!(
15839                dictionary.compare_rank(rank, value).expect("compare"),
15840                Ordering::Equal,
15841                "rank {rank} is its own value"
15842            );
15843            if rank > 0 {
15844                assert_eq!(
15845                    dictionary.compare_rank(rank - 1, value).expect("compare"),
15846                    Ordering::Less,
15847                    "rank {rank} follows the one before it"
15848                );
15849            }
15850        }
15851        fs::remove_file(path).expect("remove scratch file");
15852    }
15853
15854    /// Five text columns of different sizes close at the same time, and each comes back with its
15855    /// own values in its own order.
15856    ///
15857    /// The sizes differ so that the columns are taken in an order that is not the column order, and
15858    /// the values of each column are spelled with its number so that one column's page written in
15859    /// another's place would read back as the wrong strings rather than the right ones by chance.
15860    #[test]
15861    fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
15862        let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
15863        let path = path("dictionaries-at-once");
15864        let fields = (0..sizes.len())
15865            .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
15866            .collect::<Vec<_>>();
15867        let mut writer = Writer::create(&path, "items", fields).expect("new file");
15868        let rows = 10_000_usize;
15869        for start in (0..rows).step_by(1_024) {
15870            let columns = sizes
15871                .iter()
15872                .enumerate()
15873                .map(|(column, &size)| {
15874                    let values = (start..(start + 1_024).min(rows))
15875                        .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
15876                        .collect::<Vec<_>>();
15877                    Vector::from_values(LogicalType::Varchar, &values).expect("strings")
15878                })
15879                .collect::<Vec<_>>();
15880            writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
15881        }
15882        writer.finish().expect("commit");
15883
15884        let reader = Reader::open(&path).expect("valid directory");
15885        for (column, &size) in sizes.iter().enumerate() {
15886            let dictionary =
15887                reader.dictionary(column).expect("read").expect("a string column has one");
15888            let count = dictionary.ranks().expect("a v10 file stores one");
15889            assert_eq!(count, size, "column {column} has its own distinct count");
15890            let ranked = (0..count)
15891                .map(|rank| {
15892                    let code = dictionary.code_at_rank(rank).expect("a code");
15893                    dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15894                })
15895                .collect::<Vec<_>>();
15896            let expected = (0..size)
15897                .map(|value| format!("c{column}-{value:05}").into_bytes())
15898                .collect::<Vec<_>>();
15899            assert_eq!(ranked, expected, "column {column} ranks its own values in order");
15900        }
15901        fs::remove_file(path).expect("remove scratch file");
15902    }
15903
15904    /// A dictionary large enough to be decoded and sorted on several threads ranks the way one small
15905    /// enough for one thread does.
15906    ///
15907    /// Seventy thousand values over sixty nine blocks, in no order and each four times over so the
15908    /// column is worth a dictionary, written and ranked in the close.
15909    /// Some share a long prefix and some differ only in the last byte, so the buckets of the sort cut
15910    /// through runs of values that agree for a long way.
15911    #[test]
15912    fn a_large_dictionary_ranks_in_value_order() {
15913        let path = path("dictionary-large-rank");
15914        let value = |row: u64| {
15915            let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
15916            match row % 3 {
15917                0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
15918                1 => format!("{mixed}"),
15919                _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
15920            }
15921        };
15922        let distinct = 70_000;
15923        let parts = 4 * distinct / 1000;
15924        let per_part = 1000;
15925        let mut writer =
15926            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15927                .expect("new file");
15928        for part in 0..parts {
15929            let values = (0..per_part)
15930                .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
15931                .collect::<Vec<_>>();
15932            let chunk = Chunk::new(vec![
15933                Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15934            ])
15935            .expect("matching rows");
15936            writer.append(&chunk).expect("a part");
15937        }
15938        writer.finish().expect("commit");
15939
15940        let reader = Reader::open(&path).expect("reopen from disk");
15941        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15942        let count = dictionary.ranks().expect("a ranked dictionary");
15943        assert_eq!(count, distinct as usize, "every distinct value has a rank");
15944        assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
15945        let ranked = (0..count)
15946            .map(|rank| {
15947                let code = dictionary.code_at_rank(rank).expect("a code");
15948                dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15949            })
15950            .collect::<Vec<_>>();
15951        let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
15952        expected.sort();
15953        assert_eq!(ranked, expected, "rank order is value order");
15954        fs::remove_file(path).expect("remove scratch file");
15955    }
15956
15957    /// A string column's synopsis is turned into values without keeping the blocks it went through.
15958    ///
15959    /// Three thousand values, every fifth of them four times over, so the synopsis is a prefix of
15960    /// five hundred and twelve codes spread over all three payload blocks. Reading it used to leave
15961    /// all three decoded for as long as the reader lived. It leaves none of them now, and the second
15962    /// read answers out of what the first remembered.
15963    /// A directory read out of the file a window at a time is the directory read whole.
15964    ///
15965    /// The windows here are far smaller than any field is long, so every kind of field is split
15966    /// across a refill somewhere, and a bound is offered to its codec short more than once. The
15967    /// synopses are left in the file, and each one read back from where it was left is the one the
15968    /// whole read decoded.
15969    #[test]
15970    fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
15971        let path = path("windowed-directory");
15972        let fields = vec![
15973            Field::required("id", LogicalType::BigInt),
15974            Field::required("word", LogicalType::Varchar),
15975            Field::new("score", LogicalType::Double),
15976        ];
15977        let mut writer = Writer::create(&path, "items", fields).expect("new file");
15978        for part in 0..70_i64 {
15979            let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
15980            let words = (0..100)
15981                .map(|row| Value::Varchar(format!("word {}", row % 13)))
15982                .collect::<Vec<_>>();
15983            let scores = (0..100)
15984                .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
15985                .collect::<Vec<_>>();
15986            let chunk = Chunk::new(vec![
15987                Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
15988                Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
15989                Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
15990            ])
15991            .expect("three columns");
15992            writer.append(&chunk).expect("a part");
15993        }
15994        writer.finish().expect("commit");
15995
15996        let catalog = Catalog::open(&path).expect("reopen");
15997        let entry = catalog.entries.first().expect("one table").directory;
15998        let (offset, length) = (entry.offset, entry.length as usize);
15999        let mut bytes = vec![0; length];
16000        read_at(&catalog.file, offset, &mut bytes).expect("the directory");
16001        assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
16002        let whole = decode_directory(&bytes, catalog.size).expect("whole");
16003        assert!(whole.stripes.len() > 1, "the table should span stripes");
16004        for size in [1, 7, 33, 4_096] {
16005            let mut cursor = Cursor::over(&catalog.file, offset, length);
16006            cursor.window.as_mut().expect("a window").size = size;
16007            let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
16008            assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
16009            assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
16010            let mut stored = 0;
16011            for (column, (left, held)) in
16012                windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
16013            {
16014                match (left, held) {
16015                    (None, None) => {}
16016                    (
16017                        Some(super::Frequencies::Stored { span, values }),
16018                        Some(super::Frequencies::Held(summary)),
16019                    ) => {
16020                        let mut one = vec![0; span.length as usize];
16021                        read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
16022                        let read = decode_summary(
16023                            &mut Cursor::new(&one),
16024                            &whole.fields[column],
16025                            whole.rows,
16026                            *values,
16027                        )
16028                        .expect("a valid synopsis")
16029                        .expect("one is there");
16030                        assert_eq!(format!("{read:?}"), format!("{summary:?}"));
16031                        stored += 1;
16032                    }
16033                    other => panic!("column {column} came back as {other:?}"),
16034                }
16035            }
16036            assert!(stored >= 2, "only {stored} synopses were left in the file");
16037        }
16038        let reader = catalog.table("items").expect("the table");
16039        assert!(reader.frequency_summaries[1].get().is_none());
16040        assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
16041        let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
16042        let clone = reader.clone();
16043        assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
16044        assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
16045        fs::remove_file(path).expect("remove scratch file");
16046    }
16047
16048    #[test]
16049    fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
16050        let path = path("file-checksum");
16051        let bytes = (0..200_000_u32)
16052            .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
16053            .collect::<Vec<_>>();
16054        fs::write(&path, &bytes).expect("scratch file");
16055        let file = File::open(&path).expect("open");
16056        for (offset, length) in [
16057            (0, 0),
16058            (3, 1),
16059            (5, 31),
16060            (0, 32),
16061            (9, 33),
16062            (1, 65_536),
16063            (7, 65_567),
16064            (0, 200_000),
16065            (11, 131_101),
16066        ] {
16067            let whole = checksum(&bytes[offset..offset + length]);
16068            assert_eq!(
16069                file_checksum(&file, offset as u64, length).expect("read"),
16070                whole,
16071                "{offset} {length}"
16072            );
16073        }
16074        fs::remove_file(path).expect("remove scratch file");
16075    }
16076
16077    #[test]
16078    fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
16079        let path = path("synopsis-keeps-no-block");
16080        let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
16081        let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
16082        for _ in 0..3 {
16083            values.extend((0..3_000).step_by(5).map(spelled));
16084        }
16085        let mut writer =
16086            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16087                .expect("new file");
16088        for part in values.chunks(1_024) {
16089            writer
16090                .append(
16091                    &Chunk::new(vec![
16092                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16093                    ])
16094                    .expect("one column"),
16095                )
16096                .expect("a part");
16097        }
16098        writer.finish().expect("commit");
16099
16100        let reader = Reader::open(&path).expect("reopen from disk");
16101        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16102        let resting = dictionary.footprint();
16103        let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16104        assert_eq!(prefix.entries.len(), 512);
16105        for (value, count) in &prefix.entries {
16106            let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
16107            let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
16108            assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
16109        }
16110        assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
16111        let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16112        assert_eq!(again.entries, prefix.entries);
16113        fs::remove_file(path).expect("remove scratch file");
16114    }
16115
16116    /// `length` over a stored column keeps a count a value rather than the blocks it counted.
16117    ///
16118    /// Reading the bytes a row at a time keeps every block it touches, so a scan of `length` over a
16119    /// whole column used to end up holding the column decoded. The counts are what is kept now, and
16120    /// they have to be the counts of characters rather than bytes, which is why the values here are
16121    /// not ASCII.
16122    #[test]
16123    fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
16124        let path = path("character-lengths");
16125        let spellings = (0..2_500)
16126            .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
16127            .collect::<Vec<_>>();
16128        let mut writer =
16129            Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16130                .expect("new file");
16131        for part in spellings.chunks(1_024) {
16132            writer
16133                .append(
16134                    &Chunk::new(vec![
16135                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16136                    ])
16137                    .expect("one column"),
16138                )
16139                .expect("a part");
16140        }
16141        writer.finish().expect("commit");
16142
16143        let reader = Reader::open(&path).expect("reopen from disk");
16144        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16145        let resting = dictionary.footprint();
16146        let mut lens = Vec::new();
16147        assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
16148        let counted = dictionary.footprint() - resting;
16149        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16150        assert!(
16151            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16152            "counting kept {counted} bytes, more than a count a value"
16153        );
16154        let expected = (0..dictionary.len())
16155            .map(|code| {
16156                let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
16157                i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
16158                    .expect("small")
16159            })
16160            .collect::<Vec<_>>();
16161        assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
16162        let mut again = Vec::new();
16163        assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
16164        assert_eq!(again, lens, "the kept counts answer the second time");
16165        fs::remove_file(path).expect("remove scratch file");
16166    }
16167
16168    /// Writes one column of strings whose code is where they sit in `spellings`, and reopens it.
16169    fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
16170        let path = path(label);
16171        let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
16172        let mut writer =
16173            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16174                .expect("new file");
16175        for part in values.chunks(1_024) {
16176            writer
16177                .append(
16178                    &Chunk::new(vec![
16179                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16180                    ])
16181                    .expect("one column"),
16182                )
16183                .expect("a part");
16184        }
16185        writer.finish().expect("commit");
16186        let reader = Reader::open(&path).expect("reopen from disk");
16187        (path, reader)
16188    }
16189
16190    /// Codes that go all over a dictionary of `len` values, and every seventh row null.
16191    ///
16192    /// The shape of a vector a scan hands out: its codes are in row order, which lands them in
16193    /// every block of the dictionary in no order at all, so a read of the whole vector has to put
16194    /// them in block order itself to read each block once.
16195    fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
16196        let codes = (0..len)
16197            .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
16198            .collect::<Vec<_>>();
16199        let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
16200        (codes, valid)
16201    }
16202
16203    /// `length` over a vector with nulls keeps the counts and not the blocks, the same as over one
16204    /// without.
16205    ///
16206    /// The whole vector count used to be taken only when no row was null, and every other vector
16207    /// went a row at a time through the bytes, which keeps every block it reads. A column with a
16208    /// null in each vector was held decoded after one `length` over it.
16209    #[test]
16210    fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
16211        let spellings = (0..2_500)
16212            .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
16213            .collect::<Vec<_>>();
16214        let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
16215        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16216        let (codes, valid) = scattered_rows(spellings.len());
16217        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
16218            .expect("every code is inside")
16219            .with_validity(Validity::from_run(&valid));
16220
16221        let resting = dictionary.footprint();
16222        let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
16223            .expect("length reads");
16224        let counted = dictionary.footprint() - resting;
16225        let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16226        assert!(
16227            counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16228            "length over a vector with nulls kept {counted} bytes, more than a count a value"
16229        );
16230        let expected = (0..rows.len())
16231            .map(|row| match valid[row] {
16232                true => Value::BigInt(
16233                    i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
16234                ),
16235                false => Value::Null,
16236            })
16237            .collect::<Vec<_>>();
16238        let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
16239        assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
16240        fs::remove_file(path).expect("remove scratch file");
16241    }
16242
16243    /// `lower`, `upper` and `substring` read a stored dictionary a block at a time and keep none of
16244    /// it while the column is at its budget, until reading without keeping stops being cheap.
16245    ///
16246    /// The three used to read a row at a time through the bytes, which keeps every block a row lands
16247    /// in for as long as the table is open. They read the whole vector in one visit now, and the
16248    /// dictionary here is opened with a budget of zero so that what a visit would keep under the
16249    /// budget of a running database is what the test sees dropped. After a column's worth of blocks
16250    /// has been decoded and dropped the visit keeps what it reads, which is what bounds its cost on
16251    /// a scan whose codes keep coming back to every block, and the end of the test holds it to that.
16252    #[test]
16253    fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
16254        let spellings = (0..2_500)
16255            .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
16256            .collect::<Vec<_>>();
16257        let (path, reader) = stored_spellings("string-kernels", &spellings);
16258        let page = reader.table.dictionaries[0].expect("a string column has one");
16259        let starved =
16260            open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
16261                .expect("a dictionary opens whatever it may keep");
16262        let starved = Arc::new(starved);
16263        let (codes, valid) = scattered_rows(spellings.len());
16264        let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
16265            .expect("every code is inside")
16266            .with_validity(Validity::from_run(&valid));
16267        let expected = |each: &dyn Fn(&str) -> String| {
16268            (0..rows.len())
16269                .map(|row| match valid[row] {
16270                    true => Value::Varchar(each(&spellings[codes[row] as usize])),
16271                    false => Value::Null,
16272                })
16273                .collect::<Vec<_>>()
16274        };
16275        let answers =
16276            |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
16277
16278        // What a visit may add is the table of where every value ends, four bytes a value, which
16279        // reading every value this often makes worth building. A block is tens of bytes a value.
16280        let resting = starved.footprint();
16281        let ends = spellings.len() * size_of::<u32>();
16282        let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
16283            .expect("lower reads");
16284        assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
16285        assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
16286
16287        let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
16288        let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
16289        let cut =
16290            rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
16291                .expect("substring reads");
16292        let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
16293        assert_eq!(answers(&cut), expected(&cut_of), "substring");
16294        assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
16295
16296        // Every block has been read twice now and dropped the second time as well, which is a
16297        // column's worth dropped for want of a budget, so the next visit keeps what it reads.
16298        let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16299            .expect("upper reads");
16300        assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
16301        let payload = spellings.iter().map(String::len).sum::<usize>();
16302        assert!(
16303            starved.footprint() >= resting + payload,
16304            "a visit that has dropped a column's worth of blocks keeps what it reads"
16305        );
16306        let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16307            .expect("upper reads kept blocks");
16308        assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
16309        fs::remove_file(path).expect("remove scratch file");
16310    }
16311
16312    /// A sweep of the dictionary reads every value, and the second sweep keeps what it read, up to
16313    /// the budget.
16314    ///
16315    /// The point of the sweep is the resident size rather than the answer, so both are checked
16316    /// here. The first sweep keeps nothing, because a process that runs one statement never reads
16317    /// a block twice. A dictionary this small is well under [`TEXT_KEEP_BUDGET`], so the second
16318    /// sweep keeps everything and a third decodes nothing, which is what makes a session asking the
16319    /// same question again cost what it should. The ceiling is the other half of it and it has its own
16320    /// test below, because a ceiling that never binds is not a ceiling anybody checked.
16321    #[test]
16322    fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
16323        let path = path("dictionary-sweep");
16324        // Two thousand five hundred distinct values is two whole payload blocks and a part of a
16325        // third, so the sweep has to be called more than once and the last call has to stop short.
16326        let spellings = (0..2_500)
16327            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16328            .collect::<Vec<_>>();
16329        let mut writer =
16330            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16331                .expect("new file");
16332        // A chunk is a part and a part is at most 1,024 rows, so the values go in three of them.
16333        // The dictionary is table wide and does not care where a value was written.
16334        for part in spellings.chunks(1_024) {
16335            writer
16336                .append(
16337                    &Chunk::new(vec![
16338                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16339                    ])
16340                    .expect("one column"),
16341                )
16342                .expect("stripe written");
16343        }
16344        writer.finish().expect("commit");
16345
16346        let reader = Reader::open(&path).expect("valid directory");
16347        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16348        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16349        for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
16350            assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
16351            assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
16352        }
16353
16354        let resting = dictionary.footprint();
16355        let sweep = || {
16356            let mut swept: Vec<Vec<u8>> = Vec::new();
16357            let mut at = 0;
16358            let mut calls = 0;
16359            while at < dictionary.len() {
16360                let stopped = dictionary
16361                    .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16362                        assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16363                        swept.push(text.to_vec());
16364                        Ok(())
16365                    })
16366                    .expect("a sweep reads");
16367                assert!(stopped > at, "a sweep moves");
16368                at = stopped;
16369                calls += 1;
16370            }
16371            assert_eq!(calls, 3, "a sweep hands over one block at a time");
16372            swept
16373        };
16374        let swept = sweep();
16375        assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
16376        assert_eq!(sweep(), swept, "a second sweep reads what the first did");
16377        let after = dictionary.footprint();
16378        assert!(after > resting, "a second sweep under the budget keeps what it decoded");
16379
16380        let read = (0..dictionary.len())
16381            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16382            .collect::<Vec<_>>();
16383        assert_eq!(swept, read, "a sweep answers what a point read answers");
16384        // A read per value is about what makes the unpacked ends worth building, so whether they
16385        // are built here depends on how many reads the sweep made on the way. They are the one thing
16386        // allowed to grow, by four bytes a value, and nothing of the payload is.
16387        let grown = dictionary.footprint() - after;
16388        assert!(
16389            grown == 0 || grown == dictionary.len() * size_of::<u32>(),
16390            "a point read of a kept block decodes nothing, and {grown} bytes grew"
16391        );
16392        fs::remove_file(path).expect("remove scratch file");
16393    }
16394
16395    #[test]
16396    fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
16397        let path = path("narrow-substring-signature");
16398        let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
16399        let mut grams = Vec::new();
16400        for text in blocks {
16401            let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
16402            for gram in text.windows(4) {
16403                for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
16404                    bits[bit / 8] |= 1 << (bit % 8);
16405                }
16406            }
16407            grams.extend(bits);
16408        }
16409        fs::write(&path, &grams).expect("scratch file");
16410        let file = File::open(&path).expect("open scratch file");
16411        let signatures = NativeGrams {
16412            start: 0,
16413            length: grams.len(),
16414            width: NARROW_GRAM_BYTES,
16415            hash: checksum(&grams),
16416            verdicts: Mutex::new(Vec::new()),
16417        };
16418        let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
16419        assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
16420        assert!(signatures.footprint() > 0, "a verdict is remembered");
16421        let again = signatures.verdicts(&file, b"google").expect("remembered");
16422        assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
16423
16424        let damaged = NativeGrams {
16425            hash: signatures.hash ^ 1,
16426            verdicts: Mutex::new(Vec::new()),
16427            ..signatures
16428        };
16429        let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
16430        assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
16431        fs::remove_file(path).expect("remove scratch file");
16432    }
16433
16434    #[test]
16435    fn a_damaged_substring_signature_is_checked_only_when_used() {
16436        let path = path("damaged-substring-signature");
16437        let mut writer =
16438            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16439                .expect("new file");
16440        let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
16441        writer
16442            .append(
16443                &Chunk::new(vec![
16444                    Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
16445                ])
16446                .expect("one column"),
16447            )
16448            .expect("stripe written");
16449        writer.finish().expect("commit");
16450
16451        let reader = Reader::open(&path).expect("valid directory");
16452        let page = reader.table.dictionaries[0].expect("string dictionary page");
16453        let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16454        file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
16455            .expect("last signature byte");
16456        file.write_all(&[255]).expect("damage signature");
16457        let reader = Reader::open(&path).expect("the directory is still valid");
16458        let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
16459        let error = dictionary
16460            .text_block_might_contain(0, b"goog")
16461            .expect_err("a used signature checks its own checksum");
16462        assert!(error.message().contains("substring signatures checksum differs"), "{error}");
16463        fs::remove_file(path).expect("remove scratch file");
16464    }
16465
16466    /// A sweep over a block whose second run of offsets is short reads the same values as a point
16467    /// read does.
16468    ///
16469    /// The sweep decodes the offsets of a whole run at a time rather than a value at a time, and a
16470    /// run holds half a block, so the count it asks for is the run length everywhere but at the end
16471    /// of the dictionary. Two thousand five hundred values, which is what the test above writes,
16472    /// never puts a short run second in its block: the last block there begins on a run boundary and
16473    /// holds one run. Two thousand eight hundred does, so the last block is a whole run of five
16474    /// hundred and twelve followed by two hundred and forty, and an off by one in either the count
16475    /// asked for or the slice taken out of the answer shows up as a wrong value or a refusal.
16476    #[test]
16477    fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
16478        let path = path("dictionary-sweep-short-run");
16479        let spellings = (0..2_800)
16480            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16481            .collect::<Vec<_>>();
16482        let mut writer =
16483            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16484                .expect("new file");
16485        for part in spellings.chunks(1_024) {
16486            writer
16487                .append(
16488                    &Chunk::new(vec![
16489                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16490                    ])
16491                    .expect("one column"),
16492                )
16493                .expect("stripe written");
16494        }
16495        writer.finish().expect("commit");
16496
16497        let reader = Reader::open(&path).expect("valid directory");
16498        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16499        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16500        let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
16501        assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
16502        assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
16503
16504        let mut swept: Vec<Vec<u8>> = Vec::new();
16505        let mut at = 0;
16506        while at < dictionary.len() {
16507            let stopped = dictionary
16508                .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16509                    assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16510                    swept.push(text.to_vec());
16511                    Ok(())
16512                })
16513                .expect("a sweep reads");
16514            assert!(stopped > at, "a sweep moves");
16515            at = stopped;
16516        }
16517        let read = (0..dictionary.len())
16518            .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16519            .collect::<Vec<_>>();
16520        assert_eq!(swept, read, "a sweep answers what a point read answers");
16521        fs::remove_file(path).expect("remove scratch file");
16522    }
16523
16524    /// The unpacked ends answer what the packed ends answer, on both sides of the switch.
16525    ///
16526    /// A column asked for one offset at a time reads them out of the packed form until the reads
16527    /// are worth a table and out of the table after that, so every value here is read twice and the
16528    /// two passes are compared against the spellings and against each other. Two thousand eight
16529    /// hundred values is two payload blocks and a bit, which puts the switch in the middle of the
16530    /// first pass and means the pass straddles a block boundary, where the start of a value is zero
16531    /// rather than the end of the value before it.
16532    #[test]
16533    fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
16534        let path = path("dictionary-unpacked-ends");
16535        let spellings = (0..2_800)
16536            .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16537            .collect::<Vec<_>>();
16538        let mut writer =
16539            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16540                .expect("new file");
16541        for part in spellings.chunks(1_024) {
16542            writer
16543                .append(
16544                    &Chunk::new(vec![
16545                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16546                    ])
16547                    .expect("one column"),
16548                )
16549                .expect("stripe written");
16550        }
16551        writer.finish().expect("commit");
16552
16553        let reader = Reader::open(&path).expect("valid directory");
16554        let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16555        assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16556        let wanted = (0..spellings.len())
16557            .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
16558            .collect::<Vec<_>>();
16559
16560        let pass = |what: &str| {
16561            for (index, value) in wanted.iter().enumerate() {
16562                let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
16563                assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
16564                let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
16565                assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
16566            }
16567        };
16568        pass("the first pass");
16569        pass("the second pass");
16570
16571        // The whole vector in one call, over the text and through codes into it, which is how a
16572        // scan of a stored column hands it out. The codes run backwards and repeat so that they are
16573        // neither the positions nor in order.
16574        let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
16575        let mut whole = vec![0i64; wanted.len()];
16576        assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
16577        assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
16578        let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
16579        let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
16580        let mut through = vec![0i64; codes.len()];
16581        assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
16582        for (row, &code) in codes.iter().enumerate() {
16583            assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
16584            let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
16585            assert_eq!(through[row], one as i64, "row {row} a row at a time");
16586        }
16587
16588        // A handful of codes over a column nobody has read yet is short of the table, so the same
16589        // call answers out of the packed ends instead, and has to answer the same.
16590        let fresh = Reader::open(&path).expect("valid directory");
16591        let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
16592        let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
16593        let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
16594        let mut short = vec![0i64; few.len()];
16595        assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
16596        let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
16597        assert_eq!(short, expected, "the packed ends answer what the table answers");
16598        fs::remove_file(path).expect("remove scratch file");
16599    }
16600
16601    /// All three block layouts come back as the same values in the same order.
16602    ///
16603    /// Blocks outside the page are what every file this build writes holds. Blocks that say where
16604    /// they are but sit inside the page behind the order are format 26, and blocks behind one
16605    /// another with only their ends recorded are older still. Nothing in the writer produces the
16606    /// last two any more, so the only way to find out whether the reader still understands those
16607    /// files is to write them here. The
16608    /// bytes go straight into a file with no directory around them, because what is under test is
16609    /// [`open_global_dictionary`], which is handed a page and a file and asks the directory for
16610    /// nothing.
16611    ///
16612    /// Three thousand values so that there are three payload blocks and a partial fourth, which is
16613    /// what makes the last block the one place where a length and an end disagree about what they
16614    /// are counting.
16615    #[test]
16616    fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
16617        let spellings = (0..3_000)
16618            .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
16619            .collect::<Vec<_>>();
16620        let mut read = Vec::new();
16621        for layout in ["outside", "inside", "behind"] {
16622            let mut dictionary = GlobalDictionary::new();
16623            for text in &spellings {
16624                dictionary.code(text).expect("a code for every spelling");
16625            }
16626            dictionary.finish_blocks().expect("the last block encodes");
16627            let order = dictionary.ranked(None).expect("a sorted order");
16628            // Where the blocks go if they start at `from` and follow one another.
16629            let laid = |from: u64| {
16630                let mut at = from;
16631                dictionary
16632                    .blocks
16633                    .iter()
16634                    .map(|block| {
16635                        let place =
16636                            Placed { start: at, length: block.len() as u64, hash: checksum(block) };
16637                        at += block.len() as u64;
16638                        place
16639                    })
16640                    .collect::<Vec<_>>()
16641            };
16642            let payload = dictionary.blocks.concat();
16643            let scattered = layout != "behind";
16644            let (bytes, encoded, offset, length) = if layout == "outside" {
16645                let mut bytes = vec![0; HEADER as usize];
16646                bytes.extend_from_slice(&payload);
16647                let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
16648                    .expect("an encoding");
16649                let offset = bytes.len() as u64;
16650                bytes.extend_from_slice(&encoded.index);
16651                bytes.extend_from_slice(&encoded.ranks);
16652                bytes.extend_from_slice(&encoded.grams);
16653                let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
16654                (bytes, encoded, offset, length)
16655            } else {
16656                // The index is the same length wherever the blocks are, so a first pass says where
16657                // the page ends and the second writes the places that follow it.
16658                let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
16659                    .expect("an encoding");
16660                let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
16661                let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
16662                    .expect("an encoding");
16663                let mut bytes = encoded.index.clone();
16664                bytes.extend_from_slice(&encoded.ranks);
16665                bytes.extend_from_slice(&encoded.grams);
16666                bytes.extend_from_slice(&payload);
16667                let length = bytes.len();
16668                (bytes, encoded, 0, length)
16669            };
16670            let path = path(&format!("blocks-{layout}"));
16671            fs::write(&path, &bytes).expect("the dictionary is written on its own");
16672            let file = Arc::new(File::open(&path).expect("it opens again"));
16673            let page = Page {
16674                offset,
16675                length: u32::try_from(length).expect("a test dictionary is small"),
16676                hash: checksum(&encoded.index),
16677            };
16678            let opened =
16679                open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
16680                    .expect("a dictionary laid out either way opens");
16681            let mut swept: Vec<Vec<u8>> = Vec::new();
16682            let mut at = 0;
16683            while at < opened.len() {
16684                at = opened
16685                    .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
16686                        swept.push(text.to_vec());
16687                        Ok(())
16688                    })
16689                    .expect("a sweep reads");
16690            }
16691            fs::remove_file(&path).expect("clean up");
16692            read.push(swept);
16693        }
16694        let wanted =
16695            spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
16696        assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
16697        assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
16698        assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
16699    }
16700
16701    /// A dictionary at its budget sweeps without keeping, and still answers what it answered.
16702    ///
16703    /// The budget is a quarter of a gigabyte in a running database, which is a fine size for a real
16704    /// column and no size at all for a test, so this opens the same dictionary a second time with a
16705    /// budget of zero. That is the shape of the hundred million row case: `URL` fills the budget
16706    /// somewhere in the middle of itself and everything past that point is read and dropped, which
16707    /// costs the decode again and holds none of it.
16708    #[test]
16709    fn a_dictionary_at_its_budget_sweeps_without_keeping() {
16710        let path = path("dictionary-budget");
16711        let spellings = (0..2_500)
16712            .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
16713            .collect::<Vec<_>>();
16714        let mut writer =
16715            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16716                .expect("new file");
16717        for part in spellings.chunks(1_024) {
16718            writer
16719                .append(
16720                    &Chunk::new(vec![
16721                        Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16722                    ])
16723                    .expect("one column"),
16724                )
16725                .expect("stripe written");
16726        }
16727        writer.finish().expect("commit");
16728
16729        let reader = Reader::open(&path).expect("valid directory");
16730        let page = reader.table.dictionaries[0].expect("a string column has one");
16731        let file = Arc::clone(&reader.file);
16732        let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
16733            .expect("a dictionary opens whatever it may keep");
16734
16735        let resting = starved.footprint();
16736        let mut swept: Vec<Vec<u8>> = Vec::new();
16737        let mut at = 0;
16738        while at < starved.len() {
16739            at = starved
16740                .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
16741                    swept.push(text.to_vec());
16742                    Ok(())
16743                })
16744                .expect("a sweep reads");
16745        }
16746        assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
16747        assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
16748
16749        let generous = reader.dictionary(0).expect("read").expect("a string column has one");
16750        let read = (0..generous.len())
16751            .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
16752            .collect::<Vec<_>>();
16753        assert_eq!(swept, read, "a starved sweep answers what a point read answers");
16754        fs::remove_file(path).expect("remove scratch file");
16755    }
16756
16757    #[test]
16758    fn damaged_membership_cannot_skip_a_string_page() {
16759        let path = path("damaged-membership");
16760        let mut writer = Writer::create(
16761            &path,
16762            "items",
16763            vec![
16764                Field::required("id", LogicalType::Integer),
16765                Field::new("text", LogicalType::Varchar),
16766            ],
16767        )
16768        .expect("new file");
16769        writer.append(&sample()).expect("stripe written");
16770        writer.finish().expect("commit");
16771
16772        let reader = Reader::open(&path).expect("valid directory");
16773        let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
16774        let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
16775        file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
16776        file.write_all(&[255]).expect("damage membership");
16777        let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
16778        assert!(error.message().contains("membership page checksum differs"), "{error}");
16779        fs::remove_file(path).expect("remove scratch file");
16780    }
16781
16782    #[test]
16783    fn membership_delta_stream_is_sorted_exact_and_bounded() {
16784        let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
16785        assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
16786        let encoded = encode_membership(&unique);
16787        assert_eq!(
16788            decode_membership(&encoded).expect("valid membership"),
16789            [4, 9, 72, 900, u32::MAX]
16790        );
16791        // A stripe's index is the union of its parts', so a code in two of them is in it once and
16792        // the result is still one ascending run of deltas.
16793        let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
16794        assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
16795        assert_eq!(
16796            decode_membership(&encode_membership(&merged)).expect("valid membership"),
16797            unique
16798        );
16799        assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
16800        assert!(
16801            decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
16802            "a value past u32 is invalid"
16803        );
16804    }
16805
16806    #[test]
16807    fn a_global_dictionary_may_be_larger_than_one_column_page() {
16808        let dictionary = Page {
16809            offset: HEADER,
16810            length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
16811            hash: 0,
16812        };
16813        let table = Table {
16814            name: "items".to_owned(),
16815            fields: vec![Field::new("text", LogicalType::Varchar)],
16816            stripes: Vec::new(),
16817            rows: 0,
16818            dictionaries: vec![Some(dictionary)],
16819            dictionary_payloads: Vec::new(),
16820            demoted: Vec::new(),
16821            distincts: vec![None],
16822            frequencies: vec![None],
16823            pair_frequencies: Vec::new(),
16824            frequency_texts: Vec::new(),
16825            host_groups: None,
16826            clustering: None,
16827            generation: 1,
16828            sections: Vec::new(),
16829        };
16830        let directory = encode_directory(&table).expect("directory");
16831        let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
16832
16833        let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
16834        assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
16835    }
16836
16837    #[test]
16838    fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
16839        let path = path("constant-codes");
16840        let mut writer =
16841            Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16842                .expect("new file");
16843        let empty = vec![Value::Varchar(String::new()); 1024];
16844        for _ in 0..4 {
16845            let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
16846            writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
16847        }
16848        writer.finish().expect("commit");
16849
16850        let reader = Reader::open(&path).expect("valid directory");
16851        let pages = reader.layout().columns.first().expect("one column").pages;
16852        // This column used to cost four bytes a row, 16,384 of them, the same as a column of four
16853        // thousand distinct URLs would. The cascade calls each part a constant, so what is left is
16854        // a tag, a count and the value, and the row count stops being what drives the number.
16855        assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
16856        let read = reader.read(3, &[0]).expect("the last part back");
16857        assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
16858        assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
16859        fs::remove_file(path).expect("remove scratch file");
16860    }
16861
16862    #[test]
16863    fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
16864        // What a damaged page looks like from here: the cascade decoded, so the bytes are not
16865        // truncated, but the values do not belong to the column the directory says they do.
16866        let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
16867        let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
16868        assert!(format!("{error}").contains("not of its type"), "{error}");
16869        let low = integer::encode(&[i64::MIN]).expect("a chunk");
16870        assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
16871        let zero = integer::encode(&[0]).expect("a chunk");
16872        assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
16873    }
16874
16875    #[test]
16876    fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
16877        // A shift register rather than a run, because an arithmetic run is the one wide shape the
16878        // cascade does shrink. This is what a column with tens of millions of distinct values hands
16879        // over: full width codes with no order to them.
16880        let mut state: u32 = 0x9e37_79b9;
16881        let spread: Vec<u32> = (0..1024)
16882            .map(|_| {
16883                state ^= state << 13;
16884                state ^= state >> 17;
16885                state ^= state << 5;
16886                state
16887            })
16888            .collect();
16889        assert_eq!(encoded_codes(&spread).expect("no failure"), None);
16890        let near: Vec<u32> = (0..1024).collect();
16891        let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
16892        assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
16893    }
16894
16895    /// The columns of a stripe are encoded on whichever thread got to them, so the one thing that
16896    /// must not depend on which thread that was is the file. Two writes of the same rows are
16897    /// compared byte for byte rather than value for value, because a dictionary that two columns
16898    /// somehow shared would still read back correctly and would hand out its codes in the order the
16899    /// threads happened to run in, which is exactly what this is here to catch.
16900    #[test]
16901    fn two_writes_of_the_same_rows_give_the_same_bytes() {
16902        fn written(path: &PathBuf) {
16903            let fields = (0..40)
16904                .map(|column| {
16905                    let ty =
16906                        if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
16907                    Field::new(format!("c{column}"), ty)
16908                })
16909                .collect::<Vec<_>>();
16910            let mut writer = Writer::create(path, "wide", fields).expect("new file");
16911            for part in 0..70_u64 {
16912                let columns = (0..40)
16913                    .map(|column| {
16914                        let values = (0..64_u64)
16915                            .map(|row| {
16916                                let seed = part.wrapping_mul(31).wrapping_add(row);
16917                                if column % 4 == 0 {
16918                                    Value::Varchar(format!("v{}", seed % 17))
16919                                } else {
16920                                    Value::BigInt(i64::try_from(seed % 97).expect("small"))
16921                                }
16922                            })
16923                            .collect::<Vec<_>>();
16924                        let ty = if column % 4 == 0 {
16925                            LogicalType::Varchar
16926                        } else {
16927                            LogicalType::BigInt
16928                        };
16929                        Vector::from_values(ty, &values).expect("a column")
16930                    })
16931                    .collect::<Vec<_>>();
16932                writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
16933            }
16934            writer.finish().expect("commit");
16935        }
16936
16937        let first = path("repeatable-one");
16938        let second = path("repeatable-two");
16939        written(&first);
16940        written(&second);
16941        let left = fs::read(&first).expect("the first file");
16942        let right = fs::read(&second).expect("the second file");
16943        assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
16944        assert!(left == right, "two writes of the same rows differ in their bytes");
16945
16946        // And the rows are still there, since a pair of identically wrong files would pass the
16947        // comparison above on its own.
16948        let reader = Reader::open(&first).expect("valid directory");
16949        assert_eq!(reader.table().rows(), 70 * 64);
16950        let read = reader.read(0, &[0, 1]).expect("the first part back");
16951        assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
16952        assert_eq!(read.value_at(0, 1), Value::BigInt(0));
16953        fs::remove_file(first).expect("remove scratch file");
16954        fs::remove_file(second).expect("remove scratch file");
16955    }
16956
16957    /// Three tables of different shapes in one file, read back by name.
16958    fn three_tables(path: &PathBuf) {
16959        let writer = Writer::create(
16960            path,
16961            "region",
16962            vec![
16963                Field::new("r_key", LogicalType::Integer),
16964                Field::new("r_name", LogicalType::Varchar),
16965            ],
16966        )
16967        .expect("new file");
16968        let mut writer = writer;
16969        writer
16970            .append(
16971                &Chunk::new(vec![
16972                    Vector::from_values(
16973                        LogicalType::Integer,
16974                        &[Value::Integer(0), Value::Integer(1)],
16975                    )
16976                    .expect("keys"),
16977                    Vector::from_values(
16978                        LogicalType::Varchar,
16979                        &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
16980                    )
16981                    .expect("names"),
16982                ])
16983                .expect("two columns"),
16984            )
16985            .expect("a part");
16986        let mut writer = writer
16987            .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
16988            .expect("a second table");
16989        writer
16990            .append(
16991                &Chunk::new(vec![
16992                    Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
16993                ])
16994                .expect("one column"),
16995            )
16996            .expect("a part");
16997        let mut writer =
16998            writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
16999        for part in 0..70_i64 {
17000            let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
17001            writer
17002                .append(
17003                    &Chunk::new(vec![
17004                        Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
17005                    ])
17006                    .expect("one column"),
17007                )
17008                .expect("a part");
17009        }
17010        writer.finish().expect("commit");
17011    }
17012
17013    #[test]
17014    fn three_tables_in_one_file_read_back_by_name() {
17015        let file = path("three-tables");
17016        three_tables(&file);
17017        let catalog = Catalog::open(&file).expect("a committed catalog");
17018        assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
17019
17020        let region = catalog.table("region").expect("the first table");
17021        assert_eq!(region.table().rows(), 2);
17022        assert_eq!(
17023            region.read(0, &[1]).expect("names").value_at(1, 0),
17024            Value::Varchar("ASIA".to_owned())
17025        );
17026
17027        let wide = catalog.table("wide").expect("the third table");
17028        assert_eq!(wide.table().rows(), 70 * 64);
17029        assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
17030
17031        // The middle table is reached without the one after it having been touched, which is what
17032        // a directory per table buys over one directory of everything.
17033        let empty = catalog.table("empty").expect("the second table");
17034        assert_eq!(empty.table().rows(), 1);
17035        assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
17036
17037        fs::remove_file(file).expect("remove scratch file");
17038    }
17039
17040    #[test]
17041    fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
17042        let file = path("three-tables-missing");
17043        three_tables(&file);
17044        let catalog = Catalog::open(&file).expect("a committed catalog");
17045        let error = catalog.table("nation").expect_err("no such table");
17046        assert!(error.message().contains("nation"), "{}", error.message());
17047        fs::remove_file(file).expect("remove scratch file");
17048    }
17049
17050    #[test]
17051    fn a_file_of_three_tables_will_not_open_as_one() {
17052        let file = path("three-tables-unnamed");
17053        three_tables(&file);
17054        let error = Reader::open(&file).expect_err("more than one table");
17055        assert!(error.message().contains("more than one table"), "{}", error.message());
17056        fs::remove_file(file).expect("remove scratch file");
17057    }
17058
17059    /// One column per storage width, because the width is what decides how many bytes a row costs.
17060    #[test]
17061    fn decimals_of_every_storage_width_round_trip() {
17062        let file = path("decimals");
17063        let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
17064        let fields = widths
17065            .iter()
17066            .enumerate()
17067            .map(|(index, (width, scale))| {
17068                Field::new(
17069                    format!("d{index}"),
17070                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
17071                )
17072            })
17073            .collect::<Vec<_>>();
17074        let mut writer = Writer::create(&file, "money", fields).expect("new file");
17075        let rows: [i128; 3] = [-1234, 0, 999];
17076        let columns = widths
17077            .iter()
17078            .map(|(width, scale)| {
17079                let values = rows
17080                    .iter()
17081                    .map(|unscaled| Value::Decimal {
17082                        unscaled: *unscaled,
17083                        width: *width,
17084                        scale: *scale,
17085                    })
17086                    .collect::<Vec<_>>();
17087                Vector::from_values(
17088                    LogicalType::decimal(*width, *scale).expect("a decimal type"),
17089                    &values,
17090                )
17091                .expect("a decimal column")
17092            })
17093            .collect::<Vec<_>>();
17094        writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
17095        writer.finish().expect("commit");
17096
17097        let reader = Reader::open(&file).expect("a committed file");
17098        for (index, (width, scale)) in widths.iter().enumerate() {
17099            assert_eq!(
17100                reader.table().fields()[index].ty,
17101                LogicalType::decimal(*width, *scale).expect("a decimal type"),
17102                "column {index} came back as another type"
17103            );
17104            let column = reader.read(0, &[index]).expect("the column");
17105            for (row, unscaled) in rows.iter().enumerate() {
17106                assert_eq!(
17107                    column.value_at(row, 0),
17108                    Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
17109                    "column {index} row {row}"
17110                );
17111            }
17112        }
17113        fs::remove_file(file).expect("remove scratch file");
17114    }
17115
17116    #[test]
17117    fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
17118        let file = path("two-of-a-name");
17119        let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
17120            .expect("new file");
17121        let error = writer
17122            .next("t", vec![Field::new("a", LogicalType::BigInt)])
17123            .expect_err("the same name twice");
17124        assert!(error.message().contains("same name"), "{}", error.message());
17125        fs::remove_file(file).expect("remove scratch file");
17126    }
17127
17128    #[test]
17129    fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
17130        let file = path("integer-tally");
17131        let mut writer =
17132            Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
17133                .expect("new file");
17134        let mut values = vec![Value::SmallInt(0); 1024];
17135        values[7] = Value::SmallInt(3);
17136        values[99] = Value::SmallInt(-2);
17137        values[1001] = Value::SmallInt(3);
17138        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
17139        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
17140        values[0] = Value::Null;
17141        let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
17142        writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
17143        writer.finish().expect("commit");
17144
17145        let reader = Reader::open(&file).expect("read file");
17146        assert_eq!(
17147            reader.integer_tally(0, 0).expect("valid part"),
17148            Some(vec![(-2, 1), (0, 1021), (3, 2)])
17149        );
17150        assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
17151        let catalog = Catalog::open(&file).expect("catalog");
17152        assert_eq!(
17153            catalog.integer_tally("events", 0).expect("nullable column"),
17154            Some(vec![(-2, 2), (0, 2041), (3, 4)])
17155        );
17156        fs::remove_file(file).expect("remove scratch file");
17157    }
17158
17159    #[test]
17160    fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
17161        let file = path("catalog-integer-tally");
17162        let mut writer = Writer::create(
17163            &file,
17164            "events",
17165            vec![
17166                Field::new("noise", LogicalType::SmallInt),
17167                Field::new("source", LogicalType::SmallInt),
17168            ],
17169        )
17170        .expect("new file");
17171        let noise = vec![Value::SmallInt(9); 1024];
17172        let mut source = vec![Value::SmallInt(0); 1024];
17173        source[7] = Value::SmallInt(3);
17174        source[99] = Value::SmallInt(-2);
17175        let chunk = Chunk::new(vec![
17176            Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
17177            Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
17178        ])
17179        .expect("two columns");
17180        writer.append(&chunk).expect("append");
17181        writer.finish().expect("commit");
17182
17183        let catalog = Catalog::open(&file).expect("catalog");
17184        assert_eq!(
17185            catalog.integer_tally("events", 1).expect("selected column"),
17186            Some(vec![(-2, 1), (0, 1022), (3, 1)])
17187        );
17188        assert_eq!(
17189            catalog.integer_tally("events", 0).expect("other column"),
17190            Some(vec![(9, 1024)])
17191        );
17192        fs::remove_file(file).expect("remove scratch file");
17193    }
17194
17195    #[test]
17196    fn opening_the_catalog_reads_no_table_directory() {
17197        let file = path("catalog-only");
17198        three_tables(&file);
17199        let catalog = Catalog::open(&file).expect("a committed catalog");
17200        // The header and one slot, and nothing under it. The third table's directory covers seventy
17201        // stripes and reading it here would be the whole point of the two levels thrown away.
17202        assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
17203        assert_eq!(catalog.names().len(), 3);
17204        fs::remove_file(file).expect("remove scratch file");
17205    }
17206
17207    /// The checksum answers what it has always answered, at every length its branches split on.
17208    ///
17209    /// This is a compatibility test rather than a correctness one. Nothing about the hash has to be
17210    /// any particular function, but a file already on disk carries the answers the version that
17211    /// wrote it gave, so a change here is a change that makes every stored file fail to verify. The
17212    /// lengths are the ones the code makes decisions about: nothing, under a block, a block exactly,
17213    /// a block and a word, a word and a half word, and a half word and a byte.
17214    ///
17215    /// The empty answer is the published xxHash64 vector for an empty input at seed zero, which is
17216    /// also a check that this is the function it says it is.
17217    #[test]
17218    fn the_checksum_answers_what_it_has_always_answered() {
17219        let bytes: Vec<u8> =
17220            (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
17221        for (length, expected) in [
17222            (0, 0xef46_db37_51d8_e999),
17223            (1, 0xa96c_7f0c_e858_bbb7),
17224            (3, 0x56e6_9576_32a4_87f9),
17225            (4, 0xc60d_15b1_e3ff_8f04),
17226            (5, 0x8088_1585_8624_dd4e),
17227            (7, 0xafbe_fc3d_6c6f_9a8e),
17228            (8, 0x3da5_c7aa_2696_83e0),
17229            (9, 0x465e_c429_b13c_3892),
17230            (15, 0xdee8_9d8a_065a_6233),
17231            (16, 0x1330_489a_7767_9c80),
17232            (31, 0x3391_303d_485e_846e),
17233            (32, 0x40b7_aff7_5d45_bbc8),
17234            (33, 0x4997_cae4_951c_17a5),
17235            (39, 0x5807_28fd_5c14_5739),
17236            (40, 0xf95c_f6f5_c08a_3d3b),
17237            (63, 0x2944_b4da_fc69_b206),
17238            (64, 0xbb76_f6ef_19bd_5a1b),
17239            (65, 0x814e_0c65_4a9f_d640),
17240            (127, 0x00de_aab1_31cf_f89b),
17241            (1000, 0x9e33_00c1_cde3_c58d),
17242        ] {
17243            assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
17244        }
17245        assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
17246    }
17247    /// A declared order survives the file, and a table that declared none stays as it was.
17248    ///
17249    /// The second half is the one worth a test. The clustering section is written only when there
17250    /// is a declaration, so a file of two tables where one is clustered exercises both the present
17251    /// and the absent branch of the decoder in one directory, which is where a length bug would
17252    /// show up as one table reading the other's bytes.
17253    #[test]
17254    fn a_declared_order_comes_back_out_of_the_file() {
17255        let path = path("clustered");
17256        let shipped = vec![
17257            Field::new("key", LogicalType::BigInt),
17258            Field::new("line", LogicalType::Integer),
17259            Field::new("shipdate", LogicalType::Date),
17260        ];
17261        let plain = vec![Field::new("a", LogicalType::Integer)];
17262        let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
17263
17264        let mut writer = Writer::create(&path, "lineitem", shipped)
17265            .expect("new file")
17266            .declare(stage_zero.clone())
17267            .expect("the columns are the table's");
17268        let column = |ty: LogicalType, values: &[Value]| {
17269            Vector::from_values(ty, values).expect("the values match the type")
17270        };
17271        writer
17272            .append(
17273                &Chunk::new(vec![
17274                    column(
17275                        LogicalType::BigInt,
17276                        &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
17277                    ),
17278                    column(
17279                        LogicalType::Integer,
17280                        &[
17281                            Value::Integer(1),
17282                            Value::Integer(1),
17283                            Value::Integer(1),
17284                            Value::Integer(1),
17285                        ],
17286                    ),
17287                    column(
17288                        LogicalType::Date,
17289                        &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
17290                    ),
17291                ])
17292                .expect("three columns"),
17293            )
17294            .expect("four rows");
17295        let mut writer = writer.next("nation", plain).expect("a second table");
17296        writer
17297            .append(
17298                &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
17299                    .expect("one column"),
17300            )
17301            .expect("one row");
17302        writer.finish().expect("commit");
17303
17304        let catalog = Catalog::open(&path).expect("reopen");
17305        let lineitem = catalog.table("lineitem").expect("the clustered table");
17306        assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
17307        let nation = catalog.table("nation").expect("the plain table");
17308        assert_eq!(nation.table().clustering(), None, "nobody declared one here");
17309
17310        // And the rows are still the rows, because the section goes on the end of the directory
17311        // and the easy way to break that is to leave the cursor somewhere the next read trusts.
17312        assert_eq!(lineitem.table().rows(), 4);
17313        assert_eq!(nation.table().rows(), 1);
17314        fs::remove_file(&path).ok();
17315    }
17316
17317    /// A declaration naming a column the table does not have is refused where it is made.
17318    #[test]
17319    fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
17320        let path = path("clustered-bad");
17321        let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
17322            .expect("new file");
17323        let four =
17324            (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
17325        let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
17326        assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
17327        fs::remove_file(&path).ok();
17328    }
17329
17330    /// The sorted order is the byte order, whatever the values do before they differ.
17331    ///
17332    /// A block handed out of the writer's lock to be encoded, and given back in whatever order the
17333    /// stripes happen to finish in, is the same block with the same signature as one encoded in
17334    /// place, and lands in the same position.
17335    #[test]
17336    fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
17337        let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
17338            .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
17339            .collect::<Vec<_>>();
17340        let filled = || {
17341            let mut dictionary = GlobalDictionary::new();
17342            for value in &values {
17343                dictionary.code(value).expect("a code for every value");
17344            }
17345            dictionary.settle().expect("a shape");
17346            dictionary
17347        };
17348        let mut in_place = filled();
17349        in_place.finish_blocks().expect("every block encodes");
17350
17351        let mut handed = filled();
17352        let out = handed.hand_out(3);
17353        assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
17354        assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
17355        for job in out.iter().rev() {
17356            assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
17357            handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
17358        }
17359        assert!(handed.early.is_empty(), "nothing is waiting on a gap");
17360        handed.finish_blocks().expect("the last block encodes");
17361
17362        assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
17363        assert_eq!(handed.grams, in_place.grams, "with the same signatures");
17364    }
17365
17366    /// A block given back twice is a bug in whoever gave it, and is said rather than written twice.
17367    #[test]
17368    fn a_block_given_back_twice_is_refused() {
17369        let mut dictionary = GlobalDictionary::new();
17370        for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
17371            dictionary.code(&format!("value {at}")).expect("a code");
17372        }
17373        dictionary.settle().expect("a shape");
17374        let out = dictionary.hand_out(0);
17375        let last = out.last().expect("blocks went out");
17376        let at = last.place().1;
17377        dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
17378        assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
17379    }
17380
17381    /// The values here are the shape the sort is built for and the shape a comparison sort is worst
17382    /// at: a common scheme, a handful of hosts, and a path that only decides the pair thirty bytes
17383    /// in. They also cover what the bucketing has to get right at the edges, which is a value that
17384    /// has run out where another carries on, the empty value, and enough entries to take the range
17385    /// down through several passes and out the bottom into the comparison that finishes it.
17386    #[test]
17387    fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
17388        let mut values = vec![String::new(), "http://".to_owned()];
17389        for host in 0..7 {
17390            for path in 0..30 {
17391                values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
17392                values.push(format!("http://example{host}.test/page/{path:04}"));
17393            }
17394        }
17395        values.push("http://example0.test/page/0000/index.htmlx".to_owned());
17396
17397        let mut dictionary = GlobalDictionary::new();
17398        for value in &values {
17399            dictionary.code(value).expect("a code for every value");
17400        }
17401        dictionary.finish_blocks().expect("the last block encodes");
17402        let ranked = dictionary.ranked(None).expect("a sorted order");
17403        assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
17404
17405        let spellings = dictionary_values(&dictionary);
17406        let seen = ranked
17407            .iter()
17408            .map(|&(_, code)| {
17409                String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
17410            })
17411            .collect::<Vec<_>>();
17412        let mut wanted = values.clone();
17413        wanted.sort_unstable();
17414        assert_eq!(seen, wanted, "the order is the order the bytes give");
17415
17416        for &(carried, code) in &ranked {
17417            let value = &spellings[code as usize];
17418            assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
17419        }
17420    }
17421
17422    /// Picking the commonest entries leaves exactly what sorting all of them and cutting left.
17423    ///
17424    /// The counts here are deliberately full of ties, including a tie that straddles the cut, which
17425    /// is where a partition and a sort can disagree if the comparison they are given is not total.
17426    #[test]
17427    fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
17428        let entry =
17429            |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
17430        let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
17431            .map(|code| entry(code, u64::from(code % 7) + 1))
17432            .collect::<Vec<_>>();
17433        all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
17434
17435        let mut sorted = all.clone();
17436        sorted.sort_unstable_by(|left, right| {
17437            right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
17438        });
17439        let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
17440        sorted.truncate(FREQUENCY_ENTRIES);
17441
17442        let mut picked = all.clone();
17443        let omitted = keep_most_frequent(&mut picked);
17444        assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
17445        assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
17446        assert!(
17447            picked
17448                .iter()
17449                .zip(&sorted)
17450                .all(|(one, two)| one.value == two.value && one.count == two.count),
17451            "the same entries in the same order"
17452        );
17453
17454        let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
17455        let omitted = keep_most_frequent(&mut short);
17456        assert_eq!(omitted, 0, "nothing is omitted when everything fits");
17457        assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
17458    }
17459
17460    /// A dictionary too small to bucket, and one with nothing in it, come back in order too.
17461    #[test]
17462    fn a_short_dictionary_sorts_without_a_bucketing_pass() {
17463        let empty = GlobalDictionary::new();
17464        assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
17465
17466        let mut dictionary = GlobalDictionary::new();
17467        for value in ["pear", "apple", "", "apples", "app"] {
17468            dictionary.code(value).expect("a code for every value");
17469        }
17470        dictionary.finish_blocks().expect("the one block encodes");
17471        let spellings = dictionary_values(&dictionary);
17472        let seen = dictionary
17473            .ranked(None)
17474            .expect("a sorted order")
17475            .iter()
17476            .map(|&(_, code)| spellings[code as usize].clone())
17477            .collect::<Vec<_>>();
17478        let wanted: Vec<Vec<u8>> =
17479            [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
17480        assert_eq!(seen, wanted, "shorter first where one runs out inside another");
17481    }
17482
17483    /// A demoted dictionary gives back what it kept for looking values up, the load profile is told,
17484    /// and it refuses any value after that.
17485    #[test]
17486    fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
17487        let profile = LoadProfile::begin("demoted");
17488        let mut dictionary = GlobalDictionary::new();
17489        for value in 0..50_000 {
17490            dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
17491        }
17492        let (_, grown) = dictionary.recharge(Some(&profile));
17493        assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
17494
17495        dictionary.demote();
17496        let (before, after) = dictionary.recharge(Some(&profile));
17497        assert_eq!(before, grown);
17498        // What stays is the ends, the counts and the blocks not yet written, which a load writes
17499        // as it goes, so here the drop is the hash tables and the check hashes.
17500        assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
17501        assert_eq!(profile.held(), after, "the profile was told about the drop");
17502        assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
17503
17504        dictionary.demote();
17505        assert_eq!(
17506            dictionary.recharge(Some(&profile)),
17507            (after, after),
17508            "demoting twice is a no-op"
17509        );
17510        assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
17511    }
17512}